Harden offline FG parser and add d2jsp collection tooling
This commit is contained in:
50
config/fg_day_estimates.json
Normal file
50
config/fg_day_estimates.json
Normal file
@@ -0,0 +1,50 @@
|
||||
{
|
||||
"notes": "Estimated Day 1-14 ladder prices derived from Day 3 snapshot using decay multipliers. Use as planning guidance, not exact market truth.",
|
||||
"source_day": 3,
|
||||
"day_multipliers": {
|
||||
"day_1": 1.45,
|
||||
"day_2": 1.2,
|
||||
"day_3": 1.0,
|
||||
"day_4": 0.93,
|
||||
"day_5": 0.88,
|
||||
"day_6": 0.84,
|
||||
"day_7": 0.8,
|
||||
"day_8": 0.77,
|
||||
"day_9": 0.74,
|
||||
"day_10": 0.72,
|
||||
"day_11": 0.7,
|
||||
"day_12": 0.68,
|
||||
"day_13": 0.66,
|
||||
"day_14": 0.64
|
||||
},
|
||||
"base_day_3_fg": {
|
||||
"Cham Rune": 800,
|
||||
"Lo Rune": 900,
|
||||
"Ohm Rune": 600,
|
||||
"Vex Rune": 400,
|
||||
"Gul Rune": 200,
|
||||
"Ist Rune": 200,
|
||||
"Mal Rune": 120,
|
||||
"Um Rune": 90,
|
||||
"Pul Rune": 50,
|
||||
"Lem Rune": 70,
|
||||
"Unid Anni": 400,
|
||||
"Unid Torch": 1500,
|
||||
"Unid Griffon": 2000,
|
||||
"Unid Eth Andy": 2000,
|
||||
"Shako": 400,
|
||||
"Mara 30": 1200,
|
||||
"Mara Mid": 750,
|
||||
"BK 5": 600,
|
||||
"War Traveler": 500,
|
||||
"Death's Fathom": 800,
|
||||
"5/5 Facet": 500,
|
||||
"5@res SC": 300,
|
||||
"20life SC": 100,
|
||||
"7mf SC": 100,
|
||||
"Pcomb SK": 200,
|
||||
"Cold SK": 250,
|
||||
"Java SK": 200,
|
||||
"Light SK": 200
|
||||
}
|
||||
}
|
||||
60
config/fg_prices.json
Normal file
60
config/fg_prices.json
Normal file
@@ -0,0 +1,60 @@
|
||||
{
|
||||
"enabled": true,
|
||||
"notes": "Day 3 ladder snapshot pricing. Values are per item in fg.",
|
||||
"rules": [
|
||||
{ "name": "Cham Rune", "match": "exact", "pattern": "CHAM RUNE", "fg": 800 },
|
||||
{ "name": "Lo Rune", "match": "exact", "pattern": "LO RUNE", "fg": 900 },
|
||||
{ "name": "Ohm Rune", "match": "exact", "pattern": "OHM RUNE", "fg": 600 },
|
||||
{ "name": "Vex Rune", "match": "exact", "pattern": "VEX RUNE", "fg": 400 },
|
||||
{ "name": "Gul Rune", "match": "exact", "pattern": "GUL RUNE", "fg": 200 },
|
||||
{ "name": "Ist Rune", "match": "exact", "pattern": "IST RUNE", "fg": 200 },
|
||||
{ "name": "Mal Rune", "match": "exact", "pattern": "MAL RUNE", "fg": 120 },
|
||||
{ "name": "Um Rune", "match": "exact", "pattern": "UM RUNE", "fg": 90 },
|
||||
{ "name": "Pul Rune", "match": "exact", "pattern": "PUL RUNE", "fg": 50 },
|
||||
{ "name": "Lem Rune", "match": "exact", "pattern": "LEM RUNE", "fg": 70 },
|
||||
|
||||
{ "name": "Unid Anni", "match": "contains", "pattern": "UNID ANNI", "fg": 400 },
|
||||
{ "name": "Unid Diablo Charm", "match": "contains", "pattern": "UNID DIABOLOS", "fg": 2500 },
|
||||
{ "name": "Unid Mephisto Charm", "match": "contains", "pattern": "UNID MEPHISTOS", "fg": 1500 },
|
||||
{ "name": "Unid Baal Charm", "match": "contains", "pattern": "UNID BAALOS", "fg": 1500 },
|
||||
{ "name": "Unid Gheed", "match": "contains", "pattern": "UNID GHEED", "fg": 650 },
|
||||
{ "name": "Unid Arach", "match": "contains", "pattern": "UNID ARACH", "fg": 500 },
|
||||
{ "name": "Unid Andy", "match": "contains", "pattern": "UNID ANDY", "fg": 300 },
|
||||
{ "name": "Unid War Traveler", "match": "contains", "pattern": "UNID WT", "fg": 500 },
|
||||
{ "name": "Unid Death's Fathom", "match": "contains", "pattern": "UNID DF", "fg": 800 },
|
||||
{ "name": "Unid Griffon", "match": "contains", "pattern": "UNID GRIFFON", "fg": 2000 },
|
||||
{ "name": "Unid CoA", "match": "contains", "pattern": "UNID COA", "fg": 600 },
|
||||
{ "name": "Unid Eth Andy", "match": "contains", "pattern": "UNID ETH ANDY", "fg": 2000 },
|
||||
{ "name": "Unid Eth Titans", "match": "contains", "pattern": "UNID ETH TITANS", "fg": 1000 },
|
||||
{ "name": "Unid Eth Treks", "match": "contains", "pattern": "UNID ETH TREKS", "fg": 700 },
|
||||
|
||||
{ "name": "Shako", "match": "contains", "pattern": "HARLEQUIN CREST", "fg": 400 },
|
||||
{ "name": "Occy", "match": "contains", "pattern": "THE OCCULUS", "fg": 250 },
|
||||
{ "name": "Highlord", "match": "contains", "pattern": "HIGHLORD", "fg": 300 },
|
||||
{ "name": "Mara 30", "match": "contains", "pattern": "MARA 30", "fg": 1200 },
|
||||
{ "name": "Mara 29", "match": "contains", "pattern": "MARA 29", "fg": 900 },
|
||||
{ "name": "Mara 27", "match": "contains", "pattern": "MARA 27", "fg": 750 },
|
||||
{ "name": "Mara 26", "match": "contains", "pattern": "MARA 26", "fg": 600 },
|
||||
{ "name": "Mara 20", "match": "contains", "pattern": "MARA 20", "fg": 500 },
|
||||
{ "name": "BK 5", "match": "contains", "pattern": "BK 5", "fg": 600 },
|
||||
{ "name": "BK 4", "match": "contains", "pattern": "BK 4", "fg": 400 },
|
||||
{ "name": "BK 3", "match": "contains", "pattern": "BK 3", "fg": 300 },
|
||||
|
||||
{ "name": "5/5 Light Facet", "match": "contains", "pattern": "5/5 LIGHT FACET", "fg": 500 },
|
||||
{ "name": "5/5 Cold Facet", "match": "contains", "pattern": "5/5 COLD FACET", "fg": 500 },
|
||||
|
||||
{ "name": "Latent Crack of the Heavens", "match": "contains", "pattern": "CRACK OF THE HEAVENS", "fg": 1000 },
|
||||
{ "name": "Latent Black Cleft", "match": "contains", "pattern": "BLACK CLEFT", "fg": 650 },
|
||||
{ "name": "Latent Bone Break", "match": "contains", "pattern": "BONE BREAK", "fg": 650 },
|
||||
{ "name": "Latent Flame Rift", "match": "contains", "pattern": "FLAME RIFT", "fg": 800 },
|
||||
|
||||
{ "name": "5@res SC", "match": "contains", "pattern": "5 RESIST", "fg": 300 },
|
||||
{ "name": "20life SC", "match": "contains", "pattern": "20 LIFE", "fg": 100 },
|
||||
{ "name": "7mf SC", "match": "contains", "pattern": "7 MF", "fg": 100 },
|
||||
|
||||
{ "name": "Pcomb SK", "match": "contains", "pattern": "PALADIN COMBAT", "fg": 200 },
|
||||
{ "name": "Cold SK", "match": "contains", "pattern": "COLD SKILLS", "fg": 250 },
|
||||
{ "name": "Java SK", "match": "contains", "pattern": "JAVELIN", "fg": 200 },
|
||||
{ "name": "Light SK", "match": "contains", "pattern": "LIGHTNING SKILLS", "fg": 200 }
|
||||
]
|
||||
}
|
||||
25
data/d2jsp_topic_urls.txt
Normal file
25
data/d2jsp_topic_urls.txt
Normal file
@@ -0,0 +1,25 @@
|
||||
https://forums.d2jsp.org/topic.php?t=109393834&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109393841&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109393817&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109387092&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109393743&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109393813&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109393750&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109393839&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109393838&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109378166&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109393710&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109343762&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109393837&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109392830&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109325785&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109350751&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109357697&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109392368&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109391282&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109393833&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109393830&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109368065&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109390974&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109393789&f=271
|
||||
https://forums.d2jsp.org/topic.php?t=109393832&f=271
|
||||
57
docs/d2jsp_scraping_guide.md
Normal file
57
docs/d2jsp_scraping_guide.md
Normal file
@@ -0,0 +1,57 @@
|
||||
# d2jsp Semi-Automated Scraping Guide
|
||||
|
||||
Due to Cloudflare's aggressive bot protection, fully automated scraping is currently restricted. This project uses a **Semi-Automated (Offline) Workflow** that leverages your authenticated browser session to safely collect market data.
|
||||
|
||||
## Prerequisites
|
||||
|
||||
1. **Python Dependencies:** Ensure `keyboard` and `requests` are installed (included in `requirements.txt`).
|
||||
2. **Browser:** Google Chrome is recommended.
|
||||
3. **Authentication:** Be logged into [forums.d2jsp.org](https://forums.d2jsp.org/) in your browser.
|
||||
|
||||
---
|
||||
|
||||
## The Workflow
|
||||
|
||||
### 1. Initial Directory Setup
|
||||
Before running the automation, you must set the default "Save As" path in your browser:
|
||||
1. Open any topic on d2jsp.
|
||||
2. Press `Ctrl + S`.
|
||||
3. Navigate to your bot folder: `data/d2jsp_pages/`.
|
||||
4. Save the file. Your browser will now remember this location as the default.
|
||||
|
||||
### 2. Collect Topic URLs
|
||||
If you don't have a fresh list of URLs, run the collector on a saved forum listing page:
|
||||
```powershell
|
||||
python tools/d2jsp_topic_collector.py --offline-dir data/d2jsp_pages --out data/d2jsp_topic_urls.txt
|
||||
```
|
||||
|
||||
### 3. Run Browser Automation
|
||||
This script will open the first 20 topics from your list, wait for Cloudflare to pass, and simulate the save command.
|
||||
```powershell
|
||||
python tools/browser_auto_save.py
|
||||
```
|
||||
**Important:**
|
||||
* Keep your browser as the active window.
|
||||
* Do not move the mouse or type while the script is running.
|
||||
* It will automatically `Ctrl + S` -> `Enter` -> `Ctrl + W` for each tab.
|
||||
|
||||
### 4. Generate Price Estimates
|
||||
Once the HTML files are saved in `data/d2jsp_pages/`, run the offline scraper to update your bot's configuration:
|
||||
```powershell
|
||||
python tools/fg_market_scraper.py --ladder-start-date 2026-05-20 --offline-dir data/d2jsp_pages --out config/fg_daily_estimates.json
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### Cloudflare "Just a Moment" Loop
|
||||
If the automation is too fast and hits a "Just a Moment" screen that doesn't resolve:
|
||||
1. Increase the `time.sleep(8)` value in `tools/browser_auto_save.py`.
|
||||
2. Manually solve one challenge in the browser to "warm up" the IP clearance.
|
||||
|
||||
### Files Not Saving to Correct Folder
|
||||
If files are saving to your "Downloads" folder instead of `data/d2jsp_pages`, the browser's default path was reset. Repeat **Step 1** to fix it.
|
||||
|
||||
### Date Parsing Errors
|
||||
If the scraper reports 0 topics scanned or dates are missing, ensure your browser language isn't translating the page, as the scraper expects English month names (Jan, Feb, Mar, etc.).
|
||||
44
docs/fg_market_scraper.md
Normal file
44
docs/fg_market_scraper.md
Normal file
@@ -0,0 +1,44 @@
|
||||
# FG Market Scraper (Day 1-14)
|
||||
|
||||
This tool estimates FG prices per ladder day by scraping public trade topics from:
|
||||
- https://forums.d2jsp.org/forum.php?f=271
|
||||
|
||||
## Script
|
||||
- `tools/fg_market_scraper.py`
|
||||
|
||||
## What it does
|
||||
1. Scans forum listing pages for topic links.
|
||||
2. Fetches topic pages (rate-limited).
|
||||
3. Detects known item/rune keywords.
|
||||
4. Extracts `fg` prices from post text.
|
||||
5. Buckets prices by ladder day index (Day 1..Day N).
|
||||
6. Writes median estimates + sample counts.
|
||||
|
||||
## Usage
|
||||
From repo root:
|
||||
|
||||
```powershell
|
||||
python tools/fg_market_scraper.py --ladder-start-date 2026-05-23 --days 14
|
||||
```
|
||||
|
||||
Output:
|
||||
- `config/fg_daily_estimates.json`
|
||||
|
||||
## Tuning
|
||||
- `--max-forum-pages`: listing pages to scan (default 40)
|
||||
- `--max-topics`: hard cap on fetched topics (default 800)
|
||||
- `--delay-s`: delay between requests (default 0.35s)
|
||||
|
||||
Example heavier run:
|
||||
|
||||
```powershell
|
||||
python tools/fg_market_scraper.py --ladder-start-date 2026-05-23 --days 14 --max-forum-pages 120 --max-topics 2400 --delay-s 0.5
|
||||
```
|
||||
|
||||
## Notes
|
||||
- This is a heuristic estimator, not a full market engine.
|
||||
- Accuracy depends on post format quality and keyword matches.
|
||||
- Keep request rate polite to avoid stressing the forum.
|
||||
- Some environments/IPs will receive HTTP `403` from d2jsp. In that case use:
|
||||
- `config/fg_day_estimates.json` (Day 1-14 estimated multiplier model from Day 3 snapshot),
|
||||
- and update `config/fg_prices.json` manually from your current market sample.
|
||||
@@ -132,7 +132,17 @@ class FoHdin(Paladin):
|
||||
pindle_pos_abs = convert_screen_to_abs(Config().path["pindle_end"][0])
|
||||
|
||||
cast_pos_abs = [pindle_pos_abs[0] * 0.80, pindle_pos_abs[1] * 0.80]
|
||||
self._generic_foh_attack_sequence(default_target_abs=cast_pos_abs, min_duration=atk_len_dur, max_duration=atk_len_dur, default_spray=10, target_detect=False, foh_to_holy_bolt_ratio=1)
|
||||
# Pindle is a short burst fight: favor repeated FoH casts over Holy Bolt alternation.
|
||||
# Also enforce a minimum cast window so we don't exit after a single FoH on low atk_len values.
|
||||
pindle_attack_window = max(atk_len_dur, self._foh_cycle_duration * 3)
|
||||
self._generic_foh_attack_sequence(
|
||||
default_target_abs=cast_pos_abs,
|
||||
min_duration=pindle_attack_window,
|
||||
max_duration=pindle_attack_window,
|
||||
default_spray=10,
|
||||
target_detect=False,
|
||||
foh_to_holy_bolt_ratio=0,
|
||||
)
|
||||
|
||||
if self.capabilities.can_teleport_natively:
|
||||
self._pather.traverse_nodes_fixed("pindle_end", self)
|
||||
|
||||
55
src/fg_price_tracker.py
Normal file
55
src/fg_price_tracker.py
Normal file
@@ -0,0 +1,55 @@
|
||||
import json
|
||||
import os
|
||||
|
||||
from logger import Logger
|
||||
|
||||
|
||||
class FGPriceTracker:
|
||||
def __init__(self, cfg_path: str = "config/fg_prices.json"):
|
||||
self._cfg_path = cfg_path
|
||||
self._rules: list[dict] = []
|
||||
self._enabled = False
|
||||
self._load()
|
||||
|
||||
@staticmethod
|
||||
def _normalize(text: str) -> str:
|
||||
return " ".join((text or "").strip().upper().split())
|
||||
|
||||
def _load(self):
|
||||
cfg_candidates = [self._cfg_path, os.path.join("..", self._cfg_path)]
|
||||
cfg_path = next((p for p in cfg_candidates if os.path.exists(p)), None)
|
||||
if not cfg_path:
|
||||
Logger.info(f"FG tracker config not found at {self._cfg_path}, tracker disabled")
|
||||
return
|
||||
try:
|
||||
with open(cfg_path, "r", encoding="utf-8") as f:
|
||||
data = json.load(f)
|
||||
self._enabled = bool(data.get("enabled", True))
|
||||
raw_rules = data.get("rules", [])
|
||||
parsed = []
|
||||
for rule in raw_rules:
|
||||
parsed.append({
|
||||
"name": rule.get("name", ""),
|
||||
"match": str(rule.get("match", "exact")).lower(),
|
||||
"fg": float(rule.get("fg", 0)),
|
||||
"pattern": self._normalize(rule.get("pattern", "")),
|
||||
})
|
||||
self._rules = [r for r in parsed if r["pattern"] and r["fg"] > 0]
|
||||
Logger.info(f"FG tracker loaded {len(self._rules)} pricing rules")
|
||||
except Exception as e:
|
||||
Logger.warning(f"Failed to load FG tracker config '{self._cfg_path}': {e}")
|
||||
self._enabled = False
|
||||
self._rules = []
|
||||
|
||||
def estimate(self, item_name: str) -> tuple[float, str] | tuple[None, None]:
|
||||
if not self._enabled or not self._rules:
|
||||
return None, None
|
||||
normalized = self._normalize(item_name)
|
||||
for rule in self._rules:
|
||||
if rule["match"] == "contains":
|
||||
if rule["pattern"] in normalized:
|
||||
return rule["fg"], rule["name"] or rule["pattern"]
|
||||
else:
|
||||
if normalized == rule["pattern"]:
|
||||
return rule["fg"], rule["name"] or rule["pattern"]
|
||||
return None, None
|
||||
@@ -10,6 +10,7 @@ from beautifultable import BeautifulTable
|
||||
from logger import Logger
|
||||
from config import Config
|
||||
from messages import Messenger
|
||||
from fg_price_tracker import FGPriceTracker
|
||||
from utils.misc import hms
|
||||
from utils.levels import get_level, get_level_from_exp
|
||||
from version import __version__
|
||||
@@ -65,13 +66,18 @@ class GameStats:
|
||||
"merc_deaths": 0,
|
||||
"failed_runs": 0,
|
||||
"runes": 0,
|
||||
"fg_estimated": 0.0,
|
||||
}
|
||||
self._valuable_item_counts = {name: 0 for name in self._TOP_VALUABLE_ITEMS}
|
||||
self._fg_tracker = FGPriceTracker()
|
||||
self._fg_item_counts = {}
|
||||
self._fg_item_totals = {}
|
||||
self._stats_filename = f'stats_{time.strftime("%Y%m%d_%H%M%S")}.log'
|
||||
self._events_filename = f'events_{time.strftime("%Y%m%d_%H%M%S")}.jsonl'
|
||||
self._mini_stats_filename = f'mini_stats_{time.strftime("%Y%m%d_%H%M%S")}.json'
|
||||
self._nopickup_active = False
|
||||
self._starting_exp = 0
|
||||
self._current_exp = 1
|
||||
self._current_exp = 0
|
||||
self._current_lvl = 0
|
||||
self._exp_logging_disabled = False
|
||||
self._exp_logging_error_warned = False
|
||||
@@ -99,6 +105,7 @@ class GameStats:
|
||||
# Keep a crash-resilient human-readable snapshot updated throughout a session.
|
||||
try:
|
||||
self._save_stats_to_file()
|
||||
self._save_mini_stats_to_file()
|
||||
except Exception as e:
|
||||
Logger.warning(f"Failed to persist stats snapshot: {e}")
|
||||
|
||||
@@ -117,6 +124,8 @@ class GameStats:
|
||||
"item_counts": {},
|
||||
"rune_counts": {},
|
||||
"valuable_counts": {},
|
||||
"fg_item_counts": {},
|
||||
"fg_item_totals": {},
|
||||
"deaths": 0,
|
||||
"chickens": 0,
|
||||
"merc_deaths": 0,
|
||||
@@ -137,6 +146,8 @@ class GameStats:
|
||||
"CHIPPED SKULL", "FLAWED SKULL", "SKULL", "FLAWLESS SKULL", "PERFECT SKULL"]
|
||||
skip_log = any(substring in item_name for substring in filtered_substrings) or any(match == item_name.strip() for match in filtered_matches)
|
||||
normalized_name = self._normalize_item_name(item_name)
|
||||
fg_value = None
|
||||
fg_name = None
|
||||
if self._location is not None and not skip_log:
|
||||
Logger.debug(f"Stashed and logged: {item_name}")
|
||||
self._location_stats[self._location]["items"].append(item_name)
|
||||
@@ -153,6 +164,16 @@ class GameStats:
|
||||
valuable_counts[normalized_name] = valuable_counts.get(normalized_name, 0) + 1
|
||||
self._valuable_item_counts[normalized_name] += 1
|
||||
self._log_event("valuable_item_kept", {"item_name": normalized_name})
|
||||
fg_value, fg_name = self._fg_tracker.estimate(normalized_name)
|
||||
if fg_value is not None and fg_name:
|
||||
fg_counts = self._location_stats[self._location]["fg_item_counts"]
|
||||
fg_totals = self._location_stats[self._location]["fg_item_totals"]
|
||||
fg_counts[fg_name] = fg_counts.get(fg_name, 0) + 1
|
||||
fg_totals[fg_name] = fg_totals.get(fg_name, 0.0) + fg_value
|
||||
self._fg_item_counts[fg_name] = self._fg_item_counts.get(fg_name, 0) + 1
|
||||
self._fg_item_totals[fg_name] = self._fg_item_totals.get(fg_name, 0.0) + fg_value
|
||||
self._location_stats["totals"]["fg_estimated"] += fg_value
|
||||
self._log_event("fg_item_kept", {"item_name": normalized_name, "fg_name": fg_name, "fg_value": fg_value})
|
||||
self._log_event("item_kept", {"item_name": item_name, "expression": expression})
|
||||
self._persist_snapshot()
|
||||
elif self._location is not None and skip_log:
|
||||
@@ -160,7 +181,10 @@ class GameStats:
|
||||
|
||||
if send_message and self._messenger.enabled and not skip_log:
|
||||
if expression[0] != "@":
|
||||
self._messenger.send_item(item_name, img, self._location, ocr_text, expression, item_props)
|
||||
expression_with_fg = expression
|
||||
if fg_value is not None and fg_name:
|
||||
expression_with_fg = f"{expression} | est_fg={fg_value:.1f}"
|
||||
self._messenger.send_item(item_name, img, self._location, ocr_text, expression_with_fg, item_props)
|
||||
|
||||
def log_death(self, img: str):
|
||||
self._death_counter += 1
|
||||
@@ -307,15 +331,16 @@ class GameStats:
|
||||
avg_length = good_games_time / float(good_games_count)
|
||||
avg_length_str = hms(avg_length)
|
||||
|
||||
curr_lvl = get_level(self._current_lvl)
|
||||
curr_lvl = get_level(self._current_lvl) if self._current_lvl > 0 else { "lvl": 0, "exp": 0, "xp_to_next": 0 }
|
||||
|
||||
msg = f'\nSession length: {elapsed_time_str}'
|
||||
msg += f'\nGames: {self._game_counter}'
|
||||
msg += f'\nAvg Game Length: {avg_length_str}'
|
||||
msg += f'\nCurrent Level: {curr_lvl["lvl"]}'
|
||||
msg += f'\nCurrent Level: {curr_lvl["lvl"] if curr_lvl["lvl"] > 0 else "n/a"}'
|
||||
msg += f'\nRunes Kept: {self._location_stats["totals"]["runes"]}'
|
||||
msg += f'\nEstimated FG: {self._location_stats["totals"]["fg_estimated"]:.1f}'
|
||||
|
||||
if curr_lvl["lvl"] < 99:
|
||||
if curr_lvl["lvl"] > 0 and curr_lvl["lvl"] < 99 and self._current_exp > 0:
|
||||
try:
|
||||
exp_gained = self._current_exp - curr_lvl['exp']
|
||||
gained_exp = self._current_exp - self._starting_exp
|
||||
@@ -400,14 +425,56 @@ class GameStats:
|
||||
sorted_valuables = sorted(stats["valuable_counts"].items(), key=lambda x: x[1], reverse=True)
|
||||
for valuable_name, count in sorted_valuables:
|
||||
msg += f"\n {valuable_name}: {count}"
|
||||
if len(stats["fg_item_totals"]) > 0:
|
||||
msg += "\n FG tracked items:"
|
||||
sorted_fg = sorted(stats["fg_item_totals"].items(), key=lambda x: x[1], reverse=True)
|
||||
for fg_name, fg_total in sorted_fg:
|
||||
count = stats["fg_item_counts"].get(fg_name, 0)
|
||||
msg += f"\n {fg_name}: {count}x, {fg_total:.1f} fg"
|
||||
|
||||
msg += "\n\nTop-10 valuable items (session totals):"
|
||||
for item_name in self._TOP_VALUABLE_ITEMS:
|
||||
msg += f"\n {item_name}: {self._valuable_item_counts[item_name]}"
|
||||
if len(self._fg_item_totals) > 0:
|
||||
msg += "\n\nFG tracker (session totals):"
|
||||
sorted_fg_totals = sorted(self._fg_item_totals.items(), key=lambda x: x[1], reverse=True)
|
||||
for fg_name, fg_total in sorted_fg_totals:
|
||||
count = self._fg_item_counts.get(fg_name, 0)
|
||||
msg += f"\n {fg_name}: {count}x, {fg_total:.1f} fg"
|
||||
|
||||
with open(file=f"log/stats/{self._stats_filename}", mode="w+", encoding="utf-8") as f:
|
||||
f.write(msg)
|
||||
|
||||
def _save_mini_stats_to_file(self):
|
||||
top_items = sorted(
|
||||
self._fg_item_totals.items(),
|
||||
key=lambda x: x[1],
|
||||
reverse=True
|
||||
)[:10]
|
||||
payload = {
|
||||
"ts": time.strftime("%Y-%m-%dT%H:%M:%S"),
|
||||
"session_seconds": round(time.time() - self._start_time, 2),
|
||||
"games": self._game_counter,
|
||||
"run_counter": self._run_counter,
|
||||
"runs_failed_total": self._runs_failed,
|
||||
"runs_failed_consecutive": self._consecutive_runs_failed,
|
||||
"current_level": self._current_lvl if self._current_lvl > 0 else None,
|
||||
"current_exp": self._current_exp if self._current_exp > 0 else None,
|
||||
"runes_kept": self._location_stats["totals"]["runes"],
|
||||
"estimated_fg_total": round(self._location_stats["totals"]["fg_estimated"], 1),
|
||||
"current_location": self._location,
|
||||
"top_fg_items": [
|
||||
{
|
||||
"item": item_name,
|
||||
"count": self._fg_item_counts.get(item_name, 0),
|
||||
"fg_total": round(fg_total, 1),
|
||||
}
|
||||
for item_name, fg_total in top_items
|
||||
],
|
||||
}
|
||||
with open(file=f"log/stats/{self._mini_stats_filename}", mode="w+", encoding="utf-8") as f:
|
||||
json.dump(payload, f, indent=2)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
game_stats = GameStats()
|
||||
|
||||
@@ -98,7 +98,7 @@ def gamble():
|
||||
Logger.warning("gamble: gamble vendor window not detected")
|
||||
return False
|
||||
|
||||
def buy_item(template_name: str, quantity: int = 1, img: np.ndarray = None, shift_click: bool = False) -> bool:
|
||||
def buy_item(template_name: str, quantity: int = 1, img: np.ndarray = None, shift_click: bool = False, log_missing: bool = True) -> bool:
|
||||
"""
|
||||
Buy desired item from vendors. Vendor inventory needs to be open!
|
||||
:param template_name: Name of template for desired item to buy; e.g., SUPER_MANA_POTION
|
||||
@@ -139,5 +139,6 @@ def buy_item(template_name: str, quantity: int = 1, img: np.ndarray = None, shif
|
||||
else:
|
||||
Logger.error("buy_item: Quantity not specified")
|
||||
return False
|
||||
Logger.error(f"buy_item: Desired item {template_name} not found")
|
||||
if log_missing:
|
||||
Logger.error(f"buy_item: Desired item {template_name} not found")
|
||||
return False
|
||||
|
||||
@@ -101,8 +101,10 @@ class TownManager:
|
||||
if not (new_loc and common.wait_for_left_inventory()): return False, items
|
||||
img=grab()
|
||||
def buy_best_available_potion(potion_templates: list[str], quantity: int, shift_click: bool) -> bool:
|
||||
for template_name in potion_templates:
|
||||
if vendor.buy_item(template_name=template_name, quantity=quantity, shift_click=shift_click, img=img):
|
||||
for idx, template_name in enumerate(potion_templates):
|
||||
# Intermediate fallback checks should not spam "not found" errors.
|
||||
log_missing = idx == (len(potion_templates) - 1)
|
||||
if vendor.buy_item(template_name=template_name, quantity=quantity, shift_click=shift_click, img=img, log_missing=log_missing):
|
||||
if template_name not in ["SUPER_HEALING_POTION", "SUPER_MANA_POTION"]:
|
||||
Logger.info(f"buy_consumables: Falling back to {template_name.replace('_', ' ')}")
|
||||
return True
|
||||
|
||||
@@ -22,7 +22,30 @@ def _parse_ocr_int(token: str) -> int:
|
||||
|
||||
def _extract_experience_values(text: str):
|
||||
upper = (text or "").upper()
|
||||
# Keep only number-like tokens from OCR and parse current/required XP from the first pair.
|
||||
# Prefer parsing left/right side around "/" to avoid grabbing stray OCR digits.
|
||||
if "/" in upper:
|
||||
left, right = upper.split("/", 1)
|
||||
left_tokens = re.findall(r"[0-9Il|Oo][0-9Il|Oo,\.]*", left)
|
||||
right_tokens = re.findall(r"[0-9Il|Oo][0-9Il|Oo,\.]*", right)
|
||||
left_val = 0
|
||||
right_val = 0
|
||||
if left_tokens:
|
||||
try:
|
||||
left_val = max((_parse_ocr_int(t) for t in left_tokens), default=0)
|
||||
except Exception:
|
||||
left_val = 0
|
||||
if right_tokens:
|
||||
try:
|
||||
right_val = max((_parse_ocr_int(t) for t in right_tokens), default=0)
|
||||
except Exception:
|
||||
right_val = 0
|
||||
if left_val > 0 and right_val > 0:
|
||||
# Common OCR artifact: one extra trailing digit on current XP.
|
||||
while left_val > right_val and left_val >= 10:
|
||||
left_val //= 10
|
||||
return left_val, right_val
|
||||
|
||||
# Fallback parser when slash split fails.
|
||||
tokens = re.findall(r"[0-9Il|Oo][0-9Il|Oo,\.]*", upper)
|
||||
numbers = []
|
||||
for token in tokens:
|
||||
@@ -31,7 +54,10 @@ def _extract_experience_values(text: str):
|
||||
except Exception:
|
||||
continue
|
||||
if len(numbers) >= 2:
|
||||
return numbers[0], numbers[1]
|
||||
current_exp, required_exp = numbers[0], numbers[1]
|
||||
while current_exp > required_exp and current_exp >= 10:
|
||||
current_exp //= 10
|
||||
return current_exp, required_exp
|
||||
return 0, 0
|
||||
|
||||
|
||||
|
||||
109
tools/d2jsp_fetch_topics_with_cookies.py
Normal file
109
tools/d2jsp_fetch_topics_with_cookies.py
Normal file
@@ -0,0 +1,109 @@
|
||||
import argparse
|
||||
import re
|
||||
import time
|
||||
from pathlib import Path
|
||||
from http.cookiejar import MozillaCookieJar
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
def topic_id_from_url(url: str) -> str:
|
||||
m = re.search(r"[?&]t=(\d+)", url)
|
||||
return m.group(1) if m else "unknown"
|
||||
|
||||
|
||||
def load_urls(url_file: Path, limit: int) -> list[str]:
|
||||
urls = []
|
||||
for line in url_file.read_text(encoding="utf-8", errors="ignore").splitlines():
|
||||
line = line.strip()
|
||||
if not line or line.startswith("#"):
|
||||
continue
|
||||
urls.append(line)
|
||||
if limit > 0:
|
||||
urls = urls[:limit]
|
||||
return urls
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="Fetch d2jsp topic pages using exported cookies.txt")
|
||||
parser.add_argument("--cookies-file", required=True, help="Path to Netscape cookies.txt export")
|
||||
parser.add_argument("--url-file", default="data/d2jsp_topic_urls.txt", help="Text file with topic URLs")
|
||||
parser.add_argument("--out-dir", default="data/d2jsp_pages", help="Output directory for saved HTML pages")
|
||||
parser.add_argument("--limit", type=int, default=0, help="Max number of topic URLs to fetch (0=all)")
|
||||
parser.add_argument("--delay-s", type=float, default=0.6, help="Delay between requests")
|
||||
parser.add_argument("--overwrite", action="store_true", help="Overwrite existing topic HTML files")
|
||||
args = parser.parse_args()
|
||||
|
||||
cookies_path = Path(args.cookies_file)
|
||||
if not cookies_path.exists():
|
||||
raise RuntimeError(f"cookies-file not found: {cookies_path}")
|
||||
|
||||
url_path = Path(args.url_file)
|
||||
if not url_path.exists():
|
||||
raise RuntimeError(f"url-file not found: {url_path}")
|
||||
|
||||
out_dir = Path(args.out_dir)
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
urls = load_urls(url_path, args.limit)
|
||||
if not urls:
|
||||
raise RuntimeError("No URLs found in url-file")
|
||||
|
||||
cookiejar = MozillaCookieJar()
|
||||
cookiejar.load(str(cookies_path), ignore_discard=True, ignore_expires=True)
|
||||
|
||||
session = requests.Session()
|
||||
session.cookies = cookiejar
|
||||
session.headers.update(
|
||||
{
|
||||
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/148.0.0.0 Safari/537.36",
|
||||
"Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8,application/signed-exchange;v=b3;q=0.7",
|
||||
"Accept-Language": "en-DK,en;q=0.9,da-DK;q=0.8,da;q=0.7,en-GB;q=0.6,en-US;q=0.5",
|
||||
"Referer": "https://forums.d2jsp.org/forum.php?f=271",
|
||||
"Cache-Control": "max-age=0",
|
||||
"sec-ch-ua": '"Chromium";v="148", "Google Chrome";v="148", "Not/A)Brand";v="99"',
|
||||
"sec-ch-ua-mobile": "?0",
|
||||
"sec-ch-ua-platform": '"Windows"',
|
||||
"sec-fetch-dest": "document",
|
||||
"sec-fetch-mode": "navigate",
|
||||
"sec-fetch-site": "none",
|
||||
"sec-fetch-user": "?1",
|
||||
"upgrade-insecure-requests": "1",
|
||||
}
|
||||
)
|
||||
|
||||
ok = 0
|
||||
failed = 0
|
||||
for idx, url in enumerate(urls, start=1):
|
||||
topic_id = topic_id_from_url(url)
|
||||
out_file = out_dir / f"topic_{topic_id}.html"
|
||||
if out_file.exists() and not args.overwrite:
|
||||
print(f"[{idx}/{len(urls)}] skip existing {out_file.name}")
|
||||
continue
|
||||
try:
|
||||
r = session.get(url, timeout=25)
|
||||
status = r.status_code
|
||||
if status == 200 and len(r.text) > 2000:
|
||||
out_file.write_text(r.text, encoding="utf-8", errors="ignore")
|
||||
ok += 1
|
||||
print(f"[{idx}/{len(urls)}] OK {topic_id} ({len(r.text)} bytes)")
|
||||
else:
|
||||
failed += 1
|
||||
print(f"[{idx}/{len(urls)}] FAIL {topic_id} status={status} len={len(r.text)}")
|
||||
if status == 403:
|
||||
debug_file = out_dir / f"debug_403_{topic_id}.html"
|
||||
debug_file.write_text(r.text, encoding="utf-8", errors="ignore")
|
||||
print(f" Cloudflare block detected. Saved response to {debug_file}")
|
||||
if "Just a moment..." in r.text:
|
||||
print(" INFO: This is a Cloudflare 'Just a moment' challenge.")
|
||||
print(" To fix: You must find the 'cf_clearance' cookie in your browser's Network tab and add it.")
|
||||
except Exception as e:
|
||||
failed += 1
|
||||
print(f"[{idx}/{len(urls)}] ERROR {topic_id}: {e}")
|
||||
time.sleep(args.delay_s)
|
||||
|
||||
print(f"Done. saved={ok}, failed={failed}, out_dir={out_dir}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
78
tools/d2jsp_topic_collector.py
Normal file
78
tools/d2jsp_topic_collector.py
Normal file
@@ -0,0 +1,78 @@
|
||||
import argparse
|
||||
import re
|
||||
import time
|
||||
import webbrowser
|
||||
from html import unescape
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
BASE_URL = "https://forums.d2jsp.org"
|
||||
|
||||
|
||||
def parse_topic_links(html: str) -> list[str]:
|
||||
matches = []
|
||||
matches.extend(re.findall(r'href="(/topic\.php\?t=\d+[^"]*)"', html))
|
||||
matches.extend(re.findall(r'href="(https?://forums\.d2jsp\.org/topic\.php\?t=\d+[^"]*)"', html))
|
||||
out = []
|
||||
seen = set()
|
||||
for m in matches:
|
||||
href = unescape(m)
|
||||
url = href if href.startswith("http") else f"{BASE_URL}{href}"
|
||||
# Canonicalize by topic id to avoid duplicate entries like &v=1.
|
||||
topic_id_match = re.search(r"[?&]t=(\d+)", url)
|
||||
topic_id = topic_id_match.group(1) if topic_id_match else url
|
||||
canonical = f"{BASE_URL}/topic.php?t={topic_id}&f=271"
|
||||
if canonical not in seen:
|
||||
seen.add(canonical)
|
||||
out.append(canonical)
|
||||
return out
|
||||
|
||||
|
||||
def collect_from_dir(offline_dir: Path) -> list[str]:
|
||||
files = sorted(list(offline_dir.glob("*.html")) + list(offline_dir.glob("*.htm")))
|
||||
urls = []
|
||||
seen = set()
|
||||
for f in files:
|
||||
text = f.read_text(encoding="utf-8", errors="ignore")
|
||||
for url in parse_topic_links(text):
|
||||
if url not in seen:
|
||||
seen.add(url)
|
||||
urls.append(url)
|
||||
return urls
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="Extract d2jsp topic URLs from saved forum listing HTML.")
|
||||
parser.add_argument("--offline-dir", default="data/d2jsp_pages", help="Directory containing saved listing HTML")
|
||||
parser.add_argument("--out", default="data/d2jsp_topic_urls.txt", help="Output text file for topic URLs")
|
||||
parser.add_argument("--open", action="store_true", help="Open extracted topic URLs in browser tabs")
|
||||
parser.add_argument("--limit", type=int, default=0, help="Max URLs to output/open (0 = all)")
|
||||
parser.add_argument("--batch-size", type=int, default=20, help="Open tabs in batches")
|
||||
parser.add_argument("--batch-delay-s", type=float, default=8.0, help="Delay between batches")
|
||||
args = parser.parse_args()
|
||||
|
||||
offline_dir = Path(args.offline_dir)
|
||||
if not offline_dir.exists() or not offline_dir.is_dir():
|
||||
raise RuntimeError(f"offline-dir does not exist or is not a directory: {offline_dir}")
|
||||
|
||||
urls = collect_from_dir(offline_dir)
|
||||
if args.limit > 0:
|
||||
urls = urls[: args.limit]
|
||||
|
||||
out_path = Path(args.out)
|
||||
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
out_path.write_text("\n".join(urls) + ("\n" if urls else ""), encoding="utf-8")
|
||||
print(f"Extracted {len(urls)} topic URLs -> {out_path}")
|
||||
|
||||
if args.open and urls:
|
||||
total = len(urls)
|
||||
for idx, url in enumerate(urls, start=1):
|
||||
webbrowser.open_new_tab(url)
|
||||
if idx % max(1, args.batch_size) == 0 and idx < total:
|
||||
print(f"Opened {idx}/{total} tabs. Waiting {args.batch_delay_s:.1f}s before next batch...")
|
||||
time.sleep(args.batch_delay_s)
|
||||
print(f"Finished opening {total} topic tabs.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
355
tools/fg_market_scraper.py
Normal file
355
tools/fg_market_scraper.py
Normal file
@@ -0,0 +1,355 @@
|
||||
import argparse
|
||||
import datetime as dt
|
||||
import json
|
||||
import re
|
||||
import statistics
|
||||
import time
|
||||
from html import unescape
|
||||
from pathlib import Path
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
BASE_URL = "https://forums.d2jsp.org"
|
||||
FORUM_PATH = "/forum.php?f=271"
|
||||
|
||||
|
||||
ITEM_PATTERNS = {
|
||||
"Cham Rune": [r"\bcham\b"],
|
||||
"Lo Rune": [r"\blo\b"],
|
||||
"Ohm Rune": [r"\bohm\b"],
|
||||
"Vex Rune": [r"\bvex\b"],
|
||||
"Gul Rune": [r"\bgul\b"],
|
||||
"Ist Rune": [r"\bist\b"],
|
||||
"Mal Rune": [r"\bmal\b"],
|
||||
"Um Rune": [r"\bum\b"],
|
||||
"Pul Rune": [r"\bpul\b"],
|
||||
"Lem Rune": [r"\blem\b"],
|
||||
"Unid Anni": [r"\bunid\s+anni\b", r"\bunid\s+annihilus\b"],
|
||||
"Unid Torch": [r"\bunid\s+(diab?olos|mephistos|baalos|torch)\b", r"\bunid\s+torch\b"],
|
||||
"Griffon": [r"\bgriffon"],
|
||||
"Shako": [r"\bshako\b", r"\bharlequin crest\b"],
|
||||
"Mara": [r"\bmara"],
|
||||
"Highlord": [r"\bhighlord"],
|
||||
"BK Ring": [r"\bbk\b", r"\bbul[\s\-']*kathos"],
|
||||
"War Traveler": [r"\bwar traveler\b", r"\bwt\b"],
|
||||
"Death's Fathom": [r"\bdeath'?s fathom\b", r"\bdf\b"],
|
||||
"Arach": [r"\barach", r"\barachnid mesh\b"],
|
||||
"5/5 Facet": [r"\b5\s*/\s*5\b.*\bfacet\b"],
|
||||
"Pcomb SK": [r"\bpcomb\b", r"\bpaladin combat\b"],
|
||||
"Cold SK": [r"\bcold sk\b", r"\bcold skills\b"],
|
||||
"Java SK": [r"\bjava sk\b", r"\bjavelin\b"],
|
||||
"Light SK": [r"\blight sk\b", r"\blightning skills\b"],
|
||||
"CTA": [r"\bcta\b", r"\bcall to arms\b"],
|
||||
"Sorc Torch": [r"\bsorc torch\b", r"\bsorc\s+torc?h\b"],
|
||||
"Stealth RW": [r"\bstealth\b"],
|
||||
"Dual Leech Ring": [r"\bdual leech ring\b"],
|
||||
"Aldur's Advance": [r"\baldur'?s advance\b"],
|
||||
"Shako": [r"\bshako\b"],
|
||||
}
|
||||
|
||||
|
||||
def normalize_text(html_text: str) -> str:
|
||||
text = re.sub(r"<script.*?</script>", " ", html_text, flags=re.IGNORECASE | re.DOTALL)
|
||||
text = re.sub(r"<style.*?</style>", " ", text, flags=re.IGNORECASE | re.DOTALL)
|
||||
text = re.sub(r"<[^>]+>", " ", text)
|
||||
text = unescape(text)
|
||||
text = re.sub(r"\s+", " ", text)
|
||||
return text.strip()
|
||||
|
||||
|
||||
def parse_prices(text: str) -> list[float]:
|
||||
values: list[float] = []
|
||||
for m in re.finditer(r"\b(?:bin\s*)?(\d{1,3}(?:,\d{3})*|\d+)(?:\.\d+)?\s*fg\b", text, flags=re.IGNORECASE):
|
||||
raw = m.group(1).replace(",", "")
|
||||
try:
|
||||
values.append(float(raw))
|
||||
except ValueError:
|
||||
continue
|
||||
return values
|
||||
|
||||
|
||||
def detect_item(text: str) -> str | None:
|
||||
lower = text.lower()
|
||||
for name, patterns in ITEM_PATTERNS.items():
|
||||
for pat in patterns:
|
||||
if re.search(pat, lower, flags=re.IGNORECASE):
|
||||
return name
|
||||
return None
|
||||
|
||||
|
||||
def is_blocked_page(html_text: str) -> bool:
|
||||
lower = (html_text or "").lower()
|
||||
return ("just a moment" in lower and "cloudflare" in lower) or "challenge-platform" in lower
|
||||
|
||||
|
||||
def is_topic_page(html_text: str) -> bool:
|
||||
lower = (html_text or "").lower()
|
||||
if "saved from url=" in lower and "topic.php?t=" in lower:
|
||||
return True
|
||||
if "<title>" in lower and "- topic - d2jsp" in lower:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def parse_forum_topic_links(html_text: str) -> list[str]:
|
||||
links = re.findall(r'href="(/topic\.php\?t=\d+[^"]*)"', html_text)
|
||||
unique = []
|
||||
seen = set()
|
||||
for link in links:
|
||||
if link not in seen:
|
||||
seen.add(link)
|
||||
unique.append(link)
|
||||
return unique
|
||||
|
||||
|
||||
def parse_topic_datetime(topic_html: str) -> dt.datetime | None:
|
||||
# Typical d2jsp pages contain explicit date strings in post headers.
|
||||
# Support both "Jan 01 2026" and "26 May 2026"
|
||||
candidates = re.findall(
|
||||
r"\b(?:\d{1,2}\s+)?(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}(?:\s+\d{1,2}:\d{2})?|"
|
||||
r"\b\d{1,2}\s+(?:Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec)[a-z]*\s+\d{4}(?:\s+\d{1,2}:\d{2})?",
|
||||
topic_html,
|
||||
flags=re.IGNORECASE,
|
||||
)
|
||||
formats = (
|
||||
"%b %d %Y, %I:%M %p", "%B %d %Y, %I:%M %p", "%b %d %Y", "%B %d %Y",
|
||||
"%d %b %Y %H:%M", "%d %B %Y %H:%M", "%d %b %Y", "%d %B %Y",
|
||||
"%b %d %Y %H:%M", "%B %d %Y %H:%M"
|
||||
)
|
||||
for token in candidates:
|
||||
for fmt in formats:
|
||||
try:
|
||||
return dt.datetime.strptime(token, fmt)
|
||||
except ValueError:
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def parse_topic_title(topic_html: str) -> str:
|
||||
m = re.search(r"<title>(.*?)</title>", topic_html, flags=re.IGNORECASE | re.DOTALL)
|
||||
if not m:
|
||||
return ""
|
||||
return normalize_text(m.group(1))
|
||||
|
||||
|
||||
def fallback_item_from_title(title: str) -> str | None:
|
||||
cleaned = (title or "").replace(" - Topic - d2jsp", "").strip()
|
||||
if not cleaned:
|
||||
return None
|
||||
# Avoid overly generic buckets.
|
||||
if cleaned.lower() in {"d2:r rotw softcore ladder trading - d2jsp", "d2:r rotw softcore ladder trading"}:
|
||||
return None
|
||||
return f"Topic:{cleaned[:80]}"
|
||||
|
||||
|
||||
def scrape_day_estimates(
|
||||
ladder_start: dt.date,
|
||||
days: int,
|
||||
max_forum_pages: int,
|
||||
max_topics: int,
|
||||
delay_s: float,
|
||||
cookie_header: str = "",
|
||||
) -> dict:
|
||||
session = requests.Session()
|
||||
session.headers.update({"User-Agent": "Mozilla/5.0 (compatible; botty-fg-scraper/1.0)"})
|
||||
if cookie_header:
|
||||
session.headers.update({"Cookie": cookie_header})
|
||||
|
||||
topic_urls: list[str] = []
|
||||
for page_idx in range(max_forum_pages):
|
||||
offset = page_idx * 25
|
||||
url = f"{BASE_URL}{FORUM_PATH}&st={offset}"
|
||||
r = session.get(url, timeout=20)
|
||||
if r.status_code == 403:
|
||||
raise RuntimeError(
|
||||
"d2jsp returned HTTP 403 Forbidden. "
|
||||
"Scraping is blocked from this environment/IP. "
|
||||
"Run this tool from a browser-authenticated environment or provide exported topic data."
|
||||
)
|
||||
r.raise_for_status()
|
||||
links = parse_forum_topic_links(r.text)
|
||||
for link in links:
|
||||
full = f"{BASE_URL}{link}"
|
||||
if full not in topic_urls:
|
||||
topic_urls.append(full)
|
||||
time.sleep(delay_s)
|
||||
if len(topic_urls) >= max_topics:
|
||||
break
|
||||
|
||||
topic_urls = topic_urls[:max_topics]
|
||||
buckets: dict[int, dict[str, list[float]]] = {d: {} for d in range(1, days + 1)}
|
||||
processed = 0
|
||||
|
||||
for topic_url in topic_urls:
|
||||
try:
|
||||
r = session.get(topic_url, timeout=20)
|
||||
r.raise_for_status()
|
||||
except Exception:
|
||||
continue
|
||||
html = r.text
|
||||
text = normalize_text(html)
|
||||
item = detect_item(text)
|
||||
prices = parse_prices(text)
|
||||
post_dt = parse_topic_datetime(html)
|
||||
processed += 1
|
||||
time.sleep(delay_s)
|
||||
|
||||
if not item or not prices or not post_dt:
|
||||
continue
|
||||
|
||||
day_idx = (post_dt.date() - ladder_start).days + 1
|
||||
if day_idx < 1 or day_idx > days:
|
||||
continue
|
||||
|
||||
if item not in buckets[day_idx]:
|
||||
buckets[day_idx][item] = []
|
||||
# Keep first few to avoid huge noisy post lists.
|
||||
buckets[day_idx][item].extend(prices[:3])
|
||||
|
||||
result = {
|
||||
"generated_at": dt.datetime.now(dt.UTC).isoformat(),
|
||||
"ladder_start_date": ladder_start.isoformat(),
|
||||
"days": days,
|
||||
"topics_scanned": processed,
|
||||
"estimates": {},
|
||||
}
|
||||
|
||||
for day_idx in range(1, days + 1):
|
||||
day_key = f"day_{day_idx}"
|
||||
result["estimates"][day_key] = {}
|
||||
for item, samples in buckets[day_idx].items():
|
||||
if not samples:
|
||||
continue
|
||||
median_fg = statistics.median(samples)
|
||||
result["estimates"][day_key][item] = {
|
||||
"median_fg": round(float(median_fg), 1),
|
||||
"samples": len(samples),
|
||||
}
|
||||
return result
|
||||
|
||||
|
||||
def scrape_day_estimates_offline(
|
||||
offline_dir: str,
|
||||
ladder_start: dt.date,
|
||||
days: int,
|
||||
) -> dict:
|
||||
offline_path = Path(offline_dir)
|
||||
if not offline_path.exists() or not offline_path.is_dir():
|
||||
raise RuntimeError(f"Offline dir does not exist or is not a directory: {offline_dir}")
|
||||
|
||||
html_files = sorted(list(offline_path.glob("*.html")) + list(offline_path.glob("*.htm")))
|
||||
if not html_files:
|
||||
return {
|
||||
"generated_at": dt.datetime.utcnow().isoformat() + "Z",
|
||||
"mode": "offline",
|
||||
"offline_dir": str(offline_path),
|
||||
"ladder_start_date": ladder_start.isoformat(),
|
||||
"days": days,
|
||||
"topics_scanned": 0,
|
||||
"warning": "No .html/.htm files found in offline dir. Save forum/topic pages into this folder and rerun.",
|
||||
"estimates": {f"day_{d}": {} for d in range(1, days + 1)},
|
||||
}
|
||||
|
||||
buckets: dict[int, dict[str, list[float]]] = {d: {} for d in range(1, days + 1)}
|
||||
processed = 0
|
||||
|
||||
for html_file in html_files:
|
||||
try:
|
||||
html = html_file.read_text(encoding="utf-8", errors="ignore")
|
||||
except Exception:
|
||||
continue
|
||||
if is_blocked_page(html):
|
||||
continue
|
||||
if not is_topic_page(html):
|
||||
continue
|
||||
text = normalize_text(html)
|
||||
title = parse_topic_title(html)
|
||||
item = (
|
||||
detect_item(text)
|
||||
or detect_item(title)
|
||||
or detect_item(html_file.stem.replace("-", " "))
|
||||
or fallback_item_from_title(title)
|
||||
)
|
||||
prices = parse_prices(text)
|
||||
if not prices:
|
||||
prices = parse_prices(title)
|
||||
# Offline mode uses file save timestamp as authoritative collection time.
|
||||
# Page content may contain many unrelated historical dates that pollute parsing.
|
||||
post_dt = dt.datetime.fromtimestamp(html_file.stat().st_mtime)
|
||||
processed += 1
|
||||
|
||||
if not item or not prices or not post_dt:
|
||||
continue
|
||||
|
||||
day_idx = (post_dt.date() - ladder_start).days + 1
|
||||
if day_idx < 1 or day_idx > days:
|
||||
continue
|
||||
|
||||
if item not in buckets[day_idx]:
|
||||
buckets[day_idx][item] = []
|
||||
buckets[day_idx][item].extend(prices[:3])
|
||||
|
||||
result = {
|
||||
"generated_at": dt.datetime.now(dt.UTC).isoformat(),
|
||||
"mode": "offline",
|
||||
"offline_dir": str(offline_path),
|
||||
"ladder_start_date": ladder_start.isoformat(),
|
||||
"days": days,
|
||||
"topics_scanned": processed,
|
||||
"estimates": {},
|
||||
}
|
||||
|
||||
for day_idx in range(1, days + 1):
|
||||
day_key = f"day_{day_idx}"
|
||||
result["estimates"][day_key] = {}
|
||||
for item, samples in buckets[day_idx].items():
|
||||
if not samples:
|
||||
continue
|
||||
median_fg = statistics.median(samples)
|
||||
result["estimates"][day_key][item] = {
|
||||
"median_fg": round(float(median_fg), 1),
|
||||
"samples": len(samples),
|
||||
}
|
||||
return result
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="Scrape d2jsp ladder forum and estimate day-by-day FG prices.")
|
||||
parser.add_argument("--ladder-start-date", required=True, help="Ladder start date in YYYY-MM-DD")
|
||||
parser.add_argument("--days", type=int, default=14, help="Number of ladder days to aggregate (default: 14)")
|
||||
parser.add_argument("--max-forum-pages", type=int, default=40, help="How many forum pages to scan (25 topics/page)")
|
||||
parser.add_argument("--max-topics", type=int, default=800, help="Hard cap for topic pages to fetch")
|
||||
parser.add_argument("--delay-s", type=float, default=0.35, help="Delay between requests (seconds)")
|
||||
parser.add_argument("--offline-dir", default="", help="Parse local saved .html files instead of live scraping")
|
||||
parser.add_argument("--cookie", default="", help="Raw Cookie header for authenticated scraping (e.g. 'member_id=...; msec=...')")
|
||||
parser.add_argument("--out", default="config/fg_daily_estimates.json", help="Output JSON path")
|
||||
args = parser.parse_args()
|
||||
|
||||
ladder_start = dt.date.fromisoformat(args.ladder_start_date)
|
||||
if args.offline_dir:
|
||||
data = scrape_day_estimates_offline(
|
||||
offline_dir=args.offline_dir,
|
||||
ladder_start=ladder_start,
|
||||
days=args.days,
|
||||
)
|
||||
else:
|
||||
data = scrape_day_estimates(
|
||||
ladder_start=ladder_start,
|
||||
days=args.days,
|
||||
max_forum_pages=args.max_forum_pages,
|
||||
max_topics=args.max_topics,
|
||||
delay_s=args.delay_s,
|
||||
cookie_header=args.cookie,
|
||||
)
|
||||
|
||||
out_path = Path(args.out)
|
||||
out_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
out_path.write_text(json.dumps(data, indent=2), encoding="utf-8")
|
||||
print(f"Wrote FG daily estimates to {out_path} (topics_scanned={data['topics_scanned']})")
|
||||
if "warning" in data:
|
||||
print(f"WARNING: {data['warning']}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user