277 lines
9.6 KiB
Python
277 lines
9.6 KiB
Python
#!/usr/bin/env python3
|
||
"""
|
||
scrape-rules-from-pdfs.py
|
||
|
||
Extracts gear/wargear rule descriptions from Deathwatch PDFs and upserts
|
||
them into the MariaDB rules table with FULL untruncated text.
|
||
|
||
Uses PyMuPDF to extract bold headings as entry titles and collects all
|
||
following paragraph text as the rule content — no length limits.
|
||
|
||
Usage:
|
||
python3 scripts/scrape-rules-from-pdfs.py # all PDFs
|
||
python3 scripts/scrape-rules-from-pdfs.py CR # only Core Rulebook
|
||
python3 scripts/scrape-rules-from-pdfs.py CR FF # multiple
|
||
python3 scripts/scrape-rules-from-pdfs.py --dry-run # print, don't save
|
||
"""
|
||
|
||
import os, re, sys
|
||
from pathlib import Path
|
||
import fitz # PyMuPDF
|
||
import mysql.connector
|
||
|
||
# ── Config ────────────────────────────────────────────────────────────────────
|
||
REPO = Path(__file__).resolve().parent.parent
|
||
PDF_DIR = REPO / 'database' / 'rules'
|
||
|
||
def load_env_file(path):
|
||
if not path.exists():
|
||
return
|
||
for line in path.read_text().splitlines():
|
||
line = line.strip()
|
||
if not line or line.startswith('#') or '=' not in line:
|
||
continue
|
||
key, value = line.split('=', 1)
|
||
os.environ.setdefault(key.strip(), value.strip().strip('"').strip("'"))
|
||
|
||
load_env_file(REPO / 'database' / '.env')
|
||
load_env_file(REPO / '.env')
|
||
if not os.environ.get('DB_PASSWORD'):
|
||
raise RuntimeError('DB_PASSWORD must be set in database/.env or .env')
|
||
|
||
DB = dict(
|
||
host=os.environ.get('DB_HOST', '192.168.1.113'),
|
||
port=int(os.environ.get('DB_PORT', '3307')),
|
||
user=os.environ.get('DB_USER', 'deathwatch'),
|
||
password=os.environ['DB_PASSWORD'],
|
||
database=os.environ.get('DB_NAME', 'deathwatch'),
|
||
)
|
||
|
||
# Book definitions: key → (filename, full_name, abbreviation, page_ranges)
|
||
# page_ranges = list of (start, end) 1-indexed PDF pages for gear/wargear sections
|
||
# None = scan entire PDF
|
||
BOOKS = {
|
||
'CR': ('CR.pdf', 'Core Rulebook', 'CR', [(139, 175), (270, 310)]),
|
||
'FF': ('FF.pdf', 'First Founding', 'FF', None),
|
||
'RoB': ('RoB.pdf', 'Rites of Battle', 'RoB', None),
|
||
'MoX': ('MoX.pdf', 'Mark of the Xenos', 'MoX', None),
|
||
'HtC': ('HtC.pdf', 'Honour the Chapter','HtC', None),
|
||
}
|
||
|
||
DRY_RUN = '--dry-run' in sys.argv
|
||
BOOK_FILTER = [a for a in sys.argv[1:] if not a.startswith('-')]
|
||
|
||
MIN_CONTENT = 50 # minimum chars to keep an entry
|
||
MAX_CONTENT = 6000 # safety cap
|
||
|
||
# Words that indicate a chapter/section header, not a rule entry
|
||
CHAPTER_WORDS = {
|
||
'chapter', 'table', 'contents', 'index', 'appendix', 'introduction',
|
||
'foreword', 'credits', 'special thanks', 'foreword', 'designer',
|
||
'developer', 'copyright',
|
||
}
|
||
|
||
# ── Helpers ───────────────────────────────────────────────────────────────────
|
||
def clean(s):
|
||
s = re.sub(r'[ \t]+', ' ', str(s))
|
||
s = re.sub(r' \n', '\n', s)
|
||
s = re.sub(r'\n ', '\n', s)
|
||
s = re.sub(r'\n{3,}', '\n\n', s)
|
||
return s.strip()
|
||
|
||
def normalize_heading(text):
|
||
"""
|
||
Fix PDF font artefact where each word is split: 'a staRtes s toRm' → 'Astartes Storm'.
|
||
Single isolated letter + next token = one word.
|
||
"""
|
||
parts = text.split()
|
||
merged = []
|
||
i = 0
|
||
while i < len(parts):
|
||
if len(parts[i]) == 1 and parts[i].isalpha() and i + 1 < len(parts):
|
||
merged.append(parts[i] + parts[i+1])
|
||
i += 2
|
||
else:
|
||
merged.append(parts[i])
|
||
i += 1
|
||
return ' '.join(w.capitalize() for w in merged)
|
||
|
||
def slug(title):
|
||
s = re.sub(r'[^a-z0-9]+', '-', title.lower()).strip('-')
|
||
return s[:80]
|
||
|
||
def looks_like_stat_row(text):
|
||
"""True if line is mostly numbers/dashes (a stat table row)."""
|
||
tokens = text.split()
|
||
if len(tokens) < 3:
|
||
return False
|
||
num = sum(1 for t in tokens if re.fullmatch(r'[\d\-–—/]+', t))
|
||
return num / len(tokens) > 0.55
|
||
|
||
def is_page_stamp(text):
|
||
"""True if text is just a page number or chapter stamp."""
|
||
t = text.strip()
|
||
return bool(re.fullmatch(r'\d{1,4}', t))
|
||
|
||
def is_rule_heading(text, is_bold, prev_text=''):
|
||
"""
|
||
Heuristic: bold, short (3–80 chars), not a pure number/symbol,
|
||
not a chapter header, not a stat row.
|
||
"""
|
||
if not is_bold:
|
||
return False
|
||
t = text.strip()
|
||
if not (3 <= len(t) <= 80):
|
||
return False
|
||
if is_page_stamp(t):
|
||
return False
|
||
if looks_like_stat_row(t):
|
||
return False
|
||
if re.fullmatch(r'[\d\W]+', t):
|
||
return False
|
||
tl = t.lower()
|
||
if any(tl.startswith(w) for w in CHAPTER_WORDS):
|
||
return False
|
||
# Roman numeral chapter headings like "IV : Talents & Traits"
|
||
if re.match(r'^[IVXivx]+\s*[:\-]', t):
|
||
return False
|
||
return True
|
||
|
||
# ── PDF extraction ────────────────────────────────────────────────────────────
|
||
def extract_entries(pdf_path, page_ranges=None):
|
||
"""
|
||
Walk the PDF's text blocks; detect bold headings as entry titles and
|
||
accumulate following paragraph text as content.
|
||
Returns list of {title, content, page} dicts.
|
||
"""
|
||
doc = fitz.open(str(pdf_path))
|
||
entries = []
|
||
cur_title = None
|
||
cur_lines = []
|
||
cur_page = 0
|
||
|
||
total = len(doc)
|
||
pages_to_scan = []
|
||
if page_ranges:
|
||
for start, end in page_ranges:
|
||
pages_to_scan.extend(range(start - 1, min(end, total)))
|
||
else:
|
||
pages_to_scan = range(total)
|
||
|
||
def flush():
|
||
nonlocal cur_title, cur_lines, cur_page
|
||
if cur_title:
|
||
content = clean(' '.join(cur_lines))
|
||
if len(content) >= MIN_CONTENT:
|
||
entries.append({
|
||
'title': cur_title,
|
||
'content': content[:MAX_CONTENT],
|
||
'page': cur_page,
|
||
})
|
||
cur_title = None
|
||
cur_lines = []
|
||
cur_page = 0
|
||
|
||
for pg_idx in pages_to_scan:
|
||
page = doc[pg_idx]
|
||
pg_no = pg_idx + 1
|
||
raw = page.get_text('dict', flags=fitz.TEXT_PRESERVE_WHITESPACE)
|
||
|
||
for block in raw['blocks']:
|
||
if block['type'] != 0:
|
||
continue
|
||
for line in block['lines']:
|
||
parts, line_bold = [], False
|
||
for span in line['spans']:
|
||
txt = span['text'].strip()
|
||
if not txt:
|
||
continue
|
||
if span['flags'] & (1 << 4):
|
||
line_bold = True
|
||
parts.append(txt)
|
||
|
||
if not parts:
|
||
continue
|
||
text = ' '.join(parts)
|
||
|
||
if is_page_stamp(text):
|
||
continue
|
||
if looks_like_stat_row(text):
|
||
continue
|
||
|
||
if is_rule_heading(text, line_bold):
|
||
flush()
|
||
cur_title = normalize_heading(text)
|
||
cur_page = pg_no
|
||
elif cur_title:
|
||
cur_lines.append(text)
|
||
|
||
flush()
|
||
doc.close()
|
||
return entries
|
||
|
||
# ── DB upsert ─────────────────────────────────────────────────────────────────
|
||
def upsert_entries(entries, source, source_abbr):
|
||
conn = mysql.connector.connect(**DB)
|
||
cur = conn.cursor()
|
||
ins = upd = 0
|
||
for e in entries:
|
||
cur.execute("""
|
||
INSERT INTO rules (rule_id, title, content, page, source, source_abbr, category)
|
||
VALUES (%s, %s, %s, %s, %s, %s, %s)
|
||
ON DUPLICATE KEY UPDATE
|
||
title = VALUES(title),
|
||
content = VALUES(content),
|
||
page = VALUES(page),
|
||
source = VALUES(source),
|
||
source_abbr= VALUES(source_abbr),
|
||
category = VALUES(category)
|
||
""", (
|
||
slug(e['title']), e['title'], e['content'],
|
||
e['page'], source, source_abbr, 'gear'
|
||
))
|
||
if cur.rowcount == 1:
|
||
ins += 1
|
||
elif cur.rowcount == 2:
|
||
upd += 1
|
||
conn.commit()
|
||
cur.close()
|
||
conn.close()
|
||
return ins, upd
|
||
|
||
# ── Main ──────────────────────────────────────────────────────────────────────
|
||
def main():
|
||
targets = BOOK_FILTER if BOOK_FILTER else list(BOOKS.keys())
|
||
|
||
for abbr in targets:
|
||
if abbr not in BOOKS:
|
||
print(f'Unknown book: {abbr} (valid: {list(BOOKS)})')
|
||
continue
|
||
|
||
fname, source, source_abbr, page_ranges = BOOKS[abbr]
|
||
pdf_path = PDF_DIR / fname
|
||
if not pdf_path.exists():
|
||
print(f'✗ Not found: {pdf_path}')
|
||
continue
|
||
|
||
range_str = str(page_ranges) if page_ranges else 'all pages'
|
||
print(f'\n── {source} [{range_str}]')
|
||
|
||
entries = extract_entries(pdf_path, page_ranges)
|
||
print(f' Found {len(entries)} entries')
|
||
|
||
if DRY_RUN:
|
||
for e in entries[:8]:
|
||
print(f' p{e["page"]:3d} [{e["title"][:45]:<45}] {len(e["content"])} chars')
|
||
print(f' {e["content"][:100]}…')
|
||
if len(entries) > 8:
|
||
print(f' … and {len(entries)-8} more')
|
||
else:
|
||
ins, upd = upsert_entries(entries, source, source_abbr)
|
||
print(f' ✓ Inserted: {ins} Updated: {upd}')
|
||
|
||
print()
|
||
|
||
if __name__ == '__main__':
|
||
main()
|