Clean rules database and add documentation

- Remove OCR noise, credits, and duplicates from rules-database.json (288→255 rules)
- Add clean_rules.py script for rule cleanup
- Add CLAUDE.md, docs/, and update README with documentation links

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
2026-04-22 17:35:36 +02:00
parent e511bff882
commit a4cabbd2b8
16 changed files with 2773 additions and 20197 deletions

View File

@@ -0,0 +1,165 @@
#!/usr/bin/env python3
"""Clean up the rules database: remove duplicates, OCR artifacts, and noise."""
import json
import re
import sys
from collections import Counter
INPUT = "database/rules/rules-database.json"
OUTPUT = "database/rules/rules-database-cleaned.json"
def is_noise(content):
"""Return True if content is likely OCR noise."""
text = content.strip()
if len(text) < 10:
return True
# Only symbols/whitespace
if re.sub(r'[\s™®©—“”•‣…]', '', text).strip() == "":
return True
# Only page numbers
if re.match(r'^\s*\d+\s*$', text):
return True
# Only index entries
if text.lower() == "index":
return True
return False
def is_credits(content):
"""Return True if content is credits/copyright."""
text = content.lower()
return any(kw in text for kw in [
"credits", "designed by", "lead developer", "art direction",
"production manager", "game designer", "game producer",
"publisher", "licensing", "graphic design",
"cover art", "copyright", "game workshop",
"ffg", "fantasy flight games",
])
def is_page_header(content):
"""Return True if content looks like a page header/number."""
text = content.strip()
# Page number lines like "88 —" or "165 115"
if re.match(r'^\d+\s*(—|—\s*\d+)?$', text):
return True
# "GAME?" or "GAME? (p.X"
if re.match(r'^GAME\?\s*\(p\.', text):
return True
return False
def fix_title(title):
"""Fix common OCR title issues."""
# Remove leading/trailing whitespace and control chars
title = title.strip()
# Remove control characters
title = re.sub(r'[\x00-\x1f\x7f]', '', title).strip()
# Fix "GAME?" -> "GAME"
title = re.sub(r'^GAME\?', 'GAME', title)
# Fix "fellowShip" -> "Fellowship"
title = re.sub(r'^(fel)\s*\(', r'\1 (', title, flags=re.IGNORECASE)
# Fix "balliStiC" -> "Ballistic"
title = re.sub(r'^(balliStiC)\s*Skill', r'Ballistic Skill', title, flags=re.IGNORECASE)
# Fix "perCeption" -> "Perception"
title = re.sub(r'^(perCeption)', r'Perception', title, flags=re.IGNORECASE)
# Fix "ExaMple" -> "Example"
title = re.sub(r'^(exaMple)', r'Example', title, flags=re.IGNORECASE)
# Fix "s" on its own line
if title.strip() == "s":
return None
# Fix "88 —" or "165 115" style page numbers
if re.match(r'^\d+\s*(—|—\s*\d+)?$', title):
return None
# Fix "therefore, and you will know no fear." (p.34, Core Rulebook)
if re.match(r'^.*\(p\.\d+,\s*(Core Rulebook|Game Master\'s Kit)\)\s*$', title):
return None
# Fix "— (p.XX, Source)"
if re.match(r'^—\s*\(p\.\d+,\s*(Core Rulebook|Game Master\'s Kit)\)\s*$', title):
return None
return title
def dedup_rules(rules):
"""Remove duplicate rules, keeping the one with the most content."""
seen = {}
kept = []
removed = []
for rule in rules:
title = fix_title(rule.get('title', ''))
if title is None:
removed.append(('title_fix_none', rule))
continue
key = title.strip().lower()
content = rule.get('content', '')
if key in seen:
existing = seen[key]
if len(content) > len(existing.get('content', '')):
removed.append(('duplicate', existing))
seen[key] = rule
else:
removed.append(('duplicate', rule))
else:
seen[key] = rule
# Update titles in the kept rules
for rule in seen.values():
rule['title'] = fix_title(rule.get('title', ''))
return list(seen.values()), removed
def main():
with open(INPUT) as f:
data = json.load(f)
rules = data['rules']
print(f"Loaded {len(rules)} rules")
# Phase 1: Remove noise
noise_rules = [r for r in rules if is_noise(r.get('content', ''))]
credits_rules = [r for r in rules if is_credits(r.get('content', ''))]
header_rules = [r for r in rules if is_page_header(r.get('content', ''))]
print(f"\nNoise (short/symbols): {len(noise_rules)}")
print(f"Credits/copyright: {len(credits_rules)}")
print(f"Page headers: {len(header_rules)}")
# Phase 2: Dedup
deduped, dups = dedup_rules(rules)
print(f"\nDuplicates removed: {len(dups)}")
# Phase 3: Combine noise removal
cleaned = []
for r in deduped:
content = r.get('content', '')
if is_noise(content) or is_credits(content) or is_page_header(content):
continue
cleaned.append(r)
print(f"\nFinal count: {len(cleaned)} (was {len(rules)})")
print(f"Removed: {len(rules) - len(cleaned)}")
# Show what was removed
print("\n--- Removed rules ---")
for reason, r in dups:
print(f" [{reason}] {r.get('title', '?')[:60]}")
for r in noise_rules + credits_rules + header_rules:
if r not in cleaned:
print(f" [noise] {r.get('title', '?')[:60]}")
# Write output
out = {'rules': cleaned}
with open(OUTPUT, 'w') as f:
json.dump(out, f, indent=2, ensure_ascii=False)
print(f"\nWrote cleaned rules to {OUTPUT}")
# Show category breakdown
from collections import Counter
cats = Counter(r.get('category', 'unknown') for r in cleaned)
print("\nCategories:")
for c, n in cats.most_common():
print(f" {c}: {n}")
if __name__ == '__main__':
main()