Fix bestiary data: sync profiles, armour, movement from stats; fix cross-contaminated entries
This commit is contained in:
@@ -3,9 +3,33 @@
|
||||
Remove duplicate/inferior skill entries now that we have clean Core Rulebook versions.
|
||||
Also recategorize clearly non-skill sanitized entries.
|
||||
"""
|
||||
import re, mysql.connector
|
||||
import os, re, mysql.connector
|
||||
from pathlib import Path
|
||||
|
||||
DB = dict(host='192.168.1.113', port=3307, user='deathwatch', password='DwRoller@2025!', database='deathwatch')
|
||||
REPO = Path(__file__).resolve().parent.parent
|
||||
|
||||
def load_env_file(path):
|
||||
if not path.exists():
|
||||
return
|
||||
for line in path.read_text().splitlines():
|
||||
line = line.strip()
|
||||
if not line or line.startswith('#') or '=' not in line:
|
||||
continue
|
||||
key, value = line.split('=', 1)
|
||||
os.environ.setdefault(key.strip(), value.strip().strip('"').strip("'"))
|
||||
|
||||
load_env_file(REPO / 'database' / '.env')
|
||||
load_env_file(REPO / '.env')
|
||||
if not os.environ.get('DB_PASSWORD'):
|
||||
raise RuntimeError('DB_PASSWORD must be set in database/.env or .env')
|
||||
|
||||
DB = dict(
|
||||
host=os.environ.get('DB_HOST', '192.168.1.113'),
|
||||
port=int(os.environ.get('DB_PORT', '3307')),
|
||||
user=os.environ.get('DB_USER', 'deathwatch'),
|
||||
password=os.environ['DB_PASSWORD'],
|
||||
database=os.environ.get('DB_NAME', 'deathwatch'),
|
||||
)
|
||||
db = mysql.connector.connect(**DB)
|
||||
c = db.cursor()
|
||||
|
||||
|
||||
@@ -5,10 +5,33 @@ Fix garbled OCR talent titles in the rules DB.
|
||||
- Set source/source_abbr/category correctly for all talent entries
|
||||
- Remove duplicate sanitized entries where a clean 'talent_*' version exists
|
||||
"""
|
||||
import re, mysql.connector
|
||||
import os, re, mysql.connector
|
||||
from pathlib import Path
|
||||
|
||||
DB = dict(host='192.168.1.113', port=3307, user='deathwatch',
|
||||
password='DwRoller@2025!', database='deathwatch')
|
||||
REPO = Path(__file__).resolve().parent.parent
|
||||
|
||||
def load_env_file(path):
|
||||
if not path.exists():
|
||||
return
|
||||
for line in path.read_text().splitlines():
|
||||
line = line.strip()
|
||||
if not line or line.startswith('#') or '=' not in line:
|
||||
continue
|
||||
key, value = line.split('=', 1)
|
||||
os.environ.setdefault(key.strip(), value.strip().strip('"').strip("'"))
|
||||
|
||||
load_env_file(REPO / 'database' / '.env')
|
||||
load_env_file(REPO / '.env')
|
||||
if not os.environ.get('DB_PASSWORD'):
|
||||
raise RuntimeError('DB_PASSWORD must be set in database/.env or .env')
|
||||
|
||||
DB = dict(
|
||||
host=os.environ.get('DB_HOST', '192.168.1.113'),
|
||||
port=int(os.environ.get('DB_PORT', '3307')),
|
||||
user=os.environ.get('DB_USER', 'deathwatch'),
|
||||
password=os.environ['DB_PASSWORD'],
|
||||
database=os.environ.get('DB_NAME', 'deathwatch'),
|
||||
)
|
||||
|
||||
def normalise_title(raw):
|
||||
"""Convert OCR-garbled title to proper Title Case."""
|
||||
|
||||
@@ -0,0 +1,327 @@
|
||||
#!/usr/bin/env node
|
||||
/*
|
||||
* Hydrate staged 40k RPG Tools index rows from local PDFs into the compact live
|
||||
* bestiary JSON. This does not scrape statblocks from the web; it uses the web
|
||||
* index as a page map and extracts only compact combat fields from local PDFs.
|
||||
*/
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const { execFileSync } = require('child_process');
|
||||
|
||||
const ROOT = path.join(__dirname, '..');
|
||||
const INDEX_PATH = path.join(ROOT, 'database', 'remote-extract', '40krpgtools-bestiary-index.json');
|
||||
const PUBLIC_PATH = path.join(ROOT, 'public', 'deathwatch-bestiary-extracted.json');
|
||||
const DB_PATH = path.join(ROOT, 'database', 'deathwatch-bestiary-extracted.json');
|
||||
|
||||
const BOOKS = {
|
||||
'Mark of the Xenos': { pdf: path.join(ROOT, 'database', 'rules', 'MoX.pdf'), pageOffset: 6 },
|
||||
'First Founding': { pdf: path.join(ROOT, 'database', 'rules', 'FF.pdf'), pageOffset: 8 },
|
||||
'The Emperor Protects': { pdf: path.join(ROOT, 'database', 'rules', 'HtC.pdf'), pageOffset: 8 },
|
||||
'Rites of Battle': { pdf: path.join(ROOT, 'database', 'rules', 'RoB.pdf'), pageOffset: 8 },
|
||||
'Deathwatch Core Rulebook': { pdf: path.join(ROOT, 'database', 'rules', 'CR.pdf'), pageOffset: 0 },
|
||||
};
|
||||
|
||||
const PROFILE_KEYS = ['ws', 'bs', 's', 't', 'ag', 'int', 'per', 'wp', 'fel'];
|
||||
const RANK_RE = '(Troops?|Elite|Master|Minion|Personal|Vehicle|Horde)';
|
||||
|
||||
function argValue(name, fallback = null) {
|
||||
const prefix = `--${name}=`;
|
||||
const found = process.argv.find(arg => arg.startsWith(prefix));
|
||||
return found ? found.slice(prefix.length) : fallback;
|
||||
}
|
||||
|
||||
function hasFlag(name) {
|
||||
return process.argv.includes(`--${name}`);
|
||||
}
|
||||
|
||||
function norm(value) {
|
||||
return String(value || '')
|
||||
.toLowerCase()
|
||||
.replace(/[’']/g, '')
|
||||
.replace(/[^a-z0-9]+/g, ' ')
|
||||
.trim()
|
||||
.replace(/\s+/g, ' ');
|
||||
}
|
||||
|
||||
function titleCase(value) {
|
||||
return String(value || '').replace(/\b\w/g, c => c.toUpperCase());
|
||||
}
|
||||
|
||||
function textForPage(book, page) {
|
||||
const cfg = BOOKS[book];
|
||||
const listedPage = Number(page);
|
||||
const physicalStart = Math.max(1, listedPage);
|
||||
const physicalEnd = listedPage + cfg.pageOffset + 2;
|
||||
return execFileSync('pdftotext', ['-raw', '-f', String(physicalStart), '-l', String(physicalEnd), cfg.pdf, '-'], {
|
||||
cwd: ROOT,
|
||||
encoding: 'utf8',
|
||||
maxBuffer: 1024 * 1024 * 8,
|
||||
});
|
||||
}
|
||||
|
||||
function cleanText(value) {
|
||||
return String(value || '')
|
||||
.replace(/\u0008/g, '')
|
||||
.replace(/[“”]/g, '"')
|
||||
.replace(/[’]/g, "'")
|
||||
.replace(/\s+/g, ' ')
|
||||
.replace(/\s+([,.;:])/g, '$1')
|
||||
.trim();
|
||||
}
|
||||
|
||||
function splitTopLevelList(text) {
|
||||
const items = [];
|
||||
let current = '';
|
||||
let depth = 0;
|
||||
for (const ch of cleanText(text)) {
|
||||
if (ch === '(') depth += 1;
|
||||
if (ch === ')' && depth > 0) depth -= 1;
|
||||
if (ch === ',' && depth === 0) {
|
||||
if (current.trim()) items.push(current.trim());
|
||||
current = '';
|
||||
} else {
|
||||
current += ch;
|
||||
}
|
||||
}
|
||||
if (current.trim()) items.push(current.trim());
|
||||
return items;
|
||||
}
|
||||
|
||||
function compactList(text, maxItems = 10, maxChars = 260) {
|
||||
const items = splitTopLevelList(text)
|
||||
.map(item => item.replace(/[.;:\s]+$/, '').trim())
|
||||
.filter(Boolean);
|
||||
if (!items.length) return '';
|
||||
const suffix = items.length > maxItems ? `, +${items.length - maxItems} more` : '';
|
||||
const out = items.slice(0, maxItems).join(', ') + suffix;
|
||||
return out.length > maxChars ? out.slice(0, maxChars - 1).trimEnd() + '...' : out;
|
||||
}
|
||||
|
||||
function compactText(text, maxChars = 260) {
|
||||
const out = cleanText(text).replace(/[.;:\s]+$/, '');
|
||||
return out.length > maxChars ? out.slice(0, maxChars - 1).trimEnd() + '...' : out;
|
||||
}
|
||||
|
||||
function section(block, start, stops) {
|
||||
const startRe = new RegExp(`${start}\\s*:`, 'i');
|
||||
const startMatch = startRe.exec(block);
|
||||
if (!startMatch) return '';
|
||||
let rest = block.slice(startMatch.index + startMatch[0].length);
|
||||
let cut = -1;
|
||||
for (const stop of stops) {
|
||||
const looseStop = /^(SpecialRules|Using|Ranged Weapons|Melee Weapons|Table|The Headsman|Tides of|Mark of|Cloud of)$/i.test(stop);
|
||||
const pattern = looseStop
|
||||
? `(?:^|\\s)${stop}(?:\\s*:|\\b)`
|
||||
: `(?:^|\\n)\\s*${stop}\\s*:`;
|
||||
const m = new RegExp(pattern, 'i').exec(rest);
|
||||
if (m && (cut === -1 || m.index < cut)) cut = m.index;
|
||||
}
|
||||
if (cut >= 0) rest = rest.slice(0, cut);
|
||||
return rest;
|
||||
}
|
||||
|
||||
function profileFromBlock(block) {
|
||||
const rows = block.split(/\r?\n/);
|
||||
for (const line of rows) {
|
||||
const tokens = line.trim().split(/\s+/).filter(Boolean);
|
||||
const nums = tokens
|
||||
.map(token => token.replace(/[–—-]{1,2}/g, '0'))
|
||||
.filter(token => /^\d{1,3}$/.test(token))
|
||||
.map(Number);
|
||||
if (nums.length >= 9) {
|
||||
return Object.fromEntries(PROFILE_KEYS.map((key, idx) => [key, nums[idx]]));
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function armourFromText(text) {
|
||||
const cleaned = cleanText(text);
|
||||
const all = /\bAll\s+(\d+)/i.exec(cleaned);
|
||||
if (all) return { all: Number(all[1]) };
|
||||
|
||||
const body = /\bBody\s+(\d+)/i.exec(cleaned);
|
||||
const head = /\bHead\s+(\d+)/i.exec(cleaned);
|
||||
const arms = /\bArms?(?:\s+and\s+Legs)?\s+(\d+)/i.exec(cleaned);
|
||||
const legs = /\bLegs?\s+(\d+)/i.exec(cleaned);
|
||||
if (body || head || arms || legs) {
|
||||
return {
|
||||
head: head ? Number(head[1]) : (body ? Number(body[1]) : 0),
|
||||
body: body ? Number(body[1]) : 0,
|
||||
arm: arms ? Number(arms[1]) : (body ? Number(body[1]) : 0),
|
||||
leg: legs ? Number(legs[1]) : (arms ? Number(arms[1]) : (body ? Number(body[1]) : 0)),
|
||||
};
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function tbFrom(profile, traitsText) {
|
||||
const base = Math.floor((Number(profile.t) || 0) / 10);
|
||||
const unnatural = /Unnatural Toughness\s*\((?:x|×)(\d+)\)/i.exec(traitsText || '');
|
||||
if (unnatural) return base * Number(unnatural[1]);
|
||||
const daemonic = /Daemonic\s*\(TB\s*(\d+)\)/i.exec(traitsText || '');
|
||||
if (daemonic) return Number(daemonic[1]);
|
||||
return base;
|
||||
}
|
||||
|
||||
function parseProfilesFromPage(text) {
|
||||
const matches = [];
|
||||
const profileRe = new RegExp(`([^\\n\\r]{2,80}?)\\s+\\(${RANK_RE}\\)\\s+Profile`, 'ig');
|
||||
let match;
|
||||
while ((match = profileRe.exec(text))) {
|
||||
matches.push({
|
||||
name: cleanText(match[1]).replace(/^[^A-Za-z0-9]+/, ''),
|
||||
rank: titleCase(match[2]),
|
||||
index: match.index,
|
||||
});
|
||||
}
|
||||
|
||||
return matches.map((item, idx) => {
|
||||
const nextStart = matches[idx + 1]?.index ?? text.length;
|
||||
const afterProfile = text.slice(item.index, nextStart);
|
||||
const searchableAfterProfile = afterProfile.replace(/\u0008/g, '');
|
||||
const profile = profileFromBlock(afterProfile);
|
||||
const woundsMatch = /Wounds\s*:?\s*(\d+)/i.exec(searchableAfterProfile);
|
||||
if (!profile || !woundsMatch) return null;
|
||||
|
||||
const movementMatch = /\b(?:Movement|Move|Speed)\s*:?\s*([0-9/]+)/i.exec(searchableAfterProfile);
|
||||
const skills = compactList(section(searchableAfterProfile, 'Skills', ['Talents', 'Traits', 'Armour', 'Armor', 'Weapons', 'Gear', 'SpecialRules', 'Special']));
|
||||
const talents = compactList(section(searchableAfterProfile, 'Talents', ['Traits', 'Armour', 'Armor', 'Weapons', 'Gear', 'SpecialRules', 'Special']));
|
||||
const traitsText = section(searchableAfterProfile, 'Traits', ['Armour', 'Armor', 'Weapons', 'Gear', 'SpecialRules', 'Special', 'Psy Rating', 'Psychic Powers']);
|
||||
const traits = compactList(traitsText);
|
||||
const armourText = section(searchableAfterProfile, 'Armou?r', ['Weapons', 'Gear', 'Special', 'Skills', 'Talents', 'Traits']);
|
||||
const weaponsText = section(searchableAfterProfile, 'Weapons', [
|
||||
'Gear',
|
||||
'SpecialRules',
|
||||
'Special',
|
||||
'Skills',
|
||||
'Talents',
|
||||
'Traits',
|
||||
'Armour',
|
||||
'Armor',
|
||||
'Mark of',
|
||||
'Cloud of',
|
||||
'The Headsman',
|
||||
'Tides of',
|
||||
'Using',
|
||||
'Ranged Weapons',
|
||||
'Melee Weapons',
|
||||
'Table',
|
||||
]);
|
||||
|
||||
return {
|
||||
name: item.name,
|
||||
rank: item.rank,
|
||||
profile,
|
||||
wounds: Number(woundsMatch[1]),
|
||||
movement: movementMatch ? movementMatch[1] : '',
|
||||
skills,
|
||||
talents,
|
||||
traits,
|
||||
armour: armourFromText(armourText),
|
||||
weapons: compactText(weaponsText),
|
||||
toughnessBonus: tbFrom(profile, traitsText),
|
||||
};
|
||||
}).filter(Boolean);
|
||||
}
|
||||
|
||||
function loadBestiary(file) {
|
||||
const raw = JSON.parse(fs.readFileSync(file, 'utf8'));
|
||||
const arr = Array.isArray(raw) ? raw : (raw.results || raw.entries || raw.items);
|
||||
if (!Array.isArray(arr)) throw new Error(`Unsupported bestiary shape in ${file}`);
|
||||
return { raw, arr };
|
||||
}
|
||||
|
||||
function toEntry(parsed, indexRow, book) {
|
||||
const stats = {
|
||||
profile: parsed.profile,
|
||||
movement: parsed.movement,
|
||||
wounds: parsed.wounds,
|
||||
toughnessBonus: parsed.toughnessBonus,
|
||||
statSource: 'pdf',
|
||||
needsReview: false,
|
||||
sourceIndex: '40krpgtools',
|
||||
threatTier: parsed.rank,
|
||||
parsedName: parsed.name,
|
||||
};
|
||||
if (parsed.armour) stats.armour = parsed.armour;
|
||||
if (parsed.weapons) stats.weapons = parsed.weapons;
|
||||
if (parsed.skills) stats.skills = parsed.skills;
|
||||
if (parsed.talents) stats.talents = parsed.talents;
|
||||
if (parsed.traits) stats.traits = parsed.traits;
|
||||
return {
|
||||
bestiaryName: indexRow.name,
|
||||
type: indexRow.type,
|
||||
affiliation: indexRow.affiliation,
|
||||
book,
|
||||
page: indexRow.page,
|
||||
pdf: path.basename(BOOKS[book].pdf),
|
||||
sourceUrl: indexRow.sourceUrl,
|
||||
wounds: parsed.wounds,
|
||||
stats,
|
||||
};
|
||||
}
|
||||
|
||||
function main() {
|
||||
const book = argValue('book', 'Mark of the Xenos');
|
||||
const limit = Number(argValue('limit', '0'));
|
||||
const dryRun = hasFlag('dry-run');
|
||||
const replaceBook = hasFlag('replace-book');
|
||||
if (!BOOKS[book]) throw new Error(`Unsupported book: ${book}`);
|
||||
|
||||
const index = JSON.parse(fs.readFileSync(INDEX_PATH, 'utf8'));
|
||||
const { raw, arr } = loadBestiary(PUBLIC_PATH);
|
||||
if (replaceBook) {
|
||||
for (let i = arr.length - 1; i >= 0; i -= 1) {
|
||||
const entry = arr[i] || {};
|
||||
if (entry.book === book && entry.stats?.sourceIndex === '40krpgtools') {
|
||||
arr.splice(i, 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
const existing = new Set(arr.map(entry => norm(entry.bestiaryName || entry.name)).filter(Boolean));
|
||||
const rows = index.entries.filter(row => row.book === book && !existing.has(norm(row.name)));
|
||||
|
||||
const pageCache = new Map();
|
||||
const parsedByPage = new Map();
|
||||
const added = [];
|
||||
const skipped = [];
|
||||
|
||||
for (const row of rows) {
|
||||
if (limit && added.length >= limit) break;
|
||||
if (!pageCache.has(row.page)) {
|
||||
const text = textForPage(book, row.page);
|
||||
pageCache.set(row.page, text);
|
||||
parsedByPage.set(row.page, parseProfilesFromPage(text));
|
||||
}
|
||||
const parsed = parsedByPage.get(row.page);
|
||||
const hit = parsed.find(candidate => {
|
||||
const a = norm(candidate.name);
|
||||
const b = norm(row.name);
|
||||
return a === b || a.includes(b) || b.includes(a);
|
||||
});
|
||||
if (!hit) {
|
||||
skipped.push(`${row.name} (${book} ${row.page})`);
|
||||
continue;
|
||||
}
|
||||
arr.push(toEntry(hit, row, book));
|
||||
existing.add(norm(row.name));
|
||||
added.push(`${row.name} (${book} ${row.page})`);
|
||||
}
|
||||
|
||||
if (!dryRun) {
|
||||
if (!Array.isArray(raw)) raw.count = arr.length;
|
||||
const serialized = JSON.stringify(raw, null, 2);
|
||||
for (const file of [PUBLIC_PATH, DB_PATH]) {
|
||||
if (fs.existsSync(file)) fs.writeFileSync(file + '.pre-hydrate.' + Date.now() + '.json', fs.readFileSync(file));
|
||||
fs.writeFileSync(file, serialized);
|
||||
}
|
||||
}
|
||||
|
||||
console.log(`${dryRun ? 'Would add' : 'Added'} ${added.length} entries from ${book}`);
|
||||
added.forEach(name => console.log(` + ${name}`));
|
||||
console.log(`Skipped ${skipped.length} rows without a parsed profile match`);
|
||||
skipped.slice(0, 20).forEach(name => console.log(` - ${name}`));
|
||||
}
|
||||
|
||||
main();
|
||||
@@ -0,0 +1,136 @@
|
||||
#!/usr/bin/env node
|
||||
/*
|
||||
* Import the public 40k RPG Tools master bestiary index into a staging file.
|
||||
*
|
||||
* The source page is an index of names, type, affiliation, setting, book, and
|
||||
* page numbers. It is not a statblock source, so this script intentionally does
|
||||
* not add rows to the live bestiary JSON. Use the staged output to decide which
|
||||
* local PDFs/pages to hydrate into compact verified entries.
|
||||
*/
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const https = require('https');
|
||||
if (typeof global.File === 'undefined') {
|
||||
global.File = class File {};
|
||||
}
|
||||
const cheerio = require('cheerio');
|
||||
|
||||
const ROOT = path.join(__dirname, '..');
|
||||
const SOURCE_URL = 'https://www.40krpgtools.com/bestiary/';
|
||||
const OUT_PATH = path.join(ROOT, 'database', 'remote-extract', '40krpgtools-bestiary-index.json');
|
||||
const ACTIVE_BESTIARY = path.join(ROOT, 'public', 'deathwatch-bestiary-extracted.json');
|
||||
|
||||
function argValue(name, fallback = null) {
|
||||
const prefix = `--${name}=`;
|
||||
const found = process.argv.find(arg => arg.startsWith(prefix));
|
||||
return found ? found.slice(prefix.length) : fallback;
|
||||
}
|
||||
|
||||
function hasFlag(name) {
|
||||
return process.argv.includes(`--${name}`);
|
||||
}
|
||||
|
||||
function norm(value) {
|
||||
return String(value || '').trim().toLowerCase().replace(/\s+/g, ' ');
|
||||
}
|
||||
|
||||
function fetchText(url) {
|
||||
return new Promise((resolve, reject) => {
|
||||
https.get(url, response => {
|
||||
if (response.statusCode >= 300 && response.statusCode < 400 && response.headers.location) {
|
||||
fetchText(new URL(response.headers.location, url).toString()).then(resolve, reject);
|
||||
return;
|
||||
}
|
||||
if (response.statusCode !== 200) {
|
||||
reject(new Error(`GET ${url} failed with HTTP ${response.statusCode}`));
|
||||
return;
|
||||
}
|
||||
let body = '';
|
||||
response.setEncoding('utf8');
|
||||
response.on('data', chunk => { body += chunk; });
|
||||
response.on('end', () => resolve(body));
|
||||
}).on('error', reject);
|
||||
});
|
||||
}
|
||||
|
||||
function loadActiveNames() {
|
||||
try {
|
||||
const raw = JSON.parse(fs.readFileSync(ACTIVE_BESTIARY, 'utf8'));
|
||||
const arr = Array.isArray(raw) ? raw : (raw.results || raw.entries || raw.items || []);
|
||||
return new Set(arr.map(entry => norm(entry.bestiaryName || entry.name)).filter(Boolean));
|
||||
} catch {
|
||||
return new Set();
|
||||
}
|
||||
}
|
||||
|
||||
function parseRows(html, activeNames) {
|
||||
const $ = cheerio.load(html);
|
||||
const rows = [];
|
||||
$('#bestiaryTable tr').each((_, tr) => {
|
||||
const cells = $(tr).find('td');
|
||||
if (cells.length < 6) return;
|
||||
const link = $(cells[0]).find('a').first();
|
||||
const href = link.attr('href') || '';
|
||||
const entry = {
|
||||
name: link.text().trim() || $(cells[0]).text().trim(),
|
||||
type: $(cells[1]).text().trim(),
|
||||
affiliation: $(cells[2]).text().trim(),
|
||||
setting: $(cells[3]).text().trim(),
|
||||
book: $(cells[4]).text().trim(),
|
||||
page: $(cells[5]).text().trim(),
|
||||
sourceUrl: href ? new URL(href, SOURCE_URL).toString() : SOURCE_URL,
|
||||
};
|
||||
entry.key = norm(entry.name);
|
||||
entry.inActiveBestiary = activeNames.has(entry.key);
|
||||
rows.push(entry);
|
||||
});
|
||||
return rows;
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const setting = argValue('setting', 'Deathwatch');
|
||||
const includeExisting = hasFlag('include-existing');
|
||||
const allSettings = hasFlag('all-settings');
|
||||
const activeNames = loadActiveNames();
|
||||
const html = await fetchText(SOURCE_URL);
|
||||
let rows = parseRows(html, activeNames);
|
||||
|
||||
if (!allSettings) rows = rows.filter(row => norm(row.setting) === norm(setting));
|
||||
if (!includeExisting) rows = rows.filter(row => !row.inActiveBestiary);
|
||||
|
||||
rows.sort((a, b) => (
|
||||
a.book.localeCompare(b.book) ||
|
||||
Number(a.page || 0) - Number(b.page || 0) ||
|
||||
a.name.localeCompare(b.name)
|
||||
));
|
||||
|
||||
const byBook = rows.reduce((acc, row) => {
|
||||
acc[row.book] = (acc[row.book] || 0) + 1;
|
||||
return acc;
|
||||
}, {});
|
||||
|
||||
const output = {
|
||||
source: SOURCE_URL,
|
||||
importedAt: new Date().toISOString(),
|
||||
mode: {
|
||||
setting: allSettings ? 'all' : setting,
|
||||
includeExisting,
|
||||
},
|
||||
count: rows.length,
|
||||
byBook,
|
||||
entries: rows,
|
||||
};
|
||||
|
||||
fs.mkdirSync(path.dirname(OUT_PATH), { recursive: true });
|
||||
fs.writeFileSync(OUT_PATH, JSON.stringify(output, null, 2));
|
||||
|
||||
console.log(`Wrote ${rows.length} staged index rows -> ${path.relative(ROOT, OUT_PATH)}`);
|
||||
Object.entries(byBook)
|
||||
.sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]))
|
||||
.forEach(([book, count]) => console.log(` ${book}: ${count}`));
|
||||
}
|
||||
|
||||
main().catch(error => {
|
||||
console.error(error);
|
||||
process.exit(1);
|
||||
});
|
||||
@@ -0,0 +1,118 @@
|
||||
#!/usr/bin/env node
|
||||
/*
|
||||
* Import horde entries from the staged 40k RPG Tools index as compact
|
||||
* reviewable bestiary rows. These are derived entries, not PDF-parsed
|
||||
* statblocks, so they are flagged as hordes and kept intentionally small.
|
||||
*/
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const { bestiaryHelpers, initializationPromise } = require('../database/mariadb');
|
||||
|
||||
const ROOT = path.join(__dirname, '..');
|
||||
const INDEX_PATH = path.join(ROOT, 'database', 'remote-extract', '40krpgtools-bestiary-index.json');
|
||||
const PUBLIC_PATH = path.join(ROOT, 'public', 'deathwatch-bestiary-extracted.json');
|
||||
const DB_PATH = path.join(ROOT, 'database', 'deathwatch-bestiary-extracted.json');
|
||||
|
||||
const HORDE_SPECS = {
|
||||
'Mutant Horde': { magnitude: 20, profile: { ws: 30, bs: 20, s: 30, t: 30, ag: 30, int: 20, per: 30, wp: 30, fel: 20 }, damage: '1d10+3 I', pen: 0, armour: 3, tb: 3 },
|
||||
'PDF Guardsman Horde': { magnitude: 20, profile: { ws: 30, bs: 35, s: 30, t: 30, ag: 30, int: 30, per: 30, wp: 30, fel: 25 }, damage: '1d10+3 I', pen: 0, armour: 4, tb: 3 },
|
||||
'Rebel Horde': { magnitude: 20, profile: { ws: 30, bs: 30, s: 30, t: 30, ag: 30, int: 25, per: 30, wp: 30, fel: 25 }, damage: '1d10+3 I', pen: 0, armour: 4, tb: 3 },
|
||||
'Kroot Carnivore Horde': { magnitude: 15, profile: { ws: 40, bs: 20, s: 35, t: 30, ag: 40, int: 20, per: 35, wp: 30, fel: 10 }, damage: '1d10+4 R', pen: 0, armour: 3, tb: 3 },
|
||||
'Regressed Kroot Carnivore Horde': { magnitude: 15, profile: { ws: 40, bs: 20, s: 35, t: 30, ag: 40, int: 20, per: 35, wp: 30, fel: 10 }, damage: '1d10+4 R', pen: 0, armour: 3, tb: 3 },
|
||||
'Hromagaunt Horde': { magnitude: 20, profile: { ws: 45, bs: 20, s: 35, t: 30, ag: 55, int: 10, per: 40, wp: 30, fel: 5 }, damage: '1d10+5 R', pen: 0, armour: 3, tb: 3 },
|
||||
'Termagaunt Horde': { magnitude: 20, profile: { ws: 30, bs: 30, s: 30, t: 30, ag: 30, int: 10, per: 30, wp: 30, fel: 5 }, damage: '1d10+4 R', pen: 0, armour: 3, tb: 3 },
|
||||
'Gargoyle Horde': { magnitude: 20, profile: { ws: 30, bs: 33, s: 32, t: 30, ag: 40, int: 10, per: 40, wp: 30, fel: 0 }, damage: '1d10+5 R', pen: 3, armour: 3, tb: 3 },
|
||||
};
|
||||
|
||||
function loadJson(file) {
|
||||
const raw = JSON.parse(fs.readFileSync(file, 'utf8'));
|
||||
const arr = Array.isArray(raw) ? raw : (raw.results || raw.entries || raw.items || []);
|
||||
if (!Array.isArray(arr)) throw new Error(`Unsupported bestiary JSON shape: ${file}`);
|
||||
return { raw, arr };
|
||||
}
|
||||
|
||||
function norm(value) {
|
||||
return String(value || '').toLowerCase().replace(/[^a-z0-9]+/g, ' ').trim();
|
||||
}
|
||||
|
||||
function appendHordeEntries(arr, indexRows) {
|
||||
const existing = new Set(arr.map(entry => norm(entry.bestiaryName || entry.name)).filter(Boolean));
|
||||
const added = [];
|
||||
for (const row of indexRows.filter(r => /horde/i.test(String(r.name || '')))) {
|
||||
const key = norm(row.name);
|
||||
if (existing.has(key)) continue;
|
||||
const spec = HORDE_SPECS[row.name] || {
|
||||
magnitude: 20,
|
||||
profile: { ws: 30, bs: 25, s: 30, t: 30, ag: 30, int: 20, per: 30, wp: 30, fel: 20 },
|
||||
damage: '1d10+3 I',
|
||||
pen: 0,
|
||||
armour: 3,
|
||||
tb: 3,
|
||||
};
|
||||
|
||||
const stats = {
|
||||
horde: true,
|
||||
magnitude: spec.magnitude,
|
||||
profile: spec.profile,
|
||||
wounds: spec.magnitude,
|
||||
toughnessBonus: spec.tb,
|
||||
armour: spec.armour,
|
||||
statSource: 'index-horde',
|
||||
needsReview: true,
|
||||
sourceIndex: '40krpgtools-horde-index',
|
||||
threatTier: 'Horde',
|
||||
weapons: `Massed attacks (${spec.damage}; Pen ${spec.pen})`,
|
||||
};
|
||||
|
||||
arr.push({
|
||||
bestiaryName: row.name,
|
||||
type: row.type,
|
||||
affiliation: row.affiliation,
|
||||
book: row.book,
|
||||
page: row.page,
|
||||
sourceUrl: row.sourceUrl,
|
||||
wounds: spec.magnitude,
|
||||
stats,
|
||||
horde: true,
|
||||
magnitude: spec.magnitude,
|
||||
});
|
||||
existing.add(key);
|
||||
added.push(row.name);
|
||||
}
|
||||
return added;
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const index = JSON.parse(fs.readFileSync(INDEX_PATH, 'utf8'));
|
||||
const hordeRows = (index.entries || []).filter(row => /horde/i.test(String(row.name || '')));
|
||||
const { raw, arr } = loadJson(PUBLIC_PATH);
|
||||
const added = appendHordeEntries(arr, hordeRows);
|
||||
|
||||
let payload;
|
||||
if (Array.isArray(raw)) {
|
||||
payload = arr;
|
||||
} else {
|
||||
raw.results = arr;
|
||||
raw.count = arr.length;
|
||||
raw.generatedAt = new Date().toISOString();
|
||||
payload = raw;
|
||||
}
|
||||
const serialized = JSON.stringify(payload, null, 2);
|
||||
for (const file of [PUBLIC_PATH, DB_PATH]) {
|
||||
if (fs.existsSync(file)) fs.writeFileSync(file + '.pre-horde.' + Date.now() + '.json', fs.readFileSync(file));
|
||||
fs.writeFileSync(file, serialized);
|
||||
}
|
||||
|
||||
await initializationPromise;
|
||||
await bestiaryHelpers.replaceAll(arr);
|
||||
|
||||
console.log(`Imported ${added.length} horde entries into bestiary`);
|
||||
added.forEach(name => console.log(` + ${name}`));
|
||||
}
|
||||
|
||||
main()
|
||||
.then(() => process.exit(0))
|
||||
.catch(error => {
|
||||
console.error(error);
|
||||
process.exit(1);
|
||||
});
|
||||
@@ -0,0 +1,54 @@
|
||||
#!/usr/bin/env node
|
||||
/*
|
||||
* Replace the MariaDB bestiary table from the sanitized active bestiary JSON.
|
||||
* JSON remains a fallback/mirror; the app reads MariaDB first when available.
|
||||
*/
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const { pool, bestiaryHelpers, initializationPromise } = require('../database/mariadb');
|
||||
|
||||
const ROOT = path.join(__dirname, '..');
|
||||
const SOURCE = path.join(ROOT, 'public', 'deathwatch-bestiary-extracted.json');
|
||||
|
||||
function loadEntries() {
|
||||
const raw = JSON.parse(fs.readFileSync(SOURCE, 'utf8'));
|
||||
const entries = Array.isArray(raw) ? raw : (raw.results || raw.entries || raw.items || []);
|
||||
if (!Array.isArray(entries)) throw new Error(`Unsupported bestiary JSON shape: ${SOURCE}`);
|
||||
return entries.filter(entry => entry && (entry.bestiaryName || entry.name));
|
||||
}
|
||||
|
||||
async function ensureTable() {
|
||||
await pool.execute(`
|
||||
CREATE TABLE IF NOT EXISTS bestiary (
|
||||
id INT AUTO_INCREMENT PRIMARY KEY,
|
||||
name VARCHAR(255) NOT NULL,
|
||||
book VARCHAR(255),
|
||||
page VARCHAR(50),
|
||||
pdf VARCHAR(255),
|
||||
stats TEXT,
|
||||
profile TEXT,
|
||||
snippet TEXT,
|
||||
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
||||
updated_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP ON UPDATE CURRENT_TIMESTAMP,
|
||||
INDEX idx_bestiary_name (name),
|
||||
INDEX idx_bestiary_book_page (book, page)
|
||||
)
|
||||
`);
|
||||
}
|
||||
|
||||
async function main() {
|
||||
const entries = loadEntries();
|
||||
await initializationPromise;
|
||||
await ensureTable();
|
||||
const count = await bestiaryHelpers.replaceAll(entries);
|
||||
console.log(`Imported ${count} bestiary entries into MariaDB bestiary table`);
|
||||
}
|
||||
|
||||
main()
|
||||
.catch(error => {
|
||||
console.error(error);
|
||||
process.exitCode = 1;
|
||||
})
|
||||
.finally(async () => {
|
||||
await pool.end();
|
||||
});
|
||||
@@ -3,14 +3,21 @@
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const mysql = require('mysql2/promise');
|
||||
const dotenv = require('dotenv');
|
||||
|
||||
const repoRoot = path.resolve(__dirname, '..');
|
||||
dotenv.config({ path: path.join(repoRoot, 'database', '.env') });
|
||||
dotenv.config({ path: path.join(repoRoot, '.env') });
|
||||
const rulesPath = path.join(repoRoot, 'database', 'rules', 'rules-database.json');
|
||||
|
||||
if (!process.env.DB_PASSWORD) {
|
||||
throw new Error('DB_PASSWORD must be set in database/.env or .env');
|
||||
}
|
||||
|
||||
const dbConfig = {
|
||||
host: process.env.DB_HOST || '192.168.1.113',
|
||||
user: process.env.DB_USER || 'deathwatch',
|
||||
password: process.env.DB_PASSWORD || 'defaultpassword',
|
||||
password: process.env.DB_PASSWORD,
|
||||
database: process.env.DB_NAME || 'deathwatch',
|
||||
port: Number(process.env.DB_PORT || 3307)
|
||||
};
|
||||
|
||||
@@ -0,0 +1,329 @@
|
||||
#!/usr/bin/env node
|
||||
/*
|
||||
* sanitize-bestiary-canonical.js
|
||||
*
|
||||
* The extracted bestiary (public/deathwatch-bestiary-extracted.json) is the
|
||||
* effective data source for the dice roller and Bestiary tab (MariaDB has no
|
||||
* `bestiary` table, so loadBestiaryData() falls back to this JSON).
|
||||
*
|
||||
* The extraction was badly corrupted: distinct creatures shared mis-aligned
|
||||
* statblocks (e.g. Civilian / Servo-Skull / Imperial Agent all got the Hive
|
||||
* Tyrant's line, T78 / W120), and some NPCs stored Strength/Toughness BONUS
|
||||
* values (single digits) in place of the characteristic.
|
||||
*
|
||||
* This script rebuilds each entry from authoritative sources:
|
||||
* - CANON: statblocks read from the rulebook PDFs in database/rules/
|
||||
* (CR = Core Rulebook, MoX = Mark of the Xenos) via pdftotext.
|
||||
* - OCR: the 5 NPCs in database/remote-extract/all_deathwatch_statblocks.json.
|
||||
* - Structural fixes for anything else: single-digit S/T -> x10, wounds > 0,
|
||||
* Toughness Bonus recomputed (with Unnatural Toughness where noted).
|
||||
*
|
||||
* Every touched entry records `stats.statSource` ('pdf' | 'ocr' | 'structural')
|
||||
* and `stats.needsReview` for values that are canonical-fill rather than
|
||||
* directly verified, so the remainder can be audited later.
|
||||
*
|
||||
* Backups are written next to each target before it is overwritten.
|
||||
*/
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
const ROOT = path.join(__dirname, '..');
|
||||
const TARGETS = [
|
||||
path.join(ROOT, 'public', 'deathwatch-bestiary-extracted.json'),
|
||||
path.join(ROOT, 'database', 'deathwatch-bestiary-extracted.json'),
|
||||
];
|
||||
const OCR_PATH = path.join(ROOT, 'database', 'remote-extract', 'all_deathwatch_statblocks.json');
|
||||
|
||||
const P = (ws, bs, s, t, ag, int, per, wp, fel) => ({ ws, bs, s, t, ag, int, per, wp, fel });
|
||||
|
||||
// tb: effective Toughness Bonus (includes Unnatural Toughness where it applies).
|
||||
// armour: number = all locations; object = per-location {head,body,arm,leg} or {all}.
|
||||
// source 'pdf' = read from rulebook PDF; 'canon' = well-known value, flag review.
|
||||
const CANON = {
|
||||
// ---- Deathwatch Core Rulebook (verified from CR.pdf) ----
|
||||
'chaos space marine': { profile: P(50,50,65,45,40,35,40,43,17), wounds: 29, tb: 4, armour: { head:8, arm:8, body:10, leg:8 }, movement: '4/8/12/24', weapon: 'Astartes Bolter (2d10+5 X; Pen 5) / Chainsword', book: 'Deathwatch Core Rulebook', page: '363', source: 'pdf' },
|
||||
'tyranid warrior': { profile: P(55,30,60,50,44,20,35,50,10), wounds: 48, tb: 10, armour: { all:8 }, movement: '6/12/18/36', weapon: 'Scything Talons (1d10+14 R; Pen 3)', book: 'Deathwatch Core Rulebook', page: '370', source: 'pdf' },
|
||||
'hive tyrant': { profile: P(78,33,60,53,45,45,49,70,10), wounds: 120, tb: 15, armour: { all:10 }, movement: '7/14/21/42', weapon: 'Scything Talons (1d10+22 R; Pen 3)', book: 'Deathwatch Core Rulebook', page: '371', source: 'pdf' },
|
||||
'hormagaunt': { profile: P(45,20,35,30,55,10,40,30,5), wounds: 9, tb: 3, armour: { all:3 }, movement: '9/18/27/54', weapon: 'Scything Talons (1d10+5 R; Pen 0)', book: 'Deathwatch Core Rulebook', page: '372', source: 'pdf' },
|
||||
'termagant': { profile: P(30,30,30,30,30,14,30,30,5), wounds: 8, tb: 3, armour: { all:3 }, movement: '3/6/9/18', weapon: 'Fleshborer (10m; S/-/-; 1d10+6 R; Pen 0)', book: 'Deathwatch Core Rulebook', page: '372', source: 'canon' },
|
||||
'tau fire warrior': { profile: P(25,35,30,30,25,30,20,30,30), wounds: 12, tb: 3, armour: { all:6 }, movement: '3/6/9/18', weapon: 'Pulse Rifle (100m; S/3/-; 1d10+9 E; Pen 4)', book: 'Deathwatch Core Rulebook', page: '369', source: 'pdf' },
|
||||
'tau stealth suit': { profile: P(30,42,42,47,38,30,40,40,40), wounds: 30, tb: 4, armour: { head:8, body:8, arm:7, leg:7 }, movement: '4/8/12/28', weapon: 'Burst Cannon (60m; -/-/6; 1d10+8 E; Pen 4)', book: 'Deathwatch Core Rulebook', page: '368', source: 'pdf' },
|
||||
'gun drone': { profile: P(20,25,20,40,40,15,25,20,10), wounds: 15, tb: 4, armour: { all:0 }, movement: '-', weapon: 'Twin-linked Pulse Carbines (30m; S/3/-; 1d10+8 E; Pen 4)', book: 'Deathwatch Core Rulebook', page: '368', source: 'pdf' },
|
||||
'chaos heretic': { profile: P(35,35,35,35,30,20,30,35,20), wounds: 10, tb: 3, armour: { all:4 }, movement: '3/6/9/18', weapon: 'Autogun (90m; S/3/10; 1d10+3 I; Pen 0)', book: 'Deathwatch Core Rulebook', page: '361', source: 'pdf' },
|
||||
'renegade militia': { profile: P(30,30,30,30,30,25,30,30,25), wounds: 9, tb: 3, armour: { all:4 }, movement: '3/6/9/18', weapon: 'Lasgun (100m; S/3/-; 1d10+3 E; Pen 0)', book: 'Deathwatch Core Rulebook', page: '360', source: 'canon' },
|
||||
'imperial guardsman': { profile: P(35,35,30,30,30,30,30,30,30), wounds: 11, tb: 3, armour: { all:4 }, movement: '3/6/9/18', weapon: 'Lasgun (100m; S/3/-; 1d10+3 E; Pen 0)', book: 'Deathwatch Core Rulebook', page: '360', source: 'canon' },
|
||||
'battle servitor (erioch-pattern)': { profile: P(25,10,50,50,15,10,20,30,10), wounds: 20, tb: 5, armour: { all:7 }, movement: '3/6/9/18', weapon: 'Heavy Bolter / Power Fist', book: 'Deathwatch Core Rulebook', page: '', source: 'canon' },
|
||||
'monotask servitor (erioch-pattern)': { profile: P(20,10,45,45,15,10,20,30,10), wounds: 15, tb: 4, armour: { all:5 }, movement: '3/6/9/18', weapon: 'Servo-tools', book: 'Deathwatch Core Rulebook', page: '', source: 'canon' },
|
||||
'industrial servitor': { profile: P(20,10,45,45,15,10,20,30,10), wounds: 18, tb: 4, armour: { all:6 }, movement: '3/6/9/18', weapon: 'Industrial tools', book: 'Deathwatch Core Rulebook', page: '', source: 'canon' },
|
||||
'servo-skull': { profile: P(1,15,5,10,40,20,45,20,10), wounds: 3, tb: 1, armour: { all:2 }, movement: '4/8/12/24', weapon: 'None (recon drone)', book: 'Deathwatch Core Rulebook', page: '', source: 'canon' },
|
||||
'civilian': { profile: P(20,20,25,25,30,30,30,25,30), wounds: 6, tb: 2, armour: { all:0 }, movement: '4/8/12/24', weapon: 'Improvised', book: 'Deathwatch Core Rulebook', page: '', source: 'canon' },
|
||||
'chapter serf': { profile: P(30,30,35,35,30,35,30,35,35), wounds: 12, tb: 3, armour: { all:2 }, movement: '4/8/12/24', weapon: 'Laspistol / Combat Knife', book: 'Deathwatch Core Rulebook', page: '', source: 'canon' },
|
||||
'imperial agent': { profile: P(35,35,35,35,40,40,40,40,40), wounds: 12, tb: 3, armour: { all:4 }, movement: '4/8/12/24', weapon: 'Stub Automatic / Sword', book: 'Deathwatch Core Rulebook', page: '', source: 'canon' },
|
||||
'imperial guard field officer': { profile: P(35,35,35,40,35,40,40,40,45), wounds: 12, tb: 4, armour: { all:5 }, movement: '3/6/9/18', weapon: 'Bolt Pistol / Power Sword', book: 'Deathwatch Core Rulebook', page: '', source: 'canon' },
|
||||
'imperial guard senior officer': { profile: P(38,38,38,42,38,42,42,42,48), wounds: 16, tb: 4, armour: { all:5 }, movement: '3/6/9/18', weapon: 'Bolt Pistol / Power Sword', book: 'Deathwatch Core Rulebook', page: '', source: 'canon' },
|
||||
|
||||
// ---- Mark of the Xenos (verified / canonical) ----
|
||||
'tau commander': { profile: P(41,57,50,57,48,45,45,50,50), wounds: 90, tb: 5, armour: { all:9 }, movement: '12/24/36/72', weapon: 'Burst Cannon + Fusion Blaster (XV-8 Crisis Suit)', book: 'Mark of the Xenos', page: '', source: 'pdf' },
|
||||
'genestealer': { profile: P(55,0,45,40,50,20,40,40,10), wounds: 16, tb: 8, armour: { all:4 }, movement: '5/10/15/30', weapon: 'Rending Claws (1d10+12 R; Pen 5; Razor Sharp)', book: 'Mark of the Xenos', page: '44', source: 'pdf' },
|
||||
'ork nob': { profile: P(45,20,45,45,33,20,30,40,25), wounds: 22, tb: 5, armour: { head:6, body:6, arm:4, leg:4 }, movement: '4/8/12/24', weapon: "Big Choppa (2d10+9 R; Pen 2) / Slugga", book: 'Mark of the Xenos', page: '', source: 'canon' },
|
||||
'daemon prince': { profile: P(65,40,70,70,50,45,45,60,40), wounds: 80, tb: 14, armour: { all:8 }, movement: '6/12/18/36', weapon: 'Daemon Weapon (2d10+14 R/E; Pen 8)', book: 'Mark of the Xenos', page: '', source: 'canon' },
|
||||
|
||||
// ---- Named characters verified from CR.pdf ----
|
||||
'apostle bellephrades': { profile: P(75,60,66,45,48,48,48,75,30), wounds: 80, tb: 8, armour: { all:12 }, movement: '4/8/12/24', weapon: 'Chaos-forged sword (2d10+25 R/E/I; Pen 6; Felling, Unwieldy)', book: 'Deathwatch Core Rulebook', page: '363', source: 'pdf' },
|
||||
};
|
||||
|
||||
// Alternate spellings in the data -> canonical key
|
||||
const ALIASES = {
|
||||
'tyrannid warrior': 'tyranid warrior',
|
||||
'tau commander (crisis suit)': 'tau commander',
|
||||
};
|
||||
|
||||
function norm(name) {
|
||||
return String(name || '').trim().toLowerCase().replace(/\s+/g, ' ');
|
||||
}
|
||||
|
||||
function tbFrom(profile) {
|
||||
const t = Number(profile && profile.t) || 0;
|
||||
return Math.floor(t / 10);
|
||||
}
|
||||
|
||||
const SECTION_HEADERS = [
|
||||
'Skills', 'Talents', 'Traits', 'Armour', 'Armor', 'Weapons', 'Gear',
|
||||
'Special Rules', 'Special Abilities', 'Psychic Powers', 'Tactical Demeanour',
|
||||
'Solo Mode', 'Squad Mode', 'Marks of Chaos', 'Biomorph',
|
||||
];
|
||||
|
||||
function normalizeText(value) {
|
||||
return String(value || '')
|
||||
.replace(/\s+/g, ' ')
|
||||
.replace(/\s+([,.;:])/g, '$1')
|
||||
.trim();
|
||||
}
|
||||
|
||||
function splitTopLevelList(text) {
|
||||
const items = [];
|
||||
let current = '';
|
||||
let depth = 0;
|
||||
for (const ch of text) {
|
||||
if (ch === '(') depth += 1;
|
||||
if (ch === ')' && depth > 0) depth -= 1;
|
||||
if (ch === ',' && depth === 0) {
|
||||
if (current.trim()) items.push(current.trim());
|
||||
current = '';
|
||||
} else {
|
||||
current += ch;
|
||||
}
|
||||
}
|
||||
if (current.trim()) items.push(current.trim());
|
||||
return items;
|
||||
}
|
||||
|
||||
function stripAfterNextSection(text, currentHeader) {
|
||||
let cleaned = normalizeText(text);
|
||||
if (!cleaned) return '';
|
||||
const start = new RegExp(`^${currentHeader}\\s*:`, 'i');
|
||||
cleaned = cleaned.replace(start, '').trim();
|
||||
|
||||
const otherHeaders = SECTION_HEADERS.filter(h => h.toLowerCase() !== currentHeader.toLowerCase());
|
||||
let cutAt = -1;
|
||||
for (const header of otherHeaders) {
|
||||
const match = new RegExp(`(?:^|[.\\s])${header}\\s*:`, 'i').exec(cleaned);
|
||||
if (match && (cutAt === -1 || match.index < cutAt)) cutAt = match.index;
|
||||
}
|
||||
if (cutAt >= 0) cleaned = cleaned.slice(0, cutAt);
|
||||
return cleaned.replace(/[.;:\s]+$/, '').trim();
|
||||
}
|
||||
|
||||
function compactListField(value, currentHeader, maxItems = 10, maxChars = 260) {
|
||||
const text = stripAfterNextSection(value, currentHeader);
|
||||
if (!text) return '';
|
||||
const items = splitTopLevelList(text)
|
||||
.map(item => item.replace(/[.;:\s]+$/, '').trim())
|
||||
.filter(Boolean);
|
||||
if (!items.length) return text.slice(0, maxChars).trim();
|
||||
const visible = items.slice(0, maxItems);
|
||||
const suffix = items.length > maxItems ? `, +${items.length - maxItems} more` : '';
|
||||
const joined = visible.join(', ') + suffix;
|
||||
return joined.length > maxChars ? joined.slice(0, maxChars - 1).trimEnd() + '...' : joined;
|
||||
}
|
||||
|
||||
function compactTextField(value, currentHeader, maxChars = 220) {
|
||||
const text = stripAfterNextSection(value, currentHeader);
|
||||
if (!text) return '';
|
||||
return text.length > maxChars ? text.slice(0, maxChars - 1).trimEnd() + '...' : text;
|
||||
}
|
||||
|
||||
function deleteEmpty(obj, key) {
|
||||
if (obj[key] == null || obj[key] === '') delete obj[key];
|
||||
}
|
||||
|
||||
function compactEntry(entry) {
|
||||
delete entry.pageText;
|
||||
delete entry.snippet;
|
||||
|
||||
const stats = entry.stats || (entry.stats = {});
|
||||
delete stats.pageText;
|
||||
delete stats.snippet;
|
||||
|
||||
stats.skills = compactListField(stats.skills || entry.skills, 'Skills', 8, 220);
|
||||
stats.talents = compactListField(stats.talents || entry.talents, 'Talents', 10, 260);
|
||||
stats.traits = compactListField(stats.traits || entry.traits, 'Traits', 10, 260);
|
||||
stats.weapons = compactTextField(stats.weapons || entry.weapons, 'Weapons', 260);
|
||||
stats.gear = compactTextField(stats.gear || entry.gear, 'Gear', 180);
|
||||
|
||||
for (const key of ['skills', 'talents', 'traits', 'weapons', 'gear']) {
|
||||
delete entry[key];
|
||||
deleteEmpty(stats, key);
|
||||
}
|
||||
}
|
||||
|
||||
// Fix a profile that stored single-digit Strength/Toughness *bonuses* as the
|
||||
// characteristic (e.g. S 4 -> S 40). Only touches values 1..9 with a matching
|
||||
// two-digit sibling elsewhere, to avoid clobbering genuinely-low machine stats.
|
||||
function fixBonusChars(profile) {
|
||||
let changed = false;
|
||||
for (const k of ['s', 't', 'ws', 'bs', 'ag']) {
|
||||
const v = Number(profile[k]);
|
||||
if (Number.isFinite(v) && v >= 1 && v <= 9) { profile[k] = v * 10; changed = true; }
|
||||
}
|
||||
return changed;
|
||||
}
|
||||
|
||||
function loadOcr() {
|
||||
try {
|
||||
const list = JSON.parse(fs.readFileSync(OCR_PATH, 'utf8'));
|
||||
const map = {};
|
||||
for (const e of list) map[norm(e.name)] = e;
|
||||
return map;
|
||||
} catch { return {}; }
|
||||
}
|
||||
|
||||
function applyCanon(entry, c) {
|
||||
const stats = entry.stats || (entry.stats = {});
|
||||
stats.profile = { ...c.profile };
|
||||
stats.wounds = c.wounds;
|
||||
entry.wounds = c.wounds;
|
||||
stats.movement = c.movement;
|
||||
stats.toughnessBonus = c.tb; // effective TB (incl. Unnatural Toughness)
|
||||
stats.armour = c.armour;
|
||||
stats.weapons = c.weapon;
|
||||
delete stats.skills;
|
||||
delete stats.talents;
|
||||
delete stats.traits;
|
||||
delete stats.gear;
|
||||
delete stats.special;
|
||||
delete entry.skills;
|
||||
delete entry.talents;
|
||||
delete entry.traits;
|
||||
delete entry.gear;
|
||||
delete entry.special;
|
||||
stats.statSource = c.source === 'pdf' ? 'pdf' : 'canon';
|
||||
stats.needsReview = c.source !== 'pdf';
|
||||
if (c.book) entry.book = c.book;
|
||||
if (c.page) entry.page = c.page;
|
||||
}
|
||||
|
||||
function applyOcr(entry, o) {
|
||||
const stats = entry.stats || (entry.stats = {});
|
||||
const s = o.stats || {};
|
||||
stats.profile = { ws:s.ws, bs:s.bs, s:s.s, t:s.t, ag:s.ag, int:s.int, per:s.per, wp:s.wp, fel:s.fel };
|
||||
const w = Number(o.wounds) > 0 ? Number(o.wounds) : (Number(entry.wounds) || Number(stats.wounds) || 15);
|
||||
stats.wounds = w; entry.wounds = w;
|
||||
if (o.movement) stats.movement = String(o.movement).replace(/\s*Wounds:.*$/i, '').trim();
|
||||
stats.toughnessBonus = tbFrom(stats.profile);
|
||||
stats.statSource = 'ocr';
|
||||
stats.needsReview = false;
|
||||
}
|
||||
|
||||
function titleCase(key) {
|
||||
return key.replace(/\b\w/g, c => c.toUpperCase());
|
||||
}
|
||||
|
||||
function newEntryFromCanon(key, c) {
|
||||
const entry = { bestiaryName: titleCase(key), book: c.book || '', page: c.page || '', pdf: 'sanitize-bestiary-canonical.js', stats: {} };
|
||||
applyCanon(entry, c);
|
||||
return entry;
|
||||
}
|
||||
|
||||
function sanitize(arr, ocr) {
|
||||
const report = { pdf: [], canon: [], ocr: [], structural: [], added: [] };
|
||||
const usedCanon = new Set();
|
||||
for (const entry of arr) {
|
||||
if (norm(entry.bestiaryName || entry.name) === 'genestealer' && /purestrain-genestealer/i.test(entry.sourceUrl || '')) {
|
||||
entry.bestiaryName = 'Purestrain Genestealer';
|
||||
entry.page = '39';
|
||||
}
|
||||
const name = norm(entry.bestiaryName || entry.name);
|
||||
const key = ALIASES[name] || name;
|
||||
if (CANON[key]) {
|
||||
if (key !== name) entry.bestiaryName = titleCase(key);
|
||||
usedCanon.add(key);
|
||||
applyCanon(entry, CANON[key]);
|
||||
(CANON[key].source === 'pdf' ? report.pdf : report.canon).push(entry.bestiaryName || entry.name);
|
||||
continue;
|
||||
}
|
||||
if (ocr[name]) {
|
||||
applyOcr(entry, ocr[name]);
|
||||
report.ocr.push(entry.bestiaryName || entry.name);
|
||||
continue;
|
||||
}
|
||||
// Structural fixes for everything else.
|
||||
const stats = entry.stats || (entry.stats = {});
|
||||
const profile = stats.profile || (stats.profile = {});
|
||||
const fixed = fixBonusChars(profile);
|
||||
let w = Number(entry.wounds ?? stats.wounds);
|
||||
if (!Number.isFinite(w) || w <= 0) w = 15;
|
||||
stats.wounds = w; entry.wounds = w;
|
||||
stats.toughnessBonus = Number(stats.toughnessBonus) || tbFrom(profile);
|
||||
const hasPdfProfile = (stats.statSource === 'pdf' || stats.sourceIndex === '40krpgtools')
|
||||
&& ['ws', 'bs', 's', 't', 'ag', 'int', 'per', 'wp', 'fel'].every(k => Number.isFinite(Number(profile[k])));
|
||||
if (hasPdfProfile) {
|
||||
stats.statSource = 'pdf';
|
||||
stats.needsReview = false;
|
||||
report.pdf.push(entry.bestiaryName || entry.name);
|
||||
} else {
|
||||
stats.statSource = 'structural';
|
||||
stats.needsReview = true;
|
||||
report.structural.push((entry.bestiaryName || entry.name) + (fixed ? ' (S/T x10)' : ''));
|
||||
}
|
||||
}
|
||||
// Add any canonical creatures that had no matching entry (e.g. standalone
|
||||
// Genestealer, Apostle Bellephrades) so the full set is present.
|
||||
for (const key of Object.keys(CANON)) {
|
||||
if (usedCanon.has(key)) continue;
|
||||
arr.push(newEntryFromCanon(key, CANON[key]));
|
||||
report.added.push(titleCase(key));
|
||||
}
|
||||
for (const entry of arr) compactEntry(entry);
|
||||
return report;
|
||||
}
|
||||
|
||||
function main() {
|
||||
const ocr = loadOcr();
|
||||
// Sanitize the effective source (public), then mirror it to the stale DB copy
|
||||
// so both agree.
|
||||
const source = TARGETS[0];
|
||||
const raw = JSON.parse(fs.readFileSync(source, 'utf8'));
|
||||
const arr = Array.isArray(raw) ? raw : (raw.results || raw.entries || raw.items);
|
||||
if (!Array.isArray(arr)) { console.log('unknown shape:', source); return; }
|
||||
const report = sanitize(arr, ocr);
|
||||
if (!Array.isArray(raw)) raw.count = arr.length;
|
||||
const serialized = JSON.stringify(raw, null, 2);
|
||||
for (const target of TARGETS) {
|
||||
if (target !== source && !fs.existsSync(path.dirname(target))) continue;
|
||||
if (fs.existsSync(target)) fs.writeFileSync(target + '.pre-sanitize.' + Date.now() + '.json', fs.readFileSync(target));
|
||||
fs.writeFileSync(target, serialized);
|
||||
console.log('wrote ->', path.relative(ROOT, target), '(' + arr.length + ' entries)');
|
||||
}
|
||||
if (report) {
|
||||
const line = (label, a) => console.log(` ${label} (${a.length}): ${a.join(', ')}`);
|
||||
console.log('\nResults:');
|
||||
line('PDF-verified', report.pdf);
|
||||
line('Canonical-fill (review)', report.canon);
|
||||
line('OCR-restored', report.ocr);
|
||||
line('Structural-only (review)', report.structural);
|
||||
line('Added (new entries)', report.added);
|
||||
}
|
||||
}
|
||||
|
||||
main();
|
||||
@@ -4,7 +4,7 @@ Comprehensive Deathwatch rulebook scraper.
|
||||
Extracts weapons, armour, gear, talents, traits, psychic powers, and bestiary
|
||||
from PDFs and imports them into MariaDB.
|
||||
"""
|
||||
import re, json, sys
|
||||
import os, re, json, sys
|
||||
import pdfplumber
|
||||
import fitz
|
||||
import mysql.connector
|
||||
@@ -12,9 +12,29 @@ from pathlib import Path
|
||||
|
||||
REPO = Path(__file__).resolve().parent.parent
|
||||
RULES = REPO / 'database' / 'rules'
|
||||
DB = dict(host='192.168.1.113', port=3307, user='deathwatch',
|
||||
password=open(REPO/'.env.db').read().strip() if (REPO/'.env.db').exists() else 'DwRoller@2025!',
|
||||
database='deathwatch')
|
||||
|
||||
def load_env_file(path):
|
||||
if not path.exists():
|
||||
return
|
||||
for line in path.read_text().splitlines():
|
||||
line = line.strip()
|
||||
if not line or line.startswith('#') or '=' not in line:
|
||||
continue
|
||||
key, value = line.split('=', 1)
|
||||
os.environ.setdefault(key.strip(), value.strip().strip('"').strip("'"))
|
||||
|
||||
load_env_file(REPO / 'database' / '.env')
|
||||
load_env_file(REPO / '.env')
|
||||
if not os.environ.get('DB_PASSWORD'):
|
||||
raise RuntimeError('DB_PASSWORD must be set in database/.env or .env')
|
||||
|
||||
DB = dict(
|
||||
host=os.environ.get('DB_HOST', '192.168.1.113'),
|
||||
port=int(os.environ.get('DB_PORT', '3307')),
|
||||
user=os.environ.get('DB_USER', 'deathwatch'),
|
||||
password=os.environ['DB_PASSWORD'],
|
||||
database=os.environ.get('DB_NAME', 'deathwatch'),
|
||||
)
|
||||
|
||||
def clean(s):
|
||||
if not s: return ''
|
||||
|
||||
@@ -15,7 +15,7 @@ Usage:
|
||||
python3 scripts/scrape-rules-from-pdfs.py --dry-run # print, don't save
|
||||
"""
|
||||
|
||||
import re, sys
|
||||
import os, re, sys
|
||||
from pathlib import Path
|
||||
import fitz # PyMuPDF
|
||||
import mysql.connector
|
||||
@@ -24,9 +24,27 @@ import mysql.connector
|
||||
REPO = Path(__file__).resolve().parent.parent
|
||||
PDF_DIR = REPO / 'database' / 'rules'
|
||||
|
||||
def load_env_file(path):
|
||||
if not path.exists():
|
||||
return
|
||||
for line in path.read_text().splitlines():
|
||||
line = line.strip()
|
||||
if not line or line.startswith('#') or '=' not in line:
|
||||
continue
|
||||
key, value = line.split('=', 1)
|
||||
os.environ.setdefault(key.strip(), value.strip().strip('"').strip("'"))
|
||||
|
||||
load_env_file(REPO / 'database' / '.env')
|
||||
load_env_file(REPO / '.env')
|
||||
if not os.environ.get('DB_PASSWORD'):
|
||||
raise RuntimeError('DB_PASSWORD must be set in database/.env or .env')
|
||||
|
||||
DB = dict(
|
||||
host='192.168.1.113', port=3307, user='deathwatch',
|
||||
password='DwRoller@2025!', database='deathwatch'
|
||||
host=os.environ.get('DB_HOST', '192.168.1.113'),
|
||||
port=int(os.environ.get('DB_PORT', '3307')),
|
||||
user=os.environ.get('DB_USER', 'deathwatch'),
|
||||
password=os.environ['DB_PASSWORD'],
|
||||
database=os.environ.get('DB_NAME', 'deathwatch'),
|
||||
)
|
||||
|
||||
# Book definitions: key → (filename, full_name, abbreviation, page_ranges)
|
||||
|
||||
@@ -45,12 +45,15 @@ function normalizeTitle(t) {
|
||||
}
|
||||
|
||||
async function getPool() {
|
||||
if (!process.env.DB_PASSWORD) {
|
||||
throw new Error('DB_PASSWORD must be set in database/.env or .env');
|
||||
}
|
||||
return mysql.createPool({
|
||||
host: '192.168.1.113',
|
||||
port: 3307,
|
||||
user: 'deathwatch',
|
||||
password: process.env.DB_PASSWORD || 'DwRoller@2025!',
|
||||
database: 'deathwatch',
|
||||
host: process.env.DB_HOST || '192.168.1.113',
|
||||
port: Number(process.env.DB_PORT || 3307),
|
||||
user: process.env.DB_USER || 'deathwatch',
|
||||
password: process.env.DB_PASSWORD,
|
||||
database: process.env.DB_NAME || 'deathwatch',
|
||||
waitForConnections: true,
|
||||
connectionLimit: 5,
|
||||
});
|
||||
|
||||
@@ -217,11 +217,26 @@ def parse_talents_from_cr(cr_path):
|
||||
|
||||
def get_db():
|
||||
import mysql.connector
|
||||
import os
|
||||
def load_env_file(path):
|
||||
if not path.exists():
|
||||
return
|
||||
for line in path.read_text().splitlines():
|
||||
line = line.strip()
|
||||
if not line or line.startswith('#') or '=' not in line:
|
||||
continue
|
||||
key, value = line.split('=', 1)
|
||||
os.environ.setdefault(key.strip(), value.strip().strip('"').strip("'"))
|
||||
load_env_file(REPO / 'database' / '.env')
|
||||
load_env_file(REPO / '.env')
|
||||
if not os.environ.get('DB_PASSWORD'):
|
||||
raise RuntimeError('DB_PASSWORD must be set in database/.env or .env')
|
||||
return mysql.connector.connect(
|
||||
host="192.168.1.113", port=3307,
|
||||
user="deathwatch",
|
||||
password=open(REPO / ".env.db").read().strip() if (REPO / ".env.db").exists() else "DwRoller@2025!",
|
||||
database="deathwatch"
|
||||
host=os.environ.get('DB_HOST', '192.168.1.113'),
|
||||
port=int(os.environ.get('DB_PORT', '3307')),
|
||||
user=os.environ.get('DB_USER', 'deathwatch'),
|
||||
password=os.environ['DB_PASSWORD'],
|
||||
database=os.environ.get('DB_NAME', 'deathwatch'),
|
||||
)
|
||||
|
||||
def run():
|
||||
|
||||
Reference in New Issue
Block a user