feat: join dump-catalog provenance in database

This commit is contained in:
Abdessamad Derraz committed 2026-08-08 02:25:44 +02:00
1 parent 933d258df5
commit f4b02c2a43
9 files changed
+2664 -1

No files matched your search

+89
View File
@@ -1144,6 +1144,7 @@ import re
_TIMESTAMP_PATTERNS = [
re.compile(r'"generated_at":\s*"[^"]*"'), # database.json
re.compile(r'"imported_at":\s*"[^"]*"'), # provenance snapshots
re.compile(r"\*Auto-generated on [^*]*\*"), # README.md
re.compile(r"\*Generated on [^*]*\*"), # docs site pages
]
@@ -1335,3 +1336,91 @@ def build_target_cores_cache(
raise
kept = [p for p in platforms if p not in skip]
return cache, kept
DEFAULT_PROVENANCE_DIR = "provenance"
def load_provenance_snapshots(provenance_dir: str = DEFAULT_PROVENANCE_DIR) -> dict:
"""Load dump-catalog snapshots from provenance/*.json.
Returns {source_name: snapshot} where snapshot holds the normalized
entries written by the redump scraper or the pack importer. Missing
directory means no snapshots: returns an empty dict.
"""
snapshots = {}
prov_path = Path(provenance_dir)
if not prov_path.is_dir():
return snapshots
for path in sorted(prov_path.glob("*.json")):
with open(path) as f:
snapshot = json.load(f)
source = snapshot.get("source")
if source and snapshot.get("entries"):
snapshots[source] = snapshot
return snapshots
def build_provenance_index(snapshots: dict) -> dict:
"""Index snapshot entries by sha1 and by (md5, size) per source.
First entry wins on hash collisions within a source; entries are
pre-sorted at snapshot write time so the outcome is deterministic.
"""
index = {}
for source, snapshot in snapshots.items():
by_sha1 = {}
by_md5_size = {}
for entry in snapshot["entries"]:
sha1 = entry.get("sha1", "")
md5 = entry.get("md5", "")
if sha1 and sha1 not in by_sha1:
by_sha1[sha1] = entry
if md5 and entry.get("size"):
key = (md5, entry["size"])
if key not in by_md5_size:
by_md5_size[key] = entry
index[source] = {"by_sha1": by_sha1, "by_md5_size": by_md5_size}
return index
def annotate_provenance(files: dict, snapshots: dict) -> dict[str, int]:
"""Attach a provenance field to database file entries.
Matches by SHA1 first, then MD5 + size. Returns per-source match
counts. Files without any catalog match keep no provenance field.
"""
index = build_provenance_index(snapshots)
counts = dict.fromkeys(index, 0)
for sha1, entry in files.items():
matches = {}
for source in sorted(index):
src_index = index[source]
hit = src_index["by_sha1"].get(sha1) or src_index["by_md5_size"].get(
(entry.get("md5", ""), entry.get("size", 0))
)
if hit:
matches[source] = {
"dat": hit.get("dat", ""),
"name": hit.get("name", ""),
"description": hit.get("description", ""),
}
counts[source] += 1
if matches:
entry["provenance"] = matches
else:
entry.pop("provenance", None)
return counts
def write_provenance_snapshot(
path: str, source: str, imported_at: str, dats: dict, entries: list[dict]
) -> bool:
"""Write a normalized provenance snapshot, sorted for determinism."""
snapshot = {
"source": source,
"imported_at": imported_at,
"dats": dict(sorted(dats.items())),
"entries": sorted(entries, key=lambda e: (e["dat"], e["name"])),
}
return write_if_changed(path, json.dumps(snapshot, indent=2) + "\n")
+22 -1
View File
@@ -18,7 +18,14 @@ from datetime import datetime, timezone
from pathlib import Path
sys.path.insert(0, os.path.dirname(__file__))
from common import compute_hashes, list_registered_platforms, write_if_changed
from common import (
DEFAULT_PROVENANCE_DIR,
annotate_provenance,
compute_hashes,
list_registered_platforms,
load_provenance_snapshots,
write_if_changed,
)
CACHE_DIR = ".cache"
CACHE_FILE = os.path.join(CACHE_DIR, "db_cache.json")
@@ -296,6 +303,11 @@ def main():
parser.add_argument(
"--output", "-o", default=DEFAULT_OUTPUT, help="Output JSON file"
)
parser.add_argument(
"--provenance-dir",
default=DEFAULT_PROVENANCE_DIR,
help="Directory with dump-catalog snapshots",
)
args = parser.parse_args()
bios_dir = Path(args.bios_dir)
@@ -323,6 +335,9 @@ def main():
aliases[sha1] = []
aliases[sha1].append(alias_entry)
snapshots = load_provenance_snapshots(args.provenance_dir)
provenance_counts = annotate_provenance(files, snapshots)
indexes = build_indexes(files, aliases)
total_size = sum(entry["size"] for entry in files.values())
@@ -344,6 +359,12 @@ def main():
status = "Generated" if written else "Unchanged"
print(f"{status} {args.output}: {len(files)} files, {total_size:,} bytes total")
print(f" Name index: {name_count} names ({alias_count} aliases)")
if provenance_counts:
matched = sum(1 for e in files.values() if "provenance" in e)
per_source = ", ".join(
f"{source} {count}" for source, count in sorted(provenance_counts.items())
)
print(f" Provenance: {matched} files catalog-matched ({per_source})")
return 0
+9
View File
@@ -3,6 +3,7 @@
Steps:
1. generate_db.py --force (rebuild database.json from bios/)
1b. provenance_report.py (dump-catalog coverage from provenance/)
2. refresh_data_dirs.py (update Dolphin Sys, PPSSPP, etc.)
3. verify.py --all (check all platforms)
4. generate_pack.py --all (build ZIP packs)
@@ -207,6 +208,14 @@ def main():
print("\nDatabase generation failed, aborting.")
sys.exit(1)
# Step 1b: Dump-catalog coverage (offline, reads provenance/ snapshots)
ok, _ = run(
[sys.executable, "scripts/provenance_report.py"],
"1b provenance report",
)
results["provenance"] = ok
all_ok = all_ok and ok
# Step 2: Refresh data directories
if not args.offline:
ok, out = run(
+118
View File
@@ -0,0 +1,118 @@
#!/usr/bin/env python3
"""Report dump-catalog coverage of the collection.
Reads provenance/*.json snapshots and database.json, then reports per
source how many catalog entries the collection holds and which are
missing. Missing entries are acquisition targets: catalog-verified
dumps the collection does not have yet.
Usage:
python scripts/provenance_report.py
python scripts/provenance_report.py --missing
python scripts/provenance_report.py --json
"""
from __future__ import annotations
import argparse
import json
import os
import sys
sys.path.insert(0, os.path.dirname(__file__))
from common import DEFAULT_PROVENANCE_DIR, load_database, load_provenance_snapshots
def build_report(db: dict, snapshots: dict) -> dict:
"""Compare each snapshot against the collection.
A DAT counts as covered when the collection holds at least one of
its entries. Missing entries from covered DATs are acquisition
targets; missing entries from DATs the collection does not cover at
all are out of scope and only counted. No-Intro tags every non-game
dump "[BIOS]", including digital title distribution, so without this
split the target list is swamped by content the project never ships.
"""
by_sha1 = db.get("files", {})
by_md5_size = {
(entry.get("md5", ""), entry.get("size", 0)) for entry in by_sha1.values()
}
report = {}
for source, snapshot in sorted(snapshots.items()):
matched = 0
covered_dats = set()
unmatched = []
for entry in snapshot["entries"]:
if entry.get("sha1") in by_sha1 or (
entry.get("md5"),
entry.get("size"),
) in by_md5_size:
matched += 1
covered_dats.add(entry.get("dat", ""))
else:
unmatched.append(entry)
missing = [e for e in unmatched if e.get("dat", "") in covered_dats]
out_of_scope = len(unmatched) - len(missing)
report[source] = {
"imported_at": snapshot.get("imported_at", ""),
"dats": snapshot.get("dats", {}),
"total": len(snapshot["entries"]),
"matched": matched,
"covered_dats": sorted(covered_dats),
"missing": missing,
"out_of_scope": out_of_scope,
}
return report
def main() -> int:
parser = argparse.ArgumentParser(description="Dump-catalog coverage report")
parser.add_argument("--db", default="database.json", help="Database path")
parser.add_argument(
"--provenance-dir",
default=DEFAULT_PROVENANCE_DIR,
help="Directory with dump-catalog snapshots",
)
parser.add_argument("--missing", action="store_true", help="List missing entries")
parser.add_argument("--json", action="store_true", help="Full report as JSON")
args = parser.parse_args()
db = load_database(args.db)
snapshots = load_provenance_snapshots(args.provenance_dir)
if not snapshots:
print(f"No provenance snapshots in {args.provenance_dir}/")
return 0
report = build_report(db, snapshots)
if args.json:
print(json.dumps(report, indent=2))
return 0
for source, data in report.items():
in_scope = data["matched"] + len(data["missing"])
pct = 100 * data["matched"] / in_scope if in_scope else 0
print(
f" {source} ({data['imported_at']}): "
f"{data['matched']}/{in_scope} in collection ({pct:.0f}%) "
f"across {len(data['covered_dats'])} covered DATs"
)
if data["out_of_scope"]:
print(
f" {data['out_of_scope']} entries in DATs the collection "
f"does not cover, not counted as targets"
)
if args.missing:
for entry in data["missing"]:
label = entry.get("description") or entry["name"]
print(f" MISSING {entry['name']} ({label}) sha1={entry.get('sha1')}")
total_missing = sum(len(d["missing"]) for d in report.values())
if total_missing:
print(f" {total_missing} catalog entries missing from collection")
return 0
if __name__ == "__main__":
sys.exit(main())
+135
View File
@@ -0,0 +1,135 @@
"""Import BIOS entries from a No-Intro or TOSEC DAT pack.
Dat-o-Matic blocks automated downloads and TOSEC ships yearly packs,
so both are imported from a locally downloaded archive:
python -m scripts.scraper.dat_pack_importer --source no-intro --pack lovepack.zip
python -m scripts.scraper.dat_pack_importer --source tosec --pack TOSEC-v2025-03-13.zip
No-Intro marks BIOS dumps with a "[BIOS]" game name prefix inside the
system DATs. TOSEC groups firmware into dedicated DATs whose name
contains "Firmware" or "BIOS". Only those entries are imported.
"""
from __future__ import annotations
import argparse
import sys
import zipfile
from datetime import datetime, timezone
from pathlib import Path
from ..common import write_provenance_snapshot
from .logiqx_parser import LogiqxDat, parse_logiqx
MAX_MEMBER_SIZE = 100 * 1024 * 1024
SOURCES = ("no-intro", "tosec")
def _bios_entries(source: str, dat: LogiqxDat) -> list[dict]:
"""Filter a parsed DAT down to its BIOS/firmware entries."""
if source == "no-intro":
roms = [r for r in dat.roms if r.game.startswith("[BIOS]")]
else:
dat_name = dat.name.casefold()
if "firmware" not in dat_name and "bios" not in dat_name:
return []
roms = dat.roms
return [
{
"dat": dat.name,
"name": rom.name,
"description": rom.description,
"size": rom.size,
"crc32": rom.crc32,
"md5": rom.md5,
"sha1": rom.sha1,
}
for rom in roms
]
def _iter_dat_contents(pack: Path):
"""Yield (member name, content bytes) for DAT files in a pack.
Accepts a ZIP archive or a directory of extracted DATs.
"""
if pack.is_dir():
for path in sorted(pack.rglob("*")):
if path.suffix.lower() in (".dat", ".xml") and path.is_file():
yield str(path), path.read_bytes()
return
with zipfile.ZipFile(pack) as zf:
for info in sorted(zf.infolist(), key=lambda i: i.filename):
suffix = Path(info.filename).suffix.lower()
if info.is_dir() or suffix not in (".dat", ".xml"):
continue
if info.file_size > MAX_MEMBER_SIZE:
print(f" Skipping {info.filename}: exceeds {MAX_MEMBER_SIZE} bytes")
continue
yield info.filename, zf.read(info)
def import_pack(source: str, pack: Path) -> tuple[dict, list[dict], int]:
"""Import BIOS entries from a pack. Returns (dats, entries, skipped)."""
dats = {}
entries = []
skipped = 0
for member, content in _iter_dat_contents(pack):
try:
dat = parse_logiqx(content)
except (ValueError, SyntaxError):
skipped += 1
continue
dat_entries = _bios_entries(source, dat)
if dat_entries:
dats[dat.name] = dat.version
entries.extend(dat_entries)
return dats, entries, skipped
def main() -> int:
parser = argparse.ArgumentParser(description="Import BIOS provenance from a DAT pack")
parser.add_argument("--source", required=True, choices=SOURCES)
parser.add_argument("--pack", required=True, help="DAT pack (ZIP or directory)")
parser.add_argument("--output", "-o", help="Snapshot output path")
parser.add_argument("--dry-run", action="store_true", help="Show imported entries")
args = parser.parse_args()
pack = Path(args.pack)
if not pack.exists():
print(f"Error: pack '{pack}' not found", file=sys.stderr)
return 1
try:
dats, entries, skipped = import_pack(args.source, pack)
except zipfile.BadZipFile as e:
print(f"Error: {pack} is not a valid ZIP archive: {e}", file=sys.stderr)
return 1
if not entries:
print(f"Error: no BIOS entries found in {pack}", file=sys.stderr)
return 1
if skipped:
print(f" Skipped {skipped} non-Logiqx members")
if args.dry_run:
for dat in sorted(dats):
count = sum(1 for e in entries if e["dat"] == dat)
print(f" {dat} ({dats[dat]}): {count} entries")
print(f"\nTotal: {len(entries)} entries across {len(dats)} DATs")
return 0
output = args.output or f"provenance/{args.source}.json"
Path(output).parent.mkdir(parents=True, exist_ok=True)
imported_at = datetime.now(timezone.utc).strftime("%Y-%m-%d")
written = write_provenance_snapshot(output, args.source, imported_at, dats, entries)
status = "Written" if written else "Unchanged"
print(f"{status} {output}: {len(entries)} entries from {len(dats)} DATs")
return 0
if __name__ == "__main__":
sys.exit(main())
+92
View File
@@ -0,0 +1,92 @@
"""Parser for Logiqx XML DAT format.
Parses files like No-Intro daily packs and TOSEC releases which use:
<datafile>
<header><name>...</name><version>...</version></header>
<game name="[BIOS] Game Boy Advance (World)">
<description>...</description>
<rom name="..." size="16384" crc="..." md5="..." sha1="..."/>
</game>
</datafile>
"""
from __future__ import annotations
import xml.etree.ElementTree as ET
from dataclasses import dataclass, field
@dataclass
class LogiqxRom:
"""A ROM entry from a Logiqx DAT file."""
game: str
description: str
name: str
size: int
crc32: str
md5: str
sha1: str
@dataclass
class LogiqxDat:
"""A parsed Logiqx DAT file."""
name: str = ""
version: str = ""
roms: list[LogiqxRom] = field(default_factory=list)
def parse_logiqx(content: str | bytes) -> LogiqxDat:
"""Parse Logiqx XML DAT content.
Handles both <game> and <machine> entries, and games with
multiple <rom> children. Content declaring XML entities is
rejected: DAT files never define them, and expanding entities
from untrusted packs opens entity-expansion attacks.
"""
haystack = content if isinstance(content, str) else content.decode("utf-8", "replace")
if "<!ENTITY" in haystack.upper():
raise ValueError("XML entity declarations are not allowed in DAT files")
root = ET.fromstring(content)
dat = LogiqxDat()
header = root.find("header")
if header is not None:
dat.name = header.findtext("name", "").strip()
dat.version = header.findtext("version", "").strip()
for game in list(root.iter("game")) + list(root.iter("machine")):
game_name = game.get("name", "")
description = game.findtext("description", "").strip()
for rom in game.iter("rom"):
name = rom.get("name", "")
if not name:
continue
try:
size = int(rom.get("size", "0"))
except ValueError:
size = 0
dat.roms.append(
LogiqxRom(
game=game_name,
description=description or game_name,
name=name,
size=size,
crc32=rom.get("crc", "").lower(),
md5=rom.get("md5", "").lower(),
sha1=rom.get("sha1", "").lower(),
)
)
return dat
def validate_logiqx_format(content: str | bytes) -> bool:
"""Validate that content parses as a Logiqx DAT with rom entries."""
try:
dat = parse_logiqx(content)
except ET.ParseError:
return False
return bool(dat.roms)
+115
View File
@@ -0,0 +1,115 @@
"""Fetch Redump BIOS DAT files and write a provenance snapshot.
Redump serves BIOS DATs as static clrmamepro files linked from
https://redump.info/downloads under /static/bios/. The downloads page
is scanned for those links, so new BIOS DATs are picked up without
code changes.
Run manually, then commit the snapshot:
python -m scripts.scraper.redump_dat_scraper --dry-run
python -m scripts.scraper.redump_dat_scraper --output provenance/redump.json
"""
from __future__ import annotations
import argparse
import re
import sys
import urllib.error
import urllib.request
from datetime import datetime, timezone
from pathlib import Path
from ..common import write_provenance_snapshot
from .base_scraper import _read_limited
from .logiqx_parser import parse_logiqx, validate_logiqx_format
BASE_URL = "https://redump.info"
DOWNLOADS_URL = f"{BASE_URL}/downloads"
def _fetch(url: str) -> str:
req = urllib.request.Request(url, headers={"User-Agent": "retrobios-scraper/1.0"})
try:
with urllib.request.urlopen(req, timeout=30) as resp:
return _read_limited(resp).decode("utf-8", "replace")
except urllib.error.URLError as e:
raise ConnectionError(f"Failed to fetch {url}: {e}") from e
def discover_bios_datfiles(downloads_html: str) -> list[str]:
"""Extract BIOS DAT paths from the downloads page."""
paths = set(re.findall(r'href="(/static/bios/[^"]+\.dat)"', downloads_html))
return sorted(paths)
def parse_redump_dat(content: str) -> tuple[str, str, list[dict]]:
"""Parse a Redump Logiqx BIOS DAT into provenance entries.
The machine description carries the useful context (hardware
model, kernel or firmware version), so it is kept per entry.
"""
dat = parse_logiqx(content)
entries = [
{
"dat": dat.name,
"name": rom.name,
"description": rom.description,
"size": rom.size,
"crc32": rom.crc32,
"md5": rom.md5,
"sha1": rom.sha1,
}
for rom in dat.roms
]
return dat.name, dat.version, entries
def fetch_snapshot() -> tuple[dict, list[dict]]:
"""Fetch all Redump BIOS DATs. Returns (dats metadata, entries)."""
paths = discover_bios_datfiles(_fetch(DOWNLOADS_URL))
if not paths:
raise ValueError("No BIOS datfile links found on downloads page")
dats = {}
entries = []
for path in paths:
content = _fetch(f"{BASE_URL}{path}")
if not validate_logiqx_format(content):
raise ValueError(f"Unexpected DAT format for {path}")
name, version, dat_entries = parse_redump_dat(content)
dats[name] = version
entries.extend(dat_entries)
return dats, entries
def main() -> int:
parser = argparse.ArgumentParser(description="Fetch Redump BIOS DAT provenance")
parser.add_argument("--dry-run", action="store_true", help="Show fetched entries")
parser.add_argument(
"--output", "-o", default="provenance/redump.json", help="Snapshot output path"
)
args = parser.parse_args()
try:
dats, entries = fetch_snapshot()
except (ConnectionError, ValueError) as e:
print(f"Error: {e}", file=sys.stderr)
return 1
if args.dry_run:
for dat in sorted(dats):
count = sum(1 for e in entries if e["dat"] == dat)
print(f" {dat} ({dats[dat]}): {count} entries")
print(f"\nTotal: {len(entries)} entries across {len(dats)} DATs")
return 0
Path(args.output).parent.mkdir(parents=True, exist_ok=True)
imported_at = datetime.now(timezone.utc).strftime("%Y-%m-%d")
written = write_provenance_snapshot(args.output, "redump", imported_at, dats, entries)
status = "Written" if written else "Unchanged"
print(f"{status} {args.output}: {len(entries)} entries from {len(dats)} DATs")
return 0
if __name__ == "__main__":
sys.exit(main())