feat: join dump-catalog provenance in database

This commit is contained in:
Abdessamad Derraz committed 2026-08-08 02:25:44 +02:00
1 parent 933d258df5
commit f4b02c2a43
9 files changed
+2664 -1

No files matched your search

File diff suppressed because it is too large. Load diff
+89
View File
@@ -1144,6 +1144,7 @@ import re
_TIMESTAMP_PATTERNS = [
re.compile(r'"generated_at":\s*"[^"]*"'), # database.json
re.compile(r'"imported_at":\s*"[^"]*"'), # provenance snapshots
re.compile(r"\*Auto-generated on [^*]*\*"), # README.md
re.compile(r"\*Generated on [^*]*\*"), # docs site pages
]
@@ -1335,3 +1336,91 @@ def build_target_cores_cache(
raise
kept = [p for p in platforms if p not in skip]
return cache, kept
DEFAULT_PROVENANCE_DIR = "provenance"
def load_provenance_snapshots(provenance_dir: str = DEFAULT_PROVENANCE_DIR) -> dict:
"""Load dump-catalog snapshots from provenance/*.json.
Returns {source_name: snapshot} where snapshot holds the normalized
entries written by the redump scraper or the pack importer. Missing
directory means no snapshots: returns an empty dict.
"""
snapshots = {}
prov_path = Path(provenance_dir)
if not prov_path.is_dir():
return snapshots
for path in sorted(prov_path.glob("*.json")):
with open(path) as f:
snapshot = json.load(f)
source = snapshot.get("source")
if source and snapshot.get("entries"):
snapshots[source] = snapshot
return snapshots
def build_provenance_index(snapshots: dict) -> dict:
"""Index snapshot entries by sha1 and by (md5, size) per source.
First entry wins on hash collisions within a source; entries are
pre-sorted at snapshot write time so the outcome is deterministic.
"""
index = {}
for source, snapshot in snapshots.items():
by_sha1 = {}
by_md5_size = {}
for entry in snapshot["entries"]:
sha1 = entry.get("sha1", "")
md5 = entry.get("md5", "")
if sha1 and sha1 not in by_sha1:
by_sha1[sha1] = entry
if md5 and entry.get("size"):
key = (md5, entry["size"])
if key not in by_md5_size:
by_md5_size[key] = entry
index[source] = {"by_sha1": by_sha1, "by_md5_size": by_md5_size}
return index
def annotate_provenance(files: dict, snapshots: dict) -> dict[str, int]:
"""Attach a provenance field to database file entries.
Matches by SHA1 first, then MD5 + size. Returns per-source match
counts. Files without any catalog match keep no provenance field.
"""
index = build_provenance_index(snapshots)
counts = dict.fromkeys(index, 0)
for sha1, entry in files.items():
matches = {}
for source in sorted(index):
src_index = index[source]
hit = src_index["by_sha1"].get(sha1) or src_index["by_md5_size"].get(
(entry.get("md5", ""), entry.get("size", 0))
)
if hit:
matches[source] = {
"dat": hit.get("dat", ""),
"name": hit.get("name", ""),
"description": hit.get("description", ""),
}
counts[source] += 1
if matches:
entry["provenance"] = matches
else:
entry.pop("provenance", None)
return counts
def write_provenance_snapshot(
path: str, source: str, imported_at: str, dats: dict, entries: list[dict]
) -> bool:
"""Write a normalized provenance snapshot, sorted for determinism."""
snapshot = {
"source": source,
"imported_at": imported_at,
"dats": dict(sorted(dats.items())),
"entries": sorted(entries, key=lambda e: (e["dat"], e["name"])),
}
return write_if_changed(path, json.dumps(snapshot, indent=2) + "\n")
+22 -1
View File
@@ -18,7 +18,14 @@ from datetime import datetime, timezone
from pathlib import Path
sys.path.insert(0, os.path.dirname(__file__))
from common import compute_hashes, list_registered_platforms, write_if_changed
from common import (
DEFAULT_PROVENANCE_DIR,
annotate_provenance,
compute_hashes,
list_registered_platforms,
load_provenance_snapshots,
write_if_changed,
)
CACHE_DIR = ".cache"
CACHE_FILE = os.path.join(CACHE_DIR, "db_cache.json")
@@ -296,6 +303,11 @@ def main():
parser.add_argument(
"--output", "-o", default=DEFAULT_OUTPUT, help="Output JSON file"
)
parser.add_argument(
"--provenance-dir",
default=DEFAULT_PROVENANCE_DIR,
help="Directory with dump-catalog snapshots",
)
args = parser.parse_args()
bios_dir = Path(args.bios_dir)
@@ -323,6 +335,9 @@ def main():
aliases[sha1] = []
aliases[sha1].append(alias_entry)
snapshots = load_provenance_snapshots(args.provenance_dir)
provenance_counts = annotate_provenance(files, snapshots)
indexes = build_indexes(files, aliases)
total_size = sum(entry["size"] for entry in files.values())
@@ -344,6 +359,12 @@ def main():
status = "Generated" if written else "Unchanged"
print(f"{status} {args.output}: {len(files)} files, {total_size:,} bytes total")
print(f" Name index: {name_count} names ({alias_count} aliases)")
if provenance_counts:
matched = sum(1 for e in files.values() if "provenance" in e)
per_source = ", ".join(
f"{source} {count}" for source, count in sorted(provenance_counts.items())
)
print(f" Provenance: {matched} files catalog-matched ({per_source})")
return 0
+9
View File
@@ -3,6 +3,7 @@
Steps:
1. generate_db.py --force (rebuild database.json from bios/)
1b. provenance_report.py (dump-catalog coverage from provenance/)
2. refresh_data_dirs.py (update Dolphin Sys, PPSSPP, etc.)
3. verify.py --all (check all platforms)
4. generate_pack.py --all (build ZIP packs)
@@ -207,6 +208,14 @@ def main():
print("\nDatabase generation failed, aborting.")
sys.exit(1)
# Step 1b: Dump-catalog coverage (offline, reads provenance/ snapshots)
ok, _ = run(
[sys.executable, "scripts/provenance_report.py"],
"1b provenance report",
)
results["provenance"] = ok
all_ok = all_ok and ok
# Step 2: Refresh data directories
if not args.offline:
ok, out = run(
+118
View File
@@ -0,0 +1,118 @@
#!/usr/bin/env python3
"""Report dump-catalog coverage of the collection.
Reads provenance/*.json snapshots and database.json, then reports per
source how many catalog entries the collection holds and which are
missing. Missing entries are acquisition targets: catalog-verified
dumps the collection does not have yet.
Usage:
python scripts/provenance_report.py
python scripts/provenance_report.py --missing
python scripts/provenance_report.py --json
"""
from __future__ import annotations
import argparse
import json
import os
import sys
sys.path.insert(0, os.path.dirname(__file__))
from common import DEFAULT_PROVENANCE_DIR, load_database, load_provenance_snapshots
def build_report(db: dict, snapshots: dict) -> dict:
"""Compare each snapshot against the collection.
A DAT counts as covered when the collection holds at least one of
its entries. Missing entries from covered DATs are acquisition
targets; missing entries from DATs the collection does not cover at
all are out of scope and only counted. No-Intro tags every non-game
dump "[BIOS]", including digital title distribution, so without this
split the target list is swamped by content the project never ships.
"""
by_sha1 = db.get("files", {})
by_md5_size = {
(entry.get("md5", ""), entry.get("size", 0)) for entry in by_sha1.values()
}
report = {}
for source, snapshot in sorted(snapshots.items()):
matched = 0
covered_dats = set()
unmatched = []
for entry in snapshot["entries"]:
if entry.get("sha1") in by_sha1 or (
entry.get("md5"),
entry.get("size"),
) in by_md5_size:
matched += 1
covered_dats.add(entry.get("dat", ""))
else:
unmatched.append(entry)
missing = [e for e in unmatched if e.get("dat", "") in covered_dats]
out_of_scope = len(unmatched) - len(missing)
report[source] = {
"imported_at": snapshot.get("imported_at", ""),
"dats": snapshot.get("dats", {}),
"total": len(snapshot["entries"]),
"matched": matched,
"covered_dats": sorted(covered_dats),
"missing": missing,
"out_of_scope": out_of_scope,
}
return report
def main() -> int:
parser = argparse.ArgumentParser(description="Dump-catalog coverage report")
parser.add_argument("--db", default="database.json", help="Database path")
parser.add_argument(
"--provenance-dir",
default=DEFAULT_PROVENANCE_DIR,
help="Directory with dump-catalog snapshots",
)
parser.add_argument("--missing", action="store_true", help="List missing entries")
parser.add_argument("--json", action="store_true", help="Full report as JSON")
args = parser.parse_args()
db = load_database(args.db)
snapshots = load_provenance_snapshots(args.provenance_dir)
if not snapshots:
print(f"No provenance snapshots in {args.provenance_dir}/")
return 0
report = build_report(db, snapshots)
if args.json:
print(json.dumps(report, indent=2))
return 0
for source, data in report.items():
in_scope = data["matched"] + len(data["missing"])
pct = 100 * data["matched"] / in_scope if in_scope else 0
print(
f" {source} ({data['imported_at']}): "
f"{data['matched']}/{in_scope} in collection ({pct:.0f}%) "
f"across {len(data['covered_dats'])} covered DATs"
)
if data["out_of_scope"]:
print(
f" {data['out_of_scope']} entries in DATs the collection "
f"does not cover, not counted as targets"
)
if args.missing:
for entry in data["missing"]:
label = entry.get("description") or entry["name"]
print(f" MISSING {entry['name']} ({label}) sha1={entry.get('sha1')}")
total_missing = sum(len(d["missing"]) for d in report.values())
if total_missing:
print(f" {total_missing} catalog entries missing from collection")
return 0
if __name__ == "__main__":
sys.exit(main())
+135
View File
@@ -0,0 +1,135 @@
"""Import BIOS entries from a No-Intro or TOSEC DAT pack.
Dat-o-Matic blocks automated downloads and TOSEC ships yearly packs,
so both are imported from a locally downloaded archive:
python -m scripts.scraper.dat_pack_importer --source no-intro --pack lovepack.zip
python -m scripts.scraper.dat_pack_importer --source tosec --pack TOSEC-v2025-03-13.zip
No-Intro marks BIOS dumps with a "[BIOS]" game name prefix inside the
system DATs. TOSEC groups firmware into dedicated DATs whose name
contains "Firmware" or "BIOS". Only those entries are imported.
"""
from __future__ import annotations
import argparse
import sys
import zipfile
from datetime import datetime, timezone
from pathlib import Path
from ..common import write_provenance_snapshot
from .logiqx_parser import LogiqxDat, parse_logiqx
MAX_MEMBER_SIZE = 100 * 1024 * 1024
SOURCES = ("no-intro", "tosec")
def _bios_entries(source: str, dat: LogiqxDat) -> list[dict]:
"""Filter a parsed DAT down to its BIOS/firmware entries."""
if source == "no-intro":
roms = [r for r in dat.roms if r.game.startswith("[BIOS]")]
else:
dat_name = dat.name.casefold()
if "firmware" not in dat_name and "bios" not in dat_name:
return []
roms = dat.roms
return [
{
"dat": dat.name,
"name": rom.name,
"description": rom.description,
"size": rom.size,
"crc32": rom.crc32,
"md5": rom.md5,
"sha1": rom.sha1,
}
for rom in roms
]
def _iter_dat_contents(pack: Path):
"""Yield (member name, content bytes) for DAT files in a pack.
Accepts a ZIP archive or a directory of extracted DATs.
"""
if pack.is_dir():
for path in sorted(pack.rglob("*")):
if path.suffix.lower() in (".dat", ".xml") and path.is_file():
yield str(path), path.read_bytes()
return
with zipfile.ZipFile(pack) as zf:
for info in sorted(zf.infolist(), key=lambda i: i.filename):
suffix = Path(info.filename).suffix.lower()
if info.is_dir() or suffix not in (".dat", ".xml"):
continue
if info.file_size > MAX_MEMBER_SIZE:
print(f" Skipping {info.filename}: exceeds {MAX_MEMBER_SIZE} bytes")
continue
yield info.filename, zf.read(info)
def import_pack(source: str, pack: Path) -> tuple[dict, list[dict], int]:
"""Import BIOS entries from a pack. Returns (dats, entries, skipped)."""
dats = {}
entries = []
skipped = 0
for member, content in _iter_dat_contents(pack):
try:
dat = parse_logiqx(content)
except (ValueError, SyntaxError):
skipped += 1
continue
dat_entries = _bios_entries(source, dat)
if dat_entries:
dats[dat.name] = dat.version
entries.extend(dat_entries)
return dats, entries, skipped
def main() -> int:
parser = argparse.ArgumentParser(description="Import BIOS provenance from a DAT pack")
parser.add_argument("--source", required=True, choices=SOURCES)
parser.add_argument("--pack", required=True, help="DAT pack (ZIP or directory)")
parser.add_argument("--output", "-o", help="Snapshot output path")
parser.add_argument("--dry-run", action="store_true", help="Show imported entries")
args = parser.parse_args()
pack = Path(args.pack)
if not pack.exists():
print(f"Error: pack '{pack}' not found", file=sys.stderr)
return 1
try:
dats, entries, skipped = import_pack(args.source, pack)
except zipfile.BadZipFile as e:
print(f"Error: {pack} is not a valid ZIP archive: {e}", file=sys.stderr)
return 1
if not entries:
print(f"Error: no BIOS entries found in {pack}", file=sys.stderr)
return 1
if skipped:
print(f" Skipped {skipped} non-Logiqx members")
if args.dry_run:
for dat in sorted(dats):
count = sum(1 for e in entries if e["dat"] == dat)
print(f" {dat} ({dats[dat]}): {count} entries")
print(f"\nTotal: {len(entries)} entries across {len(dats)} DATs")
return 0
output = args.output or f"provenance/{args.source}.json"
Path(output).parent.mkdir(parents=True, exist_ok=True)
imported_at = datetime.now(timezone.utc).strftime("%Y-%m-%d")
written = write_provenance_snapshot(output, args.source, imported_at, dats, entries)
status = "Written" if written else "Unchanged"
print(f"{status} {output}: {len(entries)} entries from {len(dats)} DATs")
return 0
if __name__ == "__main__":
sys.exit(main())
+92
View File
@@ -0,0 +1,92 @@
"""Parser for Logiqx XML DAT format.
Parses files like No-Intro daily packs and TOSEC releases which use:
<datafile>
<header><name>...</name><version>...</version></header>
<game name="[BIOS] Game Boy Advance (World)">
<description>...</description>
<rom name="..." size="16384" crc="..." md5="..." sha1="..."/>
</game>
</datafile>
"""
from __future__ import annotations
import xml.etree.ElementTree as ET
from dataclasses import dataclass, field
@dataclass
class LogiqxRom:
"""A ROM entry from a Logiqx DAT file."""
game: str
description: str
name: str
size: int
crc32: str
md5: str
sha1: str
@dataclass
class LogiqxDat:
"""A parsed Logiqx DAT file."""
name: str = ""
version: str = ""
roms: list[LogiqxRom] = field(default_factory=list)
def parse_logiqx(content: str | bytes) -> LogiqxDat:
"""Parse Logiqx XML DAT content.
Handles both <game> and <machine> entries, and games with
multiple <rom> children. Content declaring XML entities is
rejected: DAT files never define them, and expanding entities
from untrusted packs opens entity-expansion attacks.
"""
haystack = content if isinstance(content, str) else content.decode("utf-8", "replace")
if "<!ENTITY" in haystack.upper():
raise ValueError("XML entity declarations are not allowed in DAT files")
root = ET.fromstring(content)
dat = LogiqxDat()
header = root.find("header")
if header is not None:
dat.name = header.findtext("name", "").strip()
dat.version = header.findtext("version", "").strip()
for game in list(root.iter("game")) + list(root.iter("machine")):
game_name = game.get("name", "")
description = game.findtext("description", "").strip()
for rom in game.iter("rom"):
name = rom.get("name", "")
if not name:
continue
try:
size = int(rom.get("size", "0"))
except ValueError:
size = 0
dat.roms.append(
LogiqxRom(
game=game_name,
description=description or game_name,
name=name,
size=size,
crc32=rom.get("crc", "").lower(),
md5=rom.get("md5", "").lower(),
sha1=rom.get("sha1", "").lower(),
)
)
return dat
def validate_logiqx_format(content: str | bytes) -> bool:
"""Validate that content parses as a Logiqx DAT with rom entries."""
try:
dat = parse_logiqx(content)
except ET.ParseError:
return False
return bool(dat.roms)
+115
View File
@@ -0,0 +1,115 @@
"""Fetch Redump BIOS DAT files and write a provenance snapshot.
Redump serves BIOS DATs as static clrmamepro files linked from
https://redump.info/downloads under /static/bios/. The downloads page
is scanned for those links, so new BIOS DATs are picked up without
code changes.
Run manually, then commit the snapshot:
python -m scripts.scraper.redump_dat_scraper --dry-run
python -m scripts.scraper.redump_dat_scraper --output provenance/redump.json
"""
from __future__ import annotations
import argparse
import re
import sys
import urllib.error
import urllib.request
from datetime import datetime, timezone
from pathlib import Path
from ..common import write_provenance_snapshot
from .base_scraper import _read_limited
from .logiqx_parser import parse_logiqx, validate_logiqx_format
BASE_URL = "https://redump.info"
DOWNLOADS_URL = f"{BASE_URL}/downloads"
def _fetch(url: str) -> str:
req = urllib.request.Request(url, headers={"User-Agent": "retrobios-scraper/1.0"})
try:
with urllib.request.urlopen(req, timeout=30) as resp:
return _read_limited(resp).decode("utf-8", "replace")
except urllib.error.URLError as e:
raise ConnectionError(f"Failed to fetch {url}: {e}") from e
def discover_bios_datfiles(downloads_html: str) -> list[str]:
"""Extract BIOS DAT paths from the downloads page."""
paths = set(re.findall(r'href="(/static/bios/[^"]+\.dat)"', downloads_html))
return sorted(paths)
def parse_redump_dat(content: str) -> tuple[str, str, list[dict]]:
"""Parse a Redump Logiqx BIOS DAT into provenance entries.
The machine description carries the useful context (hardware
model, kernel or firmware version), so it is kept per entry.
"""
dat = parse_logiqx(content)
entries = [
{
"dat": dat.name,
"name": rom.name,
"description": rom.description,
"size": rom.size,
"crc32": rom.crc32,
"md5": rom.md5,
"sha1": rom.sha1,
}
for rom in dat.roms
]
return dat.name, dat.version, entries
def fetch_snapshot() -> tuple[dict, list[dict]]:
"""Fetch all Redump BIOS DATs. Returns (dats metadata, entries)."""
paths = discover_bios_datfiles(_fetch(DOWNLOADS_URL))
if not paths:
raise ValueError("No BIOS datfile links found on downloads page")
dats = {}
entries = []
for path in paths:
content = _fetch(f"{BASE_URL}{path}")
if not validate_logiqx_format(content):
raise ValueError(f"Unexpected DAT format for {path}")
name, version, dat_entries = parse_redump_dat(content)
dats[name] = version
entries.extend(dat_entries)
return dats, entries
def main() -> int:
parser = argparse.ArgumentParser(description="Fetch Redump BIOS DAT provenance")
parser.add_argument("--dry-run", action="store_true", help="Show fetched entries")
parser.add_argument(
"--output", "-o", default="provenance/redump.json", help="Snapshot output path"
)
args = parser.parse_args()
try:
dats, entries = fetch_snapshot()
except (ConnectionError, ValueError) as e:
print(f"Error: {e}", file=sys.stderr)
return 1
if args.dry_run:
for dat in sorted(dats):
count = sum(1 for e in entries if e["dat"] == dat)
print(f" {dat} ({dats[dat]}): {count} entries")
print(f"\nTotal: {len(entries)} entries across {len(dats)} DATs")
return 0
Path(args.output).parent.mkdir(parents=True, exist_ok=True)
imported_at = datetime.now(timezone.utc).strftime("%Y-%m-%d")
written = write_provenance_snapshot(args.output, "redump", imported_at, dats, entries)
status = "Written" if written else "Unchanged"
print(f"{status} {args.output}: {len(entries)} entries from {len(dats)} DATs")
return 0
if __name__ == "__main__":
sys.exit(main())
+353
View File
@@ -0,0 +1,353 @@
"""Tests for dump-catalog provenance: parsers, importers, join, report."""
from __future__ import annotations
import json
import tempfile
import unittest
import zipfile
from pathlib import Path
from scripts.common import (
annotate_provenance,
build_provenance_index,
load_provenance_snapshots,
write_provenance_snapshot,
)
from scripts.provenance_report import build_report
from scripts.scraper.dat_pack_importer import _bios_entries, import_pack
from scripts.scraper.logiqx_parser import parse_logiqx, validate_logiqx_format
from scripts.scraper.redump_dat_scraper import discover_bios_datfiles, parse_redump_dat
LOGIQX_GAME_FIXTURE = """<?xml version="1.0"?>
<datafile>
<header>
<name>Nintendo - Game Boy Advance</name>
<version>20260801-123456</version>
</header>
<game name="[BIOS] Game Boy Advance (World)">
<description>[BIOS] Game Boy Advance (World)</description>
<rom name="[BIOS] Game Boy Advance (World).gba" size="16384" crc="81977335"
md5="A860E8C0B6D573D191E4EC7DB1B1E4F6" sha1="300C20DF6731A33952DED8C436F7F186D25D3492"/>
</game>
<game name="Some Game (USA)">
<description>Some Game (USA)</description>
<rom name="Some Game (USA).gba" size="8388608" crc="caf2e99f"
md5="55354d9e3bc9c1fa682b5110e5ed1544" sha1="6e4e9be9a07580ef267be9c2ea1bd0730b3be44a"/>
</game>
</datafile>
"""
LOGIQX_MACHINE_FIXTURE = """<?xml version="1.0"?>
<datafile>
<header>
<name>Sony - PlayStation - BIOS Images</name>
<version>2026-06-16</version>
</header>
<machine name="ps-10j">
<description>SCPH-1000/DTL-H1000 (Version 1.0 J)</description>
<rom name="ps-10j.bin" size="524288" crc="3b601fc8"
md5="239665b1a3dade1b5a52c06338011044" sha1="343883a7b555646da8cee54aadd2795b6e7dd070"/>
</machine>
<machine name="multi-rom">
<description>Two chips</description>
<rom name="chip1.bin" size="10" crc="11111111" md5="aa" sha1="bb"/>
<rom name="chip2.bin" size="20" crc="22222222" md5="cc" sha1="dd"/>
</machine>
</datafile>
"""
TOSEC_FIRMWARE_FIXTURE = """<?xml version="1.0"?>
<datafile>
<header>
<name>Sega Dreamcast - Firmware</name>
<version>TOSEC-v2025-03-13</version>
</header>
<game name="Dreamcast BIOS v1.01d (1998)(Sega)(JP)">
<description>Dreamcast BIOS v1.01d (1998)(Sega)(JP)</description>
<rom name="dc_boot.bin" size="2097152" crc="89f2b1a1"
md5="e10c53c2f8b90bab96ead2d368858623" sha1="4e9db27eb9bcd8d19d346b7c8a1ea307debbf9c9"/>
</game>
</datafile>
"""
class TestLogiqxParser(unittest.TestCase):
def test_parse_header_and_games(self):
dat = parse_logiqx(LOGIQX_GAME_FIXTURE)
self.assertEqual(dat.name, "Nintendo - Game Boy Advance")
self.assertEqual(dat.version, "20260801-123456")
self.assertEqual(len(dat.roms), 2)
bios = dat.roms[0]
self.assertEqual(bios.game, "[BIOS] Game Boy Advance (World)")
self.assertEqual(bios.name, "[BIOS] Game Boy Advance (World).gba")
self.assertEqual(bios.size, 16384)
def test_hashes_lowercased(self):
dat = parse_logiqx(LOGIQX_GAME_FIXTURE)
bios = dat.roms[0]
self.assertEqual(bios.md5, "a860e8c0b6d573d191e4ec7db1b1e4f6")
self.assertEqual(bios.sha1, "300c20df6731a33952ded8c436f7f186d25d3492")
def test_parse_machine_entries(self):
dat = parse_logiqx(LOGIQX_MACHINE_FIXTURE)
self.assertEqual(len(dat.roms), 3)
self.assertEqual(dat.roms[0].description, "SCPH-1000/DTL-H1000 (Version 1.0 J)")
def test_multiple_roms_per_machine(self):
dat = parse_logiqx(LOGIQX_MACHINE_FIXTURE)
names = [r.name for r in dat.roms if r.game == "multi-rom"]
self.assertEqual(names, ["chip1.bin", "chip2.bin"])
def test_invalid_size_defaults_to_zero(self):
content = LOGIQX_GAME_FIXTURE.replace('size="16384"', 'size="bogus"')
dat = parse_logiqx(content)
self.assertEqual(dat.roms[0].size, 0)
def test_entity_declarations_rejected(self):
content = (
'<?xml version="1.0"?><!DOCTYPE datafile [<!ENTITY x "y">]>'
"<datafile><game name=\"&x;\"><rom name=\"a\"/></game></datafile>"
)
with self.assertRaises(ValueError):
parse_logiqx(content)
def test_validate_format(self):
self.assertTrue(validate_logiqx_format(LOGIQX_GAME_FIXTURE))
self.assertFalse(validate_logiqx_format("clrmamepro ( name x )"))
self.assertFalse(
validate_logiqx_format("<datafile><header/><game name='x'/></datafile>")
)
class TestRedumpScraper(unittest.TestCase):
def test_discover_bios_datfiles(self):
html = (
'<a href="/static/bios/Sony%20PSX.dat">x</a>'
'<a href="/static/bios/Nintendo%20GC.dat">y</a>'
'<a href="/datfile/3DO">z</a>'
'<a href="/static/bios/Sony%20PSX.dat">dup</a>'
)
self.assertEqual(
discover_bios_datfiles(html),
["/static/bios/Nintendo%20GC.dat", "/static/bios/Sony%20PSX.dat"],
)
def test_parse_redump_dat(self):
name, version, entries = parse_redump_dat(LOGIQX_MACHINE_FIXTURE)
self.assertEqual(name, "Sony - PlayStation - BIOS Images")
self.assertEqual(version, "2026-06-16")
self.assertEqual(len(entries), 3)
self.assertEqual(entries[0]["dat"], "Sony - PlayStation - BIOS Images")
self.assertEqual(entries[0]["name"], "ps-10j.bin")
self.assertEqual(entries[0]["description"], "SCPH-1000/DTL-H1000 (Version 1.0 J)")
class TestPackImporter(unittest.TestCase):
def test_no_intro_filter_keeps_bios_only(self):
dat = parse_logiqx(LOGIQX_GAME_FIXTURE)
entries = _bios_entries("no-intro", dat)
self.assertEqual(len(entries), 1)
self.assertEqual(entries[0]["name"], "[BIOS] Game Boy Advance (World).gba")
def test_tosec_filter_by_dat_name(self):
firmware = parse_logiqx(TOSEC_FIRMWARE_FIXTURE)
self.assertEqual(len(_bios_entries("tosec", firmware)), 1)
games = parse_logiqx(LOGIQX_GAME_FIXTURE)
self.assertEqual(_bios_entries("tosec", games), [])
def test_import_pack_zip(self):
with tempfile.TemporaryDirectory() as tmp:
pack = Path(tmp) / "pack.zip"
with zipfile.ZipFile(pack, "w") as zf:
zf.writestr("Nintendo - Game Boy Advance.dat", LOGIQX_GAME_FIXTURE)
zf.writestr("garbage.dat", "not xml at all")
zf.writestr("readme.txt", "ignored")
dats, entries, skipped = import_pack("no-intro", pack)
self.assertEqual(list(dats), ["Nintendo - Game Boy Advance"])
self.assertEqual(len(entries), 1)
self.assertEqual(skipped, 1)
def test_import_pack_directory(self):
with tempfile.TemporaryDirectory() as tmp:
(Path(tmp) / "Sega Dreamcast - Firmware.dat").write_text(
TOSEC_FIRMWARE_FIXTURE
)
dats, entries, skipped = import_pack("tosec", Path(tmp))
self.assertEqual(list(dats), ["Sega Dreamcast - Firmware"])
self.assertEqual(len(entries), 1)
self.assertEqual(skipped, 0)
def _snapshot_entries():
return [
{
"dat": "Sony - PlayStation - BIOS Images",
"name": "ps-30j.bin",
"description": "SCPH-5500 (Version 3.0 J)",
"size": 524288,
"crc32": "ff3eeb8c",
"md5": "8dd7d5296a650fac7319bce665a6a53c",
"sha1": "b05def971d8ec59f346f2d9ac21fb742e3eb6917",
},
{
"dat": "Sony - PlayStation - BIOS Images",
"name": "ps-41a.bin",
"description": "SCPH-7001 (Version 4.1 A)",
"size": 524288,
"crc32": "502224b6",
"md5": "1e68c231d0896b7eadcad1d7d8e76129",
"sha1": "14df4f6c1e367ce097c11deae21566b4fe5647a9",
},
{
"dat": "Sony - PlayStation - BIOS Images",
"name": "no-sha1.bin",
"description": "Entry without sha1",
"size": 1024,
"crc32": "deadbeef",
"md5": "d41d8cd98f00b204e9800998ecf8427e",
"sha1": "",
},
]
class TestProvenanceJoin(unittest.TestCase):
def _snapshots(self):
return {"redump": {"source": "redump", "entries": _snapshot_entries()}}
def test_index_by_sha1_and_md5_size(self):
index = build_provenance_index(self._snapshots())
redump = index["redump"]
self.assertIn("b05def971d8ec59f346f2d9ac21fb742e3eb6917", redump["by_sha1"])
self.assertNotIn("", redump["by_sha1"])
self.assertIn(
("d41d8cd98f00b204e9800998ecf8427e", 1024), redump["by_md5_size"]
)
def test_annotate_matches_by_sha1(self):
files = {
"b05def971d8ec59f346f2d9ac21fb742e3eb6917": {
"name": "scph5500.bin",
"size": 524288,
"md5": "8dd7d5296a650fac7319bce665a6a53c",
},
"0000000000000000000000000000000000000000": {
"name": "unrelated.bin",
"size": 42,
"md5": "ffffffffffffffffffffffffffffffff",
},
}
counts = annotate_provenance(files, self._snapshots())
self.assertEqual(counts, {"redump": 1})
match = files["b05def971d8ec59f346f2d9ac21fb742e3eb6917"]["provenance"]
self.assertEqual(match["redump"]["name"], "ps-30j.bin")
self.assertEqual(match["redump"]["description"], "SCPH-5500 (Version 3.0 J)")
self.assertNotIn(
"provenance", files["0000000000000000000000000000000000000000"]
)
def test_annotate_falls_back_to_md5_size(self):
files = {
"1111111111111111111111111111111111111111": {
"name": "other-name.bin",
"size": 1024,
"md5": "d41d8cd98f00b204e9800998ecf8427e",
}
}
counts = annotate_provenance(files, self._snapshots())
self.assertEqual(counts, {"redump": 1})
self.assertEqual(
files["1111111111111111111111111111111111111111"]["provenance"]["redump"][
"name"
],
"no-sha1.bin",
)
def test_annotate_pops_stale_provenance(self):
files = {
"2222222222222222222222222222222222222222": {
"name": "was-matched.bin",
"size": 5,
"md5": "00000000000000000000000000000000",
"provenance": {"redump": {"dat": "old", "name": "old.bin"}},
}
}
annotate_provenance(files, self._snapshots())
self.assertNotIn(
"provenance", files["2222222222222222222222222222222222222222"]
)
def test_annotate_without_snapshots(self):
files = {"aa": {"name": "x.bin", "size": 1, "md5": "bb"}}
self.assertEqual(annotate_provenance(files, {}), {})
self.assertNotIn("provenance", files["aa"])
class TestSnapshotIO(unittest.TestCase):
def test_write_and_load_roundtrip(self):
with tempfile.TemporaryDirectory() as tmp:
path = Path(tmp) / "redump.json"
written = write_provenance_snapshot(
str(path),
"redump",
"2026-08-07",
{"Sony - PlayStation - BIOS Images": "2026-06-16"},
_snapshot_entries(),
)
self.assertTrue(written)
snapshots = load_provenance_snapshots(tmp)
self.assertEqual(list(snapshots), ["redump"])
self.assertEqual(len(snapshots["redump"]["entries"]), 3)
names = [e["name"] for e in snapshots["redump"]["entries"]]
self.assertEqual(names, sorted(names))
def test_rewrite_with_new_date_is_unchanged(self):
with tempfile.TemporaryDirectory() as tmp:
path = str(Path(tmp) / "redump.json")
write_provenance_snapshot(
path, "redump", "2026-08-07", {}, _snapshot_entries()
)
rewritten = write_provenance_snapshot(
path, "redump", "2027-01-01", {}, _snapshot_entries()
)
self.assertFalse(rewritten)
def test_load_missing_dir(self):
self.assertEqual(load_provenance_snapshots("does/not/exist"), {})
def test_load_ignores_empty_snapshot(self):
with tempfile.TemporaryDirectory() as tmp:
(Path(tmp) / "empty.json").write_text(
json.dumps({"source": "tosec", "entries": []})
)
self.assertEqual(load_provenance_snapshots(tmp), {})
class TestProvenanceReport(unittest.TestCase):
def test_build_report_matched_and_missing(self):
db = {
"files": {
"b05def971d8ec59f346f2d9ac21fb742e3eb6917": {
"name": "scph5500.bin",
"size": 524288,
"md5": "8dd7d5296a650fac7319bce665a6a53c",
}
}
}
snapshots = {
"redump": {
"source": "redump",
"imported_at": "2026-08-07",
"dats": {"Sony - PlayStation - BIOS Images": "2026-06-16"},
"entries": _snapshot_entries(),
}
}
report = build_report(db, snapshots)
self.assertEqual(report["redump"]["total"], 3)
self.assertEqual(report["redump"]["matched"], 1)
missing = [e["name"] for e in report["redump"]["missing"]]
self.assertEqual(missing, ["ps-41a.bin", "no-sha1.bin"])
if __name__ == "__main__":
unittest.main()