feat: join dump-catalog provenance in database

This commit is contained in:
Abdessamad Derraz committed 2026-08-08 02:25:44 +02:00
1 parent 933d258df5
commit f4b02c2a43
9 files changed
+2664 -1

No files matched your search

+135
View File
@@ -0,0 +1,135 @@
"""Import BIOS entries from a No-Intro or TOSEC DAT pack.
Dat-o-Matic blocks automated downloads and TOSEC ships yearly packs,
so both are imported from a locally downloaded archive:
python -m scripts.scraper.dat_pack_importer --source no-intro --pack lovepack.zip
python -m scripts.scraper.dat_pack_importer --source tosec --pack TOSEC-v2025-03-13.zip
No-Intro marks BIOS dumps with a "[BIOS]" game name prefix inside the
system DATs. TOSEC groups firmware into dedicated DATs whose name
contains "Firmware" or "BIOS". Only those entries are imported.
"""
from __future__ import annotations
import argparse
import sys
import zipfile
from datetime import datetime, timezone
from pathlib import Path
from ..common import write_provenance_snapshot
from .logiqx_parser import LogiqxDat, parse_logiqx
MAX_MEMBER_SIZE = 100 * 1024 * 1024
SOURCES = ("no-intro", "tosec")
def _bios_entries(source: str, dat: LogiqxDat) -> list[dict]:
"""Filter a parsed DAT down to its BIOS/firmware entries."""
if source == "no-intro":
roms = [r for r in dat.roms if r.game.startswith("[BIOS]")]
else:
dat_name = dat.name.casefold()
if "firmware" not in dat_name and "bios" not in dat_name:
return []
roms = dat.roms
return [
{
"dat": dat.name,
"name": rom.name,
"description": rom.description,
"size": rom.size,
"crc32": rom.crc32,
"md5": rom.md5,
"sha1": rom.sha1,
}
for rom in roms
]
def _iter_dat_contents(pack: Path):
"""Yield (member name, content bytes) for DAT files in a pack.
Accepts a ZIP archive or a directory of extracted DATs.
"""
if pack.is_dir():
for path in sorted(pack.rglob("*")):
if path.suffix.lower() in (".dat", ".xml") and path.is_file():
yield str(path), path.read_bytes()
return
with zipfile.ZipFile(pack) as zf:
for info in sorted(zf.infolist(), key=lambda i: i.filename):
suffix = Path(info.filename).suffix.lower()
if info.is_dir() or suffix not in (".dat", ".xml"):
continue
if info.file_size > MAX_MEMBER_SIZE:
print(f" Skipping {info.filename}: exceeds {MAX_MEMBER_SIZE} bytes")
continue
yield info.filename, zf.read(info)
def import_pack(source: str, pack: Path) -> tuple[dict, list[dict], int]:
"""Import BIOS entries from a pack. Returns (dats, entries, skipped)."""
dats = {}
entries = []
skipped = 0
for member, content in _iter_dat_contents(pack):
try:
dat = parse_logiqx(content)
except (ValueError, SyntaxError):
skipped += 1
continue
dat_entries = _bios_entries(source, dat)
if dat_entries:
dats[dat.name] = dat.version
entries.extend(dat_entries)
return dats, entries, skipped
def main() -> int:
parser = argparse.ArgumentParser(description="Import BIOS provenance from a DAT pack")
parser.add_argument("--source", required=True, choices=SOURCES)
parser.add_argument("--pack", required=True, help="DAT pack (ZIP or directory)")
parser.add_argument("--output", "-o", help="Snapshot output path")
parser.add_argument("--dry-run", action="store_true", help="Show imported entries")
args = parser.parse_args()
pack = Path(args.pack)
if not pack.exists():
print(f"Error: pack '{pack}' not found", file=sys.stderr)
return 1
try:
dats, entries, skipped = import_pack(args.source, pack)
except zipfile.BadZipFile as e:
print(f"Error: {pack} is not a valid ZIP archive: {e}", file=sys.stderr)
return 1
if not entries:
print(f"Error: no BIOS entries found in {pack}", file=sys.stderr)
return 1
if skipped:
print(f" Skipped {skipped} non-Logiqx members")
if args.dry_run:
for dat in sorted(dats):
count = sum(1 for e in entries if e["dat"] == dat)
print(f" {dat} ({dats[dat]}): {count} entries")
print(f"\nTotal: {len(entries)} entries across {len(dats)} DATs")
return 0
output = args.output or f"provenance/{args.source}.json"
Path(output).parent.mkdir(parents=True, exist_ok=True)
imported_at = datetime.now(timezone.utc).strftime("%Y-%m-%d")
written = write_provenance_snapshot(output, args.source, imported_at, dats, entries)
status = "Written" if written else "Unchanged"
print(f"{status} {output}: {len(entries)} entries from {len(dats)} DATs")
return 0
if __name__ == "__main__":
sys.exit(main())
+92
View File
@@ -0,0 +1,92 @@
"""Parser for Logiqx XML DAT format.
Parses files like No-Intro daily packs and TOSEC releases which use:
<datafile>
<header><name>...</name><version>...</version></header>
<game name="[BIOS] Game Boy Advance (World)">
<description>...</description>
<rom name="..." size="16384" crc="..." md5="..." sha1="..."/>
</game>
</datafile>
"""
from __future__ import annotations
import xml.etree.ElementTree as ET
from dataclasses import dataclass, field
@dataclass
class LogiqxRom:
"""A ROM entry from a Logiqx DAT file."""
game: str
description: str
name: str
size: int
crc32: str
md5: str
sha1: str
@dataclass
class LogiqxDat:
"""A parsed Logiqx DAT file."""
name: str = ""
version: str = ""
roms: list[LogiqxRom] = field(default_factory=list)
def parse_logiqx(content: str | bytes) -> LogiqxDat:
"""Parse Logiqx XML DAT content.
Handles both <game> and <machine> entries, and games with
multiple <rom> children. Content declaring XML entities is
rejected: DAT files never define them, and expanding entities
from untrusted packs opens entity-expansion attacks.
"""
haystack = content if isinstance(content, str) else content.decode("utf-8", "replace")
if "<!ENTITY" in haystack.upper():
raise ValueError("XML entity declarations are not allowed in DAT files")
root = ET.fromstring(content)
dat = LogiqxDat()
header = root.find("header")
if header is not None:
dat.name = header.findtext("name", "").strip()
dat.version = header.findtext("version", "").strip()
for game in list(root.iter("game")) + list(root.iter("machine")):
game_name = game.get("name", "")
description = game.findtext("description", "").strip()
for rom in game.iter("rom"):
name = rom.get("name", "")
if not name:
continue
try:
size = int(rom.get("size", "0"))
except ValueError:
size = 0
dat.roms.append(
LogiqxRom(
game=game_name,
description=description or game_name,
name=name,
size=size,
crc32=rom.get("crc", "").lower(),
md5=rom.get("md5", "").lower(),
sha1=rom.get("sha1", "").lower(),
)
)
return dat
def validate_logiqx_format(content: str | bytes) -> bool:
"""Validate that content parses as a Logiqx DAT with rom entries."""
try:
dat = parse_logiqx(content)
except ET.ParseError:
return False
return bool(dat.roms)
+115
View File
@@ -0,0 +1,115 @@
"""Fetch Redump BIOS DAT files and write a provenance snapshot.
Redump serves BIOS DATs as static clrmamepro files linked from
https://redump.info/downloads under /static/bios/. The downloads page
is scanned for those links, so new BIOS DATs are picked up without
code changes.
Run manually, then commit the snapshot:
python -m scripts.scraper.redump_dat_scraper --dry-run
python -m scripts.scraper.redump_dat_scraper --output provenance/redump.json
"""
from __future__ import annotations
import argparse
import re
import sys
import urllib.error
import urllib.request
from datetime import datetime, timezone
from pathlib import Path
from ..common import write_provenance_snapshot
from .base_scraper import _read_limited
from .logiqx_parser import parse_logiqx, validate_logiqx_format
BASE_URL = "https://redump.info"
DOWNLOADS_URL = f"{BASE_URL}/downloads"
def _fetch(url: str) -> str:
req = urllib.request.Request(url, headers={"User-Agent": "retrobios-scraper/1.0"})
try:
with urllib.request.urlopen(req, timeout=30) as resp:
return _read_limited(resp).decode("utf-8", "replace")
except urllib.error.URLError as e:
raise ConnectionError(f"Failed to fetch {url}: {e}") from e
def discover_bios_datfiles(downloads_html: str) -> list[str]:
"""Extract BIOS DAT paths from the downloads page."""
paths = set(re.findall(r'href="(/static/bios/[^"]+\.dat)"', downloads_html))
return sorted(paths)
def parse_redump_dat(content: str) -> tuple[str, str, list[dict]]:
"""Parse a Redump Logiqx BIOS DAT into provenance entries.
The machine description carries the useful context (hardware
model, kernel or firmware version), so it is kept per entry.
"""
dat = parse_logiqx(content)
entries = [
{
"dat": dat.name,
"name": rom.name,
"description": rom.description,
"size": rom.size,
"crc32": rom.crc32,
"md5": rom.md5,
"sha1": rom.sha1,
}
for rom in dat.roms
]
return dat.name, dat.version, entries
def fetch_snapshot() -> tuple[dict, list[dict]]:
"""Fetch all Redump BIOS DATs. Returns (dats metadata, entries)."""
paths = discover_bios_datfiles(_fetch(DOWNLOADS_URL))
if not paths:
raise ValueError("No BIOS datfile links found on downloads page")
dats = {}
entries = []
for path in paths:
content = _fetch(f"{BASE_URL}{path}")
if not validate_logiqx_format(content):
raise ValueError(f"Unexpected DAT format for {path}")
name, version, dat_entries = parse_redump_dat(content)
dats[name] = version
entries.extend(dat_entries)
return dats, entries
def main() -> int:
parser = argparse.ArgumentParser(description="Fetch Redump BIOS DAT provenance")
parser.add_argument("--dry-run", action="store_true", help="Show fetched entries")
parser.add_argument(
"--output", "-o", default="provenance/redump.json", help="Snapshot output path"
)
args = parser.parse_args()
try:
dats, entries = fetch_snapshot()
except (ConnectionError, ValueError) as e:
print(f"Error: {e}", file=sys.stderr)
return 1
if args.dry_run:
for dat in sorted(dats):
count = sum(1 for e in entries if e["dat"] == dat)
print(f" {dat} ({dats[dat]}): {count} entries")
print(f"\nTotal: {len(entries)} entries across {len(dats)} DATs")
return 0
Path(args.output).parent.mkdir(parents=True, exist_ok=True)
imported_at = datetime.now(timezone.utc).strftime("%Y-%m-%d")
written = write_provenance_snapshot(args.output, "redump", imported_at, dats, entries)
status = "Written" if written else "Unchanged"
print(f"{status} {args.output}: {len(entries)} entries from {len(dats)} DATs")
return 0
if __name__ == "__main__":
sys.exit(main())