diff --git a/scripts/dumpcatalog.py b/scripts/dumpcatalog.py index 1211722e..0dd9d2d3 100644 --- a/scripts/dumpcatalog.py +++ b/scripts/dumpcatalog.py @@ -83,14 +83,25 @@ def annotate_provenance(files: dict, snapshots: dict) -> dict[str, int]: entry.pop("provenance", None) return counts -def write_provenance_snapshot( - path: str, source: str, imported_at: str, dats: dict, entries: list[dict] -) -> bool: - """Write a normalized provenance snapshot, sorted for determinism.""" - snapshot = { +def build_snapshot( + source: str, imported_at: str, dats: dict, entries: list[dict] +) -> dict: + """A provenance snapshot, sorted for determinism.""" + return { "source": source, "imported_at": imported_at, "dats": dict(sorted(dats.items())), "entries": sorted(entries, key=lambda e: (e["dat"], e["name"])), } + + +def write_snapshot(path: str, snapshot: dict) -> bool: + """Write a snapshot; timestamps alone never count as a change.""" return write_if_changed(path, json.dumps(snapshot, indent=2) + "\n") + + +def write_provenance_snapshot( + path: str, source: str, imported_at: str, dats: dict, entries: list[dict] +) -> bool: + """Write a normalized provenance snapshot.""" + return write_snapshot(path, build_snapshot(source, imported_at, dats, entries)) diff --git a/scripts/scraper/romset_dat_importer.py b/scripts/scraper/romset_dat_importer.py index f05f7a49..455d01b8 100644 --- a/scripts/scraper/romset_dat_importer.py +++ b/scripts/scraper/romset_dat_importer.py @@ -8,6 +8,7 @@ documents every set of one version at once. MAME ships its ``-listxml`` output as a release asset and FBNeo keeps its DATs in its repository, so both are fetched without a browser: + python -m scripts.scraper.romset_dat_importer --source mame --fetch python -m scripts.scraper.romset_dat_importer --source mame --fetch mame0289 python -m scripts.scraper.romset_dat_importer --source fbneo --fetch python -m scripts.scraper.romset_dat_importer --source mame --pack local.dat @@ -38,8 +39,8 @@ from ..common import ( list_registered_platforms, load_emulator_profiles, load_platform_config, - write_provenance_snapshot, ) +from ..dumpcatalog import build_snapshot, write_snapshot from .logiqx_parser import LogiqxDat, parse_logiqx MAX_MEMBER_SIZE = 200 * 1024 * 1024 @@ -252,31 +253,42 @@ def compact_entries(entries: list[dict]) -> list[dict]: def merge_snapshot( - output: str, source: str, dats: dict[str, str], entries: list[dict] + output: str, + source: str, + dats: dict[str, str], + entries: list[dict], + upstream: dict[str, str] | None = None, ) -> bool: """Accumulate recipes across versions instead of replacing them. A platform pins the archive of whichever version its list was built against, so the snapshot is a growing library of versions, not a picture - of the newest one. + of the newest one. ``upstream`` (DAT file -> git blob sha) replaces the + previous one: it describes the files this import read, not a history. """ path = Path(output) known_entries: list[dict] = [] known_dats: dict[str, str] = {} + known_upstream: dict = {} if path.is_file(): with path.open(encoding="utf-8") as handle: existing = json.load(handle) known_entries = list(existing.get("entries", [])) known_dats = dict(existing.get("dats", {})) + known_upstream = dict(existing.get("upstream", {})) known_dats.update(dats) - return write_provenance_snapshot( - output, + if upstream: + known_upstream["blobs"] = dict(sorted(upstream.items())) + snapshot = build_snapshot( source, datetime.now(timezone.utc).strftime("%Y-%m-%d"), known_dats, compact_entries(known_entries + entries), ) + if known_upstream: + snapshot["upstream"] = dict(sorted(known_upstream.items())) + return write_snapshot(output, snapshot) def _iter_dat_contents(pack: Path): @@ -301,9 +313,47 @@ def _iter_dat_contents(pack: Path): MAME_LISTXML_URL = ( "https://github.com/mamedev/mame/releases/download/{tag}/{tag}lx.zip" ) +MAME_LATEST_API = "https://api.github.com/repos/mamedev/mame/releases/latest" FBNEO_DATS_API = "https://api.github.com/repos/libretro/FBNeo/contents/dats" +def _api_json(url: str) -> object: + request = urllib.request.Request(url, headers={"User-Agent": "retrobios"}) + with urllib.request.urlopen(request, timeout=60) as response: + return json.load(response) + + +class NoReleaseError(ValueError): + def __init__(self) -> None: + super().__init__("the newest MAME release carries no tag") + + +def latest_mame_tag() -> str: + """Tag of the newest MAME release, the one ``--fetch`` takes by default.""" + payload = _api_json(MAME_LATEST_API) + tag = payload.get("tag_name") if isinstance(payload, dict) else None + if not tag: + raise NoReleaseError + return str(tag) + + +def fetch_tag(source: str, requested: str) -> str: + """The tag a ``--fetch`` names: bare, the newest MAME release.""" + if source == "mame" and requested == "latest": + tag = latest_mame_tag() + print(f" Derniere release MAME : {tag}") + return tag + return requested + + +def git_blob_sha(data: bytes) -> str: + """The sha git gives a blob, so a cached file can be checked against a + repository listing without any other state.""" + digest = hashlib.sha1(f"blob {len(data)}\0".encode()) + digest.update(data) + return digest.hexdigest() + + def _download(url: str, destination: Path) -> Path: """Fetch *url* to *destination* unless it is already there.""" if destination.is_file(): @@ -323,28 +373,38 @@ def _download(url: str, destination: Path) -> Path: return destination -def fetch_pack(source: str, tag: str, cache_dir: Path) -> Path: +def fetch_pack( + source: str, tag: str, cache_dir: Path +) -> tuple[Path, dict[str, str]]: """Download the upstream DAT for *source*. MAME ships its ``-listxml`` output as a release asset, and FBNeo keeps its - DATs in the repository, so neither needs a browser. + DATs in the repository, so neither needs a browser. The second value + names the upstream state fetched: empty for a release asset, whose tag + already says which version it is, and the git blob sha of every DAT for + FBNeo, whose files change under a constant header version. """ if source == "mame": - return _download( - MAME_LISTXML_URL.format(tag=tag), cache_dir / f"{tag}lx.zip" + return ( + _download(MAME_LISTXML_URL.format(tag=tag), cache_dir / f"{tag}lx.zip"), + {}, ) - request = urllib.request.Request( - FBNEO_DATS_API, headers={"User-Agent": "retrobios"} - ) - with urllib.request.urlopen(request, timeout=60) as response: - listing = json.load(response) + listing = _api_json(FBNEO_DATS_API) target = cache_dir / "fbneo-dats" target.mkdir(parents=True, exist_ok=True) + blobs: dict[str, str] = {} for item in listing: - if item.get("type") == "file" and item["name"].lower().endswith(".dat"): - _download(item["download_url"], target / item["name"]) - return target + if item.get("type") != "file" or not item["name"].lower().endswith(".dat"): + continue + destination = target / item["name"] + # The cache is addressed by content: a file already there is kept + # only while its blob sha is the one the repository lists. + if destination.is_file() and git_blob_sha(destination.read_bytes()) != item["sha"]: + destination.unlink() + _download(item["download_url"], destination) + blobs[item["name"]] = item["sha"] + return target, blobs def _is_listxml(pack: Path) -> bool: @@ -383,8 +443,9 @@ def main() -> int: parser.add_argument( "--fetch", nargs="?", - const="mame0289", - help="download upstream instead of using --pack (MAME tag, e.g. mame0289)", + const="latest", + help="download upstream instead of using --pack " + "(a MAME tag such as mame0289; bare, the newest release)", ) parser.add_argument("--cache-dir", default=".cache/dats") parser.add_argument("--output", default="") @@ -396,8 +457,11 @@ def main() -> int: parser.add_argument("--dry-run", action="store_true") args = parser.parse_args() + blobs: dict[str, str] = {} if args.fetch: - pack = fetch_pack(args.source, args.fetch, Path(args.cache_dir)) + pack, blobs = fetch_pack( + args.source, fetch_tag(args.source, args.fetch), Path(args.cache_dir) + ) print(f" Recupere {pack}") elif args.pack: pack = Path(args.pack) @@ -457,7 +521,7 @@ def main() -> int: if args.dry_run: return 0 - changed = merge_snapshot(output, args.source, dats, entries) + changed = merge_snapshot(output, args.source, dats, entries, blobs) print(f"{'Wrote' if changed else 'Unchanged'} {output}") return 0 diff --git a/tests/test_romset_recipes.py b/tests/test_romset_recipes.py index bbcefa9d..1cf85c26 100644 --- a/tests/test_romset_recipes.py +++ b/tests/test_romset_recipes.py @@ -29,6 +29,7 @@ from romset_recipes import ( # noqa: E402 from scripts.scraper.romset_dat_importer import ( # noqa: E402 compact_entries, listxml_entries, + git_blob_sha, merge_snapshot, recipe_entries, ) @@ -384,6 +385,36 @@ class ListXmlRecipes(unittest.TestCase): [m["name"] for m in entries[0]["members"]], ["own.rom", "shared.rom"] ) + def test_upstream_blobs_are_recorded_and_replaced(self): + """The snapshot names the DAT blobs it was read from, so a later + check can tell whether upstream moved without importing again.""" + entry = { + "dat": "FBNeo - X", + "name": "set.zip", + "set": "set", + "description": "", + "members": [{"name": "a", "crc32": "1"}], + } + with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory: + output = str(Path(directory) / "fbneo.json") + merge_snapshot(output, "fbneo", {"FBNeo - X": "1.0"}, [entry], {"X.dat": "aaa"}) + with open(output, encoding="utf-8") as handle: + self.assertEqual(json.load(handle)["upstream"], {"blobs": {"X.dat": "aaa"}}) + merge_snapshot(output, "fbneo", {"FBNeo - X": "1.0"}, [entry], {"X.dat": "bbb"}) + with open(output, encoding="utf-8") as handle: + snapshot = json.load(handle) + self.assertEqual(snapshot["upstream"], {"blobs": {"X.dat": "bbb"}}) + self.assertEqual(len(snapshot["entries"]), 1) + # An import that names no blobs (a local --pack) keeps the last ones. + merge_snapshot(output, "fbneo", {"FBNeo - X": "1.0"}, [entry]) + with open(output, encoding="utf-8") as handle: + self.assertEqual(json.load(handle)["upstream"], {"blobs": {"X.dat": "bbb"}}) + + def test_git_blob_sha_matches_git(self): + # The values git hash-object gives for the empty blob and for "hi". + self.assertEqual(git_blob_sha(b""), "e69de29bb2d1d6434b8b29ae775ad8c2e48c5391") + self.assertEqual(git_blob_sha(b"hi"), "32f95c0d1244a78b2be1bab8de17906fabb2c4a8") + def test_versions_accumulate_instead_of_replacing_each_other(self): """A platform pins the archive of whichever version it was built on.""" with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory: