feat: record upstream dat blobs in recipe snapshots

This commit is contained in:
Abdessamad Derraz committed 2026-10-04 16:35:59 +02:00
1 parent 887fc34bf6
commit 1a993edc0d
3 files changed
+132 -26

No files matched your search

+16 -5
View File
@@ -83,14 +83,25 @@ def annotate_provenance(files: dict, snapshots: dict) -> dict[str, int]:
entry.pop("provenance", None) entry.pop("provenance", None)
return counts return counts
def write_provenance_snapshot( def build_snapshot(
path: str, source: str, imported_at: str, dats: dict, entries: list[dict] source: str, imported_at: str, dats: dict, entries: list[dict]
) -> bool: ) -> dict:
"""Write a normalized provenance snapshot, sorted for determinism.""" """A provenance snapshot, sorted for determinism."""
snapshot = { return {
"source": source, "source": source,
"imported_at": imported_at, "imported_at": imported_at,
"dats": dict(sorted(dats.items())), "dats": dict(sorted(dats.items())),
"entries": sorted(entries, key=lambda e: (e["dat"], e["name"])), "entries": sorted(entries, key=lambda e: (e["dat"], e["name"])),
} }
def write_snapshot(path: str, snapshot: dict) -> bool:
"""Write a snapshot; timestamps alone never count as a change."""
return write_if_changed(path, json.dumps(snapshot, indent=2) + "\n") return write_if_changed(path, json.dumps(snapshot, indent=2) + "\n")
def write_provenance_snapshot(
path: str, source: str, imported_at: str, dats: dict, entries: list[dict]
) -> bool:
"""Write a normalized provenance snapshot."""
return write_snapshot(path, build_snapshot(source, imported_at, dats, entries))
+85 -21
View File
@@ -8,6 +8,7 @@ documents every set of one version at once.
MAME ships its ``-listxml`` output as a release asset and FBNeo keeps its DATs MAME ships its ``-listxml`` output as a release asset and FBNeo keeps its DATs
in its repository, so both are fetched without a browser: in its repository, so both are fetched without a browser:
python -m scripts.scraper.romset_dat_importer --source mame --fetch
python -m scripts.scraper.romset_dat_importer --source mame --fetch mame0289 python -m scripts.scraper.romset_dat_importer --source mame --fetch mame0289
python -m scripts.scraper.romset_dat_importer --source fbneo --fetch python -m scripts.scraper.romset_dat_importer --source fbneo --fetch
python -m scripts.scraper.romset_dat_importer --source mame --pack local.dat python -m scripts.scraper.romset_dat_importer --source mame --pack local.dat
@@ -38,8 +39,8 @@ from ..common import (
list_registered_platforms, list_registered_platforms,
load_emulator_profiles, load_emulator_profiles,
load_platform_config, load_platform_config,
write_provenance_snapshot,
) )
from ..dumpcatalog import build_snapshot, write_snapshot
from .logiqx_parser import LogiqxDat, parse_logiqx from .logiqx_parser import LogiqxDat, parse_logiqx
MAX_MEMBER_SIZE = 200 * 1024 * 1024 MAX_MEMBER_SIZE = 200 * 1024 * 1024
@@ -252,31 +253,42 @@ def compact_entries(entries: list[dict]) -> list[dict]:
def merge_snapshot( def merge_snapshot(
output: str, source: str, dats: dict[str, str], entries: list[dict] output: str,
source: str,
dats: dict[str, str],
entries: list[dict],
upstream: dict[str, str] | None = None,
) -> bool: ) -> bool:
"""Accumulate recipes across versions instead of replacing them. """Accumulate recipes across versions instead of replacing them.
A platform pins the archive of whichever version its list was built A platform pins the archive of whichever version its list was built
against, so the snapshot is a growing library of versions, not a picture against, so the snapshot is a growing library of versions, not a picture
of the newest one. of the newest one. ``upstream`` (DAT file -> git blob sha) replaces the
previous one: it describes the files this import read, not a history.
""" """
path = Path(output) path = Path(output)
known_entries: list[dict] = [] known_entries: list[dict] = []
known_dats: dict[str, str] = {} known_dats: dict[str, str] = {}
known_upstream: dict = {}
if path.is_file(): if path.is_file():
with path.open(encoding="utf-8") as handle: with path.open(encoding="utf-8") as handle:
existing = json.load(handle) existing = json.load(handle)
known_entries = list(existing.get("entries", [])) known_entries = list(existing.get("entries", []))
known_dats = dict(existing.get("dats", {})) known_dats = dict(existing.get("dats", {}))
known_upstream = dict(existing.get("upstream", {}))
known_dats.update(dats) known_dats.update(dats)
return write_provenance_snapshot( if upstream:
output, known_upstream["blobs"] = dict(sorted(upstream.items()))
snapshot = build_snapshot(
source, source,
datetime.now(timezone.utc).strftime("%Y-%m-%d"), datetime.now(timezone.utc).strftime("%Y-%m-%d"),
known_dats, known_dats,
compact_entries(known_entries + entries), compact_entries(known_entries + entries),
) )
if known_upstream:
snapshot["upstream"] = dict(sorted(known_upstream.items()))
return write_snapshot(output, snapshot)
def _iter_dat_contents(pack: Path): def _iter_dat_contents(pack: Path):
@@ -301,9 +313,47 @@ def _iter_dat_contents(pack: Path):
MAME_LISTXML_URL = ( MAME_LISTXML_URL = (
"https://github.com/mamedev/mame/releases/download/{tag}/{tag}lx.zip" "https://github.com/mamedev/mame/releases/download/{tag}/{tag}lx.zip"
) )
MAME_LATEST_API = "https://api.github.com/repos/mamedev/mame/releases/latest"
FBNEO_DATS_API = "https://api.github.com/repos/libretro/FBNeo/contents/dats" FBNEO_DATS_API = "https://api.github.com/repos/libretro/FBNeo/contents/dats"
def _api_json(url: str) -> object:
request = urllib.request.Request(url, headers={"User-Agent": "retrobios"})
with urllib.request.urlopen(request, timeout=60) as response:
return json.load(response)
class NoReleaseError(ValueError):
def __init__(self) -> None:
super().__init__("the newest MAME release carries no tag")
def latest_mame_tag() -> str:
"""Tag of the newest MAME release, the one ``--fetch`` takes by default."""
payload = _api_json(MAME_LATEST_API)
tag = payload.get("tag_name") if isinstance(payload, dict) else None
if not tag:
raise NoReleaseError
return str(tag)
def fetch_tag(source: str, requested: str) -> str:
"""The tag a ``--fetch`` names: bare, the newest MAME release."""
if source == "mame" and requested == "latest":
tag = latest_mame_tag()
print(f" Derniere release MAME : {tag}")
return tag
return requested
def git_blob_sha(data: bytes) -> str:
"""The sha git gives a blob, so a cached file can be checked against a
repository listing without any other state."""
digest = hashlib.sha1(f"blob {len(data)}\0".encode())
digest.update(data)
return digest.hexdigest()
def _download(url: str, destination: Path) -> Path: def _download(url: str, destination: Path) -> Path:
"""Fetch *url* to *destination* unless it is already there.""" """Fetch *url* to *destination* unless it is already there."""
if destination.is_file(): if destination.is_file():
@@ -323,28 +373,38 @@ def _download(url: str, destination: Path) -> Path:
return destination return destination
def fetch_pack(source: str, tag: str, cache_dir: Path) -> Path: def fetch_pack(
source: str, tag: str, cache_dir: Path
) -> tuple[Path, dict[str, str]]:
"""Download the upstream DAT for *source*. """Download the upstream DAT for *source*.
MAME ships its ``-listxml`` output as a release asset, and FBNeo keeps its MAME ships its ``-listxml`` output as a release asset, and FBNeo keeps its
DATs in the repository, so neither needs a browser. DATs in the repository, so neither needs a browser. The second value
names the upstream state fetched: empty for a release asset, whose tag
already says which version it is, and the git blob sha of every DAT for
FBNeo, whose files change under a constant header version.
""" """
if source == "mame": if source == "mame":
return _download( return (
MAME_LISTXML_URL.format(tag=tag), cache_dir / f"{tag}lx.zip" _download(MAME_LISTXML_URL.format(tag=tag), cache_dir / f"{tag}lx.zip"),
{},
) )
request = urllib.request.Request( listing = _api_json(FBNEO_DATS_API)
FBNEO_DATS_API, headers={"User-Agent": "retrobios"}
)
with urllib.request.urlopen(request, timeout=60) as response:
listing = json.load(response)
target = cache_dir / "fbneo-dats" target = cache_dir / "fbneo-dats"
target.mkdir(parents=True, exist_ok=True) target.mkdir(parents=True, exist_ok=True)
blobs: dict[str, str] = {}
for item in listing: for item in listing:
if item.get("type") == "file" and item["name"].lower().endswith(".dat"): if item.get("type") != "file" or not item["name"].lower().endswith(".dat"):
_download(item["download_url"], target / item["name"]) continue
return target destination = target / item["name"]
# The cache is addressed by content: a file already there is kept
# only while its blob sha is the one the repository lists.
if destination.is_file() and git_blob_sha(destination.read_bytes()) != item["sha"]:
destination.unlink()
_download(item["download_url"], destination)
blobs[item["name"]] = item["sha"]
return target, blobs
def _is_listxml(pack: Path) -> bool: def _is_listxml(pack: Path) -> bool:
@@ -383,8 +443,9 @@ def main() -> int:
parser.add_argument( parser.add_argument(
"--fetch", "--fetch",
nargs="?", nargs="?",
const="mame0289", const="latest",
help="download upstream instead of using --pack (MAME tag, e.g. mame0289)", help="download upstream instead of using --pack "
"(a MAME tag such as mame0289; bare, the newest release)",
) )
parser.add_argument("--cache-dir", default=".cache/dats") parser.add_argument("--cache-dir", default=".cache/dats")
parser.add_argument("--output", default="") parser.add_argument("--output", default="")
@@ -396,8 +457,11 @@ def main() -> int:
parser.add_argument("--dry-run", action="store_true") parser.add_argument("--dry-run", action="store_true")
args = parser.parse_args() args = parser.parse_args()
blobs: dict[str, str] = {}
if args.fetch: if args.fetch:
pack = fetch_pack(args.source, args.fetch, Path(args.cache_dir)) pack, blobs = fetch_pack(
args.source, fetch_tag(args.source, args.fetch), Path(args.cache_dir)
)
print(f" Recupere {pack}") print(f" Recupere {pack}")
elif args.pack: elif args.pack:
pack = Path(args.pack) pack = Path(args.pack)
@@ -457,7 +521,7 @@ def main() -> int:
if args.dry_run: if args.dry_run:
return 0 return 0
changed = merge_snapshot(output, args.source, dats, entries) changed = merge_snapshot(output, args.source, dats, entries, blobs)
print(f"{'Wrote' if changed else 'Unchanged'} {output}") print(f"{'Wrote' if changed else 'Unchanged'} {output}")
return 0 return 0
+31
View File
@@ -29,6 +29,7 @@ from romset_recipes import ( # noqa: E402
from scripts.scraper.romset_dat_importer import ( # noqa: E402 from scripts.scraper.romset_dat_importer import ( # noqa: E402
compact_entries, compact_entries,
listxml_entries, listxml_entries,
git_blob_sha,
merge_snapshot, merge_snapshot,
recipe_entries, recipe_entries,
) )
@@ -384,6 +385,36 @@ class ListXmlRecipes(unittest.TestCase):
[m["name"] for m in entries[0]["members"]], ["own.rom", "shared.rom"] [m["name"] for m in entries[0]["members"]], ["own.rom", "shared.rom"]
) )
def test_upstream_blobs_are_recorded_and_replaced(self):
"""The snapshot names the DAT blobs it was read from, so a later
check can tell whether upstream moved without importing again."""
entry = {
"dat": "FBNeo - X",
"name": "set.zip",
"set": "set",
"description": "",
"members": [{"name": "a", "crc32": "1"}],
}
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
output = str(Path(directory) / "fbneo.json")
merge_snapshot(output, "fbneo", {"FBNeo - X": "1.0"}, [entry], {"X.dat": "aaa"})
with open(output, encoding="utf-8") as handle:
self.assertEqual(json.load(handle)["upstream"], {"blobs": {"X.dat": "aaa"}})
merge_snapshot(output, "fbneo", {"FBNeo - X": "1.0"}, [entry], {"X.dat": "bbb"})
with open(output, encoding="utf-8") as handle:
snapshot = json.load(handle)
self.assertEqual(snapshot["upstream"], {"blobs": {"X.dat": "bbb"}})
self.assertEqual(len(snapshot["entries"]), 1)
# An import that names no blobs (a local --pack) keeps the last ones.
merge_snapshot(output, "fbneo", {"FBNeo - X": "1.0"}, [entry])
with open(output, encoding="utf-8") as handle:
self.assertEqual(json.load(handle)["upstream"], {"blobs": {"X.dat": "bbb"}})
def test_git_blob_sha_matches_git(self):
# The values git hash-object gives for the empty blob and for "hi".
self.assertEqual(git_blob_sha(b""), "e69de29bb2d1d6434b8b29ae775ad8c2e48c5391")
self.assertEqual(git_blob_sha(b"hi"), "32f95c0d1244a78b2be1bab8de17906fabb2c4a8")
def test_versions_accumulate_instead_of_replacing_each_other(self): def test_versions_accumulate_instead_of_replacing_each_other(self):
"""A platform pins the archive of whichever version it was built on.""" """A platform pins the archive of whichever version it was built on."""
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory: with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory: