mirror of
https://github.com/Abdess/retroarch_system.git
synced 2026-10-10 13:33:24 -05:00
feat: record upstream dat blobs in recipe snapshots
This commit is contained in:
1 parent
887fc34bf6
commit
1a993edc0d
3 files changed
+132
-26
No files matched your search
+16
-5
@@ -83,14 +83,25 @@ def annotate_provenance(files: dict, snapshots: dict) -> dict[str, int]:
|
||||
entry.pop("provenance", None)
|
||||
return counts
|
||||
|
||||
def write_provenance_snapshot(
|
||||
path: str, source: str, imported_at: str, dats: dict, entries: list[dict]
|
||||
) -> bool:
|
||||
"""Write a normalized provenance snapshot, sorted for determinism."""
|
||||
snapshot = {
|
||||
def build_snapshot(
|
||||
source: str, imported_at: str, dats: dict, entries: list[dict]
|
||||
) -> dict:
|
||||
"""A provenance snapshot, sorted for determinism."""
|
||||
return {
|
||||
"source": source,
|
||||
"imported_at": imported_at,
|
||||
"dats": dict(sorted(dats.items())),
|
||||
"entries": sorted(entries, key=lambda e: (e["dat"], e["name"])),
|
||||
}
|
||||
|
||||
|
||||
def write_snapshot(path: str, snapshot: dict) -> bool:
|
||||
"""Write a snapshot; timestamps alone never count as a change."""
|
||||
return write_if_changed(path, json.dumps(snapshot, indent=2) + "\n")
|
||||
|
||||
|
||||
def write_provenance_snapshot(
|
||||
path: str, source: str, imported_at: str, dats: dict, entries: list[dict]
|
||||
) -> bool:
|
||||
"""Write a normalized provenance snapshot."""
|
||||
return write_snapshot(path, build_snapshot(source, imported_at, dats, entries))
|
||||
@@ -8,6 +8,7 @@ documents every set of one version at once.
|
||||
MAME ships its ``-listxml`` output as a release asset and FBNeo keeps its DATs
|
||||
in its repository, so both are fetched without a browser:
|
||||
|
||||
python -m scripts.scraper.romset_dat_importer --source mame --fetch
|
||||
python -m scripts.scraper.romset_dat_importer --source mame --fetch mame0289
|
||||
python -m scripts.scraper.romset_dat_importer --source fbneo --fetch
|
||||
python -m scripts.scraper.romset_dat_importer --source mame --pack local.dat
|
||||
@@ -38,8 +39,8 @@ from ..common import (
|
||||
list_registered_platforms,
|
||||
load_emulator_profiles,
|
||||
load_platform_config,
|
||||
write_provenance_snapshot,
|
||||
)
|
||||
from ..dumpcatalog import build_snapshot, write_snapshot
|
||||
from .logiqx_parser import LogiqxDat, parse_logiqx
|
||||
|
||||
MAX_MEMBER_SIZE = 200 * 1024 * 1024
|
||||
@@ -252,31 +253,42 @@ def compact_entries(entries: list[dict]) -> list[dict]:
|
||||
|
||||
|
||||
def merge_snapshot(
|
||||
output: str, source: str, dats: dict[str, str], entries: list[dict]
|
||||
output: str,
|
||||
source: str,
|
||||
dats: dict[str, str],
|
||||
entries: list[dict],
|
||||
upstream: dict[str, str] | None = None,
|
||||
) -> bool:
|
||||
"""Accumulate recipes across versions instead of replacing them.
|
||||
|
||||
A platform pins the archive of whichever version its list was built
|
||||
against, so the snapshot is a growing library of versions, not a picture
|
||||
of the newest one.
|
||||
of the newest one. ``upstream`` (DAT file -> git blob sha) replaces the
|
||||
previous one: it describes the files this import read, not a history.
|
||||
"""
|
||||
path = Path(output)
|
||||
known_entries: list[dict] = []
|
||||
known_dats: dict[str, str] = {}
|
||||
known_upstream: dict = {}
|
||||
if path.is_file():
|
||||
with path.open(encoding="utf-8") as handle:
|
||||
existing = json.load(handle)
|
||||
known_entries = list(existing.get("entries", []))
|
||||
known_dats = dict(existing.get("dats", {}))
|
||||
known_upstream = dict(existing.get("upstream", {}))
|
||||
|
||||
known_dats.update(dats)
|
||||
return write_provenance_snapshot(
|
||||
output,
|
||||
if upstream:
|
||||
known_upstream["blobs"] = dict(sorted(upstream.items()))
|
||||
snapshot = build_snapshot(
|
||||
source,
|
||||
datetime.now(timezone.utc).strftime("%Y-%m-%d"),
|
||||
known_dats,
|
||||
compact_entries(known_entries + entries),
|
||||
)
|
||||
if known_upstream:
|
||||
snapshot["upstream"] = dict(sorted(known_upstream.items()))
|
||||
return write_snapshot(output, snapshot)
|
||||
|
||||
|
||||
def _iter_dat_contents(pack: Path):
|
||||
@@ -301,9 +313,47 @@ def _iter_dat_contents(pack: Path):
|
||||
MAME_LISTXML_URL = (
|
||||
"https://github.com/mamedev/mame/releases/download/{tag}/{tag}lx.zip"
|
||||
)
|
||||
MAME_LATEST_API = "https://api.github.com/repos/mamedev/mame/releases/latest"
|
||||
FBNEO_DATS_API = "https://api.github.com/repos/libretro/FBNeo/contents/dats"
|
||||
|
||||
|
||||
def _api_json(url: str) -> object:
|
||||
request = urllib.request.Request(url, headers={"User-Agent": "retrobios"})
|
||||
with urllib.request.urlopen(request, timeout=60) as response:
|
||||
return json.load(response)
|
||||
|
||||
|
||||
class NoReleaseError(ValueError):
|
||||
def __init__(self) -> None:
|
||||
super().__init__("the newest MAME release carries no tag")
|
||||
|
||||
|
||||
def latest_mame_tag() -> str:
|
||||
"""Tag of the newest MAME release, the one ``--fetch`` takes by default."""
|
||||
payload = _api_json(MAME_LATEST_API)
|
||||
tag = payload.get("tag_name") if isinstance(payload, dict) else None
|
||||
if not tag:
|
||||
raise NoReleaseError
|
||||
return str(tag)
|
||||
|
||||
|
||||
def fetch_tag(source: str, requested: str) -> str:
|
||||
"""The tag a ``--fetch`` names: bare, the newest MAME release."""
|
||||
if source == "mame" and requested == "latest":
|
||||
tag = latest_mame_tag()
|
||||
print(f" Derniere release MAME : {tag}")
|
||||
return tag
|
||||
return requested
|
||||
|
||||
|
||||
def git_blob_sha(data: bytes) -> str:
|
||||
"""The sha git gives a blob, so a cached file can be checked against a
|
||||
repository listing without any other state."""
|
||||
digest = hashlib.sha1(f"blob {len(data)}\0".encode())
|
||||
digest.update(data)
|
||||
return digest.hexdigest()
|
||||
|
||||
|
||||
def _download(url: str, destination: Path) -> Path:
|
||||
"""Fetch *url* to *destination* unless it is already there."""
|
||||
if destination.is_file():
|
||||
@@ -323,28 +373,38 @@ def _download(url: str, destination: Path) -> Path:
|
||||
return destination
|
||||
|
||||
|
||||
def fetch_pack(source: str, tag: str, cache_dir: Path) -> Path:
|
||||
def fetch_pack(
|
||||
source: str, tag: str, cache_dir: Path
|
||||
) -> tuple[Path, dict[str, str]]:
|
||||
"""Download the upstream DAT for *source*.
|
||||
|
||||
MAME ships its ``-listxml`` output as a release asset, and FBNeo keeps its
|
||||
DATs in the repository, so neither needs a browser.
|
||||
DATs in the repository, so neither needs a browser. The second value
|
||||
names the upstream state fetched: empty for a release asset, whose tag
|
||||
already says which version it is, and the git blob sha of every DAT for
|
||||
FBNeo, whose files change under a constant header version.
|
||||
"""
|
||||
if source == "mame":
|
||||
return _download(
|
||||
MAME_LISTXML_URL.format(tag=tag), cache_dir / f"{tag}lx.zip"
|
||||
return (
|
||||
_download(MAME_LISTXML_URL.format(tag=tag), cache_dir / f"{tag}lx.zip"),
|
||||
{},
|
||||
)
|
||||
|
||||
request = urllib.request.Request(
|
||||
FBNEO_DATS_API, headers={"User-Agent": "retrobios"}
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=60) as response:
|
||||
listing = json.load(response)
|
||||
listing = _api_json(FBNEO_DATS_API)
|
||||
target = cache_dir / "fbneo-dats"
|
||||
target.mkdir(parents=True, exist_ok=True)
|
||||
blobs: dict[str, str] = {}
|
||||
for item in listing:
|
||||
if item.get("type") == "file" and item["name"].lower().endswith(".dat"):
|
||||
_download(item["download_url"], target / item["name"])
|
||||
return target
|
||||
if item.get("type") != "file" or not item["name"].lower().endswith(".dat"):
|
||||
continue
|
||||
destination = target / item["name"]
|
||||
# The cache is addressed by content: a file already there is kept
|
||||
# only while its blob sha is the one the repository lists.
|
||||
if destination.is_file() and git_blob_sha(destination.read_bytes()) != item["sha"]:
|
||||
destination.unlink()
|
||||
_download(item["download_url"], destination)
|
||||
blobs[item["name"]] = item["sha"]
|
||||
return target, blobs
|
||||
|
||||
|
||||
def _is_listxml(pack: Path) -> bool:
|
||||
@@ -383,8 +443,9 @@ def main() -> int:
|
||||
parser.add_argument(
|
||||
"--fetch",
|
||||
nargs="?",
|
||||
const="mame0289",
|
||||
help="download upstream instead of using --pack (MAME tag, e.g. mame0289)",
|
||||
const="latest",
|
||||
help="download upstream instead of using --pack "
|
||||
"(a MAME tag such as mame0289; bare, the newest release)",
|
||||
)
|
||||
parser.add_argument("--cache-dir", default=".cache/dats")
|
||||
parser.add_argument("--output", default="")
|
||||
@@ -396,8 +457,11 @@ def main() -> int:
|
||||
parser.add_argument("--dry-run", action="store_true")
|
||||
args = parser.parse_args()
|
||||
|
||||
blobs: dict[str, str] = {}
|
||||
if args.fetch:
|
||||
pack = fetch_pack(args.source, args.fetch, Path(args.cache_dir))
|
||||
pack, blobs = fetch_pack(
|
||||
args.source, fetch_tag(args.source, args.fetch), Path(args.cache_dir)
|
||||
)
|
||||
print(f" Recupere {pack}")
|
||||
elif args.pack:
|
||||
pack = Path(args.pack)
|
||||
@@ -457,7 +521,7 @@ def main() -> int:
|
||||
if args.dry_run:
|
||||
return 0
|
||||
|
||||
changed = merge_snapshot(output, args.source, dats, entries)
|
||||
changed = merge_snapshot(output, args.source, dats, entries, blobs)
|
||||
print(f"{'Wrote' if changed else 'Unchanged'} {output}")
|
||||
return 0
|
||||
|
||||
|
||||
@@ -29,6 +29,7 @@ from romset_recipes import ( # noqa: E402
|
||||
from scripts.scraper.romset_dat_importer import ( # noqa: E402
|
||||
compact_entries,
|
||||
listxml_entries,
|
||||
git_blob_sha,
|
||||
merge_snapshot,
|
||||
recipe_entries,
|
||||
)
|
||||
@@ -384,6 +385,36 @@ class ListXmlRecipes(unittest.TestCase):
|
||||
[m["name"] for m in entries[0]["members"]], ["own.rom", "shared.rom"]
|
||||
)
|
||||
|
||||
def test_upstream_blobs_are_recorded_and_replaced(self):
|
||||
"""The snapshot names the DAT blobs it was read from, so a later
|
||||
check can tell whether upstream moved without importing again."""
|
||||
entry = {
|
||||
"dat": "FBNeo - X",
|
||||
"name": "set.zip",
|
||||
"set": "set",
|
||||
"description": "",
|
||||
"members": [{"name": "a", "crc32": "1"}],
|
||||
}
|
||||
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
|
||||
output = str(Path(directory) / "fbneo.json")
|
||||
merge_snapshot(output, "fbneo", {"FBNeo - X": "1.0"}, [entry], {"X.dat": "aaa"})
|
||||
with open(output, encoding="utf-8") as handle:
|
||||
self.assertEqual(json.load(handle)["upstream"], {"blobs": {"X.dat": "aaa"}})
|
||||
merge_snapshot(output, "fbneo", {"FBNeo - X": "1.0"}, [entry], {"X.dat": "bbb"})
|
||||
with open(output, encoding="utf-8") as handle:
|
||||
snapshot = json.load(handle)
|
||||
self.assertEqual(snapshot["upstream"], {"blobs": {"X.dat": "bbb"}})
|
||||
self.assertEqual(len(snapshot["entries"]), 1)
|
||||
# An import that names no blobs (a local --pack) keeps the last ones.
|
||||
merge_snapshot(output, "fbneo", {"FBNeo - X": "1.0"}, [entry])
|
||||
with open(output, encoding="utf-8") as handle:
|
||||
self.assertEqual(json.load(handle)["upstream"], {"blobs": {"X.dat": "bbb"}})
|
||||
|
||||
def test_git_blob_sha_matches_git(self):
|
||||
# The values git hash-object gives for the empty blob and for "hi".
|
||||
self.assertEqual(git_blob_sha(b""), "e69de29bb2d1d6434b8b29ae775ad8c2e48c5391")
|
||||
self.assertEqual(git_blob_sha(b"hi"), "32f95c0d1244a78b2be1bab8de17906fabb2c4a8")
|
||||
|
||||
def test_versions_accumulate_instead_of_replacing_each_other(self):
|
||||
"""A platform pins the archive of whichever version it was built on."""
|
||||
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
|
||||
|
||||
Reference in new issue
Block a user