mirror of
https://github.com/Abdess/retroarch_system.git
synced 2026-10-10 13:33:24 -05:00
feat: record upstream dat blobs in recipe snapshots
This commit is contained in:
1 parent
887fc34bf6
commit
1a993edc0d
3 files changed
+132
-26
No files matched your search
+16
-5
@@ -83,14 +83,25 @@ def annotate_provenance(files: dict, snapshots: dict) -> dict[str, int]:
|
|||||||
entry.pop("provenance", None)
|
entry.pop("provenance", None)
|
||||||
return counts
|
return counts
|
||||||
|
|
||||||
def write_provenance_snapshot(
|
def build_snapshot(
|
||||||
path: str, source: str, imported_at: str, dats: dict, entries: list[dict]
|
source: str, imported_at: str, dats: dict, entries: list[dict]
|
||||||
) -> bool:
|
) -> dict:
|
||||||
"""Write a normalized provenance snapshot, sorted for determinism."""
|
"""A provenance snapshot, sorted for determinism."""
|
||||||
snapshot = {
|
return {
|
||||||
"source": source,
|
"source": source,
|
||||||
"imported_at": imported_at,
|
"imported_at": imported_at,
|
||||||
"dats": dict(sorted(dats.items())),
|
"dats": dict(sorted(dats.items())),
|
||||||
"entries": sorted(entries, key=lambda e: (e["dat"], e["name"])),
|
"entries": sorted(entries, key=lambda e: (e["dat"], e["name"])),
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def write_snapshot(path: str, snapshot: dict) -> bool:
|
||||||
|
"""Write a snapshot; timestamps alone never count as a change."""
|
||||||
return write_if_changed(path, json.dumps(snapshot, indent=2) + "\n")
|
return write_if_changed(path, json.dumps(snapshot, indent=2) + "\n")
|
||||||
|
|
||||||
|
|
||||||
|
def write_provenance_snapshot(
|
||||||
|
path: str, source: str, imported_at: str, dats: dict, entries: list[dict]
|
||||||
|
) -> bool:
|
||||||
|
"""Write a normalized provenance snapshot."""
|
||||||
|
return write_snapshot(path, build_snapshot(source, imported_at, dats, entries))
|
||||||
@@ -8,6 +8,7 @@ documents every set of one version at once.
|
|||||||
MAME ships its ``-listxml`` output as a release asset and FBNeo keeps its DATs
|
MAME ships its ``-listxml`` output as a release asset and FBNeo keeps its DATs
|
||||||
in its repository, so both are fetched without a browser:
|
in its repository, so both are fetched without a browser:
|
||||||
|
|
||||||
|
python -m scripts.scraper.romset_dat_importer --source mame --fetch
|
||||||
python -m scripts.scraper.romset_dat_importer --source mame --fetch mame0289
|
python -m scripts.scraper.romset_dat_importer --source mame --fetch mame0289
|
||||||
python -m scripts.scraper.romset_dat_importer --source fbneo --fetch
|
python -m scripts.scraper.romset_dat_importer --source fbneo --fetch
|
||||||
python -m scripts.scraper.romset_dat_importer --source mame --pack local.dat
|
python -m scripts.scraper.romset_dat_importer --source mame --pack local.dat
|
||||||
@@ -38,8 +39,8 @@ from ..common import (
|
|||||||
list_registered_platforms,
|
list_registered_platforms,
|
||||||
load_emulator_profiles,
|
load_emulator_profiles,
|
||||||
load_platform_config,
|
load_platform_config,
|
||||||
write_provenance_snapshot,
|
|
||||||
)
|
)
|
||||||
|
from ..dumpcatalog import build_snapshot, write_snapshot
|
||||||
from .logiqx_parser import LogiqxDat, parse_logiqx
|
from .logiqx_parser import LogiqxDat, parse_logiqx
|
||||||
|
|
||||||
MAX_MEMBER_SIZE = 200 * 1024 * 1024
|
MAX_MEMBER_SIZE = 200 * 1024 * 1024
|
||||||
@@ -252,31 +253,42 @@ def compact_entries(entries: list[dict]) -> list[dict]:
|
|||||||
|
|
||||||
|
|
||||||
def merge_snapshot(
|
def merge_snapshot(
|
||||||
output: str, source: str, dats: dict[str, str], entries: list[dict]
|
output: str,
|
||||||
|
source: str,
|
||||||
|
dats: dict[str, str],
|
||||||
|
entries: list[dict],
|
||||||
|
upstream: dict[str, str] | None = None,
|
||||||
) -> bool:
|
) -> bool:
|
||||||
"""Accumulate recipes across versions instead of replacing them.
|
"""Accumulate recipes across versions instead of replacing them.
|
||||||
|
|
||||||
A platform pins the archive of whichever version its list was built
|
A platform pins the archive of whichever version its list was built
|
||||||
against, so the snapshot is a growing library of versions, not a picture
|
against, so the snapshot is a growing library of versions, not a picture
|
||||||
of the newest one.
|
of the newest one. ``upstream`` (DAT file -> git blob sha) replaces the
|
||||||
|
previous one: it describes the files this import read, not a history.
|
||||||
"""
|
"""
|
||||||
path = Path(output)
|
path = Path(output)
|
||||||
known_entries: list[dict] = []
|
known_entries: list[dict] = []
|
||||||
known_dats: dict[str, str] = {}
|
known_dats: dict[str, str] = {}
|
||||||
|
known_upstream: dict = {}
|
||||||
if path.is_file():
|
if path.is_file():
|
||||||
with path.open(encoding="utf-8") as handle:
|
with path.open(encoding="utf-8") as handle:
|
||||||
existing = json.load(handle)
|
existing = json.load(handle)
|
||||||
known_entries = list(existing.get("entries", []))
|
known_entries = list(existing.get("entries", []))
|
||||||
known_dats = dict(existing.get("dats", {}))
|
known_dats = dict(existing.get("dats", {}))
|
||||||
|
known_upstream = dict(existing.get("upstream", {}))
|
||||||
|
|
||||||
known_dats.update(dats)
|
known_dats.update(dats)
|
||||||
return write_provenance_snapshot(
|
if upstream:
|
||||||
output,
|
known_upstream["blobs"] = dict(sorted(upstream.items()))
|
||||||
|
snapshot = build_snapshot(
|
||||||
source,
|
source,
|
||||||
datetime.now(timezone.utc).strftime("%Y-%m-%d"),
|
datetime.now(timezone.utc).strftime("%Y-%m-%d"),
|
||||||
known_dats,
|
known_dats,
|
||||||
compact_entries(known_entries + entries),
|
compact_entries(known_entries + entries),
|
||||||
)
|
)
|
||||||
|
if known_upstream:
|
||||||
|
snapshot["upstream"] = dict(sorted(known_upstream.items()))
|
||||||
|
return write_snapshot(output, snapshot)
|
||||||
|
|
||||||
|
|
||||||
def _iter_dat_contents(pack: Path):
|
def _iter_dat_contents(pack: Path):
|
||||||
@@ -301,9 +313,47 @@ def _iter_dat_contents(pack: Path):
|
|||||||
MAME_LISTXML_URL = (
|
MAME_LISTXML_URL = (
|
||||||
"https://github.com/mamedev/mame/releases/download/{tag}/{tag}lx.zip"
|
"https://github.com/mamedev/mame/releases/download/{tag}/{tag}lx.zip"
|
||||||
)
|
)
|
||||||
|
MAME_LATEST_API = "https://api.github.com/repos/mamedev/mame/releases/latest"
|
||||||
FBNEO_DATS_API = "https://api.github.com/repos/libretro/FBNeo/contents/dats"
|
FBNEO_DATS_API = "https://api.github.com/repos/libretro/FBNeo/contents/dats"
|
||||||
|
|
||||||
|
|
||||||
|
def _api_json(url: str) -> object:
|
||||||
|
request = urllib.request.Request(url, headers={"User-Agent": "retrobios"})
|
||||||
|
with urllib.request.urlopen(request, timeout=60) as response:
|
||||||
|
return json.load(response)
|
||||||
|
|
||||||
|
|
||||||
|
class NoReleaseError(ValueError):
|
||||||
|
def __init__(self) -> None:
|
||||||
|
super().__init__("the newest MAME release carries no tag")
|
||||||
|
|
||||||
|
|
||||||
|
def latest_mame_tag() -> str:
|
||||||
|
"""Tag of the newest MAME release, the one ``--fetch`` takes by default."""
|
||||||
|
payload = _api_json(MAME_LATEST_API)
|
||||||
|
tag = payload.get("tag_name") if isinstance(payload, dict) else None
|
||||||
|
if not tag:
|
||||||
|
raise NoReleaseError
|
||||||
|
return str(tag)
|
||||||
|
|
||||||
|
|
||||||
|
def fetch_tag(source: str, requested: str) -> str:
|
||||||
|
"""The tag a ``--fetch`` names: bare, the newest MAME release."""
|
||||||
|
if source == "mame" and requested == "latest":
|
||||||
|
tag = latest_mame_tag()
|
||||||
|
print(f" Derniere release MAME : {tag}")
|
||||||
|
return tag
|
||||||
|
return requested
|
||||||
|
|
||||||
|
|
||||||
|
def git_blob_sha(data: bytes) -> str:
|
||||||
|
"""The sha git gives a blob, so a cached file can be checked against a
|
||||||
|
repository listing without any other state."""
|
||||||
|
digest = hashlib.sha1(f"blob {len(data)}\0".encode())
|
||||||
|
digest.update(data)
|
||||||
|
return digest.hexdigest()
|
||||||
|
|
||||||
|
|
||||||
def _download(url: str, destination: Path) -> Path:
|
def _download(url: str, destination: Path) -> Path:
|
||||||
"""Fetch *url* to *destination* unless it is already there."""
|
"""Fetch *url* to *destination* unless it is already there."""
|
||||||
if destination.is_file():
|
if destination.is_file():
|
||||||
@@ -323,28 +373,38 @@ def _download(url: str, destination: Path) -> Path:
|
|||||||
return destination
|
return destination
|
||||||
|
|
||||||
|
|
||||||
def fetch_pack(source: str, tag: str, cache_dir: Path) -> Path:
|
def fetch_pack(
|
||||||
|
source: str, tag: str, cache_dir: Path
|
||||||
|
) -> tuple[Path, dict[str, str]]:
|
||||||
"""Download the upstream DAT for *source*.
|
"""Download the upstream DAT for *source*.
|
||||||
|
|
||||||
MAME ships its ``-listxml`` output as a release asset, and FBNeo keeps its
|
MAME ships its ``-listxml`` output as a release asset, and FBNeo keeps its
|
||||||
DATs in the repository, so neither needs a browser.
|
DATs in the repository, so neither needs a browser. The second value
|
||||||
|
names the upstream state fetched: empty for a release asset, whose tag
|
||||||
|
already says which version it is, and the git blob sha of every DAT for
|
||||||
|
FBNeo, whose files change under a constant header version.
|
||||||
"""
|
"""
|
||||||
if source == "mame":
|
if source == "mame":
|
||||||
return _download(
|
return (
|
||||||
MAME_LISTXML_URL.format(tag=tag), cache_dir / f"{tag}lx.zip"
|
_download(MAME_LISTXML_URL.format(tag=tag), cache_dir / f"{tag}lx.zip"),
|
||||||
|
{},
|
||||||
)
|
)
|
||||||
|
|
||||||
request = urllib.request.Request(
|
listing = _api_json(FBNEO_DATS_API)
|
||||||
FBNEO_DATS_API, headers={"User-Agent": "retrobios"}
|
|
||||||
)
|
|
||||||
with urllib.request.urlopen(request, timeout=60) as response:
|
|
||||||
listing = json.load(response)
|
|
||||||
target = cache_dir / "fbneo-dats"
|
target = cache_dir / "fbneo-dats"
|
||||||
target.mkdir(parents=True, exist_ok=True)
|
target.mkdir(parents=True, exist_ok=True)
|
||||||
|
blobs: dict[str, str] = {}
|
||||||
for item in listing:
|
for item in listing:
|
||||||
if item.get("type") == "file" and item["name"].lower().endswith(".dat"):
|
if item.get("type") != "file" or not item["name"].lower().endswith(".dat"):
|
||||||
_download(item["download_url"], target / item["name"])
|
continue
|
||||||
return target
|
destination = target / item["name"]
|
||||||
|
# The cache is addressed by content: a file already there is kept
|
||||||
|
# only while its blob sha is the one the repository lists.
|
||||||
|
if destination.is_file() and git_blob_sha(destination.read_bytes()) != item["sha"]:
|
||||||
|
destination.unlink()
|
||||||
|
_download(item["download_url"], destination)
|
||||||
|
blobs[item["name"]] = item["sha"]
|
||||||
|
return target, blobs
|
||||||
|
|
||||||
|
|
||||||
def _is_listxml(pack: Path) -> bool:
|
def _is_listxml(pack: Path) -> bool:
|
||||||
@@ -383,8 +443,9 @@ def main() -> int:
|
|||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--fetch",
|
"--fetch",
|
||||||
nargs="?",
|
nargs="?",
|
||||||
const="mame0289",
|
const="latest",
|
||||||
help="download upstream instead of using --pack (MAME tag, e.g. mame0289)",
|
help="download upstream instead of using --pack "
|
||||||
|
"(a MAME tag such as mame0289; bare, the newest release)",
|
||||||
)
|
)
|
||||||
parser.add_argument("--cache-dir", default=".cache/dats")
|
parser.add_argument("--cache-dir", default=".cache/dats")
|
||||||
parser.add_argument("--output", default="")
|
parser.add_argument("--output", default="")
|
||||||
@@ -396,8 +457,11 @@ def main() -> int:
|
|||||||
parser.add_argument("--dry-run", action="store_true")
|
parser.add_argument("--dry-run", action="store_true")
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
blobs: dict[str, str] = {}
|
||||||
if args.fetch:
|
if args.fetch:
|
||||||
pack = fetch_pack(args.source, args.fetch, Path(args.cache_dir))
|
pack, blobs = fetch_pack(
|
||||||
|
args.source, fetch_tag(args.source, args.fetch), Path(args.cache_dir)
|
||||||
|
)
|
||||||
print(f" Recupere {pack}")
|
print(f" Recupere {pack}")
|
||||||
elif args.pack:
|
elif args.pack:
|
||||||
pack = Path(args.pack)
|
pack = Path(args.pack)
|
||||||
@@ -457,7 +521,7 @@ def main() -> int:
|
|||||||
if args.dry_run:
|
if args.dry_run:
|
||||||
return 0
|
return 0
|
||||||
|
|
||||||
changed = merge_snapshot(output, args.source, dats, entries)
|
changed = merge_snapshot(output, args.source, dats, entries, blobs)
|
||||||
print(f"{'Wrote' if changed else 'Unchanged'} {output}")
|
print(f"{'Wrote' if changed else 'Unchanged'} {output}")
|
||||||
return 0
|
return 0
|
||||||
|
|
||||||
|
|||||||
@@ -29,6 +29,7 @@ from romset_recipes import ( # noqa: E402
|
|||||||
from scripts.scraper.romset_dat_importer import ( # noqa: E402
|
from scripts.scraper.romset_dat_importer import ( # noqa: E402
|
||||||
compact_entries,
|
compact_entries,
|
||||||
listxml_entries,
|
listxml_entries,
|
||||||
|
git_blob_sha,
|
||||||
merge_snapshot,
|
merge_snapshot,
|
||||||
recipe_entries,
|
recipe_entries,
|
||||||
)
|
)
|
||||||
@@ -384,6 +385,36 @@ class ListXmlRecipes(unittest.TestCase):
|
|||||||
[m["name"] for m in entries[0]["members"]], ["own.rom", "shared.rom"]
|
[m["name"] for m in entries[0]["members"]], ["own.rom", "shared.rom"]
|
||||||
)
|
)
|
||||||
|
|
||||||
|
def test_upstream_blobs_are_recorded_and_replaced(self):
|
||||||
|
"""The snapshot names the DAT blobs it was read from, so a later
|
||||||
|
check can tell whether upstream moved without importing again."""
|
||||||
|
entry = {
|
||||||
|
"dat": "FBNeo - X",
|
||||||
|
"name": "set.zip",
|
||||||
|
"set": "set",
|
||||||
|
"description": "",
|
||||||
|
"members": [{"name": "a", "crc32": "1"}],
|
||||||
|
}
|
||||||
|
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
|
||||||
|
output = str(Path(directory) / "fbneo.json")
|
||||||
|
merge_snapshot(output, "fbneo", {"FBNeo - X": "1.0"}, [entry], {"X.dat": "aaa"})
|
||||||
|
with open(output, encoding="utf-8") as handle:
|
||||||
|
self.assertEqual(json.load(handle)["upstream"], {"blobs": {"X.dat": "aaa"}})
|
||||||
|
merge_snapshot(output, "fbneo", {"FBNeo - X": "1.0"}, [entry], {"X.dat": "bbb"})
|
||||||
|
with open(output, encoding="utf-8") as handle:
|
||||||
|
snapshot = json.load(handle)
|
||||||
|
self.assertEqual(snapshot["upstream"], {"blobs": {"X.dat": "bbb"}})
|
||||||
|
self.assertEqual(len(snapshot["entries"]), 1)
|
||||||
|
# An import that names no blobs (a local --pack) keeps the last ones.
|
||||||
|
merge_snapshot(output, "fbneo", {"FBNeo - X": "1.0"}, [entry])
|
||||||
|
with open(output, encoding="utf-8") as handle:
|
||||||
|
self.assertEqual(json.load(handle)["upstream"], {"blobs": {"X.dat": "bbb"}})
|
||||||
|
|
||||||
|
def test_git_blob_sha_matches_git(self):
|
||||||
|
# The values git hash-object gives for the empty blob and for "hi".
|
||||||
|
self.assertEqual(git_blob_sha(b""), "e69de29bb2d1d6434b8b29ae775ad8c2e48c5391")
|
||||||
|
self.assertEqual(git_blob_sha(b"hi"), "32f95c0d1244a78b2be1bab8de17906fabb2c4a8")
|
||||||
|
|
||||||
def test_versions_accumulate_instead_of_replacing_each_other(self):
|
def test_versions_accumulate_instead_of_replacing_each_other(self):
|
||||||
"""A platform pins the archive of whichever version it was built on."""
|
"""A platform pins the archive of whichever version it was built on."""
|
||||||
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
|
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
|
||||||
|
|||||||
Reference in new issue
Block a user