mirror of
https://github.com/Abdess/retroarch_system.git
synced 2026-10-10 21:43:23 -05:00
feat: identify romsets from version recipes
A profile's contents: block and a DAT both state, per set, the members one emulator version expects. Torrentzip makes archive bytes a function of that list alone, so a recipe plus the roms reproduces the archive exactly. MAME ships its -listxml as a release asset and FBNeo keeps its dats in its repository, so neither needs a browser. The importer streams the 311 MB document with a sliding window, resolves romof parents in two passes, drops undumped members, and accumulates versions instead of replacing them. Identical recipes shared across versions are stored once with dats listing every version that agrees: 22 MAME versions give 23679 entries for 1726 distinct recipes, 18 MB down to 2.2 MB. Snapshots live in recipes/ because load_provenance_snapshots reads every provenance/*.json as a dump catalogue, and a recipe is not one. 1113 archives now reproduce byte for byte, against 175 from profiles alone, and spec128.zip is rebuilt from roms already held.
This commit is contained in:
1 parent
e8ee8b0954
commit
33a9934a11
5 files changed
+94206
No files matched your search
+8581
File diff suppressed because it is too large.
Load diff
+84217
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,439 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Identify and reconstruct arcade romset archives from their recipes.
|
||||
|
||||
A profile's ``contents:`` block is a recipe: the member names and CRC32s one
|
||||
emulator version expects inside an archive. TorrentZip makes archive bytes a
|
||||
function of that recipe alone, so a recipe plus the ROM bytes reproduces the
|
||||
archive exactly.
|
||||
|
||||
Two questions follow, and this module answers both:
|
||||
|
||||
``--identify``
|
||||
Which version does an archive the collection holds correspond to? The
|
||||
recipe that rebuilds it byte for byte names it.
|
||||
|
||||
``--missing``
|
||||
A platform pins a container MD5 the collection does not have. If some
|
||||
recipe plus ROMs already present reproduces that MD5, the archive is
|
||||
constructible rather than absent.
|
||||
|
||||
Reconstruction is only possible when the pinned archive is itself TorrentZip.
|
||||
An archive whose bytes carry metadata unrelated to its contents cannot be
|
||||
derived from ROMs by anyone, and is reported as such rather than guessed at.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import sys
|
||||
import zipfile
|
||||
from collections import defaultdict
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
||||
|
||||
from common import (
|
||||
load_database,
|
||||
load_emulator_profiles,
|
||||
load_platform_config,
|
||||
list_registered_platforms,
|
||||
)
|
||||
from torrentzip import build_torrentzip, is_torrentzip
|
||||
|
||||
Recipe = list[tuple[str, str]]
|
||||
|
||||
|
||||
def load_dat_recipes(path: str | Path) -> dict[str, dict[str, Recipe]]:
|
||||
"""Recipes imported from MAME or FBNeo DATs.
|
||||
|
||||
A DAT documents every set of one emulator version, where profiles only
|
||||
document the few hundred the collection curates by hand. Labels are
|
||||
prefixed so an identification says which DAT decided it.
|
||||
"""
|
||||
target = Path(path)
|
||||
if target.is_dir():
|
||||
merged: dict[str, dict[str, Recipe]] = {}
|
||||
for snapshot_file in sorted(target.glob("*.json")):
|
||||
merged = merge_recipes(merged, load_dat_recipes(snapshot_file))
|
||||
return merged
|
||||
snapshot_path = target
|
||||
if not snapshot_path.is_file():
|
||||
return {}
|
||||
with snapshot_path.open(encoding="utf-8") as handle:
|
||||
snapshot = json.load(handle)
|
||||
source = snapshot.get("source", "dat")
|
||||
# A recipe shared by several versions is stored once under the earliest.
|
||||
# The label therefore reads "unchanged since", not "is this version".
|
||||
recipes: dict[str, dict[str, Recipe]] = defaultdict(dict)
|
||||
for entry in snapshot.get("entries", []):
|
||||
archive = entry.get("name", "")
|
||||
members = entry.get("members", []) or []
|
||||
recipe = [
|
||||
(str(member["name"]), str(member.get("crc32") or "").lower())
|
||||
for member in members
|
||||
if member.get("name") is not None
|
||||
]
|
||||
if not archive or not recipe or not all(crc for _name, crc in recipe):
|
||||
continue
|
||||
label = f"{source}:{entry.get('dat', 'unnamed')}"
|
||||
recipes[archive][label] = recipe
|
||||
return dict(recipes)
|
||||
|
||||
|
||||
def merge_recipes(*sources: dict[str, dict[str, Recipe]]) -> dict[str, dict[str, Recipe]]:
|
||||
"""Combine recipe sets, keeping every label distinct."""
|
||||
merged: dict[str, dict[str, Recipe]] = defaultdict(dict)
|
||||
for source in sources:
|
||||
for archive, per_label in source.items():
|
||||
merged[archive].update(per_label)
|
||||
return dict(merged)
|
||||
|
||||
|
||||
def collect_recipes(profiles: dict) -> dict[str, dict[str, Recipe]]:
|
||||
"""Map each archive name to one recipe per profile that documents it.
|
||||
|
||||
A recipe is only usable when every member carries a CRC32: the pool is
|
||||
addressed by content, so a member without one cannot be located.
|
||||
"""
|
||||
recipes: dict[str, dict[str, Recipe]] = defaultdict(dict)
|
||||
for profile_name, profile in sorted(profiles.items()):
|
||||
for entry in profile.get("files", []) or []:
|
||||
contents = entry.get("contents")
|
||||
archive = entry.get("name", "")
|
||||
if not contents or not archive:
|
||||
continue
|
||||
recipe = [
|
||||
(str(member["name"]), str(member.get("crc32") or "").lower())
|
||||
for member in contents
|
||||
if member.get("name") is not None
|
||||
]
|
||||
if recipe and all(crc for _name, crc in recipe):
|
||||
recipes[archive][profile_name] = recipe
|
||||
return dict(recipes)
|
||||
|
||||
|
||||
class AtomPool:
|
||||
"""ROM bytes in the collection, addressed by CRC32.
|
||||
|
||||
Members are located once and read on demand: arcade sample sets reach
|
||||
hundreds of megabytes and a whole-pool preload buys nothing.
|
||||
"""
|
||||
|
||||
def __init__(self, bios_dir: Path):
|
||||
self._index: dict[str, tuple[Path, str]] = {}
|
||||
# A recipe is built once: identification and reconstruction ask for the
|
||||
# same archives, and every build re-reads and re-deflates its ROMs.
|
||||
self._built: dict[tuple[tuple[str, str], ...], bytes | None] = {}
|
||||
self.unreadable: list[str] = []
|
||||
for archive in sorted(bios_dir.rglob("*.zip")):
|
||||
try:
|
||||
with zipfile.ZipFile(archive) as handle:
|
||||
for info in handle.infolist():
|
||||
if not info.is_dir():
|
||||
self._index.setdefault(
|
||||
f"{info.CRC:08x}", (archive, info.filename)
|
||||
)
|
||||
except (OSError, zipfile.BadZipFile) as exc:
|
||||
self.unreadable.append(f"{archive}: {exc}")
|
||||
|
||||
def __len__(self) -> int:
|
||||
return len(self._index)
|
||||
|
||||
def __contains__(self, crc32: str) -> bool:
|
||||
return crc32.lower() in self._index
|
||||
|
||||
def read(self, crc32: str) -> bytes:
|
||||
archive, member = self._index[crc32.lower()]
|
||||
with zipfile.ZipFile(archive) as handle:
|
||||
return handle.read(member)
|
||||
|
||||
def build(self, recipe: Recipe) -> bytes | None:
|
||||
"""Return the TorrentZip bytes for *recipe*, or None if incomplete."""
|
||||
key = tuple(recipe)
|
||||
if key not in self._built:
|
||||
if all(crc in self for _name, crc in recipe):
|
||||
self._built[key] = build_torrentzip(
|
||||
[(name, self.read(crc)) for name, crc in recipe]
|
||||
)
|
||||
else:
|
||||
self._built[key] = None
|
||||
return self._built[key]
|
||||
|
||||
|
||||
def identify_archive(
|
||||
path: Path, recipes: dict[str, Recipe], pool: AtomPool
|
||||
) -> str | None:
|
||||
"""Return the recipe label that reproduces *path* byte for byte."""
|
||||
original = path.read_bytes()
|
||||
for label, recipe in sorted(recipes.items()):
|
||||
blob = pool.build(recipe)
|
||||
if blob is not None and blob == original:
|
||||
return label
|
||||
return None
|
||||
|
||||
|
||||
def reconstruct(
|
||||
target_md5: str, recipes: dict[str, Recipe], pool: AtomPool
|
||||
) -> tuple[str, bytes] | None:
|
||||
"""Return the (label, bytes) whose TorrentZip build matches *target_md5*."""
|
||||
target = target_md5.lower()
|
||||
for label, recipe in sorted(recipes.items()):
|
||||
blob = pool.build(recipe)
|
||||
if blob is not None and hashlib.md5(blob).hexdigest() == target:
|
||||
return label, blob
|
||||
return None
|
||||
|
||||
|
||||
def platform_targets(platforms_dir: str) -> dict[str, set[str]]:
|
||||
"""Container MD5s the platforms pin, per archive name.
|
||||
|
||||
``zipped_file`` entries pin an inner ROM instead of the container, so
|
||||
their MD5 is not an archive hash and is skipped.
|
||||
"""
|
||||
targets: dict[str, set[str]] = defaultdict(set)
|
||||
for name in list_registered_platforms(platforms_dir, include_archived=True):
|
||||
config = load_platform_config(name, platforms_dir)
|
||||
for system in config.get("systems", {}).values():
|
||||
for entry in system.get("files", []) or []:
|
||||
archive = entry.get("name", "")
|
||||
if not archive.endswith(".zip") or entry.get("zipped_file"):
|
||||
continue
|
||||
for value in str(entry.get("md5") or "").split(","):
|
||||
value = value.strip().lower()
|
||||
if len(value) == 32:
|
||||
targets[archive].add(value)
|
||||
return dict(targets)
|
||||
|
||||
|
||||
def _is_archive(path: Path) -> bool:
|
||||
"""Whether *path* is a readable ZIP.
|
||||
|
||||
A name index entry can be an alias: ``d2fdc.zip`` names a loose Apple II
|
||||
ROM, not an archive. Recipes only describe archives.
|
||||
"""
|
||||
try:
|
||||
with zipfile.ZipFile(path):
|
||||
return True
|
||||
except (OSError, zipfile.BadZipFile):
|
||||
return False
|
||||
|
||||
|
||||
def _local_paths(archive: str, db: dict) -> list[Path]:
|
||||
matches = db["indexes"]["by_name"].get(archive) or []
|
||||
if isinstance(matches, str):
|
||||
matches = [matches]
|
||||
paths = [Path(db["files"][sha1]["path"]) for sha1 in matches if sha1 in db["files"]]
|
||||
return [path for path in paths if path.is_file() and _is_archive(path)]
|
||||
|
||||
|
||||
def identify_report(
|
||||
recipes: dict[str, dict[str, Recipe]], pool: AtomPool, db: dict
|
||||
) -> dict:
|
||||
"""Label every archive the collection holds with the version it matches."""
|
||||
identified: list[dict] = []
|
||||
unidentified: list[dict] = []
|
||||
for archive, per_profile in sorted(recipes.items()):
|
||||
for path in _local_paths(archive, db):
|
||||
label = identify_archive(path, per_profile, pool)
|
||||
record = {
|
||||
"archive": archive,
|
||||
"path": str(path),
|
||||
"torrentzip": is_torrentzip(path),
|
||||
}
|
||||
if label:
|
||||
record["recipe"] = label
|
||||
identified.append(record)
|
||||
else:
|
||||
record["candidates"] = sorted(per_profile)
|
||||
unidentified.append(record)
|
||||
return {"identified": identified, "unidentified": unidentified}
|
||||
|
||||
|
||||
def missing_report(
|
||||
recipes: dict[str, dict[str, Recipe]],
|
||||
pool: AtomPool,
|
||||
db: dict,
|
||||
platforms_dir: str,
|
||||
) -> dict:
|
||||
"""Pinned archives the collection lacks, split by why they are absent."""
|
||||
have = set(db["indexes"]["by_md5"])
|
||||
constructible: list[dict] = []
|
||||
no_recipe: list[dict] = []
|
||||
unreproducible: list[dict] = []
|
||||
for archive, pinned in sorted(platform_targets(platforms_dir).items()):
|
||||
per_profile = recipes.get(archive, {})
|
||||
for target in sorted(pinned - have):
|
||||
if not per_profile:
|
||||
no_recipe.append({"archive": archive, "md5": target})
|
||||
continue
|
||||
found = reconstruct(target, per_profile, pool)
|
||||
if found:
|
||||
label, blob = found
|
||||
constructible.append(
|
||||
{
|
||||
"archive": archive,
|
||||
"md5": target,
|
||||
"recipe": label,
|
||||
"size": len(blob),
|
||||
}
|
||||
)
|
||||
else:
|
||||
usable = [
|
||||
label
|
||||
for label, recipe in per_profile.items()
|
||||
if all(crc in pool for _name, crc in recipe)
|
||||
]
|
||||
unreproducible.append(
|
||||
{
|
||||
"archive": archive,
|
||||
"md5": target,
|
||||
"recipes_tried": sorted(usable),
|
||||
"reason": (
|
||||
"no recipe reproduces this MD5; the pinned archive is "
|
||||
"not TorrentZip or follows a recipe not documented here"
|
||||
if usable
|
||||
else "ROMs for every documented recipe are missing"
|
||||
),
|
||||
}
|
||||
)
|
||||
return {
|
||||
"constructible": constructible,
|
||||
"unreproducible": unreproducible,
|
||||
"no_recipe": no_recipe,
|
||||
}
|
||||
|
||||
|
||||
def write_reconstructions(
|
||||
entries: list[dict],
|
||||
recipes: dict[str, dict[str, Recipe]],
|
||||
pool: AtomPool,
|
||||
bios_dir: Path,
|
||||
db: dict,
|
||||
) -> list[str]:
|
||||
"""Write each constructible archive beside its set, as a variant."""
|
||||
written: list[str] = []
|
||||
for entry in entries:
|
||||
archive = entry["archive"]
|
||||
recipe = recipes[archive][entry["recipe"]]
|
||||
blob = pool.build(recipe)
|
||||
if blob is None:
|
||||
continue
|
||||
siblings = _local_paths(archive, db)
|
||||
parent = siblings[0].parent if siblings else bios_dir
|
||||
destination = parent / ".variants" / f"{archive}.{entry['md5'][:8]}"
|
||||
destination.parent.mkdir(parents=True, exist_ok=True)
|
||||
destination.write_bytes(blob)
|
||||
written.append(str(destination))
|
||||
return written
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--identify", action="store_true", help="label local archives")
|
||||
parser.add_argument(
|
||||
"--missing", action="store_true", help="attempt pinned archives we lack"
|
||||
)
|
||||
parser.add_argument(
|
||||
"--write", action="store_true", help="write reconstructions to .variants/"
|
||||
)
|
||||
parser.add_argument("--json", action="store_true", help="machine-readable output")
|
||||
parser.add_argument("--bios-dir", default="bios")
|
||||
parser.add_argument("--platforms-dir", default="platforms")
|
||||
parser.add_argument("--emulators-dir", default="emulators")
|
||||
parser.add_argument("--db", default="database.json")
|
||||
parser.add_argument(
|
||||
"--dat-recipes",
|
||||
default="recipes",
|
||||
help="recipe snapshot file, or a directory of them",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
if not args.identify and not args.missing:
|
||||
args.identify = args.missing = True
|
||||
|
||||
db = load_database(args.db)
|
||||
profiles = load_emulator_profiles(args.emulators_dir, skip_aliases=False)
|
||||
profile_recipes = collect_recipes(profiles)
|
||||
dat_recipes = load_dat_recipes(args.dat_recipes)
|
||||
recipes = merge_recipes(profile_recipes, dat_recipes)
|
||||
pool = AtomPool(Path(args.bios_dir))
|
||||
|
||||
result: dict = {
|
||||
"recipes": sum(len(v) for v in recipes.values()),
|
||||
"profile_recipes": sum(len(v) for v in profile_recipes.values()),
|
||||
"dat_recipes": sum(len(v) for v in dat_recipes.values()),
|
||||
"archives_with_recipes": len(recipes),
|
||||
"atoms": len(pool),
|
||||
}
|
||||
if pool.unreadable:
|
||||
result["unreadable_archives"] = pool.unreadable
|
||||
if args.identify:
|
||||
result["identify"] = identify_report(recipes, pool, db)
|
||||
if args.missing:
|
||||
result["missing"] = missing_report(recipes, pool, db, args.platforms_dir)
|
||||
if args.write:
|
||||
result["written"] = write_reconstructions(
|
||||
result["missing"]["constructible"], recipes, pool,
|
||||
Path(args.bios_dir), db,
|
||||
)
|
||||
|
||||
if args.json:
|
||||
print(json.dumps(result, indent=2, sort_keys=True))
|
||||
return 0
|
||||
|
||||
print(
|
||||
f"{result['archives_with_recipes']} archives documentees par "
|
||||
f"{result['recipes']} recettes "
|
||||
f"({result['profile_recipes']} de profils, {result['dat_recipes']} de DAT), "
|
||||
f"{result['atoms']} atomes ROM disponibles"
|
||||
)
|
||||
if not result["dat_recipes"]:
|
||||
print(
|
||||
" Aucune recette DAT: importer avec "
|
||||
"`python -m scripts.scraper.romset_dat_importer --source mame --fetch mame0289`"
|
||||
)
|
||||
for problem in result.get("unreadable_archives", []):
|
||||
print(f" UNREADABLE: {problem}")
|
||||
|
||||
if args.identify:
|
||||
ident = result["identify"]
|
||||
print(
|
||||
f"\nIdentification: {len(ident['identified'])} archives reproduites "
|
||||
f"par une recette, {len(ident['unidentified'])} non identifiees"
|
||||
)
|
||||
print(
|
||||
" L'etiquette est la plus ancienne version produisant ces octets: "
|
||||
"une recette inchangee depuis 0.190 est comptee la, pas a sa version "
|
||||
"d'origine."
|
||||
)
|
||||
by_recipe: dict[str, int] = defaultdict(int)
|
||||
for record in ident["identified"]:
|
||||
by_recipe[record["recipe"]] += 1
|
||||
for label, count in sorted(by_recipe.items(), key=lambda kv: (-kv[1], kv[0])):
|
||||
print(f" {label}: {count}")
|
||||
for record in ident["unidentified"][:10]:
|
||||
print(f" UNIDENTIFIED: {record['path']}")
|
||||
|
||||
if args.missing:
|
||||
gap = result["missing"]
|
||||
print(
|
||||
f"\nArchives epinglees absentes: "
|
||||
f"{len(gap['constructible'])} reconstructibles, "
|
||||
f"{len(gap['unreproducible'])} irreproductibles, "
|
||||
f"{len(gap['no_recipe'])} sans recette"
|
||||
)
|
||||
for record in gap["constructible"]:
|
||||
print(
|
||||
f" BUILDABLE: {record['archive']} md5={record['md5'][:8]} "
|
||||
f"via {record['recipe']} ({record['size']} octets)"
|
||||
)
|
||||
for path in result.get("written", []):
|
||||
print(f" WROTE: {path}")
|
||||
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,466 @@
|
||||
"""Import romset recipes from a MAME or FBNeo DAT.
|
||||
|
||||
A DAT states, per set, the members an emulator version expects: name, size
|
||||
and CRC32. That is a recipe, and TorrentZip turns a recipe plus the ROM bytes
|
||||
into the exact archive. Profiles document a few hundred sets by hand; a DAT
|
||||
documents every set of one version at once.
|
||||
|
||||
MAME ships its ``-listxml`` output as a release asset and FBNeo keeps its DATs
|
||||
in its repository, so both are fetched without a browser:
|
||||
|
||||
python -m scripts.scraper.romset_dat_importer --source mame --fetch mame0289
|
||||
python -m scripts.scraper.romset_dat_importer --source fbneo --fetch
|
||||
python -m scripts.scraper.romset_dat_importer --source mame --pack local.dat
|
||||
|
||||
Only sets the collection has a reason to know are kept, so the committed
|
||||
snapshot stays small: the archive names that emulator profiles or platform
|
||||
BIOS lists reference. ``--all`` keeps everything.
|
||||
|
||||
Members without a CRC32 are undumped. A real romset does not carry them, so
|
||||
they are dropped from the recipe rather than disqualifying the whole set.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import hashlib
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import sys
|
||||
import urllib.request
|
||||
import zipfile
|
||||
from datetime import datetime, timezone
|
||||
from pathlib import Path
|
||||
|
||||
from ..common import (
|
||||
list_registered_platforms,
|
||||
load_emulator_profiles,
|
||||
load_platform_config,
|
||||
write_provenance_snapshot,
|
||||
)
|
||||
from .logiqx_parser import LogiqxDat, parse_logiqx
|
||||
|
||||
MAX_MEMBER_SIZE = 200 * 1024 * 1024
|
||||
|
||||
SOURCES = ("mame", "fbneo")
|
||||
DEFAULT_OUTPUT = "recipes/{source}.json"
|
||||
|
||||
|
||||
def relevant_archives(emulators_dir: str, platforms_dir: str) -> set[str]:
|
||||
"""Archive names any profile or platform refers to."""
|
||||
names: set[str] = set()
|
||||
for _profile_name, profile in load_emulator_profiles(
|
||||
emulators_dir, skip_aliases=False
|
||||
).items():
|
||||
for entry in profile.get("files", []) or []:
|
||||
name = entry.get("name", "")
|
||||
if name.endswith(".zip"):
|
||||
names.add(name)
|
||||
archive = entry.get("archive", "")
|
||||
if archive:
|
||||
names.add(archive)
|
||||
for platform_name in list_registered_platforms(platforms_dir, include_archived=True):
|
||||
config = load_platform_config(platform_name, platforms_dir)
|
||||
for system in config.get("systems", {}).values():
|
||||
for entry in system.get("files", []) or []:
|
||||
name = entry.get("name", "")
|
||||
if name.endswith(".zip"):
|
||||
names.add(name)
|
||||
return names
|
||||
|
||||
|
||||
def recipe_entries(
|
||||
dat: LogiqxDat, keep: set[str] | None = None
|
||||
) -> list[dict]:
|
||||
"""Group a parsed DAT into one recipe per set."""
|
||||
by_set: dict[str, list] = {}
|
||||
for rom in dat.roms:
|
||||
if not rom.crc32:
|
||||
continue
|
||||
by_set.setdefault(rom.game, []).append(rom)
|
||||
|
||||
entries: list[dict] = []
|
||||
for set_name, roms in sorted(by_set.items()):
|
||||
archive = f"{set_name}.zip"
|
||||
if keep is not None and archive not in keep:
|
||||
continue
|
||||
entries.append(
|
||||
{
|
||||
"dat": dat.name or "unnamed",
|
||||
"name": archive,
|
||||
"set": set_name,
|
||||
"description": roms[0].description,
|
||||
"members": sorted(
|
||||
(
|
||||
{
|
||||
"name": rom.name,
|
||||
"size": rom.size,
|
||||
"crc32": rom.crc32,
|
||||
"sha1": rom.sha1,
|
||||
}
|
||||
for rom in roms
|
||||
),
|
||||
key=lambda member: member["name"],
|
||||
),
|
||||
}
|
||||
)
|
||||
return entries
|
||||
|
||||
|
||||
_MACHINE = re.compile(rb'<machine name="([^"]+)"([^>]*)>')
|
||||
_ROM = re.compile(
|
||||
rb'<rom name="([^"]+)"(?=[^>]*\bcrc="([0-9a-fA-F]+)")[^>]*>'
|
||||
)
|
||||
_ROM_ATTR = re.compile(rb'(\w+)="([^"]*)"')
|
||||
_ROMOF = re.compile(r'romof="([^"]+)"')
|
||||
LISTXML_CHUNK = 8 * 1024 * 1024
|
||||
|
||||
|
||||
def _machine_roms(blob: bytes) -> list[dict]:
|
||||
"""ROM members of one machine element, undumped ones dropped."""
|
||||
members: list[dict] = []
|
||||
for match in re.finditer(rb"<rom\b([^>]*)>", blob):
|
||||
attrs = {
|
||||
key.decode(): value.decode()
|
||||
for key, value in _ROM_ATTR.findall(match.group(1))
|
||||
}
|
||||
crc = attrs.get("crc", "").lower()
|
||||
name = attrs.get("name", "")
|
||||
if not name or not crc:
|
||||
continue
|
||||
try:
|
||||
size = int(attrs.get("size", "0"))
|
||||
except ValueError:
|
||||
size = 0
|
||||
members.append(
|
||||
{
|
||||
"name": name,
|
||||
"size": size,
|
||||
"crc32": crc,
|
||||
"sha1": attrs.get("sha1", "").lower(),
|
||||
}
|
||||
)
|
||||
return members
|
||||
|
||||
|
||||
def _scan_listxml(stream, selector) -> dict[str, tuple[str | None, list[dict]]]:
|
||||
"""Stream a MAME -listxml document, keeping the machines *selector* wants.
|
||||
|
||||
The document is 311 MB for MAME 0.289, so it is never held whole: the
|
||||
reader keeps a sliding window just large enough to close the element it
|
||||
is in the middle of.
|
||||
"""
|
||||
found: dict[str, tuple[str | None, list[dict]]] = {}
|
||||
buffer = b""
|
||||
while chunk := stream.read(LISTXML_CHUNK):
|
||||
buffer += chunk
|
||||
position = 0
|
||||
while True:
|
||||
match = _MACHINE.search(buffer, position)
|
||||
if not match:
|
||||
break
|
||||
end = buffer.find(b"</machine>", match.end())
|
||||
if end == -1:
|
||||
break
|
||||
name = match.group(1).decode()
|
||||
if selector(name):
|
||||
parent = _ROMOF.search(match.group(2).decode())
|
||||
found[name] = (
|
||||
parent.group(1) if parent else None,
|
||||
_machine_roms(buffer[match.end() : end]),
|
||||
)
|
||||
position = end + len(b"</machine>")
|
||||
buffer = buffer[position:] if position else buffer[-(2 * LISTXML_CHUNK) :]
|
||||
return found
|
||||
|
||||
|
||||
def listxml_entries(
|
||||
open_stream, label: str, keep: set[str] | None
|
||||
) -> list[dict]:
|
||||
"""Recipes from a MAME -listxml document, parents resolved.
|
||||
|
||||
A non-merged archive carries its parent set's ROMs as well as its own, so
|
||||
a machine with ``romof`` is read as the union, its own members winning a
|
||||
name collision. Two passes: the parents a set needs are only known once
|
||||
that set has been seen.
|
||||
"""
|
||||
with open_stream() as stream:
|
||||
wanted = _scan_listxml(
|
||||
stream, lambda name: keep is None or f"{name}.zip" in keep
|
||||
)
|
||||
requested = set(wanted)
|
||||
parents = {
|
||||
parent for parent, _roms in wanted.values() if parent and parent not in wanted
|
||||
}
|
||||
if parents:
|
||||
with open_stream() as stream:
|
||||
for name, value in _scan_listxml(stream, parents.__contains__).items():
|
||||
wanted.setdefault(name, value)
|
||||
|
||||
entries: list[dict] = []
|
||||
for name, (parent, roms) in sorted(wanted.items()):
|
||||
# A parent read only to complete a child is not itself an entry.
|
||||
if name not in requested:
|
||||
continue
|
||||
members = {member["name"]: member for member in wanted.get(parent, (None, []))[1]} if parent else {}
|
||||
members.update({member["name"]: member for member in roms})
|
||||
if not members:
|
||||
continue
|
||||
entries.append(
|
||||
{
|
||||
"dat": label,
|
||||
"name": f"{name}.zip",
|
||||
"set": name,
|
||||
"description": f"parent {parent}" if parent else name,
|
||||
"members": sorted(members.values(), key=lambda m: m["name"]),
|
||||
}
|
||||
)
|
||||
return entries
|
||||
|
||||
|
||||
def _recipe_key(entry: dict) -> tuple[str, str]:
|
||||
"""Identity of a recipe: which archive, and exactly which members."""
|
||||
members = json.dumps(entry.get("members", []), sort_keys=True)
|
||||
return entry.get("name", ""), hashlib.sha1(members.encode()).hexdigest()
|
||||
|
||||
|
||||
def compact_entries(entries: list[dict]) -> list[dict]:
|
||||
"""Collapse identical recipes shared by several versions.
|
||||
|
||||
Most sets do not change between MAME releases: 22 imported versions give
|
||||
23,679 entries but only 1,726 distinct recipes. Each is stored once, with
|
||||
``dats`` listing every version that agrees and ``dat`` naming the earliest,
|
||||
so a match reads as "unchanged since that version" rather than "is that
|
||||
version".
|
||||
"""
|
||||
grouped: dict[tuple[str, str], dict] = {}
|
||||
for entry in entries:
|
||||
key = _recipe_key(entry)
|
||||
existing = grouped.get(key)
|
||||
labels = entry.get("dats") or ([entry["dat"]] if entry.get("dat") else [])
|
||||
if existing is None:
|
||||
merged = dict(entry)
|
||||
merged["dats"] = sorted(set(labels))
|
||||
grouped[key] = merged
|
||||
else:
|
||||
existing["dats"] = sorted(set(existing["dats"]) | set(labels))
|
||||
for entry in grouped.values():
|
||||
entry["dat"] = entry["dats"][0]
|
||||
return sorted(grouped.values(), key=lambda e: (e["name"], e["dat"]))
|
||||
|
||||
|
||||
def merge_snapshot(
|
||||
output: str, source: str, dats: dict[str, str], entries: list[dict]
|
||||
) -> bool:
|
||||
"""Accumulate recipes across versions instead of replacing them.
|
||||
|
||||
A platform pins the archive of whichever version its list was built
|
||||
against, so the snapshot is a growing library of versions, not a picture
|
||||
of the newest one.
|
||||
"""
|
||||
path = Path(output)
|
||||
known_entries: list[dict] = []
|
||||
known_dats: dict[str, str] = {}
|
||||
if path.is_file():
|
||||
with path.open(encoding="utf-8") as handle:
|
||||
existing = json.load(handle)
|
||||
known_entries = list(existing.get("entries", []))
|
||||
known_dats = dict(existing.get("dats", {}))
|
||||
|
||||
known_dats.update(dats)
|
||||
return write_provenance_snapshot(
|
||||
output,
|
||||
source,
|
||||
datetime.now(timezone.utc).strftime("%Y-%m-%d"),
|
||||
known_dats,
|
||||
compact_entries(known_entries + entries),
|
||||
)
|
||||
|
||||
|
||||
def _iter_dat_contents(pack: Path):
|
||||
"""Yield (member name, content bytes) for DAT files in *pack*."""
|
||||
if pack.is_dir():
|
||||
for path in sorted(pack.rglob("*")):
|
||||
if path.suffix.lower() in (".dat", ".xml") and path.is_file():
|
||||
yield path.name, path.read_bytes()
|
||||
return
|
||||
if pack.suffix.lower() == ".zip":
|
||||
with zipfile.ZipFile(pack) as archive:
|
||||
for info in sorted(archive.infolist(), key=lambda i: i.filename):
|
||||
if info.is_dir() or info.file_size > MAX_MEMBER_SIZE:
|
||||
continue
|
||||
if Path(info.filename).suffix.lower() not in (".dat", ".xml"):
|
||||
continue
|
||||
yield info.filename, archive.read(info)
|
||||
return
|
||||
yield pack.name, pack.read_bytes()
|
||||
|
||||
|
||||
MAME_LISTXML_URL = (
|
||||
"https://github.com/mamedev/mame/releases/download/{tag}/{tag}lx.zip"
|
||||
)
|
||||
FBNEO_DATS_API = "https://api.github.com/repos/libretro/FBNeo/contents/dats"
|
||||
|
||||
|
||||
def _download(url: str, destination: Path) -> Path:
|
||||
"""Fetch *url* to *destination* unless it is already there."""
|
||||
if destination.is_file():
|
||||
return destination
|
||||
destination.parent.mkdir(parents=True, exist_ok=True)
|
||||
request = urllib.request.Request(url, headers={"User-Agent": "retrobios"})
|
||||
scratch = destination.with_suffix(destination.suffix + f".{os.getpid()}.part")
|
||||
try:
|
||||
with urllib.request.urlopen(request, timeout=300) as response, scratch.open(
|
||||
"wb"
|
||||
) as handle:
|
||||
shutil.copyfileobj(response, handle)
|
||||
scratch.replace(destination)
|
||||
finally:
|
||||
if scratch.exists():
|
||||
scratch.unlink()
|
||||
return destination
|
||||
|
||||
|
||||
def fetch_pack(source: str, tag: str, cache_dir: Path) -> Path:
|
||||
"""Download the upstream DAT for *source*.
|
||||
|
||||
MAME ships its ``-listxml`` output as a release asset, and FBNeo keeps its
|
||||
DATs in the repository, so neither needs a browser.
|
||||
"""
|
||||
if source == "mame":
|
||||
return _download(
|
||||
MAME_LISTXML_URL.format(tag=tag), cache_dir / f"{tag}lx.zip"
|
||||
)
|
||||
|
||||
request = urllib.request.Request(
|
||||
FBNEO_DATS_API, headers={"User-Agent": "retrobios"}
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=60) as response:
|
||||
listing = json.load(response)
|
||||
target = cache_dir / "fbneo-dats"
|
||||
target.mkdir(parents=True, exist_ok=True)
|
||||
for item in listing:
|
||||
if item.get("type") == "file" and item["name"].lower().endswith(".dat"):
|
||||
_download(item["download_url"], target / item["name"])
|
||||
return target
|
||||
|
||||
|
||||
def _is_listxml(pack: Path) -> bool:
|
||||
"""Whether *pack* holds a MAME -listxml document rather than a DAT."""
|
||||
if pack.is_dir():
|
||||
return False
|
||||
if pack.suffix.lower() == ".zip":
|
||||
with zipfile.ZipFile(pack) as archive:
|
||||
names = [n for n in archive.namelist() if n.lower().endswith(".xml")]
|
||||
if not names:
|
||||
return False
|
||||
with archive.open(names[0]) as handle:
|
||||
return b"<!ELEMENT mame" in handle.read(4096)
|
||||
with pack.open("rb") as handle:
|
||||
return b"<!ELEMENT mame" in handle.read(4096)
|
||||
|
||||
|
||||
def _listxml_opener(pack: Path):
|
||||
"""A callable giving a fresh byte stream over the listxml document."""
|
||||
if pack.suffix.lower() == ".zip":
|
||||
def opener():
|
||||
archive = zipfile.ZipFile(pack)
|
||||
name = next(n for n in archive.namelist() if n.lower().endswith(".xml"))
|
||||
stream = archive.open(name)
|
||||
stream.close_archive = archive # keep the archive alive
|
||||
return stream
|
||||
|
||||
return opener
|
||||
return lambda: pack.open("rb")
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--source", choices=SOURCES, required=True)
|
||||
parser.add_argument("--pack", help="DAT file, ZIP or directory")
|
||||
parser.add_argument(
|
||||
"--fetch",
|
||||
nargs="?",
|
||||
const="mame0289",
|
||||
help="download upstream instead of using --pack (MAME tag, e.g. mame0289)",
|
||||
)
|
||||
parser.add_argument("--cache-dir", default=".cache/dats")
|
||||
parser.add_argument("--output", default="")
|
||||
parser.add_argument("--emulators-dir", default="emulators")
|
||||
parser.add_argument("--platforms-dir", default="platforms")
|
||||
parser.add_argument(
|
||||
"--all", action="store_true", help="keep every set, not only referenced ones"
|
||||
)
|
||||
parser.add_argument("--dry-run", action="store_true")
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.fetch:
|
||||
pack = fetch_pack(args.source, args.fetch, Path(args.cache_dir))
|
||||
print(f" Recupere {pack}")
|
||||
elif args.pack:
|
||||
pack = Path(args.pack)
|
||||
else:
|
||||
parser.error("either --pack or --fetch is required")
|
||||
if not pack.exists():
|
||||
print(f"ERROR: {pack} does not exist", file=sys.stderr)
|
||||
return 1
|
||||
|
||||
output = args.output or DEFAULT_OUTPUT.format(source=args.source)
|
||||
Path(output).parent.mkdir(parents=True, exist_ok=True)
|
||||
keep = (
|
||||
None
|
||||
if args.all
|
||||
else relevant_archives(args.emulators_dir, args.platforms_dir)
|
||||
)
|
||||
|
||||
dats: dict[str, str] = {}
|
||||
entries: list[dict] = []
|
||||
skipped = 0
|
||||
|
||||
if _is_listxml(pack):
|
||||
label = f"MAME {pack.stem.replace('lx', '')}"
|
||||
entries = listxml_entries(_listxml_opener(pack), label, keep)
|
||||
dats[label] = pack.stem
|
||||
print(f" {label}: {len(entries)} sets retenus")
|
||||
members = sum(len(entry["members"]) for entry in entries)
|
||||
print(f"{len(entries)} recettes, {members} membres, depuis 1 listxml")
|
||||
if args.dry_run:
|
||||
return 0
|
||||
changed = merge_snapshot(output, args.source, dats, entries)
|
||||
print(f"{'Wrote' if changed else 'Unchanged'} {output}")
|
||||
return 0
|
||||
|
||||
for member, content in _iter_dat_contents(pack):
|
||||
try:
|
||||
dat = parse_logiqx(content)
|
||||
except (ValueError, SyntaxError) as exc:
|
||||
print(f" Skipped {member}: {exc}", file=sys.stderr)
|
||||
skipped += 1
|
||||
continue
|
||||
if not dat.roms:
|
||||
skipped += 1
|
||||
continue
|
||||
found = recipe_entries(dat, keep)
|
||||
if not found:
|
||||
continue
|
||||
dats[dat.name or member] = dat.version
|
||||
entries.extend(found)
|
||||
print(f" {dat.name or member}: {len(found)} sets retenus")
|
||||
|
||||
if skipped:
|
||||
print(f" {skipped} fichier(s) ignore(s) (non-Logiqx ou vides)")
|
||||
members = sum(len(entry["members"]) for entry in entries)
|
||||
print(f"{len(entries)} recettes, {members} membres, depuis {len(dats)} DAT")
|
||||
|
||||
if args.dry_run:
|
||||
return 0
|
||||
|
||||
changed = merge_snapshot(output, args.source, dats, entries)
|
||||
print(f"{'Wrote' if changed else 'Unchanged'} {output}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,503 @@
|
||||
"""Tests for romset identification and reconstruction from recipes."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
import sys
|
||||
import tempfile
|
||||
import unittest
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
TMP_ROOT = ROOT / "tmp" / "tests"
|
||||
TMP_ROOT.mkdir(parents=True, exist_ok=True)
|
||||
sys.path.insert(0, str(ROOT / "scripts"))
|
||||
|
||||
from romset_recipes import ( # noqa: E402
|
||||
AtomPool,
|
||||
collect_recipes,
|
||||
identify_archive,
|
||||
identify_report,
|
||||
load_dat_recipes,
|
||||
merge_recipes,
|
||||
missing_report,
|
||||
reconstruct,
|
||||
write_reconstructions,
|
||||
)
|
||||
from scripts.scraper.romset_dat_importer import ( # noqa: E402
|
||||
compact_entries,
|
||||
listxml_entries,
|
||||
merge_snapshot,
|
||||
recipe_entries,
|
||||
)
|
||||
from scripts.scraper.logiqx_parser import parse_logiqx # noqa: E402
|
||||
from torrentzip import build_torrentzip, is_torrentzip # noqa: E402
|
||||
|
||||
ROM_A = b"first rom payload"
|
||||
ROM_B = b"second rom payload"
|
||||
|
||||
|
||||
def _crc(data: bytes) -> str:
|
||||
import zlib
|
||||
|
||||
return f"{zlib.crc32(data) & 0xFFFFFFFF:08x}"
|
||||
|
||||
|
||||
class RecipeCollection(unittest.TestCase):
|
||||
def test_only_recipes_with_every_crc_are_usable(self):
|
||||
profiles = {
|
||||
"complete": {
|
||||
"files": [
|
||||
{
|
||||
"name": "set.zip",
|
||||
"contents": [
|
||||
{"name": "a.rom", "crc32": "11111111"},
|
||||
{"name": "b.rom", "crc32": "22222222"},
|
||||
],
|
||||
}
|
||||
]
|
||||
},
|
||||
"partial": {
|
||||
"files": [
|
||||
{
|
||||
"name": "set.zip",
|
||||
"contents": [
|
||||
{"name": "a.rom", "crc32": "11111111"},
|
||||
{"name": "b.rom"},
|
||||
],
|
||||
}
|
||||
]
|
||||
},
|
||||
}
|
||||
recipes = collect_recipes(profiles)
|
||||
self.assertEqual(sorted(recipes["set.zip"]), ["complete"])
|
||||
|
||||
def test_an_archive_without_contents_has_no_recipe(self):
|
||||
recipes = collect_recipes({"p": {"files": [{"name": "set.zip"}]}})
|
||||
self.assertEqual(recipes, {})
|
||||
|
||||
|
||||
class PoolAndBuild(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.temp = tempfile.TemporaryDirectory(dir=TMP_ROOT)
|
||||
self.bios = Path(self.temp.name)
|
||||
with zipfile.ZipFile(self.bios / "source.zip", "w") as archive:
|
||||
archive.writestr("a.rom", ROM_A)
|
||||
archive.writestr("b.rom", ROM_B)
|
||||
self.pool = AtomPool(self.bios)
|
||||
|
||||
def tearDown(self):
|
||||
self.temp.cleanup()
|
||||
|
||||
def test_atoms_are_addressed_by_crc32(self):
|
||||
self.assertIn(_crc(ROM_A), self.pool)
|
||||
self.assertEqual(self.pool.read(_crc(ROM_A)), ROM_A)
|
||||
|
||||
def test_a_recipe_missing_an_atom_builds_nothing(self):
|
||||
self.assertIsNone(self.pool.build([("x.rom", "deadbeef")]))
|
||||
|
||||
def test_build_is_torrentzip_and_reproducible(self):
|
||||
recipe = [("a.rom", _crc(ROM_A)), ("b.rom", _crc(ROM_B))]
|
||||
first = self.pool.build(recipe)
|
||||
self.assertEqual(first, build_torrentzip([("a.rom", ROM_A), ("b.rom", ROM_B)]))
|
||||
target = self.bios / "built.zip"
|
||||
target.write_bytes(first)
|
||||
self.assertTrue(is_torrentzip(target))
|
||||
|
||||
def test_identify_names_the_recipe_that_rebuilds_the_archive(self):
|
||||
recipe = [("a.rom", _crc(ROM_A)), ("b.rom", _crc(ROM_B))]
|
||||
built = self.bios / "built.zip"
|
||||
built.write_bytes(self.pool.build(recipe))
|
||||
label = identify_archive(
|
||||
built,
|
||||
{"v2": recipe, "v1": [("a.rom", _crc(ROM_A))]},
|
||||
self.pool,
|
||||
)
|
||||
self.assertEqual(label, "v2")
|
||||
|
||||
def test_identify_returns_nothing_when_no_recipe_matches(self):
|
||||
other = self.bios / "other.zip"
|
||||
other.write_bytes(build_torrentzip([("z.rom", b"unrelated")]))
|
||||
self.assertIsNone(
|
||||
identify_archive(other, {"v1": [("a.rom", _crc(ROM_A))]}, self.pool)
|
||||
)
|
||||
|
||||
def test_reconstruct_finds_the_recipe_behind_a_pinned_md5(self):
|
||||
recipe = [("a.rom", _crc(ROM_A))]
|
||||
target = hashlib.md5(self.pool.build(recipe)).hexdigest()
|
||||
found = reconstruct(target, {"v1": recipe}, self.pool)
|
||||
self.assertIsNotNone(found)
|
||||
self.assertEqual(found[0], "v1")
|
||||
self.assertEqual(hashlib.md5(found[1]).hexdigest(), target)
|
||||
|
||||
|
||||
class ReportsOverAFixture(unittest.TestCase):
|
||||
"""A whole run over a synthetic collection, database and platform."""
|
||||
|
||||
def setUp(self):
|
||||
self.temp = tempfile.TemporaryDirectory(dir=TMP_ROOT)
|
||||
self.root = Path(self.temp.name)
|
||||
self.bios = self.root / "bios"
|
||||
self.bios.mkdir()
|
||||
self.pool_source = self.bios / "source.zip"
|
||||
with zipfile.ZipFile(self.pool_source, "w") as archive:
|
||||
archive.writestr("a.rom", ROM_A)
|
||||
archive.writestr("b.rom", ROM_B)
|
||||
self.pool = AtomPool(self.bios)
|
||||
self.recipe = [("a.rom", _crc(ROM_A)), ("b.rom", _crc(ROM_B))]
|
||||
self.set_path = self.bios / "set.zip"
|
||||
self.set_path.write_bytes(self.pool.build(self.recipe))
|
||||
# The pool must see the archive it will be asked to identify.
|
||||
self.pool = AtomPool(self.bios)
|
||||
|
||||
blob = self.set_path.read_bytes()
|
||||
self.db = {
|
||||
"files": {
|
||||
hashlib.sha1(blob).hexdigest(): {
|
||||
"path": str(self.set_path),
|
||||
"name": "set.zip",
|
||||
"md5": hashlib.md5(blob).hexdigest(),
|
||||
}
|
||||
},
|
||||
"indexes": {
|
||||
"by_name": {"set.zip": [hashlib.sha1(blob).hexdigest()]},
|
||||
"by_md5": {hashlib.md5(blob).hexdigest(): hashlib.sha1(blob).hexdigest()},
|
||||
},
|
||||
}
|
||||
self.recipes = {"set.zip": {"v1": self.recipe}}
|
||||
|
||||
def tearDown(self):
|
||||
self.temp.cleanup()
|
||||
|
||||
def test_identify_report_labels_the_local_archive(self):
|
||||
report = identify_report(self.recipes, self.pool, self.db)
|
||||
self.assertEqual(len(report["identified"]), 1)
|
||||
self.assertEqual(report["identified"][0]["recipe"], "v1")
|
||||
self.assertTrue(report["identified"][0]["torrentzip"])
|
||||
self.assertEqual(report["unidentified"], [])
|
||||
|
||||
def test_an_aliased_non_archive_is_not_treated_as_a_romset(self):
|
||||
loose = self.bios / "state-machine.rom"
|
||||
loose.write_bytes(b"not an archive")
|
||||
sha1 = hashlib.sha1(loose.read_bytes()).hexdigest()
|
||||
self.db["files"][sha1] = {"path": str(loose), "name": "state-machine.rom"}
|
||||
self.db["indexes"]["by_name"]["set.zip"].append(sha1)
|
||||
|
||||
report = identify_report(self.recipes, self.pool, self.db)
|
||||
paths = {r["path"] for r in report["identified"] + report["unidentified"]}
|
||||
self.assertNotIn(str(loose), paths)
|
||||
|
||||
def test_missing_report_reconstructs_a_pinned_archive(self):
|
||||
platforms = self.root / "platforms"
|
||||
platforms.mkdir()
|
||||
pinned = hashlib.md5(self.pool.build(self.recipe)).hexdigest()
|
||||
(platforms / "_registry.yml").write_text(
|
||||
"platforms:\n demo:\n config: demo.yml\n status: active\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
(platforms / "demo.yml").write_text(
|
||||
"platform: Demo\nsystems:\n arcade:\n files:\n"
|
||||
" - name: set.zip\n destination: set.zip\n"
|
||||
f" md5: {pinned}\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
# The collection does not hold the pinned archive yet.
|
||||
self.db["indexes"]["by_md5"] = {}
|
||||
|
||||
report = missing_report(self.recipes, self.pool, self.db, str(platforms))
|
||||
self.assertEqual(len(report["constructible"]), 1)
|
||||
entry = report["constructible"][0]
|
||||
self.assertEqual(entry["recipe"], "v1")
|
||||
self.assertEqual(entry["md5"], pinned)
|
||||
|
||||
written = write_reconstructions(
|
||||
report["constructible"], self.recipes, self.pool, self.bios, self.db
|
||||
)
|
||||
self.assertEqual(len(written), 1)
|
||||
produced = Path(written[0])
|
||||
self.assertEqual(hashlib.md5(produced.read_bytes()).hexdigest(), pinned)
|
||||
self.assertTrue(is_torrentzip(produced))
|
||||
|
||||
def test_an_unreproducible_pin_is_reported_not_guessed(self):
|
||||
platforms = self.root / "platforms"
|
||||
platforms.mkdir()
|
||||
(platforms / "_registry.yml").write_text(
|
||||
"platforms:\n demo:\n config: demo.yml\n status: active\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
(platforms / "demo.yml").write_text(
|
||||
"platform: Demo\nsystems:\n arcade:\n files:\n"
|
||||
" - name: set.zip\n destination: set.zip\n"
|
||||
f" md5: {'f' * 32}\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
self.db["indexes"]["by_md5"] = {}
|
||||
|
||||
report = missing_report(self.recipes, self.pool, self.db, str(platforms))
|
||||
self.assertEqual(report["constructible"], [])
|
||||
self.assertEqual(len(report["unreproducible"]), 1)
|
||||
self.assertEqual(report["unreproducible"][0]["recipes_tried"], ["v1"])
|
||||
|
||||
|
||||
DAT = """<?xml version="1.0"?>
|
||||
<datafile>
|
||||
<header><name>MAME</name><version>0.289</version></header>
|
||||
<machine name="manager">
|
||||
<description>Salora Manager</description>
|
||||
<rom name="01" size="17" crc="{crc_a}" sha1=""/>
|
||||
<rom name="23" size="18" crc="{crc_b}" sha1=""/>
|
||||
</machine>
|
||||
<machine name="undumped">
|
||||
<description>Nothing dumped yet</description>
|
||||
<rom name="x.rom" size="0" status="nodump"/>
|
||||
</machine>
|
||||
</datafile>
|
||||
"""
|
||||
|
||||
|
||||
class DatRecipes(unittest.TestCase):
|
||||
"""A DAT states, per set, the members one emulator version expects."""
|
||||
|
||||
def _dat(self) -> str:
|
||||
return DAT.format(crc_a=_crc(ROM_A), crc_b=_crc(ROM_B))
|
||||
|
||||
def test_a_set_becomes_one_recipe_named_after_its_archive(self):
|
||||
entries = recipe_entries(parse_logiqx(self._dat()))
|
||||
self.assertEqual([e["name"] for e in entries], ["manager.zip"])
|
||||
self.assertEqual(
|
||||
[m["name"] for m in entries[0]["members"]], ["01", "23"]
|
||||
)
|
||||
|
||||
def test_an_undumped_member_is_dropped_not_the_whole_set(self):
|
||||
"""A real romset does not carry a ROM nobody has dumped."""
|
||||
entries = recipe_entries(parse_logiqx(self._dat()))
|
||||
self.assertNotIn("undumped.zip", [e["name"] for e in entries])
|
||||
|
||||
def test_only_referenced_archives_are_kept(self):
|
||||
entries = recipe_entries(parse_logiqx(self._dat()), keep={"other.zip"})
|
||||
self.assertEqual(entries, [])
|
||||
|
||||
def test_snapshot_round_trips_into_usable_recipes(self):
|
||||
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
|
||||
root = Path(directory)
|
||||
snapshot = root / "romset-recipes.json"
|
||||
entries = recipe_entries(parse_logiqx(self._dat()))
|
||||
snapshot.write_text(
|
||||
json.dumps(
|
||||
{
|
||||
"source": "mame",
|
||||
"imported_at": "2026-08-10",
|
||||
"dats": {"MAME": "0.289"},
|
||||
"entries": entries,
|
||||
}
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
recipes = load_dat_recipes(snapshot)
|
||||
self.assertEqual(sorted(recipes), ["manager.zip"])
|
||||
self.assertEqual(sorted(recipes["manager.zip"]), ["mame:MAME"])
|
||||
|
||||
bios = root / "bios"
|
||||
bios.mkdir()
|
||||
with zipfile.ZipFile(bios / "source.zip", "w") as archive:
|
||||
archive.writestr("01", ROM_A)
|
||||
archive.writestr("23", ROM_B)
|
||||
pool = AtomPool(bios)
|
||||
built = bios / "manager.zip"
|
||||
built.write_bytes(pool.build(recipes["manager.zip"]["mame:MAME"]))
|
||||
self.assertTrue(is_torrentzip(built))
|
||||
self.assertEqual(
|
||||
identify_archive(built, recipes["manager.zip"], AtomPool(bios)),
|
||||
"mame:MAME",
|
||||
)
|
||||
|
||||
def test_a_missing_snapshot_is_simply_no_recipes(self):
|
||||
self.assertEqual(load_dat_recipes(TMP_ROOT / "absent.json"), {})
|
||||
|
||||
def test_merge_keeps_profile_and_dat_labels_side_by_side(self):
|
||||
merged = merge_recipes(
|
||||
{"set.zip": {"mame": [("a", "1")]}},
|
||||
{"set.zip": {"mame:MAME": [("a", "1")]}},
|
||||
)
|
||||
self.assertEqual(sorted(merged["set.zip"]), ["mame", "mame:MAME"])
|
||||
|
||||
|
||||
LISTXML = """<?xml version="1.0"?>
|
||||
<!DOCTYPE mame [<!ELEMENT mame (machine+)>]>
|
||||
<mame build="0.289">
|
||||
<machine name="parentset" isbios="yes">
|
||||
<description>Parent</description>
|
||||
<rom name="shared.rom" size="17" crc="{crc_a}" sha1=""/>
|
||||
</machine>
|
||||
<machine name="childset" romof="parentset">
|
||||
<description>Child</description>
|
||||
<rom name="own.rom" size="18" crc="{crc_b}" sha1=""/>
|
||||
</machine>
|
||||
<machine name="undumped">
|
||||
<description>Nothing dumped</description>
|
||||
<rom name="x.rom" size="0" status="nodump"/>
|
||||
</machine>
|
||||
</mame>
|
||||
"""
|
||||
|
||||
|
||||
class ListXmlRecipes(unittest.TestCase):
|
||||
"""MAME ships its -listxml output as a release asset."""
|
||||
|
||||
def _opener(self, text: str):
|
||||
import io
|
||||
|
||||
return lambda: io.BytesIO(text.encode("utf-8"))
|
||||
|
||||
def _document(self) -> str:
|
||||
return LISTXML.format(crc_a=_crc(ROM_A), crc_b=_crc(ROM_B))
|
||||
|
||||
def test_a_child_inherits_its_parent_members(self):
|
||||
"""A non-merged archive carries the parent set's ROMs as well."""
|
||||
entries = listxml_entries(self._opener(self._document()), "MAME 0.289", None)
|
||||
by_name = {entry["name"]: entry for entry in entries}
|
||||
self.assertEqual(
|
||||
[m["name"] for m in by_name["childset.zip"]["members"]],
|
||||
["own.rom", "shared.rom"],
|
||||
)
|
||||
|
||||
def test_a_set_without_a_parent_keeps_its_own_members(self):
|
||||
entries = listxml_entries(self._opener(self._document()), "MAME 0.289", None)
|
||||
by_name = {entry["name"]: entry for entry in entries}
|
||||
self.assertEqual(
|
||||
[m["name"] for m in by_name["parentset.zip"]["members"]], ["shared.rom"]
|
||||
)
|
||||
|
||||
def test_an_undumped_only_set_is_dropped(self):
|
||||
entries = listxml_entries(self._opener(self._document()), "MAME 0.289", None)
|
||||
self.assertNotIn("undumped.zip", {entry["name"] for entry in entries})
|
||||
|
||||
def test_a_parent_outside_the_keep_set_is_still_resolved(self):
|
||||
"""The child is wanted, its parent is not, and the union still holds."""
|
||||
entries = listxml_entries(
|
||||
self._opener(self._document()), "MAME 0.289", {"childset.zip"}
|
||||
)
|
||||
self.assertEqual([entry["name"] for entry in entries], ["childset.zip"])
|
||||
self.assertEqual(
|
||||
[m["name"] for m in entries[0]["members"]], ["own.rom", "shared.rom"]
|
||||
)
|
||||
|
||||
def test_versions_accumulate_instead_of_replacing_each_other(self):
|
||||
"""A platform pins the archive of whichever version it was built on."""
|
||||
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
|
||||
output = str(Path(directory) / "romset-recipes.json")
|
||||
merge_snapshot(
|
||||
output,
|
||||
"mame",
|
||||
{"MAME 0.288": "mame0288"},
|
||||
[
|
||||
{
|
||||
"dat": "MAME 0.288",
|
||||
"name": "set.zip",
|
||||
"set": "set",
|
||||
"description": "",
|
||||
"members": [{"name": "a", "crc32": "1"}],
|
||||
}
|
||||
],
|
||||
)
|
||||
merge_snapshot(
|
||||
output,
|
||||
"mame",
|
||||
{"MAME 0.289": "mame0289"},
|
||||
[
|
||||
{
|
||||
"dat": "MAME 0.289",
|
||||
"name": "set.zip",
|
||||
"set": "set",
|
||||
"description": "",
|
||||
"members": [{"name": "b", "crc32": "2"}],
|
||||
}
|
||||
],
|
||||
)
|
||||
recipes = load_dat_recipes(output)
|
||||
self.assertEqual(
|
||||
sorted(recipes["set.zip"]), ["mame:MAME 0.288", "mame:MAME 0.289"]
|
||||
)
|
||||
|
||||
|
||||
class SnapshotCompaction(unittest.TestCase):
|
||||
"""Most sets do not change between MAME releases."""
|
||||
|
||||
def _entry(self, dat: str, members: list[dict]) -> dict:
|
||||
return {
|
||||
"dat": dat,
|
||||
"name": "set.zip",
|
||||
"set": "set",
|
||||
"description": "",
|
||||
"members": members,
|
||||
}
|
||||
|
||||
def test_identical_recipes_collapse_to_one_entry(self):
|
||||
members = [{"name": "a", "crc32": "1"}]
|
||||
compact = compact_entries(
|
||||
[
|
||||
self._entry("MAME 0.200", members),
|
||||
self._entry("MAME 0.190", members),
|
||||
self._entry("MAME 0.210", members),
|
||||
]
|
||||
)
|
||||
self.assertEqual(len(compact), 1)
|
||||
self.assertEqual(
|
||||
compact[0]["dats"], ["MAME 0.190", "MAME 0.200", "MAME 0.210"]
|
||||
)
|
||||
|
||||
def test_the_representative_label_is_the_earliest_version(self):
|
||||
"""A match reads as 'unchanged since', not 'is this version'."""
|
||||
members = [{"name": "a", "crc32": "1"}]
|
||||
compact = compact_entries(
|
||||
[self._entry("MAME 0.280", members), self._entry("MAME 0.190", members)]
|
||||
)
|
||||
self.assertEqual(compact[0]["dat"], "MAME 0.190")
|
||||
|
||||
def test_a_changed_recipe_stays_a_separate_entry(self):
|
||||
compact = compact_entries(
|
||||
[
|
||||
self._entry("MAME 0.190", [{"name": "a", "crc32": "1"}]),
|
||||
self._entry("MAME 0.280", [{"name": "a", "crc32": "2"}]),
|
||||
]
|
||||
)
|
||||
self.assertEqual(len(compact), 2)
|
||||
|
||||
def test_compaction_is_idempotent(self):
|
||||
members = [{"name": "a", "crc32": "1"}]
|
||||
once = compact_entries(
|
||||
[self._entry("MAME 0.190", members), self._entry("MAME 0.200", members)]
|
||||
)
|
||||
self.assertEqual(compact_entries(once), once)
|
||||
|
||||
|
||||
class RepositoryRun(unittest.TestCase):
|
||||
"""The real collection, skipped when its data is not present."""
|
||||
|
||||
def test_identification_covers_the_documented_archives(self):
|
||||
database = ROOT / "database.json"
|
||||
bios = ROOT / "bios"
|
||||
if not database.is_file() or not bios.is_dir():
|
||||
self.skipTest("collection not present in this checkout")
|
||||
|
||||
from common import load_database, load_emulator_profiles
|
||||
|
||||
db = load_database(str(database))
|
||||
recipes = collect_recipes(
|
||||
load_emulator_profiles(str(ROOT / "emulators"), skip_aliases=False)
|
||||
)
|
||||
self.assertGreater(len(recipes), 0)
|
||||
pool = AtomPool(bios)
|
||||
self.assertGreater(len(pool), 0)
|
||||
self.assertEqual(pool.unreadable, [])
|
||||
|
||||
report = identify_report(recipes, pool, db)
|
||||
self.assertGreater(len(report["identified"]), 0)
|
||||
for record in report["identified"]:
|
||||
self.assertIn(record["recipe"], recipes[record["archive"]])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in new issue
Block a user