feat: identify romsets from version recipes

A profile's contents: block and a DAT both state, per set, the members
one emulator version expects. Torrentzip makes archive bytes a function
of that list alone, so a recipe plus the roms reproduces the archive
exactly.

MAME ships its -listxml as a release asset and FBNeo keeps its dats in
its repository, so neither needs a browser. The importer streams the
311 MB document with a sliding window, resolves romof parents in two
passes, drops undumped members, and accumulates versions instead of
replacing them. Identical recipes shared across versions are stored once
with dats listing every version that agrees: 22 MAME versions give 23679
entries for 1726 distinct recipes, 18 MB down to 2.2 MB.

Snapshots live in recipes/ because load_provenance_snapshots reads every
provenance/*.json as a dump catalogue, and a recipe is not one.

1113 archives now reproduce byte for byte, against 175 from profiles
alone, and spec128.zip is rebuilt from roms already held.
This commit is contained in:
Abdessamad Derraz committed 2026-08-10 13:36:12 +02:00
1 parent e8ee8b0954
commit 33a9934a11
5 files changed
+94206

No files matched your search

+439
View File
@@ -0,0 +1,439 @@
#!/usr/bin/env python3
"""Identify and reconstruct arcade romset archives from their recipes.
A profile's ``contents:`` block is a recipe: the member names and CRC32s one
emulator version expects inside an archive. TorrentZip makes archive bytes a
function of that recipe alone, so a recipe plus the ROM bytes reproduces the
archive exactly.
Two questions follow, and this module answers both:
``--identify``
Which version does an archive the collection holds correspond to? The
recipe that rebuilds it byte for byte names it.
``--missing``
A platform pins a container MD5 the collection does not have. If some
recipe plus ROMs already present reproduces that MD5, the archive is
constructible rather than absent.
Reconstruction is only possible when the pinned archive is itself TorrentZip.
An archive whose bytes carry metadata unrelated to its contents cannot be
derived from ROMs by anyone, and is reported as such rather than guessed at.
"""
from __future__ import annotations
import argparse
import hashlib
import json
import sys
import zipfile
from collections import defaultdict
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
from common import (
load_database,
load_emulator_profiles,
load_platform_config,
list_registered_platforms,
)
from torrentzip import build_torrentzip, is_torrentzip
Recipe = list[tuple[str, str]]
def load_dat_recipes(path: str | Path) -> dict[str, dict[str, Recipe]]:
"""Recipes imported from MAME or FBNeo DATs.
A DAT documents every set of one emulator version, where profiles only
document the few hundred the collection curates by hand. Labels are
prefixed so an identification says which DAT decided it.
"""
target = Path(path)
if target.is_dir():
merged: dict[str, dict[str, Recipe]] = {}
for snapshot_file in sorted(target.glob("*.json")):
merged = merge_recipes(merged, load_dat_recipes(snapshot_file))
return merged
snapshot_path = target
if not snapshot_path.is_file():
return {}
with snapshot_path.open(encoding="utf-8") as handle:
snapshot = json.load(handle)
source = snapshot.get("source", "dat")
# A recipe shared by several versions is stored once under the earliest.
# The label therefore reads "unchanged since", not "is this version".
recipes: dict[str, dict[str, Recipe]] = defaultdict(dict)
for entry in snapshot.get("entries", []):
archive = entry.get("name", "")
members = entry.get("members", []) or []
recipe = [
(str(member["name"]), str(member.get("crc32") or "").lower())
for member in members
if member.get("name") is not None
]
if not archive or not recipe or not all(crc for _name, crc in recipe):
continue
label = f"{source}:{entry.get('dat', 'unnamed')}"
recipes[archive][label] = recipe
return dict(recipes)
def merge_recipes(*sources: dict[str, dict[str, Recipe]]) -> dict[str, dict[str, Recipe]]:
"""Combine recipe sets, keeping every label distinct."""
merged: dict[str, dict[str, Recipe]] = defaultdict(dict)
for source in sources:
for archive, per_label in source.items():
merged[archive].update(per_label)
return dict(merged)
def collect_recipes(profiles: dict) -> dict[str, dict[str, Recipe]]:
"""Map each archive name to one recipe per profile that documents it.
A recipe is only usable when every member carries a CRC32: the pool is
addressed by content, so a member without one cannot be located.
"""
recipes: dict[str, dict[str, Recipe]] = defaultdict(dict)
for profile_name, profile in sorted(profiles.items()):
for entry in profile.get("files", []) or []:
contents = entry.get("contents")
archive = entry.get("name", "")
if not contents or not archive:
continue
recipe = [
(str(member["name"]), str(member.get("crc32") or "").lower())
for member in contents
if member.get("name") is not None
]
if recipe and all(crc for _name, crc in recipe):
recipes[archive][profile_name] = recipe
return dict(recipes)
class AtomPool:
"""ROM bytes in the collection, addressed by CRC32.
Members are located once and read on demand: arcade sample sets reach
hundreds of megabytes and a whole-pool preload buys nothing.
"""
def __init__(self, bios_dir: Path):
self._index: dict[str, tuple[Path, str]] = {}
# A recipe is built once: identification and reconstruction ask for the
# same archives, and every build re-reads and re-deflates its ROMs.
self._built: dict[tuple[tuple[str, str], ...], bytes | None] = {}
self.unreadable: list[str] = []
for archive in sorted(bios_dir.rglob("*.zip")):
try:
with zipfile.ZipFile(archive) as handle:
for info in handle.infolist():
if not info.is_dir():
self._index.setdefault(
f"{info.CRC:08x}", (archive, info.filename)
)
except (OSError, zipfile.BadZipFile) as exc:
self.unreadable.append(f"{archive}: {exc}")
def __len__(self) -> int:
return len(self._index)
def __contains__(self, crc32: str) -> bool:
return crc32.lower() in self._index
def read(self, crc32: str) -> bytes:
archive, member = self._index[crc32.lower()]
with zipfile.ZipFile(archive) as handle:
return handle.read(member)
def build(self, recipe: Recipe) -> bytes | None:
"""Return the TorrentZip bytes for *recipe*, or None if incomplete."""
key = tuple(recipe)
if key not in self._built:
if all(crc in self for _name, crc in recipe):
self._built[key] = build_torrentzip(
[(name, self.read(crc)) for name, crc in recipe]
)
else:
self._built[key] = None
return self._built[key]
def identify_archive(
path: Path, recipes: dict[str, Recipe], pool: AtomPool
) -> str | None:
"""Return the recipe label that reproduces *path* byte for byte."""
original = path.read_bytes()
for label, recipe in sorted(recipes.items()):
blob = pool.build(recipe)
if blob is not None and blob == original:
return label
return None
def reconstruct(
target_md5: str, recipes: dict[str, Recipe], pool: AtomPool
) -> tuple[str, bytes] | None:
"""Return the (label, bytes) whose TorrentZip build matches *target_md5*."""
target = target_md5.lower()
for label, recipe in sorted(recipes.items()):
blob = pool.build(recipe)
if blob is not None and hashlib.md5(blob).hexdigest() == target:
return label, blob
return None
def platform_targets(platforms_dir: str) -> dict[str, set[str]]:
"""Container MD5s the platforms pin, per archive name.
``zipped_file`` entries pin an inner ROM instead of the container, so
their MD5 is not an archive hash and is skipped.
"""
targets: dict[str, set[str]] = defaultdict(set)
for name in list_registered_platforms(platforms_dir, include_archived=True):
config = load_platform_config(name, platforms_dir)
for system in config.get("systems", {}).values():
for entry in system.get("files", []) or []:
archive = entry.get("name", "")
if not archive.endswith(".zip") or entry.get("zipped_file"):
continue
for value in str(entry.get("md5") or "").split(","):
value = value.strip().lower()
if len(value) == 32:
targets[archive].add(value)
return dict(targets)
def _is_archive(path: Path) -> bool:
"""Whether *path* is a readable ZIP.
A name index entry can be an alias: ``d2fdc.zip`` names a loose Apple II
ROM, not an archive. Recipes only describe archives.
"""
try:
with zipfile.ZipFile(path):
return True
except (OSError, zipfile.BadZipFile):
return False
def _local_paths(archive: str, db: dict) -> list[Path]:
matches = db["indexes"]["by_name"].get(archive) or []
if isinstance(matches, str):
matches = [matches]
paths = [Path(db["files"][sha1]["path"]) for sha1 in matches if sha1 in db["files"]]
return [path for path in paths if path.is_file() and _is_archive(path)]
def identify_report(
recipes: dict[str, dict[str, Recipe]], pool: AtomPool, db: dict
) -> dict:
"""Label every archive the collection holds with the version it matches."""
identified: list[dict] = []
unidentified: list[dict] = []
for archive, per_profile in sorted(recipes.items()):
for path in _local_paths(archive, db):
label = identify_archive(path, per_profile, pool)
record = {
"archive": archive,
"path": str(path),
"torrentzip": is_torrentzip(path),
}
if label:
record["recipe"] = label
identified.append(record)
else:
record["candidates"] = sorted(per_profile)
unidentified.append(record)
return {"identified": identified, "unidentified": unidentified}
def missing_report(
recipes: dict[str, dict[str, Recipe]],
pool: AtomPool,
db: dict,
platforms_dir: str,
) -> dict:
"""Pinned archives the collection lacks, split by why they are absent."""
have = set(db["indexes"]["by_md5"])
constructible: list[dict] = []
no_recipe: list[dict] = []
unreproducible: list[dict] = []
for archive, pinned in sorted(platform_targets(platforms_dir).items()):
per_profile = recipes.get(archive, {})
for target in sorted(pinned - have):
if not per_profile:
no_recipe.append({"archive": archive, "md5": target})
continue
found = reconstruct(target, per_profile, pool)
if found:
label, blob = found
constructible.append(
{
"archive": archive,
"md5": target,
"recipe": label,
"size": len(blob),
}
)
else:
usable = [
label
for label, recipe in per_profile.items()
if all(crc in pool for _name, crc in recipe)
]
unreproducible.append(
{
"archive": archive,
"md5": target,
"recipes_tried": sorted(usable),
"reason": (
"no recipe reproduces this MD5; the pinned archive is "
"not TorrentZip or follows a recipe not documented here"
if usable
else "ROMs for every documented recipe are missing"
),
}
)
return {
"constructible": constructible,
"unreproducible": unreproducible,
"no_recipe": no_recipe,
}
def write_reconstructions(
entries: list[dict],
recipes: dict[str, dict[str, Recipe]],
pool: AtomPool,
bios_dir: Path,
db: dict,
) -> list[str]:
"""Write each constructible archive beside its set, as a variant."""
written: list[str] = []
for entry in entries:
archive = entry["archive"]
recipe = recipes[archive][entry["recipe"]]
blob = pool.build(recipe)
if blob is None:
continue
siblings = _local_paths(archive, db)
parent = siblings[0].parent if siblings else bios_dir
destination = parent / ".variants" / f"{archive}.{entry['md5'][:8]}"
destination.parent.mkdir(parents=True, exist_ok=True)
destination.write_bytes(blob)
written.append(str(destination))
return written
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--identify", action="store_true", help="label local archives")
parser.add_argument(
"--missing", action="store_true", help="attempt pinned archives we lack"
)
parser.add_argument(
"--write", action="store_true", help="write reconstructions to .variants/"
)
parser.add_argument("--json", action="store_true", help="machine-readable output")
parser.add_argument("--bios-dir", default="bios")
parser.add_argument("--platforms-dir", default="platforms")
parser.add_argument("--emulators-dir", default="emulators")
parser.add_argument("--db", default="database.json")
parser.add_argument(
"--dat-recipes",
default="recipes",
help="recipe snapshot file, or a directory of them",
)
args = parser.parse_args()
if not args.identify and not args.missing:
args.identify = args.missing = True
db = load_database(args.db)
profiles = load_emulator_profiles(args.emulators_dir, skip_aliases=False)
profile_recipes = collect_recipes(profiles)
dat_recipes = load_dat_recipes(args.dat_recipes)
recipes = merge_recipes(profile_recipes, dat_recipes)
pool = AtomPool(Path(args.bios_dir))
result: dict = {
"recipes": sum(len(v) for v in recipes.values()),
"profile_recipes": sum(len(v) for v in profile_recipes.values()),
"dat_recipes": sum(len(v) for v in dat_recipes.values()),
"archives_with_recipes": len(recipes),
"atoms": len(pool),
}
if pool.unreadable:
result["unreadable_archives"] = pool.unreadable
if args.identify:
result["identify"] = identify_report(recipes, pool, db)
if args.missing:
result["missing"] = missing_report(recipes, pool, db, args.platforms_dir)
if args.write:
result["written"] = write_reconstructions(
result["missing"]["constructible"], recipes, pool,
Path(args.bios_dir), db,
)
if args.json:
print(json.dumps(result, indent=2, sort_keys=True))
return 0
print(
f"{result['archives_with_recipes']} archives documentees par "
f"{result['recipes']} recettes "
f"({result['profile_recipes']} de profils, {result['dat_recipes']} de DAT), "
f"{result['atoms']} atomes ROM disponibles"
)
if not result["dat_recipes"]:
print(
" Aucune recette DAT: importer avec "
"`python -m scripts.scraper.romset_dat_importer --source mame --fetch mame0289`"
)
for problem in result.get("unreadable_archives", []):
print(f" UNREADABLE: {problem}")
if args.identify:
ident = result["identify"]
print(
f"\nIdentification: {len(ident['identified'])} archives reproduites "
f"par une recette, {len(ident['unidentified'])} non identifiees"
)
print(
" L'etiquette est la plus ancienne version produisant ces octets: "
"une recette inchangee depuis 0.190 est comptee la, pas a sa version "
"d'origine."
)
by_recipe: dict[str, int] = defaultdict(int)
for record in ident["identified"]:
by_recipe[record["recipe"]] += 1
for label, count in sorted(by_recipe.items(), key=lambda kv: (-kv[1], kv[0])):
print(f" {label}: {count}")
for record in ident["unidentified"][:10]:
print(f" UNIDENTIFIED: {record['path']}")
if args.missing:
gap = result["missing"]
print(
f"\nArchives epinglees absentes: "
f"{len(gap['constructible'])} reconstructibles, "
f"{len(gap['unreproducible'])} irreproductibles, "
f"{len(gap['no_recipe'])} sans recette"
)
for record in gap["constructible"]:
print(
f" BUILDABLE: {record['archive']} md5={record['md5'][:8]} "
f"via {record['recipe']} ({record['size']} octets)"
)
for path in result.get("written", []):
print(f" WROTE: {path}")
return 0
if __name__ == "__main__":
sys.exit(main())
+466
View File
@@ -0,0 +1,466 @@
"""Import romset recipes from a MAME or FBNeo DAT.
A DAT states, per set, the members an emulator version expects: name, size
and CRC32. That is a recipe, and TorrentZip turns a recipe plus the ROM bytes
into the exact archive. Profiles document a few hundred sets by hand; a DAT
documents every set of one version at once.
MAME ships its ``-listxml`` output as a release asset and FBNeo keeps its DATs
in its repository, so both are fetched without a browser:
python -m scripts.scraper.romset_dat_importer --source mame --fetch mame0289
python -m scripts.scraper.romset_dat_importer --source fbneo --fetch
python -m scripts.scraper.romset_dat_importer --source mame --pack local.dat
Only sets the collection has a reason to know are kept, so the committed
snapshot stays small: the archive names that emulator profiles or platform
BIOS lists reference. ``--all`` keeps everything.
Members without a CRC32 are undumped. A real romset does not carry them, so
they are dropped from the recipe rather than disqualifying the whole set.
"""
from __future__ import annotations
import argparse
import hashlib
import json
import os
import re
import shutil
import sys
import urllib.request
import zipfile
from datetime import datetime, timezone
from pathlib import Path
from ..common import (
list_registered_platforms,
load_emulator_profiles,
load_platform_config,
write_provenance_snapshot,
)
from .logiqx_parser import LogiqxDat, parse_logiqx
MAX_MEMBER_SIZE = 200 * 1024 * 1024
SOURCES = ("mame", "fbneo")
DEFAULT_OUTPUT = "recipes/{source}.json"
def relevant_archives(emulators_dir: str, platforms_dir: str) -> set[str]:
"""Archive names any profile or platform refers to."""
names: set[str] = set()
for _profile_name, profile in load_emulator_profiles(
emulators_dir, skip_aliases=False
).items():
for entry in profile.get("files", []) or []:
name = entry.get("name", "")
if name.endswith(".zip"):
names.add(name)
archive = entry.get("archive", "")
if archive:
names.add(archive)
for platform_name in list_registered_platforms(platforms_dir, include_archived=True):
config = load_platform_config(platform_name, platforms_dir)
for system in config.get("systems", {}).values():
for entry in system.get("files", []) or []:
name = entry.get("name", "")
if name.endswith(".zip"):
names.add(name)
return names
def recipe_entries(
dat: LogiqxDat, keep: set[str] | None = None
) -> list[dict]:
"""Group a parsed DAT into one recipe per set."""
by_set: dict[str, list] = {}
for rom in dat.roms:
if not rom.crc32:
continue
by_set.setdefault(rom.game, []).append(rom)
entries: list[dict] = []
for set_name, roms in sorted(by_set.items()):
archive = f"{set_name}.zip"
if keep is not None and archive not in keep:
continue
entries.append(
{
"dat": dat.name or "unnamed",
"name": archive,
"set": set_name,
"description": roms[0].description,
"members": sorted(
(
{
"name": rom.name,
"size": rom.size,
"crc32": rom.crc32,
"sha1": rom.sha1,
}
for rom in roms
),
key=lambda member: member["name"],
),
}
)
return entries
_MACHINE = re.compile(rb'<machine name="([^"]+)"([^>]*)>')
_ROM = re.compile(
rb'<rom name="([^"]+)"(?=[^>]*\bcrc="([0-9a-fA-F]+)")[^>]*>'
)
_ROM_ATTR = re.compile(rb'(\w+)="([^"]*)"')
_ROMOF = re.compile(r'romof="([^"]+)"')
LISTXML_CHUNK = 8 * 1024 * 1024
def _machine_roms(blob: bytes) -> list[dict]:
"""ROM members of one machine element, undumped ones dropped."""
members: list[dict] = []
for match in re.finditer(rb"<rom\b([^>]*)>", blob):
attrs = {
key.decode(): value.decode()
for key, value in _ROM_ATTR.findall(match.group(1))
}
crc = attrs.get("crc", "").lower()
name = attrs.get("name", "")
if not name or not crc:
continue
try:
size = int(attrs.get("size", "0"))
except ValueError:
size = 0
members.append(
{
"name": name,
"size": size,
"crc32": crc,
"sha1": attrs.get("sha1", "").lower(),
}
)
return members
def _scan_listxml(stream, selector) -> dict[str, tuple[str | None, list[dict]]]:
"""Stream a MAME -listxml document, keeping the machines *selector* wants.
The document is 311 MB for MAME 0.289, so it is never held whole: the
reader keeps a sliding window just large enough to close the element it
is in the middle of.
"""
found: dict[str, tuple[str | None, list[dict]]] = {}
buffer = b""
while chunk := stream.read(LISTXML_CHUNK):
buffer += chunk
position = 0
while True:
match = _MACHINE.search(buffer, position)
if not match:
break
end = buffer.find(b"</machine>", match.end())
if end == -1:
break
name = match.group(1).decode()
if selector(name):
parent = _ROMOF.search(match.group(2).decode())
found[name] = (
parent.group(1) if parent else None,
_machine_roms(buffer[match.end() : end]),
)
position = end + len(b"</machine>")
buffer = buffer[position:] if position else buffer[-(2 * LISTXML_CHUNK) :]
return found
def listxml_entries(
open_stream, label: str, keep: set[str] | None
) -> list[dict]:
"""Recipes from a MAME -listxml document, parents resolved.
A non-merged archive carries its parent set's ROMs as well as its own, so
a machine with ``romof`` is read as the union, its own members winning a
name collision. Two passes: the parents a set needs are only known once
that set has been seen.
"""
with open_stream() as stream:
wanted = _scan_listxml(
stream, lambda name: keep is None or f"{name}.zip" in keep
)
requested = set(wanted)
parents = {
parent for parent, _roms in wanted.values() if parent and parent not in wanted
}
if parents:
with open_stream() as stream:
for name, value in _scan_listxml(stream, parents.__contains__).items():
wanted.setdefault(name, value)
entries: list[dict] = []
for name, (parent, roms) in sorted(wanted.items()):
# A parent read only to complete a child is not itself an entry.
if name not in requested:
continue
members = {member["name"]: member for member in wanted.get(parent, (None, []))[1]} if parent else {}
members.update({member["name"]: member for member in roms})
if not members:
continue
entries.append(
{
"dat": label,
"name": f"{name}.zip",
"set": name,
"description": f"parent {parent}" if parent else name,
"members": sorted(members.values(), key=lambda m: m["name"]),
}
)
return entries
def _recipe_key(entry: dict) -> tuple[str, str]:
"""Identity of a recipe: which archive, and exactly which members."""
members = json.dumps(entry.get("members", []), sort_keys=True)
return entry.get("name", ""), hashlib.sha1(members.encode()).hexdigest()
def compact_entries(entries: list[dict]) -> list[dict]:
"""Collapse identical recipes shared by several versions.
Most sets do not change between MAME releases: 22 imported versions give
23,679 entries but only 1,726 distinct recipes. Each is stored once, with
``dats`` listing every version that agrees and ``dat`` naming the earliest,
so a match reads as "unchanged since that version" rather than "is that
version".
"""
grouped: dict[tuple[str, str], dict] = {}
for entry in entries:
key = _recipe_key(entry)
existing = grouped.get(key)
labels = entry.get("dats") or ([entry["dat"]] if entry.get("dat") else [])
if existing is None:
merged = dict(entry)
merged["dats"] = sorted(set(labels))
grouped[key] = merged
else:
existing["dats"] = sorted(set(existing["dats"]) | set(labels))
for entry in grouped.values():
entry["dat"] = entry["dats"][0]
return sorted(grouped.values(), key=lambda e: (e["name"], e["dat"]))
def merge_snapshot(
output: str, source: str, dats: dict[str, str], entries: list[dict]
) -> bool:
"""Accumulate recipes across versions instead of replacing them.
A platform pins the archive of whichever version its list was built
against, so the snapshot is a growing library of versions, not a picture
of the newest one.
"""
path = Path(output)
known_entries: list[dict] = []
known_dats: dict[str, str] = {}
if path.is_file():
with path.open(encoding="utf-8") as handle:
existing = json.load(handle)
known_entries = list(existing.get("entries", []))
known_dats = dict(existing.get("dats", {}))
known_dats.update(dats)
return write_provenance_snapshot(
output,
source,
datetime.now(timezone.utc).strftime("%Y-%m-%d"),
known_dats,
compact_entries(known_entries + entries),
)
def _iter_dat_contents(pack: Path):
"""Yield (member name, content bytes) for DAT files in *pack*."""
if pack.is_dir():
for path in sorted(pack.rglob("*")):
if path.suffix.lower() in (".dat", ".xml") and path.is_file():
yield path.name, path.read_bytes()
return
if pack.suffix.lower() == ".zip":
with zipfile.ZipFile(pack) as archive:
for info in sorted(archive.infolist(), key=lambda i: i.filename):
if info.is_dir() or info.file_size > MAX_MEMBER_SIZE:
continue
if Path(info.filename).suffix.lower() not in (".dat", ".xml"):
continue
yield info.filename, archive.read(info)
return
yield pack.name, pack.read_bytes()
MAME_LISTXML_URL = (
"https://github.com/mamedev/mame/releases/download/{tag}/{tag}lx.zip"
)
FBNEO_DATS_API = "https://api.github.com/repos/libretro/FBNeo/contents/dats"
def _download(url: str, destination: Path) -> Path:
"""Fetch *url* to *destination* unless it is already there."""
if destination.is_file():
return destination
destination.parent.mkdir(parents=True, exist_ok=True)
request = urllib.request.Request(url, headers={"User-Agent": "retrobios"})
scratch = destination.with_suffix(destination.suffix + f".{os.getpid()}.part")
try:
with urllib.request.urlopen(request, timeout=300) as response, scratch.open(
"wb"
) as handle:
shutil.copyfileobj(response, handle)
scratch.replace(destination)
finally:
if scratch.exists():
scratch.unlink()
return destination
def fetch_pack(source: str, tag: str, cache_dir: Path) -> Path:
"""Download the upstream DAT for *source*.
MAME ships its ``-listxml`` output as a release asset, and FBNeo keeps its
DATs in the repository, so neither needs a browser.
"""
if source == "mame":
return _download(
MAME_LISTXML_URL.format(tag=tag), cache_dir / f"{tag}lx.zip"
)
request = urllib.request.Request(
FBNEO_DATS_API, headers={"User-Agent": "retrobios"}
)
with urllib.request.urlopen(request, timeout=60) as response:
listing = json.load(response)
target = cache_dir / "fbneo-dats"
target.mkdir(parents=True, exist_ok=True)
for item in listing:
if item.get("type") == "file" and item["name"].lower().endswith(".dat"):
_download(item["download_url"], target / item["name"])
return target
def _is_listxml(pack: Path) -> bool:
"""Whether *pack* holds a MAME -listxml document rather than a DAT."""
if pack.is_dir():
return False
if pack.suffix.lower() == ".zip":
with zipfile.ZipFile(pack) as archive:
names = [n for n in archive.namelist() if n.lower().endswith(".xml")]
if not names:
return False
with archive.open(names[0]) as handle:
return b"<!ELEMENT mame" in handle.read(4096)
with pack.open("rb") as handle:
return b"<!ELEMENT mame" in handle.read(4096)
def _listxml_opener(pack: Path):
"""A callable giving a fresh byte stream over the listxml document."""
if pack.suffix.lower() == ".zip":
def opener():
archive = zipfile.ZipFile(pack)
name = next(n for n in archive.namelist() if n.lower().endswith(".xml"))
stream = archive.open(name)
stream.close_archive = archive # keep the archive alive
return stream
return opener
return lambda: pack.open("rb")
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--source", choices=SOURCES, required=True)
parser.add_argument("--pack", help="DAT file, ZIP or directory")
parser.add_argument(
"--fetch",
nargs="?",
const="mame0289",
help="download upstream instead of using --pack (MAME tag, e.g. mame0289)",
)
parser.add_argument("--cache-dir", default=".cache/dats")
parser.add_argument("--output", default="")
parser.add_argument("--emulators-dir", default="emulators")
parser.add_argument("--platforms-dir", default="platforms")
parser.add_argument(
"--all", action="store_true", help="keep every set, not only referenced ones"
)
parser.add_argument("--dry-run", action="store_true")
args = parser.parse_args()
if args.fetch:
pack = fetch_pack(args.source, args.fetch, Path(args.cache_dir))
print(f" Recupere {pack}")
elif args.pack:
pack = Path(args.pack)
else:
parser.error("either --pack or --fetch is required")
if not pack.exists():
print(f"ERROR: {pack} does not exist", file=sys.stderr)
return 1
output = args.output or DEFAULT_OUTPUT.format(source=args.source)
Path(output).parent.mkdir(parents=True, exist_ok=True)
keep = (
None
if args.all
else relevant_archives(args.emulators_dir, args.platforms_dir)
)
dats: dict[str, str] = {}
entries: list[dict] = []
skipped = 0
if _is_listxml(pack):
label = f"MAME {pack.stem.replace('lx', '')}"
entries = listxml_entries(_listxml_opener(pack), label, keep)
dats[label] = pack.stem
print(f" {label}: {len(entries)} sets retenus")
members = sum(len(entry["members"]) for entry in entries)
print(f"{len(entries)} recettes, {members} membres, depuis 1 listxml")
if args.dry_run:
return 0
changed = merge_snapshot(output, args.source, dats, entries)
print(f"{'Wrote' if changed else 'Unchanged'} {output}")
return 0
for member, content in _iter_dat_contents(pack):
try:
dat = parse_logiqx(content)
except (ValueError, SyntaxError) as exc:
print(f" Skipped {member}: {exc}", file=sys.stderr)
skipped += 1
continue
if not dat.roms:
skipped += 1
continue
found = recipe_entries(dat, keep)
if not found:
continue
dats[dat.name or member] = dat.version
entries.extend(found)
print(f" {dat.name or member}: {len(found)} sets retenus")
if skipped:
print(f" {skipped} fichier(s) ignore(s) (non-Logiqx ou vides)")
members = sum(len(entry["members"]) for entry in entries)
print(f"{len(entries)} recettes, {members} membres, depuis {len(dats)} DAT")
if args.dry_run:
return 0
changed = merge_snapshot(output, args.source, dats, entries)
print(f"{'Wrote' if changed else 'Unchanged'} {output}")
return 0
if __name__ == "__main__":
sys.exit(main())