feat: make the native export round-trip a platform file

This commit is contained in:
Abdessamad Derraz committed 2026-09-05 17:32:48 +02:00
1 parent 7b93285e1d
commit 691ccbfca7
65 files changed
+18417 -2115

No files matched your search

+253 -74
View File
@@ -1,40 +1,223 @@
"""Export truth data to native platform formats."""
#!/usr/bin/env python3
"""Rewrite each platform's own BIOS file, corrected by the ground truth.
The output is the file the platform maintains, not a rendering of our data
in its syntax: what the truth can prove is applied, what it says nothing
about is left alone, and what it knows and the platform lacks is added. The
formats that carry code are patched rather than regenerated, so the checker
a platform ships keeps working.
Usage:
python scripts/export_native.py --all --fetch
python scripts/export_native.py --platform recalbox --upstream-dir up/
"""
from __future__ import annotations
import argparse
import sys
import urllib.error
import urllib.request
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
import yaml
from common import list_registered_platforms, load_platform_config, yaml_load
from exporter import discover_exporters
from exporter.baseline import build_native_model
OUTPUT_FILENAMES: dict[str, str] = {
"retroarch": "System.dat",
"lakka": "System.dat",
"retropie": "System.dat",
"batocera": "batocera-systems",
"recalbox": "es_bios.xml",
"retrobat": "batocera-systems.json",
"emudeck": "checkBIOS.sh",
"retrodeck": "component_manifest.json",
"romm": "known_bios_files.json",
}
DEFAULT_CACHE = ".cache/upstream-native"
_USER_AGENT = "retrobios-exporter/1.0"
_MAX_BYTES = 64 * 1024 * 1024
def output_path(platform: str, output_dir: str) -> str:
"""Return the full output path for a platform's native export.
def fetch(url: str, destination: Path) -> bytes:
"""Download an original once, then read it from the cache."""
if destination.exists():
return destination.read_bytes()
request = urllib.request.Request(url, headers={"User-Agent": _USER_AGENT})
with urllib.request.urlopen(request, timeout=60) as response:
payload = response.read(_MAX_BYTES + 1)
if len(payload) > _MAX_BYTES:
raise ValueError(f"{url}: response larger than {_MAX_BYTES} bytes")
destination.parent.mkdir(parents=True, exist_ok=True)
destination.write_bytes(payload)
return payload
Each platform gets its own subdirectory to avoid filename collisions
(e.g. retroarch, lakka, retropie all produce System.dat).
def pinned_base(wanted: dict[str, str], scraped: dict | None) -> str:
"""The revision our transcription came from, when the YAML records it.
A scraper pins a stable tag (batocera-43.1, BizHawk 2.11.1) and writes
that URL into the platform YAML. Patching the branch tip instead would
correct a file our data never described.
"""
filename = OUTPUT_FILENAMES.get(platform, f"{platform}_bios.dat")
plat_dir = Path(output_dir) / platform
plat_dir.mkdir(parents=True, exist_ok=True)
return str(plat_dir / filename)
source = _raw_url(str((scraped or {}).get("source", "")))
for relative in wanted:
if source.endswith("/" + relative):
return source[: -len(relative)]
return ""
def _raw_url(url: str) -> str:
"""A GitHub blob page names a revision but serves HTML; raw serves bytes."""
marker = "/blob/"
if url.startswith("https://github.com/") and marker in url:
owner_repo, _, path = url[len("https://github.com/") :].partition(marker)
return f"https://raw.githubusercontent.com/{owner_repo}/{path}"
return url
def collect_originals(
exporter: object,
systems: dict,
upstream_dir: Path,
allow_fetch: bool,
scraped: dict | None = None,
) -> tuple[dict[str, str], list[str]]:
"""Gather the platform's own files, from disk or from upstream."""
wanted = dict(exporter.native_sources())
components = getattr(exporter, "components", None)
if callable(components):
for component in components(systems):
wanted[f"{component}/component_manifest.json"] = exporter.component_url(
component
)
base = pinned_base(wanted, scraped)
if base:
wanted = {relative: base + relative for relative in wanted}
root = upstream_dir / exporter.platform_name()
originals: dict[str, str] = {}
missing: list[str] = []
for relative, url in wanted.items():
path = root / relative
payload: bytes | None = None
if path.exists():
payload = path.read_bytes()
elif allow_fetch:
try:
payload = fetch(url, path)
except (urllib.error.URLError, urllib.error.HTTPError, OSError) as exc:
missing.append(f"{relative}: {exc}")
continue
else:
missing.append(f"{relative}: absent and fetching is off")
continue
unpack = getattr(exporter, "unpack", None)
if callable(unpack) and relative.endswith((".zip", ".tar.gz")):
originals.update(unpack(payload))
else:
originals[relative] = payload.decode("utf-8", errors="replace")
return originals, missing
def export_platform(
platform: str,
exporter_class: type,
truth_dir: Path,
output_dir: Path,
platforms_dir: str,
upstream_dir: Path,
allow_fetch: bool,
) -> tuple[bool, list[str]]:
"""Write one platform's corrected file. Returns (ok, messages)."""
messages: list[str] = []
truth_file = truth_dir / f"{platform}.yml"
truth: dict = {}
if truth_file.exists():
with open(truth_file) as handle:
truth = yaml_load(handle) or {}
else:
messages.append(f"no truth for {platform}, only the platform's own data")
try:
scraped = load_platform_config(platform, platforms_dir)
except (FileNotFoundError, OSError):
scraped = None
systems, report = build_native_model(truth, scraped)
if not systems:
return False, ["nothing to write: neither the platform nor the truth has data"]
exporter = exporter_class()
originals, missing = collect_originals(
exporter, systems, upstream_dir, allow_fetch, scraped
)
if missing and exporter.needs_original():
return False, [f"the platform's own file is required: {m}" for m in missing]
messages.extend(
f"original unavailable, written from our data: {m}" for m in missing
)
try:
produced = exporter.render(systems, report, originals, scraped)
except ValueError as exc:
return False, [str(exc)]
if not produced and not exporter.may_write_nothing():
return False, ["the exporter produced no file"]
issues = exporter.validate(systems, produced)
platform_dir = output_dir / platform
for relative, content in produced.items():
path = platform_dir / relative
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(content, encoding="utf-8")
# Only what the format can state: a correction to a field the file has
# no place for is not a change the maintainer will find in the diff, and
# counting it would announce work the export did not do.
carried = exporter.carries()
applied = [
correction
for correction in report.hashes_corrected
if correction.rsplit(" ", 1)[-1] in carried
]
requirements = len(report.required_corrected) if "required" in carried else 0
landed = 0
refused = 0
lost = 0
for system in systems.values():
for entry in system.files:
if not entry.name:
continue
if exporter.writable(entry):
landed += entry.platform is None
elif entry.platform is None:
refused += 1
else:
# The platform declares it and the format still cannot state
# it, so their own file loses a line. Worth saying out loud.
lost += 1
summary = exporter.outcome(systems, produced)
if summary is None:
summary = (
f"{report.files_kept} kept, {landed} added, "
f"{len(applied)} hashes corrected, {requirements} requirements corrected"
)
if refused:
summary += f", {refused} the format cannot state"
if lost:
summary += f", {lost} of theirs dropped"
messages.append(summary)
for correction in applied[:5]:
messages.append(f"hash corrected: {correction}")
if len(applied) > 5:
messages.append(f"and {len(applied) - 5} more hash corrections")
messages.extend(f"INVALID: {issue}" for issue in issues[:10])
if len(issues) > 10:
messages.append(f"and {len(issues) - 10} more validation failures")
return not issues, messages
def run(
@@ -42,87 +225,83 @@ def run(
truth_dir: str,
output_dir: str,
platforms_dir: str,
upstream_dir: str,
allow_fetch: bool,
) -> int:
"""Export truth to native formats, return exit code."""
exporters = discover_exporters()
errors = 0
failures = 0
skipped: list[str] = []
for platform in sorted(platforms):
exporter_cls = exporters.get(platform)
if not exporter_cls:
print(f" SKIP {platform}: no exporter available")
exporter_class = exporters.get(platform)
if not exporter_class:
skipped.append(platform)
print(f" SKIP {platform}: no exporter")
continue
truth_file = Path(truth_dir) / f"{platform}.yml"
if not truth_file.exists():
print(f" SKIP {platform}: {truth_file} not found")
continue
ok, messages = export_platform(
platform,
exporter_class,
Path(truth_dir),
Path(output_dir),
platforms_dir,
Path(upstream_dir),
allow_fetch,
)
label = "OK " if ok else "FAIL"
print(f" {label} {platform}")
for message in messages:
print(f" {message}")
if not ok:
failures += 1
with open(truth_file) as f:
truth_data = yaml_load(f) or {}
if skipped:
print(f"\n{len(skipped)} platform(s) without an exporter: {', '.join(skipped)}")
failures += len(skipped)
scraped: dict | None = None
try:
scraped = load_platform_config(platform, platforms_dir)
except (FileNotFoundError, OSError):
pass
dest = output_path(platform, output_dir)
exporter = exporter_cls()
exporter.export(truth_data, dest, scraped_data=scraped)
issues = exporter.validate(truth_data, dest)
if issues:
print(f" WARN {platform}: {len(issues)} validation issue(s)")
for issue in issues:
print(f" {issue}")
errors += 1
else:
print(f" OK {platform} -> {dest}")
return 1 if errors else 0
return 1 if failures else 0
def main() -> None:
parser = argparse.ArgumentParser(
description="Export truth data to native platform formats.",
description="Rewrite each platform's own BIOS file, corrected.",
)
group = parser.add_mutually_exclusive_group(required=True)
group.add_argument("--all", action="store_true", help="export all platforms")
group.add_argument("--platform", help="export a single platform")
group.add_argument("--all", action="store_true", help="every platform")
group.add_argument("--platform", help="a single platform")
parser.add_argument("--output-dir", default="dist/upstream")
parser.add_argument("--truth-dir", default="dist/truth")
parser.add_argument("--platforms-dir", default="platforms")
parser.add_argument(
"--output-dir",
default="dist/upstream",
help="output directory",
"--upstream-dir",
default=DEFAULT_CACHE,
help="where the platforms' own files are read and cached",
)
parser.add_argument(
"--truth-dir",
default="dist/truth",
help="truth YAML directory",
)
parser.add_argument(
"--platforms-dir",
default="platforms",
help="platform configs directory",
)
parser.add_argument(
"--include-archived",
"--fetch",
action="store_true",
help="include archived platforms",
help="download a platform's file when it is not in the cache",
)
args = parser.parse_args()
if args.all:
platforms = list_registered_platforms(
args.platforms_dir,
include_archived=args.include_archived,
include_archived=True,
)
else:
platforms = [args.platform]
code = run(platforms, args.truth_dir, args.output_dir, args.platforms_dir)
sys.exit(code)
sys.exit(
run(
platforms,
args.truth_dir,
args.output_dir,
args.platforms_dir,
args.upstream_dir,
args.fetch,
)
)
if __name__ == "__main__":