mirror of
https://github.com/Abdess/retroarch_system.git
synced 2026-10-10 21:43:23 -05:00
feat: make the native export round-trip a platform file
This commit is contained in:
1 parent
7b93285e1d
commit
691ccbfca7
65 files changed
+18417
-2115
No files matched your search
@@ -168,12 +168,11 @@ def main() -> None:
|
||||
)
|
||||
parser.add_argument("--truth-dir", default="dist/truth")
|
||||
parser.add_argument("--platforms-dir", default="platforms")
|
||||
parser.add_argument("--include-archived", action="store_true")
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.all:
|
||||
platforms = list_registered_platforms(
|
||||
args.platforms_dir, include_archived=args.include_archived
|
||||
args.platforms_dir, include_archived=True
|
||||
)
|
||||
else:
|
||||
platforms = [args.platform]
|
||||
|
||||
+253
-74
@@ -1,40 +1,223 @@
|
||||
"""Export truth data to native platform formats."""
|
||||
#!/usr/bin/env python3
|
||||
"""Rewrite each platform's own BIOS file, corrected by the ground truth.
|
||||
|
||||
The output is the file the platform maintains, not a rendering of our data
|
||||
in its syntax: what the truth can prove is applied, what it says nothing
|
||||
about is left alone, and what it knows and the platform lacks is added. The
|
||||
formats that carry code are patched rather than regenerated, so the checker
|
||||
a platform ships keeps working.
|
||||
|
||||
Usage:
|
||||
python scripts/export_native.py --all --fetch
|
||||
python scripts/export_native.py --platform recalbox --upstream-dir up/
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import sys
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
||||
|
||||
import yaml
|
||||
from common import list_registered_platforms, load_platform_config, yaml_load
|
||||
from exporter import discover_exporters
|
||||
from exporter.baseline import build_native_model
|
||||
|
||||
OUTPUT_FILENAMES: dict[str, str] = {
|
||||
"retroarch": "System.dat",
|
||||
"lakka": "System.dat",
|
||||
"retropie": "System.dat",
|
||||
"batocera": "batocera-systems",
|
||||
"recalbox": "es_bios.xml",
|
||||
"retrobat": "batocera-systems.json",
|
||||
"emudeck": "checkBIOS.sh",
|
||||
"retrodeck": "component_manifest.json",
|
||||
"romm": "known_bios_files.json",
|
||||
}
|
||||
DEFAULT_CACHE = ".cache/upstream-native"
|
||||
_USER_AGENT = "retrobios-exporter/1.0"
|
||||
_MAX_BYTES = 64 * 1024 * 1024
|
||||
|
||||
|
||||
def output_path(platform: str, output_dir: str) -> str:
|
||||
"""Return the full output path for a platform's native export.
|
||||
def fetch(url: str, destination: Path) -> bytes:
|
||||
"""Download an original once, then read it from the cache."""
|
||||
if destination.exists():
|
||||
return destination.read_bytes()
|
||||
request = urllib.request.Request(url, headers={"User-Agent": _USER_AGENT})
|
||||
with urllib.request.urlopen(request, timeout=60) as response:
|
||||
payload = response.read(_MAX_BYTES + 1)
|
||||
if len(payload) > _MAX_BYTES:
|
||||
raise ValueError(f"{url}: response larger than {_MAX_BYTES} bytes")
|
||||
destination.parent.mkdir(parents=True, exist_ok=True)
|
||||
destination.write_bytes(payload)
|
||||
return payload
|
||||
|
||||
Each platform gets its own subdirectory to avoid filename collisions
|
||||
(e.g. retroarch, lakka, retropie all produce System.dat).
|
||||
|
||||
def pinned_base(wanted: dict[str, str], scraped: dict | None) -> str:
|
||||
"""The revision our transcription came from, when the YAML records it.
|
||||
|
||||
A scraper pins a stable tag (batocera-43.1, BizHawk 2.11.1) and writes
|
||||
that URL into the platform YAML. Patching the branch tip instead would
|
||||
correct a file our data never described.
|
||||
"""
|
||||
filename = OUTPUT_FILENAMES.get(platform, f"{platform}_bios.dat")
|
||||
plat_dir = Path(output_dir) / platform
|
||||
plat_dir.mkdir(parents=True, exist_ok=True)
|
||||
return str(plat_dir / filename)
|
||||
source = _raw_url(str((scraped or {}).get("source", "")))
|
||||
for relative in wanted:
|
||||
if source.endswith("/" + relative):
|
||||
return source[: -len(relative)]
|
||||
return ""
|
||||
|
||||
|
||||
def _raw_url(url: str) -> str:
|
||||
"""A GitHub blob page names a revision but serves HTML; raw serves bytes."""
|
||||
marker = "/blob/"
|
||||
if url.startswith("https://github.com/") and marker in url:
|
||||
owner_repo, _, path = url[len("https://github.com/") :].partition(marker)
|
||||
return f"https://raw.githubusercontent.com/{owner_repo}/{path}"
|
||||
return url
|
||||
|
||||
|
||||
def collect_originals(
|
||||
exporter: object,
|
||||
systems: dict,
|
||||
upstream_dir: Path,
|
||||
allow_fetch: bool,
|
||||
scraped: dict | None = None,
|
||||
) -> tuple[dict[str, str], list[str]]:
|
||||
"""Gather the platform's own files, from disk or from upstream."""
|
||||
wanted = dict(exporter.native_sources())
|
||||
components = getattr(exporter, "components", None)
|
||||
if callable(components):
|
||||
for component in components(systems):
|
||||
wanted[f"{component}/component_manifest.json"] = exporter.component_url(
|
||||
component
|
||||
)
|
||||
|
||||
base = pinned_base(wanted, scraped)
|
||||
if base:
|
||||
wanted = {relative: base + relative for relative in wanted}
|
||||
|
||||
root = upstream_dir / exporter.platform_name()
|
||||
originals: dict[str, str] = {}
|
||||
missing: list[str] = []
|
||||
for relative, url in wanted.items():
|
||||
path = root / relative
|
||||
payload: bytes | None = None
|
||||
if path.exists():
|
||||
payload = path.read_bytes()
|
||||
elif allow_fetch:
|
||||
try:
|
||||
payload = fetch(url, path)
|
||||
except (urllib.error.URLError, urllib.error.HTTPError, OSError) as exc:
|
||||
missing.append(f"{relative}: {exc}")
|
||||
continue
|
||||
else:
|
||||
missing.append(f"{relative}: absent and fetching is off")
|
||||
continue
|
||||
|
||||
unpack = getattr(exporter, "unpack", None)
|
||||
if callable(unpack) and relative.endswith((".zip", ".tar.gz")):
|
||||
originals.update(unpack(payload))
|
||||
else:
|
||||
originals[relative] = payload.decode("utf-8", errors="replace")
|
||||
|
||||
return originals, missing
|
||||
|
||||
|
||||
def export_platform(
|
||||
platform: str,
|
||||
exporter_class: type,
|
||||
truth_dir: Path,
|
||||
output_dir: Path,
|
||||
platforms_dir: str,
|
||||
upstream_dir: Path,
|
||||
allow_fetch: bool,
|
||||
) -> tuple[bool, list[str]]:
|
||||
"""Write one platform's corrected file. Returns (ok, messages)."""
|
||||
messages: list[str] = []
|
||||
|
||||
truth_file = truth_dir / f"{platform}.yml"
|
||||
truth: dict = {}
|
||||
if truth_file.exists():
|
||||
with open(truth_file) as handle:
|
||||
truth = yaml_load(handle) or {}
|
||||
else:
|
||||
messages.append(f"no truth for {platform}, only the platform's own data")
|
||||
|
||||
try:
|
||||
scraped = load_platform_config(platform, platforms_dir)
|
||||
except (FileNotFoundError, OSError):
|
||||
scraped = None
|
||||
|
||||
systems, report = build_native_model(truth, scraped)
|
||||
if not systems:
|
||||
return False, ["nothing to write: neither the platform nor the truth has data"]
|
||||
|
||||
exporter = exporter_class()
|
||||
originals, missing = collect_originals(
|
||||
exporter, systems, upstream_dir, allow_fetch, scraped
|
||||
)
|
||||
if missing and exporter.needs_original():
|
||||
return False, [f"the platform's own file is required: {m}" for m in missing]
|
||||
messages.extend(
|
||||
f"original unavailable, written from our data: {m}" for m in missing
|
||||
)
|
||||
|
||||
try:
|
||||
produced = exporter.render(systems, report, originals, scraped)
|
||||
except ValueError as exc:
|
||||
return False, [str(exc)]
|
||||
|
||||
if not produced and not exporter.may_write_nothing():
|
||||
return False, ["the exporter produced no file"]
|
||||
|
||||
issues = exporter.validate(systems, produced)
|
||||
|
||||
platform_dir = output_dir / platform
|
||||
for relative, content in produced.items():
|
||||
path = platform_dir / relative
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(content, encoding="utf-8")
|
||||
|
||||
# Only what the format can state: a correction to a field the file has
|
||||
# no place for is not a change the maintainer will find in the diff, and
|
||||
# counting it would announce work the export did not do.
|
||||
carried = exporter.carries()
|
||||
applied = [
|
||||
correction
|
||||
for correction in report.hashes_corrected
|
||||
if correction.rsplit(" ", 1)[-1] in carried
|
||||
]
|
||||
requirements = len(report.required_corrected) if "required" in carried else 0
|
||||
|
||||
landed = 0
|
||||
refused = 0
|
||||
lost = 0
|
||||
for system in systems.values():
|
||||
for entry in system.files:
|
||||
if not entry.name:
|
||||
continue
|
||||
if exporter.writable(entry):
|
||||
landed += entry.platform is None
|
||||
elif entry.platform is None:
|
||||
refused += 1
|
||||
else:
|
||||
# The platform declares it and the format still cannot state
|
||||
# it, so their own file loses a line. Worth saying out loud.
|
||||
lost += 1
|
||||
|
||||
summary = exporter.outcome(systems, produced)
|
||||
if summary is None:
|
||||
summary = (
|
||||
f"{report.files_kept} kept, {landed} added, "
|
||||
f"{len(applied)} hashes corrected, {requirements} requirements corrected"
|
||||
)
|
||||
if refused:
|
||||
summary += f", {refused} the format cannot state"
|
||||
if lost:
|
||||
summary += f", {lost} of theirs dropped"
|
||||
messages.append(summary)
|
||||
for correction in applied[:5]:
|
||||
messages.append(f"hash corrected: {correction}")
|
||||
if len(applied) > 5:
|
||||
messages.append(f"and {len(applied) - 5} more hash corrections")
|
||||
|
||||
messages.extend(f"INVALID: {issue}" for issue in issues[:10])
|
||||
if len(issues) > 10:
|
||||
messages.append(f"and {len(issues) - 10} more validation failures")
|
||||
|
||||
return not issues, messages
|
||||
|
||||
|
||||
def run(
|
||||
@@ -42,87 +225,83 @@ def run(
|
||||
truth_dir: str,
|
||||
output_dir: str,
|
||||
platforms_dir: str,
|
||||
upstream_dir: str,
|
||||
allow_fetch: bool,
|
||||
) -> int:
|
||||
"""Export truth to native formats, return exit code."""
|
||||
exporters = discover_exporters()
|
||||
|
||||
errors = 0
|
||||
failures = 0
|
||||
skipped: list[str] = []
|
||||
|
||||
for platform in sorted(platforms):
|
||||
exporter_cls = exporters.get(platform)
|
||||
if not exporter_cls:
|
||||
print(f" SKIP {platform}: no exporter available")
|
||||
exporter_class = exporters.get(platform)
|
||||
if not exporter_class:
|
||||
skipped.append(platform)
|
||||
print(f" SKIP {platform}: no exporter")
|
||||
continue
|
||||
|
||||
truth_file = Path(truth_dir) / f"{platform}.yml"
|
||||
if not truth_file.exists():
|
||||
print(f" SKIP {platform}: {truth_file} not found")
|
||||
continue
|
||||
ok, messages = export_platform(
|
||||
platform,
|
||||
exporter_class,
|
||||
Path(truth_dir),
|
||||
Path(output_dir),
|
||||
platforms_dir,
|
||||
Path(upstream_dir),
|
||||
allow_fetch,
|
||||
)
|
||||
label = "OK " if ok else "FAIL"
|
||||
print(f" {label} {platform}")
|
||||
for message in messages:
|
||||
print(f" {message}")
|
||||
if not ok:
|
||||
failures += 1
|
||||
|
||||
with open(truth_file) as f:
|
||||
truth_data = yaml_load(f) or {}
|
||||
if skipped:
|
||||
print(f"\n{len(skipped)} platform(s) without an exporter: {', '.join(skipped)}")
|
||||
failures += len(skipped)
|
||||
|
||||
scraped: dict | None = None
|
||||
try:
|
||||
scraped = load_platform_config(platform, platforms_dir)
|
||||
except (FileNotFoundError, OSError):
|
||||
pass
|
||||
|
||||
dest = output_path(platform, output_dir)
|
||||
exporter = exporter_cls()
|
||||
exporter.export(truth_data, dest, scraped_data=scraped)
|
||||
|
||||
issues = exporter.validate(truth_data, dest)
|
||||
if issues:
|
||||
print(f" WARN {platform}: {len(issues)} validation issue(s)")
|
||||
for issue in issues:
|
||||
print(f" {issue}")
|
||||
errors += 1
|
||||
else:
|
||||
print(f" OK {platform} -> {dest}")
|
||||
|
||||
return 1 if errors else 0
|
||||
return 1 if failures else 0
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Export truth data to native platform formats.",
|
||||
description="Rewrite each platform's own BIOS file, corrected.",
|
||||
)
|
||||
group = parser.add_mutually_exclusive_group(required=True)
|
||||
group.add_argument("--all", action="store_true", help="export all platforms")
|
||||
group.add_argument("--platform", help="export a single platform")
|
||||
group.add_argument("--all", action="store_true", help="every platform")
|
||||
group.add_argument("--platform", help="a single platform")
|
||||
parser.add_argument("--output-dir", default="dist/upstream")
|
||||
parser.add_argument("--truth-dir", default="dist/truth")
|
||||
parser.add_argument("--platforms-dir", default="platforms")
|
||||
parser.add_argument(
|
||||
"--output-dir",
|
||||
default="dist/upstream",
|
||||
help="output directory",
|
||||
"--upstream-dir",
|
||||
default=DEFAULT_CACHE,
|
||||
help="where the platforms' own files are read and cached",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--truth-dir",
|
||||
default="dist/truth",
|
||||
help="truth YAML directory",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--platforms-dir",
|
||||
default="platforms",
|
||||
help="platform configs directory",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--include-archived",
|
||||
"--fetch",
|
||||
action="store_true",
|
||||
help="include archived platforms",
|
||||
help="download a platform's file when it is not in the cache",
|
||||
)
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.all:
|
||||
platforms = list_registered_platforms(
|
||||
args.platforms_dir,
|
||||
include_archived=args.include_archived,
|
||||
include_archived=True,
|
||||
)
|
||||
else:
|
||||
platforms = [args.platform]
|
||||
|
||||
code = run(platforms, args.truth_dir, args.output_dir, args.platforms_dir)
|
||||
sys.exit(code)
|
||||
sys.exit(
|
||||
run(
|
||||
platforms,
|
||||
args.truth_dir,
|
||||
args.output_dir,
|
||||
args.platforms_dir,
|
||||
args.upstream_dir,
|
||||
args.fetch,
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -1,81 +1,166 @@
|
||||
"""Abstract base class for platform exporters."""
|
||||
"""Contract shared by the platform exporters.
|
||||
|
||||
An exporter answers one question: what would this platform's own file look
|
||||
like if it were corrected. It is handed the platform's file when we have it,
|
||||
because several of these formats carry code, and a generator that emits only
|
||||
the data block hands the maintainer something that no longer runs.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from abc import ABC, abstractmethod
|
||||
|
||||
from .baseline import NativeFile, NativeSystem, Report
|
||||
|
||||
|
||||
class BaseExporter(ABC):
|
||||
"""Base class for exporting truth data to native platform formats."""
|
||||
"""Base class for writing a platform's own BIOS file, corrected."""
|
||||
|
||||
@staticmethod
|
||||
@abstractmethod
|
||||
def platform_name() -> str:
|
||||
"""Return the platform identifier this exporter targets."""
|
||||
|
||||
@staticmethod
|
||||
@abstractmethod
|
||||
def export(
|
||||
def native_filename() -> str:
|
||||
"""Return the name the platform gives this file."""
|
||||
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
"""Fields this format has somewhere to put.
|
||||
|
||||
A correction the file cannot state is not a correction it delivers,
|
||||
and counting it would announce a change the maintainer will not find
|
||||
in the diff.
|
||||
"""
|
||||
return frozenset({"md5"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
"""Return {relative path: URL} of the originals this exporter patches.
|
||||
|
||||
An exporter that can build its file from nothing returns an empty
|
||||
mapping. One whose format carries code cannot, and names the file it
|
||||
needs so the caller can fetch it.
|
||||
"""
|
||||
return {}
|
||||
|
||||
@staticmethod
|
||||
def needs_original() -> bool:
|
||||
"""Whether a faithful export requires the platform's own file."""
|
||||
return False
|
||||
|
||||
@staticmethod
|
||||
def may_write_nothing() -> bool:
|
||||
"""Whether producing no file is an outcome rather than a failure.
|
||||
|
||||
True only where the unit is a correction rather than a document:
|
||||
RetroPie has no BIOS list to rewrite, so a run with nothing to
|
||||
correct writes nothing and is right to.
|
||||
"""
|
||||
return False
|
||||
|
||||
@abstractmethod
|
||||
def render(
|
||||
self,
|
||||
truth_data: dict,
|
||||
output_path: str,
|
||||
scraped_data: dict | None = None,
|
||||
) -> None:
|
||||
"""Export truth data to the native platform format."""
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
"""Return {relative output path: file content}."""
|
||||
|
||||
@abstractmethod
|
||||
def validate(self, truth_data: dict, output_path: str) -> list[str]:
|
||||
"""Validate exported file against truth data, return list of issues."""
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
"""Check the produced files against what was asked of them."""
|
||||
|
||||
# Shared helpers
|
||||
|
||||
@classmethod
|
||||
def exportable(
|
||||
cls,
|
||||
systems: dict[str, NativeSystem],
|
||||
require: str = "",
|
||||
) -> list[tuple[NativeSystem, list[NativeFile]]]:
|
||||
"""Systems paired with the files this format can actually carry.
|
||||
|
||||
`require` names a hash the format cannot omit. An entry the format
|
||||
has no way to express is dropped here and counted by the caller,
|
||||
never written as a half entry the platform would read as a file that
|
||||
can never verify.
|
||||
"""
|
||||
result: list[tuple[NativeSystem, list[NativeFile]]] = []
|
||||
for system in systems.values():
|
||||
files = [fe for fe in system.files if fe.name and cls.writable(fe, require)]
|
||||
if files:
|
||||
result.append((system, files))
|
||||
return result
|
||||
|
||||
@staticmethod
|
||||
def _is_pattern(name: str) -> bool:
|
||||
"""Check if a filename is a placeholder pattern (not a real file)."""
|
||||
return "<" in name or ">" in name or "*" in name
|
||||
def requires() -> str:
|
||||
"""The hash this format cannot write a new entry without."""
|
||||
return ""
|
||||
|
||||
@staticmethod
|
||||
def _dest(fe: dict) -> str:
|
||||
"""Get destination path for a file entry, falling back to name."""
|
||||
return fe.get("path") or fe.get("destination") or fe.get("name", "")
|
||||
def can_add() -> bool:
|
||||
"""Whether a file the platform does not declare can be stated at all.
|
||||
|
||||
False where an entry is a declaration in code rather than a row:
|
||||
BizHawk wires each firmware into an option list and a status that
|
||||
only its source expresses, and writing C# for one is not something
|
||||
an exporter can do safely.
|
||||
"""
|
||||
return True
|
||||
|
||||
@classmethod
|
||||
def writable(cls, fe: NativeFile, require: str = "") -> bool:
|
||||
"""Whether this format can carry the entry.
|
||||
|
||||
An entry the platform already declares is always carried, hash or
|
||||
no hash: it is in their file today, and an export that drops it
|
||||
hands back a file poorer than the one it corrects. The requirement
|
||||
only gates what we would be adding.
|
||||
"""
|
||||
if fe.platform is not None:
|
||||
return True
|
||||
if not cls.can_add():
|
||||
return False
|
||||
require = require or cls.requires()
|
||||
return not require or bool(fe.hash(require))
|
||||
|
||||
def outcome(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> str | None:
|
||||
"""What this export did, when counting file entries would not say it.
|
||||
|
||||
Most formats state one entry per file, so the caller's own count is
|
||||
the answer. RetroPie states a sentence per package, and a count of
|
||||
file entries would describe work it did not do.
|
||||
"""
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def _display_name(
|
||||
sys_id: str,
|
||||
scraped_sys: dict | None = None,
|
||||
) -> str:
|
||||
"""Get display name for a system from scraped data or slug."""
|
||||
if scraped_sys:
|
||||
name = scraped_sys.get("name")
|
||||
if name:
|
||||
return name
|
||||
# Fallback: convert slug to display name with acronym handling
|
||||
def display_name(system: NativeSystem) -> str:
|
||||
"""The name the platform shows for a system."""
|
||||
if system.name:
|
||||
return system.name
|
||||
for fe in system.files:
|
||||
native = fe.native("native_name", "")
|
||||
if native:
|
||||
return str(native)
|
||||
_UPPER = {
|
||||
"3do",
|
||||
"cdi",
|
||||
"cpc",
|
||||
"cps1",
|
||||
"cps2",
|
||||
"cps3",
|
||||
"dos",
|
||||
"gba",
|
||||
"gbc",
|
||||
"hle",
|
||||
"msx",
|
||||
"nes",
|
||||
"nds",
|
||||
"ngp",
|
||||
"psp",
|
||||
"psx",
|
||||
"sms",
|
||||
"snes",
|
||||
"stv",
|
||||
"tvc",
|
||||
"vb",
|
||||
"zx",
|
||||
"3do", "cdi", "cpc", "cps1", "cps2", "cps3", "dos", "gba", "gbc",
|
||||
"hle", "msx", "nes", "nds", "ngp", "psp", "psx", "sms", "snes",
|
||||
"stv", "tvc", "vb", "zx",
|
||||
}
|
||||
parts = sys_id.replace("-", " ").split()
|
||||
result = []
|
||||
for p in parts:
|
||||
if p.lower() in _UPPER:
|
||||
result.append(p.upper())
|
||||
else:
|
||||
result.append(p.capitalize())
|
||||
return " ".join(result)
|
||||
parts = system.native_id.replace("-", " ").replace("_", " ").split()
|
||||
return " ".join(
|
||||
p.upper() if p.lower() in _UPPER else p.capitalize() for p in parts
|
||||
)
|
||||
@@ -0,0 +1,373 @@
|
||||
"""Reconciliation of a platform's own declarations with the ground truth.
|
||||
|
||||
An export is not the truth rendered in a native syntax. It is the platform's
|
||||
file, corrected: what the truth can prove is applied, what it says nothing
|
||||
about is left alone, and what it knows and the platform lacks is added. A
|
||||
platform that loses two thirds of its systems to an export cannot use it.
|
||||
|
||||
Systems are keyed by the identifier the platform itself uses. Several of our
|
||||
slugs collapse onto one native id (Recalbox files pcengine, pcenginecd and
|
||||
supergrafx under one slug) and one slug can carry several native ids, so the
|
||||
grouping is rebuilt from the per-file native_system the scrapers record.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from common import _norm_system_id
|
||||
|
||||
HASH_FIELDS = ("sha1", "md5", "sha256", "crc32")
|
||||
|
||||
|
||||
def _hash_values(entry: dict, field_name: str) -> list[str]:
|
||||
"""Return a hash field as a list, whatever shape it was written in."""
|
||||
raw = entry.get(field_name)
|
||||
if not raw:
|
||||
return []
|
||||
if isinstance(raw, list):
|
||||
return [str(v).strip().lower() for v in raw if str(v).strip()]
|
||||
return [v.strip().lower() for v in str(raw).split(",") if v.strip()]
|
||||
|
||||
|
||||
def _is_placeholder(name: str) -> bool:
|
||||
return "<" in name or ">" in name or "*" in name
|
||||
|
||||
|
||||
@dataclass
|
||||
class NativeFile:
|
||||
"""One file as the platform will read it, after correction."""
|
||||
|
||||
name: str
|
||||
destination: str
|
||||
native_system: str
|
||||
platform: dict | None = None
|
||||
truth: dict | None = None
|
||||
corrections: list[str] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def origin(self) -> str:
|
||||
if self.platform is not None and self.truth is not None:
|
||||
return "both"
|
||||
return "platform" if self.platform is not None else "truth"
|
||||
|
||||
def hashes(self, field_name: str) -> list[str]:
|
||||
"""Accepted values for a hash, truth first when it has an opinion.
|
||||
|
||||
The truth is read from the emulator's source; the platform list is a
|
||||
secondary source. When both speak and disagree, the truth decides and
|
||||
the divergence is recorded, never silently merged: an emulator that
|
||||
rejects a file will reject it whatever the platform declares.
|
||||
"""
|
||||
truth_values = _hash_values(self.truth or {}, field_name)
|
||||
platform_values = _hash_values(self.platform or {}, field_name)
|
||||
if truth_values and platform_values and not set(truth_values) & set(
|
||||
platform_values
|
||||
):
|
||||
return truth_values
|
||||
if truth_values:
|
||||
# Keep the platform's extra accepted revisions alongside ours.
|
||||
merged = list(truth_values)
|
||||
merged.extend(v for v in platform_values if v not in merged)
|
||||
return merged
|
||||
return platform_values
|
||||
|
||||
def hash(self, field_name: str) -> str:
|
||||
values = self.hashes(field_name)
|
||||
return values[0] if values else ""
|
||||
|
||||
@property
|
||||
def required(self) -> bool:
|
||||
if self.truth is not None and self.truth.get("required") is not None:
|
||||
return bool(self.truth["required"])
|
||||
if self.platform is not None and self.platform.get("required") is not None:
|
||||
return bool(self.platform["required"])
|
||||
return True
|
||||
|
||||
@property
|
||||
def priority(self) -> int | None:
|
||||
"""Where the code looks for this file, lowest first.
|
||||
|
||||
DuckStation's FindBIOSImageInDirectory keeps the image whose
|
||||
priority is lower, and a search list a core walks in order maps
|
||||
onto it as 1, 2, 3. Where cores disagree this is the best rank any
|
||||
of them gives: nothing is dropped on it, so a file that is some
|
||||
core's first choice is read early. None when nothing states one.
|
||||
"""
|
||||
for entry in (self.truth, self.platform):
|
||||
if entry and entry.get("priority") is not None:
|
||||
return int(entry["priority"])
|
||||
return None
|
||||
|
||||
def size(self) -> int | None:
|
||||
for entry in (self.truth, self.platform):
|
||||
if entry and entry.get("size"):
|
||||
return int(entry["size"])
|
||||
return None
|
||||
|
||||
def native(self, key: str, default: object = "") -> object:
|
||||
"""Read a field the platform declares and we only carry through."""
|
||||
for entry in (self.platform, self.truth):
|
||||
if entry and entry.get(key) not in (None, ""):
|
||||
return entry[key]
|
||||
return default
|
||||
|
||||
def cores(self) -> list[str]:
|
||||
"""Cores that want this file, the platform's naming preferred."""
|
||||
declared = self.native("core", "")
|
||||
if declared:
|
||||
return [c.strip() for c in str(declared).split(",") if c.strip()]
|
||||
if self.truth:
|
||||
return [f"libretro/{c}" for c in self.truth.get("_cores", [])]
|
||||
return []
|
||||
|
||||
|
||||
@dataclass
|
||||
class NativeSystem:
|
||||
"""One system as the platform names it."""
|
||||
|
||||
native_id: str
|
||||
name: str = ""
|
||||
files: list[NativeFile] = field(default_factory=list)
|
||||
from_platform: bool = False
|
||||
|
||||
@property
|
||||
def origin(self) -> str:
|
||||
return "platform" if self.from_platform else "truth"
|
||||
|
||||
|
||||
@dataclass
|
||||
class Report:
|
||||
"""What the reconciliation changed, so the caller can say it out loud."""
|
||||
|
||||
systems_kept: int = 0
|
||||
systems_added: int = 0
|
||||
files_kept: int = 0
|
||||
files_added: int = 0
|
||||
hashes_corrected: list[str] = field(default_factory=list)
|
||||
required_corrected: list[str] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def corrections(self) -> int:
|
||||
return len(self.hashes_corrected) + len(self.required_corrected)
|
||||
|
||||
|
||||
def _native_id_of(sys_key: str, sys_data: dict, file_entry: dict) -> str:
|
||||
"""The platform's own id for the system a file belongs to."""
|
||||
return (
|
||||
file_entry.get("native_system")
|
||||
or sys_data.get("native_id")
|
||||
or sys_key
|
||||
)
|
||||
|
||||
|
||||
def _match_key(entry: dict) -> tuple[str, str]:
|
||||
dest = str(entry.get("destination") or entry.get("path") or entry.get("name", ""))
|
||||
return dest.casefold(), str(entry.get("name", "")).casefold()
|
||||
|
||||
|
||||
def build_native_model(
|
||||
truth: dict,
|
||||
scraped: dict | None,
|
||||
) -> tuple[dict[str, NativeSystem], Report]:
|
||||
"""Rebuild the platform's systems, corrected by the truth.
|
||||
|
||||
Returns the systems keyed by native id, in the platform's own order
|
||||
first and truth-only additions after, plus what changed.
|
||||
"""
|
||||
report = Report()
|
||||
systems: dict[str, NativeSystem] = {}
|
||||
|
||||
scraped_systems = (scraped or {}).get("systems", {})
|
||||
|
||||
# Pass 1: the platform's own file, grouped as the platform groups it.
|
||||
for sys_key, sys_data in scraped_systems.items():
|
||||
for file_entry in sys_data.get("files", []):
|
||||
native_id = _native_id_of(sys_key, sys_data, file_entry)
|
||||
system = systems.get(native_id)
|
||||
if system is None:
|
||||
system = NativeSystem(
|
||||
native_id=native_id,
|
||||
name=str(file_entry.get("native_name") or sys_data.get("name", "")),
|
||||
from_platform=True,
|
||||
)
|
||||
systems[native_id] = system
|
||||
report.systems_kept += 1
|
||||
name = str(file_entry.get("name", ""))
|
||||
destination = str(file_entry.get("destination") or name)
|
||||
system.files.append(
|
||||
NativeFile(
|
||||
name=name,
|
||||
destination=destination,
|
||||
native_system=native_id,
|
||||
platform=file_entry,
|
||||
)
|
||||
)
|
||||
report.files_kept += 1
|
||||
|
||||
# Which native ids a truth system may contribute to.
|
||||
norm_to_scraped: dict[str, str] = {
|
||||
_norm_system_id(key): key for key in scraped_systems
|
||||
}
|
||||
native_by_norm: dict[str, str] = {}
|
||||
for system in systems.values():
|
||||
native_by_norm.setdefault(_norm_system_id(system.native_id), system.native_id)
|
||||
|
||||
def _target_native_ids(truth_sid: str) -> list[str]:
|
||||
scraped_key = truth_sid if truth_sid in scraped_systems else None
|
||||
if scraped_key is None:
|
||||
scraped_key = norm_to_scraped.get(_norm_system_id(truth_sid))
|
||||
if scraped_key is not None:
|
||||
sys_data = scraped_systems[scraped_key]
|
||||
ids = {
|
||||
_native_id_of(scraped_key, sys_data, fe)
|
||||
for fe in sys_data.get("files", [])
|
||||
}
|
||||
if not ids:
|
||||
return [sys_data.get("native_id") or scraped_key]
|
||||
# A file the platform does not declare joins the system's primary
|
||||
# id, not whichever of its native ids sorts first: Recalbox files
|
||||
# pcengine, pcenginecd and supergrafx under one slug, and an
|
||||
# addition belongs to the one the system is named for.
|
||||
primary = sys_data.get("native_id")
|
||||
ordered = sorted(ids)
|
||||
if primary in ids:
|
||||
ordered.remove(primary)
|
||||
ordered.insert(0, primary)
|
||||
return ordered
|
||||
direct = native_by_norm.get(_norm_system_id(truth_sid))
|
||||
return [direct] if direct else [truth_sid]
|
||||
|
||||
# Pass 2: apply the truth onto that grouping.
|
||||
for truth_sid in sorted(truth.get("systems", {})):
|
||||
truth_sys = truth["systems"][truth_sid]
|
||||
truth_files = truth_sys.get("files", [])
|
||||
if not truth_files:
|
||||
continue
|
||||
|
||||
target_ids = _target_native_ids(truth_sid)
|
||||
|
||||
for truth_entry in truth_files:
|
||||
name = str(truth_entry.get("name", ""))
|
||||
if not name or name.startswith("_") or _is_placeholder(name):
|
||||
continue
|
||||
|
||||
t_dest, t_name = _match_key(truth_entry)
|
||||
t_hashes = {
|
||||
value
|
||||
for field_name in HASH_FIELDS
|
||||
for value in _hash_values(truth_entry, field_name)
|
||||
}
|
||||
|
||||
candidates = [
|
||||
candidate
|
||||
for native_id in target_ids
|
||||
for candidate in systems.get(
|
||||
native_id, NativeSystem(native_id)
|
||||
).files
|
||||
if candidate.truth is None
|
||||
]
|
||||
|
||||
def by_destination(candidate: NativeFile) -> bool:
|
||||
theirs = _match_key(candidate.platform or {})[0]
|
||||
return bool(t_dest) and theirs == t_dest
|
||||
|
||||
def by_name(candidate: NativeFile) -> bool:
|
||||
theirs = _match_key(candidate.platform or {})[1]
|
||||
return bool(t_name) and theirs == t_name
|
||||
|
||||
def by_hash(candidate: NativeFile) -> bool:
|
||||
if not t_hashes:
|
||||
return False
|
||||
theirs = {
|
||||
value
|
||||
for field_name in HASH_FIELDS
|
||||
for value in _hash_values(candidate.platform or {}, field_name)
|
||||
}
|
||||
return bool(t_hashes & theirs)
|
||||
|
||||
# Tried in order across every candidate, not per candidate: with
|
||||
# three IPL.bin under one system, separated only by their path, a
|
||||
# first-match-wins scan would attach the truth to whichever came
|
||||
# first and correct the wrong region's file.
|
||||
matched: NativeFile | None = None
|
||||
for test in (by_destination, by_name, by_hash):
|
||||
matched = next((c for c in candidates if test(c)), None)
|
||||
if matched is not None:
|
||||
break
|
||||
|
||||
if matched is not None:
|
||||
matched.truth = truth_entry
|
||||
for field_name in HASH_FIELDS:
|
||||
ours = set(_hash_values(truth_entry, field_name))
|
||||
theirs = set(_hash_values(matched.platform or {}, field_name))
|
||||
if ours and theirs and not ours & theirs:
|
||||
matched.corrections.append(field_name)
|
||||
report.hashes_corrected.append(
|
||||
f"{matched.native_system}/{matched.name} {field_name}"
|
||||
)
|
||||
t_req = truth_entry.get("required")
|
||||
p_req = (matched.platform or {}).get("required")
|
||||
if (
|
||||
t_req is not None
|
||||
and p_req is not None
|
||||
and bool(t_req) != bool(p_req)
|
||||
):
|
||||
matched.corrections.append("required")
|
||||
report.required_corrected.append(
|
||||
f"{matched.native_system}/{matched.name}"
|
||||
)
|
||||
continue
|
||||
|
||||
# The truth knows a file the platform does not declare.
|
||||
native_id = target_ids[0]
|
||||
system = systems.get(native_id)
|
||||
if system is None:
|
||||
system = NativeSystem(native_id=native_id, from_platform=False)
|
||||
systems[native_id] = system
|
||||
report.systems_added += 1
|
||||
destination = str(
|
||||
truth_entry.get("path") or truth_entry.get("destination") or name
|
||||
)
|
||||
system.files.append(
|
||||
NativeFile(
|
||||
name=name,
|
||||
destination=destination,
|
||||
native_system=native_id,
|
||||
truth=truth_entry,
|
||||
)
|
||||
)
|
||||
report.files_added += 1
|
||||
|
||||
# A reader takes the files in the order the list gives, so the one the
|
||||
# code looks for first is named first. Only what we add is ordered: what
|
||||
# the platform already wrote keeps the place the platform gave it.
|
||||
for system in systems.values():
|
||||
head = [fe for fe in system.files if fe.platform is not None]
|
||||
tail = [fe for fe in system.files if fe.platform is None]
|
||||
system.files = head + search_order(tail)
|
||||
|
||||
return systems, report
|
||||
|
||||
|
||||
def search_order(files: list[NativeFile]) -> list[NativeFile]:
|
||||
"""Order files the way the code looks for them, best first.
|
||||
|
||||
`priority:` is that order where the source states it, lowest first.
|
||||
Where it does not, the order the entries were declared in is the order
|
||||
the code walks, so it is left alone.
|
||||
"""
|
||||
ranked = [(fe.priority, position, fe) for position, fe in enumerate(files)]
|
||||
return [
|
||||
fe
|
||||
for _, _, fe in sorted(
|
||||
ranked,
|
||||
key=lambda item: (
|
||||
(0, item[0]) if item[0] is not None else (1, item[1])
|
||||
),
|
||||
)
|
||||
]
|
||||
@@ -1,109 +1,301 @@
|
||||
"""Exporter for Batocera batocera-systems format.
|
||||
"""Exporter for Batocera's batocera-systems.
|
||||
|
||||
Produces a Python dict matching the exact format of
|
||||
batocera-linux/batocera-scripts/scripts/batocera-systems.
|
||||
batocera-systems is an executable script: the systems dict is one block
|
||||
inside it, the rest is the checker Batocera runs. Only the block is
|
||||
rewritten, entry by entry, so the comments the maintainers wrote between
|
||||
entries and the code around them survive untouched.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
import re
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeFile, NativeSystem, Report
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://raw.githubusercontent.com/batocera-linux/batocera.linux/master"
|
||||
"/package/batocera/core/batocera-scripts/scripts/batocera-systems"
|
||||
)
|
||||
|
||||
_ENTRY_START = re.compile(r'^(\s{4})"([^"]+)":\s*\{')
|
||||
_INDENT = " " * 4
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Export truth data to Batocera batocera-systems format."""
|
||||
"""Write Batocera's batocera-systems, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "batocera"
|
||||
|
||||
def export(
|
||||
self,
|
||||
truth_data: dict,
|
||||
output_path: str,
|
||||
scraped_data: dict | None = None,
|
||||
) -> None:
|
||||
# Build native_id and display name maps from scraped data
|
||||
native_map: dict[str, str] = {}
|
||||
if scraped_data:
|
||||
for sys_id, sys_data in scraped_data.get("systems", {}).items():
|
||||
nid = sys_data.get("native_id")
|
||||
if nid:
|
||||
native_map[sys_id] = nid
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "batocera-systems"
|
||||
|
||||
lines: list[str] = ["systems = {", ""]
|
||||
@staticmethod
|
||||
def requires() -> str:
|
||||
return "md5"
|
||||
|
||||
systems = truth_data.get("systems", {})
|
||||
for sys_id in sorted(systems):
|
||||
sys_data = systems[sys_id]
|
||||
files = sys_data.get("files", [])
|
||||
if not files:
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"md5"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"batocera-systems": SOURCE_URL}
|
||||
|
||||
@staticmethod
|
||||
def needs_original() -> bool:
|
||||
# The data block is a fraction of the file; the rest is the checker.
|
||||
return True
|
||||
|
||||
def _bios_files(self, files: list[NativeFile]) -> str:
|
||||
return ", ".join(self._bios_items(files))
|
||||
|
||||
def _bios_items(self, files: list[NativeFile]) -> list[str]:
|
||||
parts: list[str] = []
|
||||
for fe in files:
|
||||
# The platform states an unhashed file as an empty md5 rather
|
||||
# than leaving it out, and so do we.
|
||||
item = [f'"md5": "{fe.hash("md5")}"']
|
||||
alt = fe.native("alt_md5", "")
|
||||
if alt:
|
||||
item.append(f'"altmd5": "{alt}"')
|
||||
declared = fe.native("native_path", "")
|
||||
path = str(declared) if declared else f"bios/{fe.destination}"
|
||||
item.append(f'"file": "{path}"')
|
||||
zipped = fe.native("zipped_file", "")
|
||||
if zipped:
|
||||
item.append(f'"zippedFile": "{zipped}"')
|
||||
parts.append("{ " + ", ".join(item) + " }")
|
||||
return parts
|
||||
|
||||
def _entry_line(self, system: NativeSystem, files: list[NativeFile]) -> str:
|
||||
return (
|
||||
f'{_INDENT}"{system.native_id}": '
|
||||
f'{{ "name": "{self.display_name(system)}", '
|
||||
f'"biosFiles": [ {self._bios_files(files)} ] }},'
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _item_notes(lines: list[str]) -> tuple[dict[str, str], dict[str, list[str]]]:
|
||||
"""The notes a maintainer wrote about each file, keyed by its path.
|
||||
|
||||
Rewriting an entry as one line would drop them, and a note like
|
||||
"ideally - 94bc50..." is the only record of why a hash is blank.
|
||||
A comment on its own line belongs to the file below it; a comment at
|
||||
the end of a line belongs to the file on it.
|
||||
"""
|
||||
trailing: dict[str, str] = {}
|
||||
leading: dict[str, list[str]] = {}
|
||||
pending: list[str] = []
|
||||
for line in lines:
|
||||
stripped = line.strip()
|
||||
hash_at = line.find("#")
|
||||
if stripped.startswith("#"):
|
||||
pending.append(stripped)
|
||||
continue
|
||||
path = re.search(r'"file"\s*:\s*"([^"]+)"', line)
|
||||
if not path:
|
||||
continue
|
||||
key = path.group(1)
|
||||
if pending:
|
||||
leading[key] = pending
|
||||
pending = []
|
||||
if 0 <= hash_at and hash_at > line.index(key):
|
||||
trailing[key] = line[hash_at:].rstrip()
|
||||
return trailing, leading
|
||||
|
||||
native_id = native_map.get(sys_id, sys_id)
|
||||
scraped_sys = (
|
||||
scraped_data.get("systems", {}).get(sys_id) if scraped_data else None
|
||||
def _entry_block(
|
||||
self,
|
||||
system: NativeSystem,
|
||||
files: list[NativeFile],
|
||||
original: list[str],
|
||||
) -> list[str]:
|
||||
"""One entry, keeping the layout and the notes it was written with."""
|
||||
trailing, leading = self._item_notes(original)
|
||||
items = self._bios_items(files)
|
||||
one_line = len(original) == 1 and not trailing and not leading
|
||||
if one_line:
|
||||
return [self._entry_line(system, files)]
|
||||
|
||||
head = (
|
||||
f'{_INDENT}"{system.native_id}": '
|
||||
f'{{ "name": "{self.display_name(system)}", "biosFiles": ['
|
||||
)
|
||||
pad = " " * (len(_INDENT) + 4)
|
||||
lines = [head]
|
||||
for index, item in enumerate(items):
|
||||
key = self._item_path(item)
|
||||
lines.extend(f"{pad}{note}" for note in leading.get(key, []))
|
||||
separator = "," if index < len(items) - 1 else ""
|
||||
note = trailing.get(key, "")
|
||||
suffix = f" {note}" if note else ""
|
||||
lines.append(f"{pad}{item}{separator}{suffix}")
|
||||
lines.append(f"{_INDENT}] }},")
|
||||
return lines
|
||||
|
||||
@staticmethod
|
||||
def _item_path(item: str) -> str:
|
||||
match = re.search(r'"file": "([^"]+)"', item)
|
||||
return match.group(1) if match else ""
|
||||
|
||||
@staticmethod
|
||||
def _split_original(original: str) -> tuple[list[str], list[str], list[str]]:
|
||||
"""Cut the script into what precedes the dict, the dict, what follows."""
|
||||
lines = original.split("\n")
|
||||
start = next(
|
||||
(i for i, line in enumerate(lines) if line.startswith("systems = {")),
|
||||
None,
|
||||
)
|
||||
if start is None:
|
||||
raise ValueError("batocera-systems: no systems dict found")
|
||||
end = next(
|
||||
(i for i in range(start + 1, len(lines)) if lines[i].startswith("}")),
|
||||
None,
|
||||
)
|
||||
if end is None:
|
||||
raise ValueError("batocera-systems: the systems dict is not closed")
|
||||
return lines[: start + 1], lines[start + 1 : end], lines[end:]
|
||||
|
||||
@staticmethod
|
||||
def _entry_spans(body: list[str]) -> dict[str, tuple[int, int]]:
|
||||
"""Locate each top-level entry, which may wrap over several lines."""
|
||||
spans: dict[str, tuple[int, int]] = {}
|
||||
index = 0
|
||||
while index < len(body):
|
||||
match = _ENTRY_START.match(body[index])
|
||||
if not match:
|
||||
index += 1
|
||||
continue
|
||||
key = match.group(2)
|
||||
depth = 0
|
||||
end = index
|
||||
for cursor in range(index, len(body)):
|
||||
depth += body[cursor].count("{") + body[cursor].count("[")
|
||||
depth -= body[cursor].count("}") + body[cursor].count("]")
|
||||
if depth <= 0:
|
||||
end = cursor
|
||||
break
|
||||
else:
|
||||
end = len(body) - 1
|
||||
spans[key] = (index, end)
|
||||
index = end + 1
|
||||
return spans
|
||||
|
||||
@staticmethod
|
||||
def _parse_entry(lines: list[str]) -> dict | None:
|
||||
"""Evaluate one entry so two spellings of the same data compare equal."""
|
||||
text = "\n".join(lines).strip().rstrip(",")
|
||||
namespace: dict[str, object] = {}
|
||||
try:
|
||||
exec(f"entry = {{{text}}}", {}, namespace) # noqa: S102
|
||||
except (SyntaxError, ValueError, TypeError):
|
||||
return None
|
||||
entry = namespace.get("entry")
|
||||
return entry if isinstance(entry, dict) else None
|
||||
|
||||
def render(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
exportable = dict(
|
||||
(system.native_id, (system, files))
|
||||
for system, files in self.exportable(systems, require="md5")
|
||||
)
|
||||
|
||||
original = originals.get(self.native_filename(), "")
|
||||
if not original:
|
||||
raise ValueError(
|
||||
f"{self.native_filename()} cannot be written without the "
|
||||
"platform's own file: the systems dict is a fraction of a "
|
||||
"script, and the rest of it is the checker"
|
||||
)
|
||||
display_name = self._display_name(sys_id, scraped_sys)
|
||||
|
||||
# Build md5 lookup from scraped data for this system
|
||||
scraped_md5: dict[str, str] = {}
|
||||
if scraped_data:
|
||||
s_sys = scraped_data.get("systems", {}).get(sys_id, {})
|
||||
for sf in s_sys.get("files", []):
|
||||
sname = sf.get("name", "").lower()
|
||||
smd5 = sf.get("md5", "")
|
||||
if sname and smd5:
|
||||
scraped_md5[sname] = smd5
|
||||
head, body, tail = self._split_original(original)
|
||||
spans = self._entry_spans(body)
|
||||
|
||||
# Build biosFiles entries as compact single-line dicts
|
||||
# Original format ALWAYS has md5 — use scraped md5 as fallback
|
||||
bios_parts: list[str] = []
|
||||
for fe in files:
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
continue
|
||||
dest = self._dest(fe)
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = md5[0] if md5 else ""
|
||||
if not md5:
|
||||
md5 = scraped_md5.get(name.lower(), "")
|
||||
rebuilt: list[str] = []
|
||||
written: set[str] = set()
|
||||
index = 0
|
||||
for key, (start, end) in sorted(spans.items(), key=lambda kv: kv[1][0]):
|
||||
rebuilt.extend(body[index:start])
|
||||
pair = exportable.get(key)
|
||||
index = end + 1
|
||||
if pair is None:
|
||||
# The truth has nothing to say and no file left to declare.
|
||||
rebuilt.extend(body[start:end + 1])
|
||||
continue
|
||||
written.add(key)
|
||||
original_lines = body[start:end + 1]
|
||||
replacement = self._entry_line(*pair)
|
||||
before = self._parse_entry(original_lines)
|
||||
after = self._parse_entry([replacement])
|
||||
if before is not None and before == after:
|
||||
# Nothing changed: keep the maintainer's own lines, comments
|
||||
# and alignment included, so the diff shows only corrections.
|
||||
rebuilt.extend(original_lines)
|
||||
else:
|
||||
rebuilt.extend(self._entry_block(*pair, original_lines))
|
||||
rebuilt.extend(body[index:])
|
||||
|
||||
# Original format requires md5 for every entry — skip without
|
||||
if not md5:
|
||||
continue
|
||||
bios_parts.append(f'{{ "md5": "{md5}", "file": "bios/{dest}" }}')
|
||||
added = [
|
||||
self._entry_line(system, files)
|
||||
for native_id, (system, files) in exportable.items()
|
||||
if native_id not in written
|
||||
]
|
||||
if added:
|
||||
while rebuilt and not rebuilt[-1].strip():
|
||||
rebuilt.pop()
|
||||
rebuilt.append("")
|
||||
rebuilt.extend(sorted(added))
|
||||
rebuilt.append("")
|
||||
|
||||
bios_str = ", ".join(bios_parts)
|
||||
line = (
|
||||
f' "{native_id}": '
|
||||
f'{{ "name": "{display_name}", '
|
||||
f'"biosFiles": [ {bios_str} ] }},'
|
||||
)
|
||||
lines.append(line)
|
||||
return {self.native_filename(): "\n".join([*head, *rebuilt, *tail])}
|
||||
|
||||
lines.append("")
|
||||
lines.append("}")
|
||||
lines.append("")
|
||||
Path(output_path).write_text("\n".join(lines), encoding="utf-8")
|
||||
|
||||
def validate(self, truth_data: dict, output_path: str) -> list[str]:
|
||||
content = Path(output_path).read_text(encoding="utf-8")
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
content = produced[self.native_filename()]
|
||||
issues: list[str] = []
|
||||
for sys_data in truth_data.get("systems", {}).values():
|
||||
for fe in sys_data.get("files", []):
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
continue
|
||||
# Skip entries without md5 (not exportable in this format)
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = md5[0] if md5 else ""
|
||||
if not md5:
|
||||
continue
|
||||
dest = self._dest(fe)
|
||||
if dest not in content and name not in content:
|
||||
issues.append(f"missing: {name}")
|
||||
|
||||
namespace: dict[str, object] = {}
|
||||
block = content.split("\nsystems = {", 1)
|
||||
if len(block) != 2:
|
||||
return ["no systems dict in the output"]
|
||||
end = block[1].find("\n}")
|
||||
if end < 0:
|
||||
return ["the systems dict is not closed"]
|
||||
try:
|
||||
exec("systems = {" + block[1][:end] + "\n}", {}, namespace) # noqa: S102
|
||||
except SyntaxError as exc:
|
||||
return [f"the systems dict does not parse: {exc}"]
|
||||
|
||||
exported = namespace.get("systems", {})
|
||||
if not isinstance(exported, dict):
|
||||
return ["the systems dict did not evaluate to a dict"]
|
||||
|
||||
for system, files in self.exportable(systems, require="md5"):
|
||||
entry = exported.get(system.native_id)
|
||||
if entry is None:
|
||||
issues.append(f"system absent: {system.native_id}")
|
||||
continue
|
||||
declared = {bios.get("file", "") for bios in entry.get("biosFiles", [])}
|
||||
for fe in files:
|
||||
path = str(fe.native("native_path", "")) or f"bios/{fe.destination}"
|
||||
if path not in declared:
|
||||
issues.append(f"absent: {system.native_id}/{fe.name}")
|
||||
|
||||
for native_id, entry in exported.items():
|
||||
if not entry.get("biosFiles"):
|
||||
issues.append(f"empty entry: {native_id}")
|
||||
|
||||
if "def checkBios(" not in content:
|
||||
issues.append("the checker the script exists for is missing")
|
||||
return issues
|
||||
@@ -0,0 +1,153 @@
|
||||
"""Exporter for BizHawk's FirmwareDatabase.cs.
|
||||
|
||||
The database is C#: every firmware is a call whose arguments are the SHA1,
|
||||
the size, the file name and a description, wired into option lists and
|
||||
status flags that only the source expresses. The calls are rewritten in
|
||||
place, and nothing else in the file is touched.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeFile, NativeSystem, Report
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://raw.githubusercontent.com/TASEmulators/BizHawk/master"
|
||||
"/src/BizHawk.Emulation.Common/Database/FirmwareDatabase.cs"
|
||||
)
|
||||
|
||||
# File("<sha1>", <size>, "<name>", ...) and the same three arguments inside
|
||||
# FirmwareAndOption(<sha1>, <size>, <system>, <id>, <name>, ...).
|
||||
_FILE_CALL = re.compile(
|
||||
r'(File\(\s*")([0-9A-Fa-f]{40})("\s*,\s*)(\d+)(\s*,\s*")([^"]+)(")'
|
||||
)
|
||||
_FIRMWARE_AND_OPTION = re.compile(
|
||||
r'(FirmwareAndOption\(\s*")([0-9A-Fa-f]{40})("\s*,\s*)(\d+)'
|
||||
r'(\s*,\s*"[^"]*"\s*,\s*"[^"]*"\s*,\s*")([^"]+)(")'
|
||||
)
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Write BizHawk's FirmwareDatabase.cs, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "bizhawk"
|
||||
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "FirmwareDatabase.cs"
|
||||
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"sha1", "size"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"FirmwareDatabase.cs": SOURCE_URL}
|
||||
|
||||
@staticmethod
|
||||
def can_add() -> bool:
|
||||
# A firmware is a call wired into an option list and a status; the
|
||||
# exporter corrects the calls that exist, it does not write C#.
|
||||
return False
|
||||
|
||||
@staticmethod
|
||||
def needs_original() -> bool:
|
||||
# The database is code: option lists, statuses and the systems they
|
||||
# hang off exist nowhere else.
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def _unambiguous(systems: dict[str, NativeSystem]) -> dict[str, NativeFile]:
|
||||
"""Files whose name identifies exactly one entry with a SHA1.
|
||||
|
||||
BizHawk names a firmware by file name inside a system, and the same
|
||||
name recurs across systems. Correcting on a name that resolves to
|
||||
two different sets of bytes would corrupt the database, so only the
|
||||
names that resolve to one are touched.
|
||||
"""
|
||||
seen: dict[str, list[NativeFile]] = {}
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
if fe.hash("sha1"):
|
||||
seen.setdefault(fe.name.casefold(), []).append(fe)
|
||||
resolved: dict[str, NativeFile] = {}
|
||||
for name, entries in seen.items():
|
||||
hashes = {fe.hash("sha1").lower() for fe in entries}
|
||||
if len(hashes) == 1:
|
||||
resolved[name] = entries[0]
|
||||
return resolved
|
||||
|
||||
def render(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
source = originals.get(self.native_filename(), "")
|
||||
if not source:
|
||||
raise ValueError(
|
||||
"FirmwareDatabase.cs cannot be written without BizHawk's own "
|
||||
"file: the database is C#, not data"
|
||||
)
|
||||
|
||||
index = self._unambiguous(systems)
|
||||
|
||||
def commented_out(text: str, position: int) -> bool:
|
||||
"""Whether the call sits on a line the compiler never sees.
|
||||
|
||||
BizHawk keeps disabled entries in place behind //, and a hash
|
||||
written into one of those is a change to a comment.
|
||||
"""
|
||||
line_start = text.rfind("\n", 0, position) + 1
|
||||
return text[line_start:position].lstrip().startswith("//")
|
||||
|
||||
def rewrite(match: re.Match[str], name_group: int) -> str:
|
||||
if commented_out(match.string, match.start()):
|
||||
return match.group(0)
|
||||
name = match.group(name_group)
|
||||
fe = index.get(name.casefold())
|
||||
if fe is None:
|
||||
return match.group(0)
|
||||
sha1 = fe.hash("sha1").upper()
|
||||
size = fe.size() or int(match.group(4))
|
||||
groups = list(match.groups())
|
||||
groups[1] = sha1
|
||||
groups[3] = str(size)
|
||||
return "".join(groups)
|
||||
|
||||
patched = _FILE_CALL.sub(lambda m: rewrite(m, 6), source)
|
||||
patched = _FIRMWARE_AND_OPTION.sub(lambda m: rewrite(m, 6), patched)
|
||||
return {self.native_filename(): patched}
|
||||
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
content = produced[self.native_filename()]
|
||||
issues: list[str] = []
|
||||
|
||||
if "FirmwareDatabase" not in content:
|
||||
issues.append("the class the database lives in is missing")
|
||||
if content.count("{") != content.count("}"):
|
||||
issues.append("braces are unbalanced, the file would not compile")
|
||||
|
||||
index = self._unambiguous(systems)
|
||||
declared: dict[str, str] = {}
|
||||
for pattern in (_FILE_CALL, _FIRMWARE_AND_OPTION):
|
||||
for match in pattern.finditer(content):
|
||||
line_start = content.rfind("\n", 0, match.start()) + 1
|
||||
if content[line_start : match.start()].lstrip().startswith("//"):
|
||||
continue
|
||||
declared[match.group(6).casefold()] = match.group(2).lower()
|
||||
|
||||
for name, fe in index.items():
|
||||
written = declared.get(name)
|
||||
if written is not None and written != fe.hash("sha1").lower():
|
||||
issues.append(f"hash not applied: {fe.name}")
|
||||
return issues
|
||||
@@ -1,215 +1,166 @@
|
||||
"""Exporter for EmuDeck checkBIOS.sh format.
|
||||
"""Exporter for EmuDeck's checkBIOS.sh.
|
||||
|
||||
Produces a bash script matching the exact pattern of EmuDeck's
|
||||
functions/checkBIOS.sh: per-system check functions with MD5 arrays
|
||||
inside the function body, iterating over $biosPath/* files.
|
||||
|
||||
Two patterns:
|
||||
- MD5 pattern: systems with known hashes, loop $biosPath/*, md5sum each, match
|
||||
- File-exists pattern: systems with specific paths, check -f
|
||||
checkBIOS.sh is a shell library: each system is a function EmuDeck calls by
|
||||
name, and the only data in it is the MD5 list each function matches against.
|
||||
Writing the functions from a table would publish a file missing whichever
|
||||
checks the table forgot, so the original is patched instead and every
|
||||
function it defines keeps its shape, its scan directory and its output.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from scraper.emudeck_scraper import FUNCTION_HASH_MAP, _RE_FUNC, _RE_LOCAL_HASHES
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeSystem, Report
|
||||
|
||||
# Map our system IDs to EmuDeck function naming conventions
|
||||
_SYSTEM_CONFIG: dict[str, dict] = {
|
||||
"sony-playstation": {
|
||||
"func": "checkPS1BIOS",
|
||||
"var": "PSXBIOS",
|
||||
"array": "PSBios",
|
||||
"pattern": "md5",
|
||||
},
|
||||
"sony-playstation-2": {
|
||||
"func": "checkPS2BIOS",
|
||||
"var": "PS2BIOS",
|
||||
"array": "PS2Bios",
|
||||
"pattern": "md5",
|
||||
},
|
||||
"sega-mega-cd": {
|
||||
"func": "checkSegaCDBios",
|
||||
"var": "SEGACDBIOS",
|
||||
"array": "CDBios",
|
||||
"pattern": "md5",
|
||||
},
|
||||
"sega-saturn": {
|
||||
"func": "checkSaturnBios",
|
||||
"var": "SATURNBIOS",
|
||||
"array": "SaturnBios",
|
||||
"pattern": "md5",
|
||||
},
|
||||
"sega-dreamcast": {
|
||||
"func": "checkDreamcastBios",
|
||||
"var": "BIOS",
|
||||
"array": "hashes",
|
||||
"pattern": "md5",
|
||||
},
|
||||
"nintendo-ds": {
|
||||
"func": "checkDSBios",
|
||||
"var": "BIOS",
|
||||
"array": "hashes",
|
||||
"pattern": "md5",
|
||||
},
|
||||
"nintendo-switch": {
|
||||
"func": "checkCitronBios",
|
||||
"pattern": "file-exists",
|
||||
"firmware_path": "$biosPath/citron/firmware",
|
||||
"keys_path": "$biosPath/citron/keys/prod.keys",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _make_md5_function(cfg: dict, md5s: list[str]) -> list[str]:
|
||||
"""Generate a MD5-checking function matching EmuDeck's exact pattern."""
|
||||
func = cfg["func"]
|
||||
var = cfg["var"]
|
||||
array = cfg["array"]
|
||||
md5_str = " ".join(md5s)
|
||||
|
||||
return [
|
||||
f"{func}(){{",
|
||||
"",
|
||||
f'\t{var}="NULL"',
|
||||
"",
|
||||
'\tfor entry in "$biosPath/"*',
|
||||
"\tdo",
|
||||
'\t\tif [ -f "$entry" ]; then',
|
||||
'\t\t\tmd5=($(md5sum "$entry"))',
|
||||
f'\t\t\tif [[ "${var}" != true ]]; then',
|
||||
f"\t\t\t\t{array}=({md5_str})",
|
||||
f'\t\t\t\tfor i in "${{{array}[@]}}"',
|
||||
"\t\t\t\tdo",
|
||||
'\t\t\t\tif [[ "$md5" == *"${i}"* ]]; then',
|
||||
f"\t\t\t\t\t{var}=true",
|
||||
"\t\t\t\t\tbreak",
|
||||
"\t\t\t\telse",
|
||||
f"\t\t\t\t\t{var}=false",
|
||||
"\t\t\t\tfi",
|
||||
"\t\t\t\tdone",
|
||||
"\t\t\tfi",
|
||||
"\t\tfi",
|
||||
"\tdone",
|
||||
"",
|
||||
"",
|
||||
f"\tif [ ${var} == true ]; then",
|
||||
'\t\techo "$entry true";',
|
||||
"\telse",
|
||||
'\t\techo "false";',
|
||||
"\tfi",
|
||||
"}",
|
||||
]
|
||||
|
||||
|
||||
def _make_file_exists_function(cfg: dict) -> list[str]:
|
||||
"""Generate a file-exists function matching EmuDeck's pattern."""
|
||||
func = cfg["func"]
|
||||
firmware = cfg.get("firmware_path", "")
|
||||
keys = cfg.get("keys_path", "")
|
||||
|
||||
return [
|
||||
f"{func}(){{",
|
||||
"",
|
||||
f'\tlocal FIRMWARE="{firmware}"',
|
||||
f'\tlocal KEYS="{keys}"',
|
||||
'\tif [[ -f "$KEYS" ]] && [[ "$( ls -A "$FIRMWARE")" ]]; then',
|
||||
'\t\t\techo "true";',
|
||||
"\telse",
|
||||
'\t\t\techo "false";',
|
||||
"\tfi",
|
||||
"}",
|
||||
]
|
||||
SOURCE_URL = (
|
||||
"https://raw.githubusercontent.com/dragoonDorise/EmuDeck/main"
|
||||
"/functions/checkBIOS.sh"
|
||||
)
|
||||
_MD5 = re.compile(r"^[0-9a-f]{32}$")
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Export truth data to EmuDeck checkBIOS.sh format."""
|
||||
"""Write EmuDeck's checkBIOS.sh, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "emudeck"
|
||||
|
||||
def export(
|
||||
self,
|
||||
truth_data: dict,
|
||||
output_path: str,
|
||||
scraped_data: dict | None = None,
|
||||
) -> None:
|
||||
lines: list[str] = ["#!/bin/bash"]
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "checkBIOS.sh"
|
||||
|
||||
systems = truth_data.get("systems", {})
|
||||
@staticmethod
|
||||
def requires() -> str:
|
||||
return "md5"
|
||||
|
||||
for sys_id, cfg in sorted(_SYSTEM_CONFIG.items(), key=lambda x: x[1]["func"]):
|
||||
sys_data = systems.get(sys_id)
|
||||
if not sys_data:
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"md5"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"checkBIOS.sh": SOURCE_URL}
|
||||
|
||||
@staticmethod
|
||||
def needs_original() -> bool:
|
||||
# The checks are code, and EmuDeck calls them by name.
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def can_add() -> bool:
|
||||
"""A hash is corrected in place, never added to an array.
|
||||
|
||||
EmuDeck's frontend calls one check for several emulators
|
||||
(EmulatorsDetailPage.jsx: 'ra' asks checkPS1BIOS, checkSegaCDBios,
|
||||
checkSaturnBios, checkDSBios and checkDreamcastBios, and
|
||||
'duckstation' asks checkPS1BIOS as well), and nothing in
|
||||
checkBIOS.sh names them. Growing an array therefore changes answers
|
||||
for consumers the array does not list: DuckStation boots from a PS2
|
||||
image, the RetroArch PSX cores do not, so adding the ones
|
||||
DuckStation accepts would report a BIOS to a card that has none.
|
||||
Correcting a value in place changes no consumer's set.
|
||||
"""
|
||||
return False
|
||||
|
||||
@classmethod
|
||||
def _md5s(cls, systems: dict[str, NativeSystem], system_id: str) -> list[str]:
|
||||
"""Every MD5 the system accepts, in a stable order, deduplicated."""
|
||||
seen: list[str] = []
|
||||
for system in systems.values():
|
||||
if system.native_id != system_id:
|
||||
continue
|
||||
for fe in system.files:
|
||||
if not cls.writable(fe):
|
||||
continue
|
||||
for value in fe.hashes("md5"):
|
||||
if _MD5.match(value) and value not in seen:
|
||||
seen.append(value)
|
||||
return seen
|
||||
|
||||
lines.append("")
|
||||
def _function_spans(self, script: str) -> list[tuple[str, int, int]]:
|
||||
"""Name and byte span of every check the script defines."""
|
||||
matches = list(_RE_FUNC.finditer(script))
|
||||
spans: list[tuple[str, int, int]] = []
|
||||
for index, match in enumerate(matches):
|
||||
end = (
|
||||
matches[index + 1].start()
|
||||
if index + 1 < len(matches)
|
||||
else len(script)
|
||||
)
|
||||
spans.append((match.group(1), match.start(), end))
|
||||
return spans
|
||||
|
||||
if cfg["pattern"] == "md5":
|
||||
md5s: list[str] = []
|
||||
for fe in sys_data.get("files", []):
|
||||
name = fe.get("name", "")
|
||||
if self._is_pattern(name) or name.startswith("_"):
|
||||
continue
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5s.extend(
|
||||
m for m in md5 if m and re.fullmatch(r"[a-f0-9]{32}", m)
|
||||
)
|
||||
elif md5 and re.fullmatch(r"[a-f0-9]{32}", md5):
|
||||
md5s.append(md5)
|
||||
if md5s:
|
||||
lines.extend(_make_md5_function(cfg, md5s))
|
||||
elif cfg["pattern"] == "file-exists":
|
||||
lines.extend(_make_file_exists_function(cfg))
|
||||
def render(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
script = originals.get(self.native_filename(), "")
|
||||
if not script:
|
||||
raise ValueError(
|
||||
"checkBIOS.sh cannot be written without EmuDeck's own file: "
|
||||
"the checks are code, not data"
|
||||
)
|
||||
|
||||
lines.append("")
|
||||
Path(output_path).write_text("\n".join(lines), encoding="utf-8")
|
||||
pieces: list[str] = []
|
||||
cursor = 0
|
||||
for name, start, end in self._function_spans(script):
|
||||
pieces.append(script[cursor:start])
|
||||
body = script[start:end]
|
||||
system_id = FUNCTION_HASH_MAP.get(name)
|
||||
md5s = self._md5s(systems, system_id) if system_id else []
|
||||
match = _RE_LOCAL_HASHES.search(body)
|
||||
if md5s and match:
|
||||
# An array compared by membership says nothing about order,
|
||||
# so the same set is left as the maintainer wrote it.
|
||||
if set(md5s) != set(match.group(1).split()):
|
||||
body = (
|
||||
body[: match.start(1)]
|
||||
+ " ".join(md5s)
|
||||
+ body[match.end(1) :]
|
||||
)
|
||||
pieces.append(body)
|
||||
cursor = end
|
||||
pieces.append(script[cursor:])
|
||||
|
||||
def validate(self, truth_data: dict, output_path: str) -> list[str]:
|
||||
content = Path(output_path).read_text(encoding="utf-8")
|
||||
return {self.native_filename(): "".join(pieces)}
|
||||
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
content = produced[self.native_filename()]
|
||||
issues: list[str] = []
|
||||
|
||||
systems = truth_data.get("systems", {})
|
||||
for sys_id, cfg in _SYSTEM_CONFIG.items():
|
||||
if cfg["pattern"] != "md5":
|
||||
defined = {name for name, _, _ in self._function_spans(content)}
|
||||
for name, system_id in FUNCTION_HASH_MAP.items():
|
||||
if name not in defined:
|
||||
issues.append(f"check absent from the output: {name}")
|
||||
continue
|
||||
sys_data = systems.get(sys_id)
|
||||
if not sys_data:
|
||||
md5s = self._md5s(systems, system_id)
|
||||
if not md5s:
|
||||
continue
|
||||
for fe in sys_data.get("files", []):
|
||||
# export skips placeholders and private entries, so looking
|
||||
# for them here makes the exporter reject its own output.
|
||||
name = fe.get("name", "")
|
||||
if self._is_pattern(name) or name.startswith("_"):
|
||||
continue
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = md5[0] if md5 else ""
|
||||
if md5 and re.fullmatch(r"[a-f0-9]{32}", md5) and md5 not in content:
|
||||
issues.append(f"missing md5: {md5} ({name})")
|
||||
|
||||
for sys_id, cfg in _SYSTEM_CONFIG.items():
|
||||
func = cfg["func"]
|
||||
if func in content:
|
||||
body = next(
|
||||
content[start:end]
|
||||
for fname, start, end in self._function_spans(content)
|
||||
if fname == name
|
||||
)
|
||||
if not _RE_LOCAL_HASHES.search(body):
|
||||
# A check with no hash list is a path check, not a hash check.
|
||||
continue
|
||||
sys_data = systems.get(sys_id)
|
||||
if not sys_data or not sys_data.get("files"):
|
||||
continue
|
||||
# Only flag if the system has usable data for the function type
|
||||
if cfg["pattern"] == "md5":
|
||||
has_md5 = any(
|
||||
fe.get("md5")
|
||||
and isinstance(fe.get("md5"), str)
|
||||
and re.fullmatch(r"[a-f0-9]{32}", fe["md5"])
|
||||
for fe in sys_data["files"]
|
||||
)
|
||||
if has_md5:
|
||||
issues.append(f"missing function: {func}")
|
||||
elif cfg["pattern"] == "file-exists":
|
||||
issues.append(f"missing function: {func}")
|
||||
|
||||
for md5 in md5s:
|
||||
if md5 not in body:
|
||||
issues.append(f"absent from {name}: {md5}")
|
||||
return issues
|
||||
@@ -1,8 +1,4 @@
|
||||
"""Exporter for Lakka (System.dat format, same as RetroArch).
|
||||
|
||||
Lakka inherits RetroArch cores and uses the same System.dat format.
|
||||
Delegates to systemdat_exporter for export and validation.
|
||||
"""
|
||||
"""Exporter for Lakka, which reads RetroArch's System.dat unchanged."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -10,7 +6,7 @@ from .systemdat_exporter import Exporter as SystemDatExporter
|
||||
|
||||
|
||||
class Exporter(SystemDatExporter):
|
||||
"""Export truth data to Lakka System.dat format."""
|
||||
"""Write Lakka's System.dat, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
|
||||
@@ -0,0 +1,125 @@
|
||||
"""Exporter for MiSTer's BiosDB (bios_db.json, shipped zipped).
|
||||
|
||||
A Downloader database entry carries the download URL, the install path and
|
||||
the tag ids alongside the hash, and none of those are ours to invent. Only
|
||||
the hash and the size are rewritten, in MiSTer's own database.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import json
|
||||
import zipfile
|
||||
from collections import OrderedDict
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeSystem, Report
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://raw.githubusercontent.com/ajgowans/BiosDB_MiSTer/db/bios_db.json.zip"
|
||||
)
|
||||
_DB_NAME = "bios_db.json"
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Write MiSTer's bios_db.json, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "misterfpga"
|
||||
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return _DB_NAME
|
||||
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"md5", "size"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"bios_db.json.zip": SOURCE_URL}
|
||||
|
||||
@staticmethod
|
||||
def can_add() -> bool:
|
||||
# Every entry carries the URL MiSTer installs it from, and that is
|
||||
# not ours to invent.
|
||||
return False
|
||||
|
||||
@staticmethod
|
||||
def needs_original() -> bool:
|
||||
# Entries carry a URL and a tag vocabulary the database owns.
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def unpack(raw: bytes) -> dict[str, str]:
|
||||
"""Read the database out of the archive MiSTer publishes."""
|
||||
with zipfile.ZipFile(io.BytesIO(raw)) as archive:
|
||||
return {_DB_NAME: archive.read(_DB_NAME).decode("utf-8")}
|
||||
|
||||
def _by_path(self, systems: dict[str, NativeSystem]) -> dict[str, object]:
|
||||
indexed: dict[str, object] = {}
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
if fe.destination:
|
||||
indexed[f"games/{fe.destination}"] = fe
|
||||
return indexed
|
||||
|
||||
def render(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
original = originals.get(_DB_NAME) or originals.get("bios_db.json.zip")
|
||||
if not original:
|
||||
raise ValueError(
|
||||
"bios_db.json cannot be written without MiSTer's own database: "
|
||||
"every entry carries a URL and tag ids that are not ours"
|
||||
)
|
||||
database = json.loads(original, object_pairs_hook=OrderedDict)
|
||||
indexed = self._by_path(systems)
|
||||
|
||||
for path, entry in database.get("files", {}).items():
|
||||
fe = indexed.get(path)
|
||||
if fe is None:
|
||||
continue
|
||||
md5 = fe.hash("md5")
|
||||
if md5:
|
||||
entry["hash"] = md5
|
||||
size = fe.size()
|
||||
if size:
|
||||
entry["size"] = size
|
||||
|
||||
return {_DB_NAME: json.dumps(database, indent=2, ensure_ascii=False) + "\n"}
|
||||
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
try:
|
||||
database = json.loads(produced[_DB_NAME])
|
||||
except json.JSONDecodeError as exc:
|
||||
return [f"the database does not parse: {exc}"]
|
||||
|
||||
issues: list[str] = []
|
||||
if not database.get("db_id"):
|
||||
issues.append("the database lost its db_id")
|
||||
files = database.get("files", {})
|
||||
if not files:
|
||||
issues.append("the database has no files left")
|
||||
for path, entry in files.items():
|
||||
if not entry.get("hash"):
|
||||
issues.append(f"entry without a hash: {path}")
|
||||
if not entry.get("url"):
|
||||
issues.append(f"entry without a URL, uninstallable: {path}")
|
||||
|
||||
indexed = self._by_path(systems)
|
||||
for path, fe in indexed.items():
|
||||
declared = files.get(path)
|
||||
md5 = fe.hash("md5")
|
||||
if declared is not None and md5 and declared.get("hash") != md5:
|
||||
issues.append(f"hash not applied: {path}")
|
||||
return issues
|
||||
@@ -1,135 +1,165 @@
|
||||
"""Exporter for Recalbox es_bios.xml format.
|
||||
"""Exporter for Recalbox's es_bios.xml.
|
||||
|
||||
Produces XML matching the exact format of recalbox's es_bios.xml:
|
||||
- XML namespace declaration
|
||||
- <system fullname="..." platform="...">
|
||||
- <bios path="system/file" md5="..." core="..." /> with optional mandatory, hashMatchMandatory, note
|
||||
- mandatory absent = true (only explicit when false)
|
||||
- 2-space indentation
|
||||
The file is validated by es_bios.xsd, which makes path, md5 and core
|
||||
required on every bios element. An entry we cannot give all three to is not
|
||||
written: Recalbox would reject the file whole.
|
||||
|
||||
mandatory and hashMatchMandatory are separate axes. Recalbox reads a missing
|
||||
attribute as true for both, so each is written only when it is false, or
|
||||
when the platform stated it explicitly.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from xml.etree.ElementTree import ParseError
|
||||
from xml.sax.saxutils import quoteattr
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from common import parse_untrusted_xml
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeFile, NativeSystem, Report
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://gitlab.com/recalbox/recalbox/-/raw/master/board/recalbox/fsoverlay"
|
||||
"/recalbox/share_init/system/.emulationstation/es_bios.xml"
|
||||
)
|
||||
SCHEMA_URL = (
|
||||
"https://gitlab.com/recalbox/recalbox/-/raw/master/board/recalbox/fsoverlay"
|
||||
"/recalbox/share_init/system/.emulationstation/es_bios.xsd"
|
||||
)
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Export truth data to Recalbox es_bios.xml format."""
|
||||
"""Write Recalbox's es_bios.xml, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "recalbox"
|
||||
|
||||
def export(
|
||||
self,
|
||||
truth_data: dict,
|
||||
output_path: str,
|
||||
scraped_data: dict | None = None,
|
||||
) -> None:
|
||||
native_map: dict[str, str] = {}
|
||||
if scraped_data:
|
||||
for sys_id, sys_data in scraped_data.get("systems", {}).items():
|
||||
nid = sys_data.get("native_id")
|
||||
if nid:
|
||||
native_map[sys_id] = nid
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "es_bios.xml"
|
||||
|
||||
lines: list[str] = [
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"md5", "required"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"es_bios.xml": SOURCE_URL, "es_bios.xsd": SCHEMA_URL}
|
||||
|
||||
def _path(self, fe: NativeFile, native_id: str) -> str:
|
||||
"""The path Recalbox reads, pipe-joined when it accepts several.
|
||||
|
||||
A path Recalbox already states is reproduced exactly: several of its
|
||||
entries sit at the BIOS root with no directory at all, and prefixing
|
||||
them with the system would point the frontend somewhere else.
|
||||
"""
|
||||
if fe.platform is not None:
|
||||
path = str(fe.platform.get("destination") or fe.name)
|
||||
alternatives = fe.platform.get("alt_paths") or []
|
||||
if alternatives:
|
||||
return "|".join([path, *[str(a) for a in alternatives]])
|
||||
return path
|
||||
dest = fe.destination or fe.name
|
||||
return dest if "/" in dest else f"{native_id}/{dest}"
|
||||
|
||||
def _bios_element(self, fe: NativeFile, native_id: str) -> str:
|
||||
attrs = [f"path={quoteattr(self._path(fe, native_id))}"]
|
||||
attrs.append(f'md5={quoteattr(",".join(fe.hashes("md5")))}')
|
||||
attrs.append(f'core={quoteattr(",".join(fe.cores()))}')
|
||||
|
||||
if not fe.required:
|
||||
attrs.append('mandatory="false"')
|
||||
elif fe.native("mandatory_declared", None) is True:
|
||||
attrs.append('mandatory="true"')
|
||||
|
||||
hash_match = fe.native("hash_match_mandatory", None)
|
||||
if hash_match is False:
|
||||
attrs.append('hashMatchMandatory="false"')
|
||||
elif hash_match is True:
|
||||
attrs.append('hashMatchMandatory="true"')
|
||||
|
||||
note = " ".join(str(fe.native("note", "")).split())
|
||||
if note:
|
||||
attrs.append(f"note={quoteattr(note)}")
|
||||
|
||||
return f" <bios {' '.join(attrs)} />"
|
||||
|
||||
@classmethod
|
||||
def writable(cls, fe: NativeFile, require: str = "") -> bool:
|
||||
"""es_bios.xsd makes md5 and core required; without them, no element.
|
||||
|
||||
An entry Recalbox already ships has both, so this only ever gates
|
||||
what we would be adding.
|
||||
"""
|
||||
return bool(fe.hashes("md5")) and bool(fe.cores())
|
||||
|
||||
def render(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
lines = [
|
||||
'<?xml version="1.0" encoding="UTF-8"?>',
|
||||
'<biosList xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"'
|
||||
' xsi:noNamespaceSchemaLocation="es_bios.xsd">',
|
||||
]
|
||||
|
||||
systems = truth_data.get("systems", {})
|
||||
for sys_id in sorted(systems):
|
||||
sys_data = systems[sys_id]
|
||||
files = sys_data.get("files", [])
|
||||
if not files:
|
||||
for system in sorted(systems.values(), key=lambda s: s.native_id):
|
||||
writable = [fe for fe in system.files if self.writable(fe)]
|
||||
if not writable:
|
||||
continue
|
||||
|
||||
native_id = native_map.get(sys_id, sys_id)
|
||||
scraped_sys = (
|
||||
scraped_data.get("systems", {}).get(sys_id) if scraped_data else None
|
||||
)
|
||||
display_name = self._display_name(sys_id, scraped_sys)
|
||||
|
||||
lines.append(f' <system fullname="{display_name}" platform="{native_id}">')
|
||||
|
||||
# Build path lookup from scraped data for this system
|
||||
scraped_paths: dict[str, str] = {}
|
||||
if scraped_data:
|
||||
s_sys = scraped_data.get("systems", {}).get(sys_id, {})
|
||||
for sf in s_sys.get("files", []):
|
||||
sname = sf.get("name", "").lower()
|
||||
spath = sf.get("destination", sf.get("name", ""))
|
||||
if sname and spath:
|
||||
scraped_paths[sname] = spath
|
||||
|
||||
for fe in files:
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
continue
|
||||
|
||||
# Use scraped path when available (preserves original format)
|
||||
path = scraped_paths.get(name.lower())
|
||||
if not path:
|
||||
dest = self._dest(fe)
|
||||
path = f"{native_id}/{dest}" if "/" not in dest else dest
|
||||
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = ",".join(md5)
|
||||
|
||||
required = fe.get("required", True)
|
||||
|
||||
# Build cores string from _cores
|
||||
cores_list = fe.get("_cores", [])
|
||||
core_str = (
|
||||
",".join(f"libretro/{c}" for c in cores_list) if cores_list else ""
|
||||
)
|
||||
|
||||
attrs = [f'path="{path}"']
|
||||
if md5:
|
||||
attrs.append(f'md5="{md5}"')
|
||||
if not required:
|
||||
attrs.append('mandatory="false"')
|
||||
if not required:
|
||||
attrs.append('hashMatchMandatory="true"')
|
||||
if core_str:
|
||||
attrs.append(f'core="{core_str}"')
|
||||
|
||||
lines.append(f" <bios {' '.join(attrs)} />")
|
||||
|
||||
fullname = quoteattr(self.display_name(system))
|
||||
platform = quoteattr(system.native_id)
|
||||
lines.append(f" <system fullname={fullname} platform={platform}>")
|
||||
for fe in writable:
|
||||
lines.append(self._bios_element(fe, system.native_id))
|
||||
lines.append(" </system>")
|
||||
|
||||
lines.append("</biosList>")
|
||||
lines.append("")
|
||||
Path(output_path).write_text("\n".join(lines), encoding="utf-8")
|
||||
|
||||
def validate(self, truth_data: dict, output_path: str) -> list[str]:
|
||||
from xml.etree.ElementTree import parse as xml_parse
|
||||
|
||||
tree = xml_parse(output_path)
|
||||
root = tree.getroot()
|
||||
|
||||
exported_paths: set[str] = set()
|
||||
for bios_el in root.iter("bios"):
|
||||
path = bios_el.get("path", "")
|
||||
if path:
|
||||
exported_paths.add(path.lower())
|
||||
exported_paths.add(path.split("/")[-1].lower())
|
||||
return {self.native_filename(): "\n".join(lines)}
|
||||
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
content = produced[self.native_filename()]
|
||||
issues: list[str] = []
|
||||
for sys_data in truth_data.get("systems", {}).values():
|
||||
for fe in sys_data.get("files", []):
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
try:
|
||||
root = parse_untrusted_xml(content, self.native_filename())
|
||||
except (ParseError, ValueError) as exc:
|
||||
return [f"the XML does not parse: {exc}"]
|
||||
|
||||
for element in root.iter("bios"):
|
||||
for attribute in ("path", "md5", "core"):
|
||||
if not element.get(attribute):
|
||||
issues.append(
|
||||
f"es_bios.xsd requires {attribute}: "
|
||||
f"{element.get('path', '?')}"
|
||||
)
|
||||
for element in root.iter("system"):
|
||||
if not list(element):
|
||||
issues.append(f"empty system: {element.get('platform', '?')}")
|
||||
for attribute in ("fullname", "platform"):
|
||||
if not element.get(attribute):
|
||||
issues.append(f"es_bios.xsd requires {attribute} on system")
|
||||
|
||||
exported = {
|
||||
element.get("path", "").casefold() for element in root.iter("bios")
|
||||
}
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
if not self.writable(fe):
|
||||
continue
|
||||
dest = self._dest(fe)
|
||||
if (
|
||||
name.lower() not in exported_paths
|
||||
and dest.lower() not in exported_paths
|
||||
):
|
||||
issues.append(f"missing: {name}")
|
||||
if self._path(fe, system.native_id).casefold() not in exported:
|
||||
issues.append(f"absent: {system.native_id}/{fe.name}")
|
||||
return issues
|
||||
@@ -1,116 +1,118 @@
|
||||
"""Exporter for RetroBat batocera-systems.json format.
|
||||
"""Exporter for RetroBat's batocera-systems.json.
|
||||
|
||||
Produces JSON matching the exact format of
|
||||
RetroBat-Official/emulatorlauncher/batocera-systems/Resources/batocera-systems.json:
|
||||
- System keys with "name" and "biosFiles" fields
|
||||
- Each biosFile has "md5" before "file" (matching original key order)
|
||||
Pure data: a system key carrying a name and a biosFiles array whose entries
|
||||
state md5 then file, in that order.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from collections import OrderedDict
|
||||
from pathlib import Path
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeSystem, Report
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://raw.githubusercontent.com/RetroBat-Official/emulatorlauncher/master"
|
||||
"/batocera-systems/Resources/batocera-systems.json"
|
||||
)
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Export truth data to RetroBat batocera-systems.json format."""
|
||||
"""Write RetroBat's batocera-systems.json, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "retrobat"
|
||||
|
||||
def export(
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "batocera-systems.json"
|
||||
|
||||
@staticmethod
|
||||
def requires() -> str:
|
||||
return "md5"
|
||||
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"md5"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"batocera-systems.json": SOURCE_URL}
|
||||
|
||||
def render(
|
||||
self,
|
||||
truth_data: dict,
|
||||
output_path: str,
|
||||
scraped_data: dict | None = None,
|
||||
) -> None:
|
||||
native_map: dict[str, str] = {}
|
||||
if scraped_data:
|
||||
for sys_id, sys_data in scraped_data.get("systems", {}).items():
|
||||
nid = sys_data.get("native_id")
|
||||
if nid:
|
||||
native_map[sys_id] = nid
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
# Keep the platform's own key order when we have its file, so the
|
||||
# diff a maintainer reads is the corrections and nothing else.
|
||||
order: list[str] = []
|
||||
original = originals.get(self.native_filename(), "")
|
||||
if original:
|
||||
try:
|
||||
order = list(json.loads(original))
|
||||
except json.JSONDecodeError:
|
||||
order = []
|
||||
|
||||
output: OrderedDict[str, dict] = OrderedDict()
|
||||
|
||||
systems = truth_data.get("systems", {})
|
||||
for sys_id in sorted(systems):
|
||||
sys_data = systems[sys_id]
|
||||
files = sys_data.get("files", [])
|
||||
if not files:
|
||||
continue
|
||||
|
||||
native_id = native_map.get(sys_id, sys_id)
|
||||
scraped_sys = (
|
||||
scraped_data.get("systems", {}).get(sys_id) if scraped_data else None
|
||||
)
|
||||
display_name = self._display_name(sys_id, scraped_sys)
|
||||
bios_files: list[OrderedDict] = []
|
||||
exportable = {
|
||||
system.native_id: (system, files)
|
||||
for system, files in self.exportable(systems, require="md5")
|
||||
}
|
||||
keys = [k for k in order if k in exportable]
|
||||
keys.extend(sorted(k for k in exportable if k not in keys))
|
||||
|
||||
output: OrderedDict[str, object] = OrderedDict()
|
||||
for key in keys:
|
||||
system, files = exportable[key]
|
||||
bios_files = []
|
||||
for fe in files:
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
continue
|
||||
dest = self._dest(fe)
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = md5[0] if md5 else ""
|
||||
|
||||
# Original format requires md5 for every entry
|
||||
if not md5:
|
||||
continue
|
||||
entry: OrderedDict[str, str] = OrderedDict()
|
||||
entry["md5"] = md5
|
||||
entry["file"] = f"bios/{dest}"
|
||||
entry["md5"] = fe.hash("md5")
|
||||
declared = fe.native("native_path", "")
|
||||
entry["file"] = str(declared) if declared else f"bios/{fe.destination}"
|
||||
bios_files.append(entry)
|
||||
system_entry: OrderedDict[str, object] = OrderedDict()
|
||||
system_entry["name"] = self.display_name(system)
|
||||
system_entry["biosFiles"] = bios_files
|
||||
output[key] = system_entry
|
||||
|
||||
if bios_files:
|
||||
if native_id in output:
|
||||
existing_files = {
|
||||
e.get("file") for e in output[native_id]["biosFiles"]
|
||||
}
|
||||
for entry in bios_files:
|
||||
if entry.get("file") not in existing_files:
|
||||
output[native_id]["biosFiles"].append(entry)
|
||||
else:
|
||||
sys_entry: OrderedDict[str, object] = OrderedDict()
|
||||
sys_entry["name"] = display_name
|
||||
sys_entry["biosFiles"] = bios_files
|
||||
output[native_id] = sys_entry
|
||||
text = json.dumps(output, indent=2, ensure_ascii=False) + "\n"
|
||||
return {self.native_filename(): text}
|
||||
|
||||
Path(output_path).write_text(
|
||||
json.dumps(output, indent=2, ensure_ascii=False) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
def validate(self, truth_data: dict, output_path: str) -> list[str]:
|
||||
data = json.loads(Path(output_path).read_text(encoding="utf-8"))
|
||||
|
||||
exported_files: set[str] = set()
|
||||
for sys_data in data.values():
|
||||
for bf in sys_data.get("biosFiles", []):
|
||||
path = bf.get("file", "")
|
||||
stripped = path.removeprefix("bios/")
|
||||
exported_files.add(stripped)
|
||||
basename = path.split("/")[-1] if "/" in path else path
|
||||
exported_files.add(basename)
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
try:
|
||||
data = json.loads(produced[self.native_filename()])
|
||||
except json.JSONDecodeError as exc:
|
||||
return [f"the JSON does not parse: {exc}"]
|
||||
|
||||
issues: list[str] = []
|
||||
for sys_data in truth_data.get("systems", {}).values():
|
||||
for fe in sys_data.get("files", []):
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
continue
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = md5[0] if md5 else ""
|
||||
if not md5:
|
||||
continue
|
||||
dest = self._dest(fe)
|
||||
if name not in exported_files and dest not in exported_files:
|
||||
issues.append(f"missing: {name}")
|
||||
for key, entry in data.items():
|
||||
if not entry.get("name"):
|
||||
issues.append(f"system without a name: {key}")
|
||||
if not entry.get("biosFiles"):
|
||||
issues.append(f"empty entry: {key}")
|
||||
for bios in entry.get("biosFiles", []):
|
||||
# RetroBat states an unhashed file with an empty md5, so only
|
||||
# a missing path makes an entry unusable.
|
||||
if "md5" not in bios or not bios.get("file"):
|
||||
issues.append(f"incomplete entry: {key}/{bios.get('file', '?')}")
|
||||
|
||||
for system, files in self.exportable(systems, require="md5"):
|
||||
entry = data.get(system.native_id)
|
||||
if entry is None:
|
||||
issues.append(f"system absent: {system.native_id}")
|
||||
continue
|
||||
declared = {bios.get("file") for bios in entry.get("biosFiles", [])}
|
||||
for fe in files:
|
||||
path = str(fe.native("native_path", "")) or f"bios/{fe.destination}"
|
||||
if path not in declared:
|
||||
issues.append(f"absent: {system.native_id}/{fe.name}")
|
||||
return issues
|
||||
@@ -1,210 +1,182 @@
|
||||
"""Exporter for RetroDECK component_manifest.json format.
|
||||
"""Exporter for RetroDECK's component manifests.
|
||||
|
||||
Produces a JSON file compatible with RetroDECK's component manifests.
|
||||
Each system maps to a component with BIOS entries containing filename,
|
||||
md5 (comma-separated if multiple), paths ($bios_path default), and
|
||||
required status.
|
||||
|
||||
Path tokens: $bios_path for bios/, $roms_path for roms/.
|
||||
Entries without an explicit path default to $bios_path.
|
||||
RetroDECK has no single BIOS file. Each component carries its own
|
||||
component_manifest.json, and the BIOS list sits inside it next to the
|
||||
component's name, description and presets, at one of three keys. Only that
|
||||
list is rewritten, in the component's own file, so everything else the
|
||||
manifest drives is left alone.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from collections import OrderedDict
|
||||
from pathlib import Path
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeFile, NativeSystem, Report
|
||||
|
||||
# retrobios slug -> RetroDECK system ID (reverse of scraper SYSTEM_SLUG_MAP)
|
||||
_REVERSE_SLUG: dict[str, str] = {
|
||||
"nintendo-nes": "nes",
|
||||
"nintendo-snes": "snes",
|
||||
"nintendo-64": "n64",
|
||||
"nintendo-64dd": "n64dd",
|
||||
"nintendo-gamecube": "gc",
|
||||
"nintendo-wii": "wii",
|
||||
"nintendo-wii-u": "wiiu",
|
||||
"nintendo-switch": "switch",
|
||||
"nintendo-gb": "gb",
|
||||
"nintendo-gbc": "gbc",
|
||||
"nintendo-gba": "gba",
|
||||
"nintendo-ds": "nds",
|
||||
"nintendo-3ds": "3ds",
|
||||
"nintendo-fds": "fds",
|
||||
"nintendo-sgb": "sgb",
|
||||
"nintendo-virtual-boy": "virtualboy",
|
||||
"nintendo-pokemon-mini": "pokemini",
|
||||
"sony-playstation": "psx",
|
||||
"sony-playstation-2": "ps2",
|
||||
"sony-playstation-3": "ps3",
|
||||
"sony-psp": "psp",
|
||||
"sony-psvita": "psvita",
|
||||
"sega-mega-drive": "megadrive",
|
||||
"sega-mega-cd": "megacd",
|
||||
"sega-saturn": "saturn",
|
||||
"sega-dreamcast": "dreamcast",
|
||||
"sega-dreamcast-arcade": "naomi",
|
||||
"sega-game-gear": "gamegear",
|
||||
"sega-master-system": "mastersystem",
|
||||
"nec-pc-engine": "pcengine",
|
||||
"nec-pc-fx": "pcfx",
|
||||
"nec-pc-98": "pc98",
|
||||
"nec-pc-88": "pc88",
|
||||
"3do": "3do",
|
||||
"amstrad-cpc": "amstradcpc",
|
||||
"arcade": "arcade",
|
||||
"atari-400-800": "atari800",
|
||||
"atari-5200": "atari5200",
|
||||
"atari-7800": "atari7800",
|
||||
"atari-jaguar": "atarijaguar",
|
||||
"atari-lynx": "atarilynx",
|
||||
"atari-st": "atarist",
|
||||
"commodore-c64": "c64",
|
||||
"commodore-amiga": "amiga",
|
||||
"philips-cdi": "cdimono1",
|
||||
"fairchild-channel-f": "channelf",
|
||||
"coleco-colecovision": "colecovision",
|
||||
"mattel-intellivision": "intellivision",
|
||||
"microsoft-msx": "msx",
|
||||
"microsoft-xbox": "xbox",
|
||||
"doom": "doom",
|
||||
"j2me": "j2me",
|
||||
"apple-macintosh-ii": "macintosh",
|
||||
"apple-ii": "apple2",
|
||||
"apple-iigs": "apple2gs",
|
||||
"enterprise-64-128": "enterprise",
|
||||
"tiger-game-com": "gamecom",
|
||||
"hartung-game-master": "gmaster",
|
||||
"epoch-scv": "scv",
|
||||
"watara-supervision": "supervision",
|
||||
"bandai-wonderswan": "wonderswan",
|
||||
"snk-neogeo-cd": "neogeocd",
|
||||
"tandy-coco": "coco",
|
||||
"tandy-trs-80": "trs80",
|
||||
"dragon-32-64": "dragon",
|
||||
"pico8": "pico8",
|
||||
"wolfenstein-3d": "wolfenstein",
|
||||
"sinclair-zx-spectrum": "zxspectrum",
|
||||
}
|
||||
|
||||
|
||||
def _dest_to_path_token(destination: str) -> str:
|
||||
"""Convert a truth destination path to a RetroDECK path token."""
|
||||
if destination.startswith("roms/"):
|
||||
return "$roms_path/" + destination.removeprefix("roms/")
|
||||
if destination.startswith("bios/"):
|
||||
return "$bios_path/" + destination.removeprefix("bios/")
|
||||
# Default: bios path
|
||||
return "$bios_path/" + destination
|
||||
COMPONENTS_REPO = "RetroDECK/components"
|
||||
COMPONENTS_BRANCH = "main"
|
||||
RAW_BASE = f"https://raw.githubusercontent.com/{COMPONENTS_REPO}/{COMPONENTS_BRANCH}"
|
||||
MANIFEST = "component_manifest.json"
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Export truth data to RetroDECK component_manifest.json format."""
|
||||
"""Write RetroDECK's component manifests, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "retrodeck"
|
||||
|
||||
def export(
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return MANIFEST
|
||||
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"md5", "sha256", "required"})
|
||||
|
||||
@staticmethod
|
||||
def needs_original() -> bool:
|
||||
# A manifest is mostly presets and launch configuration; rebuilding
|
||||
# one from BIOS data alone would throw the component away.
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def component_url(component: str) -> str:
|
||||
return f"{RAW_BASE}/{component}/{MANIFEST}"
|
||||
|
||||
def components(self, systems: dict[str, NativeSystem]) -> list[str]:
|
||||
"""Components the corrected data touches."""
|
||||
found: set[str] = set()
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
component = str(fe.native("component", ""))
|
||||
if component:
|
||||
found.add(component)
|
||||
return sorted(found)
|
||||
|
||||
@staticmethod
|
||||
def _entry(fe: NativeFile) -> OrderedDict:
|
||||
entry: OrderedDict[str, object] = OrderedDict()
|
||||
entry["filename"] = fe.name
|
||||
md5 = ",".join(fe.hashes("md5"))
|
||||
if md5:
|
||||
entry["md5"] = md5
|
||||
sha256 = fe.hash("sha256")
|
||||
if sha256:
|
||||
entry["sha256"] = sha256
|
||||
entry["system"] = fe.native_system
|
||||
description = fe.native("description", "")
|
||||
if description:
|
||||
entry["description"] = str(description)
|
||||
# RetroDECK words the requirement in prose ("Required", "At least one
|
||||
# BIOS file required"), so the platform's own wording is kept and a
|
||||
# boolean is only rendered when there is none to keep.
|
||||
label = fe.native("required_label", "")
|
||||
if label:
|
||||
entry["required"] = str(label)
|
||||
elif fe.required:
|
||||
entry["required"] = "Required"
|
||||
destination = fe.destination
|
||||
if destination and destination not in (fe.name, f"bios/{fe.name}"):
|
||||
directory = destination.rsplit("/", 1)[0]
|
||||
entry["paths"] = "$bios_path/" + directory.removeprefix("bios/")
|
||||
return entry
|
||||
|
||||
def _by_component(
|
||||
self, systems: dict[str, NativeSystem]
|
||||
) -> dict[str, list[NativeFile]]:
|
||||
grouped: dict[str, list[NativeFile]] = {}
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
component = str(fe.native("component", ""))
|
||||
if component:
|
||||
grouped.setdefault(component, []).append(fe)
|
||||
return grouped
|
||||
|
||||
@staticmethod
|
||||
def _bios_holder(component_value: dict) -> tuple[dict, str] | None:
|
||||
"""Where in a manifest the BIOS list lives, if it has one."""
|
||||
if "bios" in component_value:
|
||||
return component_value, "bios"
|
||||
for key in ("preset_actions", "cores"):
|
||||
nested = component_value.get(key)
|
||||
if isinstance(nested, dict) and "bios" in nested:
|
||||
return nested, "bios"
|
||||
return None
|
||||
|
||||
def render(
|
||||
self,
|
||||
truth_data: dict,
|
||||
output_path: str,
|
||||
scraped_data: dict | None = None,
|
||||
) -> None:
|
||||
native_map: dict[str, str] = {}
|
||||
if scraped_data:
|
||||
for sys_id, sys_data in scraped_data.get("systems", {}).items():
|
||||
nid = sys_data.get("native_id")
|
||||
if nid:
|
||||
native_map[sys_id] = nid
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
grouped = self._by_component(systems)
|
||||
produced: dict[str, str] = {}
|
||||
|
||||
manifest: OrderedDict[str, dict] = OrderedDict()
|
||||
|
||||
systems = truth_data.get("systems", {})
|
||||
for sys_id in sorted(systems):
|
||||
sys_data = systems[sys_id]
|
||||
files = sys_data.get("files", [])
|
||||
if not files:
|
||||
for component, files in sorted(grouped.items()):
|
||||
path = f"{component}/{MANIFEST}"
|
||||
original = originals.get(path)
|
||||
if not original:
|
||||
continue
|
||||
try:
|
||||
manifest = json.loads(original, object_pairs_hook=OrderedDict)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
|
||||
native_id = native_map.get(sys_id, _REVERSE_SLUG.get(sys_id, sys_id))
|
||||
|
||||
bios_entries: list[OrderedDict] = []
|
||||
for fe in files:
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
entries = [self._entry(fe) for fe in files]
|
||||
for component_value in manifest.values():
|
||||
if not isinstance(component_value, dict):
|
||||
continue
|
||||
|
||||
dest = self._dest(fe)
|
||||
path_token = _dest_to_path_token(dest)
|
||||
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = ",".join(m for m in md5 if m)
|
||||
|
||||
required = fe.get("required", True)
|
||||
|
||||
entry: OrderedDict[str, object] = OrderedDict()
|
||||
entry["filename"] = name
|
||||
if md5:
|
||||
# Validate MD5 entries
|
||||
parts = [
|
||||
m.strip().lower()
|
||||
for m in str(md5).split(",")
|
||||
if re.fullmatch(r"[0-9a-f]{32}", m.strip())
|
||||
]
|
||||
if parts:
|
||||
entry["md5"] = ",".join(parts) if len(parts) > 1 else parts[0]
|
||||
entry["paths"] = path_token
|
||||
entry["required"] = required
|
||||
|
||||
system_val = native_id
|
||||
entry["system"] = system_val
|
||||
|
||||
bios_entries.append(entry)
|
||||
|
||||
if bios_entries:
|
||||
if native_id in manifest:
|
||||
# Merge into existing component (multiple truth systems
|
||||
# may map to the same native ID)
|
||||
existing_names = {
|
||||
e["filename"] for e in manifest[native_id]["bios"]
|
||||
}
|
||||
for entry in bios_entries:
|
||||
if entry["filename"] not in existing_names:
|
||||
manifest[native_id]["bios"].append(entry)
|
||||
holder = self._bios_holder(component_value)
|
||||
if holder is None:
|
||||
component_value["bios"] = entries
|
||||
else:
|
||||
component = OrderedDict()
|
||||
component["system"] = native_id
|
||||
component["bios"] = bios_entries
|
||||
manifest[native_id] = component
|
||||
container, key = holder
|
||||
container[key] = entries
|
||||
break
|
||||
|
||||
Path(output_path).write_text(
|
||||
json.dumps(manifest, indent=2, ensure_ascii=False) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
produced[path] = json.dumps(manifest, indent=2, ensure_ascii=False) + "\n"
|
||||
|
||||
def validate(self, truth_data: dict, output_path: str) -> list[str]:
|
||||
data = json.loads(Path(output_path).read_text(encoding="utf-8"))
|
||||
|
||||
exported_names: set[str] = set()
|
||||
for comp_data in data.values():
|
||||
bios = comp_data.get("bios", [])
|
||||
if isinstance(bios, list):
|
||||
for entry in bios:
|
||||
fn = entry.get("filename", "")
|
||||
if fn:
|
||||
exported_names.add(fn)
|
||||
return produced
|
||||
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
issues: list[str] = []
|
||||
for sys_data in truth_data.get("systems", {}).values():
|
||||
for fe in sys_data.get("files", []):
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
grouped = self._by_component(systems)
|
||||
|
||||
for component, files in grouped.items():
|
||||
path = f"{component}/{MANIFEST}"
|
||||
if path not in produced:
|
||||
issues.append(f"manifest not written: {path}")
|
||||
continue
|
||||
try:
|
||||
manifest = json.loads(produced[path])
|
||||
except json.JSONDecodeError as exc:
|
||||
issues.append(f"{path} does not parse: {exc}")
|
||||
continue
|
||||
|
||||
declared: set[str] = set()
|
||||
for component_value in manifest.values():
|
||||
if not isinstance(component_value, dict):
|
||||
continue
|
||||
if name not in exported_names:
|
||||
issues.append(f"missing: {name}")
|
||||
holder = self._bios_holder(component_value)
|
||||
if holder is None:
|
||||
continue
|
||||
container, key = holder
|
||||
for entry in container[key]:
|
||||
declared.add(entry.get("filename", ""))
|
||||
if not component_value.get("name") and not component_value.get(
|
||||
"system"
|
||||
):
|
||||
issues.append(f"{path}: the component lost its identity")
|
||||
|
||||
for fe in files:
|
||||
if fe.name not in declared:
|
||||
issues.append(f"absent from {path}: {fe.name}")
|
||||
return issues
|
||||
@@ -1,17 +1,312 @@
|
||||
"""Exporter for RetroPie (System.dat format, same as RetroArch).
|
||||
"""Exporter for RetroPie's scriptmodules.
|
||||
|
||||
RetroPie inherits RetroArch cores and uses the same System.dat format.
|
||||
Delegates to systemdat_exporter for export and validation.
|
||||
RetroPie ships no BIOS list. What it maintains is one shell script per
|
||||
package, and the BIOS files a package needs are named in its
|
||||
`rp_module_help` string, in prose a person reads before copying files.
|
||||
platforms.cfg carries extensions and full names only.
|
||||
|
||||
So the correctable unit is that sentence, and the correction is a file name
|
||||
missing from it. A name RetroPie already writes is never removed, and no
|
||||
sentence is invented where a maintainer wrote none: a package we would have
|
||||
to document from scratch is reported, not drafted.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from .systemdat_exporter import Exporter as SystemDatExporter
|
||||
import io
|
||||
import re
|
||||
import sys
|
||||
import tarfile
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from common import load_emulator_profiles
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeFile, NativeSystem, Report, search_order
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://codeload.github.com/RetroPie/RetroPie-Setup/tar.gz/refs/heads/master"
|
||||
)
|
||||
ARCHIVE = "RetroPie-Setup.tar.gz"
|
||||
# Resolved from the file rather than the working directory: the profiles
|
||||
# are what say which files a package needs.
|
||||
EMULATORS_DIR = Path(__file__).resolve().parents[2] / "emulators"
|
||||
|
||||
# The longest list RetroPie writes is six names (lr-atari800). Past that the
|
||||
# sentence stops being something a reader uses, so the names are reported
|
||||
# instead of appended.
|
||||
MAX_NAMES = 6
|
||||
|
||||
_MODULE_ID = re.compile(r'rp_module_id="([^"]+)"')
|
||||
_MODULE_HELP = re.compile(r'rp_module_help="((?:[^"\\]|\\.)*)"')
|
||||
_FILENAME = re.compile(r"[A-Za-z0-9][\w.+-]*\.[A-Za-z0-9]{1,5}\b")
|
||||
# RetroPie words the instruction several ways ("Copy the required BIOS files
|
||||
# a.bin and b.bin to $biosdir", "The Sega CD requires the BIOS files a.bin,
|
||||
# b.bin copied to $biosdir"), so the clause is found by the word BIOS rather
|
||||
# than by a sentence template, and the names in it are the list to extend.
|
||||
_BIOS_CLAUSE = re.compile(r"BIOS\b.*?(?=\\n|$)", re.DOTALL)
|
||||
_BIOS_NOUN = re.compile(r"BIOS (files?)\b")
|
||||
|
||||
|
||||
class Exporter(SystemDatExporter):
|
||||
"""Export truth data to RetroPie System.dat format."""
|
||||
class Exporter(BaseExporter):
|
||||
"""Write RetroPie's scriptmodules, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "retropie"
|
||||
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "scriptmodules"
|
||||
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
# The help names files; it states no hash and no requirement flag.
|
||||
return frozenset({"name"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {ARCHIVE: SOURCE_URL}
|
||||
|
||||
@staticmethod
|
||||
def needs_original() -> bool:
|
||||
# The declaration is a sentence inside a shell script.
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def may_write_nothing() -> bool:
|
||||
# Nothing to correct is a result, not a failure.
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def unpack(raw: bytes) -> dict[str, str]:
|
||||
"""Keep the scriptmodules that say anything about BIOS files."""
|
||||
found: dict[str, str] = {}
|
||||
with tarfile.open(fileobj=io.BytesIO(raw), mode="r:gz") as archive:
|
||||
for member in archive:
|
||||
if not member.isfile() or not member.name.endswith(".sh"):
|
||||
continue
|
||||
relative = member.name.split("/", 1)[-1]
|
||||
if not relative.startswith("scriptmodules/"):
|
||||
continue
|
||||
handle = archive.extractfile(member)
|
||||
if handle is None:
|
||||
continue
|
||||
text = handle.read().decode("utf-8", errors="replace")
|
||||
if "BIOS" in text:
|
||||
found[relative] = text
|
||||
return found
|
||||
|
||||
def _core_index(self) -> dict[str, str]:
|
||||
"""Module id (without its lr- prefix) to the profile it stands for."""
|
||||
index: dict[str, str] = {}
|
||||
for key, profile in load_emulator_profiles(str(EMULATORS_DIR)).items():
|
||||
index[key.replace("-", "_").lower()] = key
|
||||
for name in profile.get("cores", []) or []:
|
||||
index[str(name).replace("-", "_").lower()] = key
|
||||
return index
|
||||
|
||||
@staticmethod
|
||||
def _files_by_core(systems: dict[str, NativeSystem]) -> dict[str, list[NativeFile]]:
|
||||
"""Which files each core asks for, as the truth read its source."""
|
||||
grouped: dict[str, list[NativeFile]] = {}
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
for core in (fe.truth or {}).get("_cores", []):
|
||||
grouped.setdefault(str(core), []).append(fe)
|
||||
return grouped
|
||||
|
||||
@staticmethod
|
||||
def _names_in(text: str) -> list[re.Match[str]]:
|
||||
"""File names in a fragment, skipping what only looks like one.
|
||||
|
||||
A token preceded by a dot is the tail of a ROM extension list
|
||||
(.atr.gz), and one preceded by a separator is part of a path
|
||||
($biosdir/Machines/COL/coleco.rom): neither is a name in a list.
|
||||
"""
|
||||
found: list[re.Match[str]] = []
|
||||
for match in _FILENAME.finditer(text):
|
||||
before = text[match.start() - 1 : match.start()]
|
||||
if before in (".", "/", "\\"):
|
||||
continue
|
||||
found.append(match)
|
||||
return found
|
||||
|
||||
@classmethod
|
||||
def _listed(cls, help_text: str) -> set[str]:
|
||||
"""File names the help already writes, wherever in the string."""
|
||||
plain = help_text.replace("\\n", " ")
|
||||
return {match.group(0).lower() for match in cls._names_in(plain)}
|
||||
|
||||
@classmethod
|
||||
def _insertion_point(cls, help_text: str) -> int | None:
|
||||
"""Where a name joins the list, or None when there is no list."""
|
||||
clause = _BIOS_CLAUSE.search(help_text)
|
||||
if clause is None:
|
||||
return None
|
||||
names = cls._names_in(clause.group(0))
|
||||
return clause.start() + names[-1].end() if names else None
|
||||
|
||||
@staticmethod
|
||||
def _in_search_order(candidates: list[NativeFile]) -> list[str]:
|
||||
"""The names in the order the code looks for them, best first.
|
||||
|
||||
Someone reading the sentence copies the files in the order it
|
||||
gives, so the one the emulator prefers is named first. The model
|
||||
already holds them in that order; this only removes the duplicates
|
||||
a name can pick up from several cores.
|
||||
"""
|
||||
names: list[str] = []
|
||||
for fe in search_order(candidates):
|
||||
if fe.name not in names:
|
||||
names.append(fe.name)
|
||||
return names
|
||||
|
||||
@staticmethod
|
||||
def _join(names: list[str]) -> str:
|
||||
"""RetroPie's own idiom: a, b and c."""
|
||||
if len(names) == 1:
|
||||
return names[0]
|
||||
return ", ".join(names[:-1]) + " and " + names[-1]
|
||||
|
||||
def modules(self, originals: dict[str, str]) -> dict[str, tuple[str, str]]:
|
||||
"""Module id and help string of every script that mentions BIOS."""
|
||||
found: dict[str, tuple[str, str]] = {}
|
||||
for relative, text in originals.items():
|
||||
if not relative.startswith("scriptmodules/"):
|
||||
continue
|
||||
module = _MODULE_ID.search(text)
|
||||
help_text = _MODULE_HELP.search(text)
|
||||
if not module or not help_text or "BIOS" not in help_text.group(1):
|
||||
continue
|
||||
found[relative] = (module.group(1), help_text.group(1))
|
||||
return found
|
||||
|
||||
def render(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
modules = self.modules(originals)
|
||||
if not modules:
|
||||
raise ValueError(
|
||||
"the scriptmodules cannot be written without RetroPie's own "
|
||||
"repository: the BIOS list is a sentence inside a shell script"
|
||||
)
|
||||
|
||||
index = self._core_index()
|
||||
by_core = self._files_by_core(systems)
|
||||
# Aliases count as names we know: RetroPie writes dc_flash.bin where
|
||||
# flycast's profile files it as an alias of another primary.
|
||||
known: set[str] = set()
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
known.add(fe.name.lower())
|
||||
known.update(
|
||||
str(a).lower() for a in (fe.native("aliases", []) or [])
|
||||
)
|
||||
self._skipped: dict[str, list[str]] = {}
|
||||
produced: dict[str, str] = {}
|
||||
|
||||
def skip(reason: str, module: str) -> None:
|
||||
self._skipped.setdefault(reason, []).append(module)
|
||||
|
||||
for relative, (module_id, help_text) in sorted(modules.items()):
|
||||
core = index.get(module_id.removeprefix("lr-").replace("-", "_").lower())
|
||||
files = by_core.get(core or "", [])
|
||||
if not files:
|
||||
skip("no core of ours packages it", module_id)
|
||||
continue
|
||||
|
||||
listed = self._listed(help_text)
|
||||
unknown = sorted(listed - known)
|
||||
if unknown:
|
||||
skip(
|
||||
f"names a file we have never seen ({', '.join(unknown)})",
|
||||
module_id,
|
||||
)
|
||||
|
||||
# A file already listed under one of its other names is not
|
||||
# missing: proposing the primary would name the same bytes twice.
|
||||
candidates = [
|
||||
fe
|
||||
for fe in files
|
||||
if fe.required
|
||||
and not (
|
||||
{fe.name.lower()}
|
||||
| {str(a).lower() for a in (fe.native("aliases", []) or [])}
|
||||
)
|
||||
& listed
|
||||
]
|
||||
missing = self._in_search_order(candidates)
|
||||
if not missing:
|
||||
continue
|
||||
|
||||
insert_at = self._insertion_point(help_text)
|
||||
if insert_at is None:
|
||||
# No enumeration to extend, and a sentence we would have to
|
||||
# write ourselves is a documentation change, not a correction.
|
||||
skip("names no file to extend", module_id)
|
||||
continue
|
||||
|
||||
if len(listed) + len(missing) > MAX_NAMES:
|
||||
skip("more names than the help enumerates", module_id)
|
||||
continue
|
||||
|
||||
new_help = (
|
||||
help_text[:insert_at]
|
||||
+ ", "
|
||||
+ self._join(missing)
|
||||
+ help_text[insert_at:]
|
||||
)
|
||||
if len(listed) + len(missing) > 1:
|
||||
new_help = _BIOS_NOUN.sub("BIOS files", new_help, count=1)
|
||||
produced[relative] = originals[relative].replace(
|
||||
f'rp_module_help="{help_text}"',
|
||||
f'rp_module_help="{new_help}"',
|
||||
1,
|
||||
)
|
||||
|
||||
return produced
|
||||
|
||||
def outcome(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> str:
|
||||
"""Packages are the unit here, not file entries."""
|
||||
skipped: dict[str, list[str]] = getattr(self, "_skipped", {})
|
||||
parts = [f"{len(produced)} packages corrected"]
|
||||
for reason, modules in sorted(skipped.items()):
|
||||
shown = ", ".join(sorted(modules)[:4])
|
||||
if len(modules) > 4:
|
||||
shown += f" and {len(modules) - 4} more"
|
||||
parts.append(f"{len(modules)} {reason} ({shown})")
|
||||
return "; ".join(parts)
|
||||
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
issues: list[str] = []
|
||||
for relative, text in produced.items():
|
||||
module = _MODULE_ID.search(text)
|
||||
help_text = _MODULE_HELP.search(text)
|
||||
if not module:
|
||||
issues.append(f"{relative}: the module lost its id")
|
||||
if not help_text:
|
||||
issues.append(f"{relative}: the help string is not closed")
|
||||
continue
|
||||
if "BIOS" not in help_text.group(1):
|
||||
issues.append(f"{relative}: the BIOS sentence is gone")
|
||||
names = self._listed(help_text.group(1))
|
||||
if len(names) > MAX_NAMES:
|
||||
issues.append(
|
||||
f"{relative}: {len(names)} names, past what the help enumerates"
|
||||
)
|
||||
return issues
|
||||
@@ -0,0 +1,30 @@
|
||||
"""Exporter for ROCKNIX's rocknix-systems.
|
||||
|
||||
Same shape as Batocera's script: a systems mapping inside a checker ROCKNIX
|
||||
runs, so only the mapping is rewritten.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from .batocera_exporter import Exporter as BatoceraExporter
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://raw.githubusercontent.com/ROCKNIX/distribution/next/projects/ROCKNIX"
|
||||
"/packages/rocknix/sources/scripts/rocknix-systems"
|
||||
)
|
||||
|
||||
|
||||
class Exporter(BatoceraExporter):
|
||||
"""Write ROCKNIX's rocknix-systems, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "rocknix"
|
||||
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "rocknix-systems"
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"rocknix-systems": SOURCE_URL}
|
||||
+105
-132
@@ -1,160 +1,133 @@
|
||||
"""Exporter for RomM known_bios_files.json format.
|
||||
"""Exporter for RomM's known_bios_files.json.
|
||||
|
||||
Produces JSON matching the exact format of
|
||||
rommapp/romm/backend/models/fixtures/known_bios_files.json:
|
||||
- Keys are "igdb_slug:filename"
|
||||
- Values contain size, crc, md5, sha1 (all optional but at least one hash)
|
||||
- Hashes are lowercase hex strings
|
||||
- Size is an integer
|
||||
Keys are "<igdb slug>:<filename>". RomM verifies a firmware file with
|
||||
`file_size_bytes == int(entry.get("size", 0))` and then one hash among md5,
|
||||
sha1 and crc, so an entry without a size can never match and an entry
|
||||
without a hash can never match either. Neither is written.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import sys
|
||||
from collections import OrderedDict
|
||||
from pathlib import Path
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
# retrobios slug -> IGDB slug (reverse of scraper SLUG_MAP)
|
||||
_REVERSE_SLUG: dict[str, str] = {
|
||||
"3do": "3do",
|
||||
"nintendo-64dd": "64dd",
|
||||
"amstrad-cpc": "acpc",
|
||||
"commodore-amiga": "amiga",
|
||||
"arcade": "arcade",
|
||||
"atari-st": "atari-st",
|
||||
"atari-5200": "atari5200",
|
||||
"atari-7800": "atari7800",
|
||||
"atari-400-800": "atari8bit",
|
||||
"coleco-colecovision": "colecovision",
|
||||
"sega-dreamcast": "dc",
|
||||
"doom": "doom",
|
||||
"enterprise-64-128": "enterprise",
|
||||
"fairchild-channel-f": "fairchild-channel-f",
|
||||
"nintendo-fds": "fds",
|
||||
"sega-game-gear": "gamegear",
|
||||
"nintendo-gb": "gb",
|
||||
"nintendo-gba": "gba",
|
||||
"nintendo-gbc": "gbc",
|
||||
"sega-mega-drive": "genesis",
|
||||
"mattel-intellivision": "intellivision",
|
||||
"j2me": "j2me",
|
||||
"atari-lynx": "lynx",
|
||||
"apple-macintosh-ii": "mac",
|
||||
"microsoft-msx": "msx",
|
||||
"nintendo-ds": "nds",
|
||||
"snk-neogeo-cd": "neo-geo-cd",
|
||||
"nintendo-nes": "nes",
|
||||
"nintendo-gamecube": "ngc",
|
||||
"magnavox-odyssey2": "odyssey-2-slash-videopac-g7000",
|
||||
"nec-pc-98": "pc-9800-series",
|
||||
"nec-pc-fx": "pc-fx",
|
||||
"nintendo-pokemon-mini": "pokemon-mini",
|
||||
"sony-playstation-2": "ps2",
|
||||
"sony-psp": "psp",
|
||||
"sony-playstation": "psx",
|
||||
"nintendo-satellaview": "satellaview",
|
||||
"sega-saturn": "saturn",
|
||||
"scummvm": "scummvm",
|
||||
"sega-mega-cd": "segacd",
|
||||
"sharp-x68000": "sharp-x68000",
|
||||
"sega-master-system": "sms",
|
||||
"nintendo-snes": "snes",
|
||||
"nintendo-sufami-turbo": "sufami-turbo",
|
||||
"nintendo-sgb": "super-gb",
|
||||
"nec-pc-engine": "tg16",
|
||||
"videoton-tvc": "tvc",
|
||||
"philips-videopac": "videopac-g7400",
|
||||
"wolfenstein-3d": "wolfenstein",
|
||||
"sharp-x1": "x1",
|
||||
"microsoft-xbox": "xbox",
|
||||
"sinclair-zx-spectrum": "zxs",
|
||||
}
|
||||
from scraper.romm_scraper import SLUG_MAP
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeFile, NativeSystem, Report
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://raw.githubusercontent.com/rommapp/romm/master/backend/models"
|
||||
"/fixtures/known_bios_files.json"
|
||||
)
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Export truth data to RomM known_bios_files.json format."""
|
||||
"""Write RomM's known_bios_files.json, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "romm"
|
||||
|
||||
def export(
|
||||
self,
|
||||
truth_data: dict,
|
||||
output_path: str,
|
||||
scraped_data: dict | None = None,
|
||||
) -> None:
|
||||
native_map: dict[str, str] = {}
|
||||
if scraped_data:
|
||||
for sys_id, sys_data in scraped_data.get("systems", {}).items():
|
||||
nid = sys_data.get("native_id")
|
||||
if nid:
|
||||
native_map[sys_id] = nid
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "known_bios_files.json"
|
||||
|
||||
output: OrderedDict[str, dict] = OrderedDict()
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"size", "crc32", "md5", "sha1"})
|
||||
|
||||
systems = truth_data.get("systems", {})
|
||||
for sys_id in sorted(systems):
|
||||
sys_data = systems[sys_id]
|
||||
files = sys_data.get("files", [])
|
||||
if not files:
|
||||
continue
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"known_bios_files.json": SOURCE_URL}
|
||||
|
||||
igdb_slug = native_map.get(sys_id, _REVERSE_SLUG.get(sys_id, sys_id))
|
||||
|
||||
for fe in files:
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
continue
|
||||
|
||||
key = f"{igdb_slug}:{name}"
|
||||
|
||||
entry: OrderedDict[str, object] = OrderedDict()
|
||||
|
||||
size = fe.get("size")
|
||||
if size is not None:
|
||||
entry["size"] = int(size)
|
||||
|
||||
crc = fe.get("crc32", "")
|
||||
if crc:
|
||||
entry["crc"] = str(crc).strip().lower()
|
||||
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = md5[0] if md5 else ""
|
||||
if md5:
|
||||
entry["md5"] = str(md5).strip().lower()
|
||||
|
||||
sha1 = fe.get("sha1", "")
|
||||
if isinstance(sha1, list):
|
||||
sha1 = sha1[0] if sha1 else ""
|
||||
if sha1:
|
||||
entry["sha1"] = str(sha1).strip().lower()
|
||||
|
||||
output[key] = entry
|
||||
|
||||
Path(output_path).write_text(
|
||||
json.dumps(output, indent=2, ensure_ascii=False) + "\n",
|
||||
encoding="utf-8",
|
||||
@staticmethod
|
||||
def _verifiable(fe: NativeFile) -> bool:
|
||||
return bool(fe.size()) and any(
|
||||
fe.hash(h) for h in ("md5", "sha1", "crc32")
|
||||
)
|
||||
|
||||
def validate(self, truth_data: dict, output_path: str) -> list[str]:
|
||||
data = json.loads(Path(output_path).read_text(encoding="utf-8"))
|
||||
@staticmethod
|
||||
def _known_platform(native_id: str) -> bool:
|
||||
"""RomM keys by IGDB platform slug, and only looks up its own.
|
||||
|
||||
exported_names: set[str] = set()
|
||||
for key in data:
|
||||
if ":" in key:
|
||||
_, filename = key.split(":", 1)
|
||||
exported_names.add(filename)
|
||||
A key spelled with one of our slugs (capcom-cps3, snk-neogeo-mvs)
|
||||
matches nothing on their side, so it is reported rather than
|
||||
written.
|
||||
"""
|
||||
return native_id in SLUG_MAP
|
||||
|
||||
@classmethod
|
||||
def writable(cls, fe: NativeFile, require: str = "") -> bool:
|
||||
"""What RomM already ships stays; the conditions gate additions.
|
||||
|
||||
An entry of theirs that could never verify is still theirs, and the
|
||||
round trip is not the place to decide otherwise.
|
||||
"""
|
||||
if fe.platform is not None:
|
||||
return True
|
||||
return cls._verifiable(fe) and cls._known_platform(fe.native_system)
|
||||
|
||||
def render(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
output: OrderedDict[str, dict] = OrderedDict()
|
||||
|
||||
for system in sorted(systems.values(), key=lambda s: s.native_id):
|
||||
for fe in sorted(system.files, key=lambda f: f.name):
|
||||
if not self.writable(fe):
|
||||
continue
|
||||
entry: OrderedDict[str, str] = OrderedDict()
|
||||
# The fixture states every value as a string, size included.
|
||||
entry["size"] = str(fe.size())
|
||||
crc = fe.hash("crc32")
|
||||
if crc:
|
||||
entry["crc"] = crc
|
||||
md5 = fe.hash("md5")
|
||||
if md5:
|
||||
entry["md5"] = md5
|
||||
sha1 = fe.hash("sha1")
|
||||
if sha1:
|
||||
entry["sha1"] = sha1
|
||||
output[f"{system.native_id}:{fe.name}"] = entry
|
||||
|
||||
text = json.dumps(output, indent=2, ensure_ascii=False) + "\n"
|
||||
return {self.native_filename(): text}
|
||||
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
try:
|
||||
data = json.loads(produced[self.native_filename()])
|
||||
except json.JSONDecodeError as exc:
|
||||
return [f"the JSON does not parse: {exc}"]
|
||||
|
||||
issues: list[str] = []
|
||||
for sys_data in truth_data.get("systems", {}).values():
|
||||
for fe in sys_data.get("files", []):
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
for key, entry in data.items():
|
||||
slug = key.split(":", 1)[0] if ":" in key else ""
|
||||
if not slug:
|
||||
issues.append(f"key without a platform slug: {key}")
|
||||
elif not self._known_platform(slug):
|
||||
issues.append(f"platform slug RomM does not know: {slug}")
|
||||
if not entry.get("size"):
|
||||
issues.append(f"entry without a size, never verifiable: {key}")
|
||||
if not any(entry.get(h) for h in ("md5", "sha1", "crc")):
|
||||
issues.append(f"entry without a hash, never verifiable: {key}")
|
||||
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
if not self.writable(fe):
|
||||
continue
|
||||
if name not in exported_names:
|
||||
issues.append(f"missing: {name}")
|
||||
if f"{system.native_id}:{fe.name}" not in data:
|
||||
issues.append(f"absent: {system.native_id}/{fe.name}")
|
||||
return issues
|
||||
@@ -1,7 +1,7 @@
|
||||
"""Exporter for libretro System.dat (clrmamepro DAT format).
|
||||
|
||||
Produces a single 'game' block with all ROMs grouped by system,
|
||||
matching the exact format of libretro-database/dat/System.dat.
|
||||
One 'game' block, systems separated by a comment line carrying the name
|
||||
libretro gives them, matching libretro-database/dat/System.dat.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -14,43 +14,71 @@ sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
from scraper.dat_parser import parse_dat
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeSystem, Report
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://raw.githubusercontent.com/libretro/libretro-database/master/dat/System.dat"
|
||||
)
|
||||
|
||||
|
||||
def _slug_to_native(slug: str) -> str:
|
||||
"""Convert a system slug to 'Manufacturer - Console' format."""
|
||||
parts = slug.split("-", 1)
|
||||
if len(parts) == 1:
|
||||
return parts[0].title()
|
||||
manufacturer = parts[0].replace("-", " ").title()
|
||||
console = parts[1].replace("-", " ").title()
|
||||
return f"{manufacturer} - {console}"
|
||||
def _quote(name: str) -> str:
|
||||
"""Quote a ROM name the way the original does: only when it must be."""
|
||||
return f'"{name}"' if any(c in name for c in ' ()') else name
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Export truth data to libretro System.dat format."""
|
||||
"""Write libretro's System.dat, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "retroarch"
|
||||
|
||||
def export(
|
||||
self,
|
||||
truth_data: dict,
|
||||
output_path: str,
|
||||
scraped_data: dict | None = None,
|
||||
) -> None:
|
||||
native_map: dict[str, str] = {}
|
||||
if scraped_data:
|
||||
for sys_id, sys_data in scraped_data.get("systems", {}).items():
|
||||
nid = sys_data.get("native_id")
|
||||
if nid:
|
||||
native_map[sys_id] = nid
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "System.dat"
|
||||
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"size", "crc32", "md5", "sha1"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"System.dat": SOURCE_URL}
|
||||
|
||||
@classmethod
|
||||
def writable(cls, fe, require: str = "") -> bool:
|
||||
"""A clrmamepro rom line is a hash record, and the DAT has no other.
|
||||
|
||||
The original never states a rom it cannot hash, so neither do we.
|
||||
"""
|
||||
if fe.platform is not None:
|
||||
return True
|
||||
return any(fe.hash(h) for h in ("crc32", "md5", "sha1"))
|
||||
|
||||
@staticmethod
|
||||
def _rom_name(fe) -> str:
|
||||
"""The name the DAT gives a rom.
|
||||
|
||||
libretro writes the path for some entries (ep128emu/roms/cpc464.rom)
|
||||
and the bare name for others whose destination has a directory all
|
||||
the same (iplromco.dat, which lives under keropi/), so its own
|
||||
spelling is recorded rather than derived.
|
||||
"""
|
||||
declared = fe.native("native_path", "")
|
||||
return str(declared) if declared else fe.name
|
||||
|
||||
def _header(self, originals: dict[str, str], scraped: dict | None) -> list[str]:
|
||||
"""Reuse the original header verbatim when we have the original."""
|
||||
original = originals.get(self.native_filename(), "")
|
||||
if original:
|
||||
head, sep, _ = original.partition("\ngame (")
|
||||
if sep:
|
||||
return head.split("\n")
|
||||
|
||||
# Match exact header format of libretro-database/dat/System.dat
|
||||
version = ""
|
||||
if scraped_data:
|
||||
version = scraped_data.get("dat_version", scraped_data.get("version", ""))
|
||||
lines: list[str] = [
|
||||
if scraped:
|
||||
version = scraped.get("dat_version", scraped.get("version", ""))
|
||||
lines = [
|
||||
"clrmamepro (",
|
||||
'\tname "System"',
|
||||
'\tdescription "System"',
|
||||
@@ -61,73 +89,76 @@ class Exporter(BaseExporter):
|
||||
lines.extend(
|
||||
[
|
||||
'\tauthor "libretro"',
|
||||
'\thomepage "https://github.com/libretro/libretro-database/blob/master/dat/System.dat"',
|
||||
'\turl "https://raw.githubusercontent.com/libretro/libretro-database/master/dat/System.dat"',
|
||||
'\thomepage "https://github.com/libretro/libretro-database/blob/master'
|
||||
'/dat/System.dat"',
|
||||
'\turl "https://raw.githubusercontent.com/libretro/libretro-database'
|
||||
'/master/dat/System.dat"',
|
||||
")",
|
||||
"",
|
||||
"game (",
|
||||
'\tname "System"',
|
||||
'\tcomment "System"',
|
||||
]
|
||||
)
|
||||
return lines
|
||||
|
||||
systems = truth_data.get("systems", {})
|
||||
for sys_id in sorted(systems):
|
||||
sys_data = systems[sys_id]
|
||||
files = sys_data.get("files", [])
|
||||
if not files:
|
||||
continue
|
||||
|
||||
native_name = native_map.get(sys_id, _slug_to_native(sys_id))
|
||||
lines.append("")
|
||||
lines.append(f'\tcomment "{native_name}"')
|
||||
def render(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
lines = self._header(originals, scraped)
|
||||
lines.extend(["game (", '\tname "System"', '\tcomment "System"'])
|
||||
|
||||
for system, files in sorted(
|
||||
self.exportable(systems), key=lambda pair: pair[0].native_id
|
||||
):
|
||||
rendered: list[str] = []
|
||||
for fe in files:
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
if not any(fe.hash(h) for h in ("crc32", "md5", "sha1")):
|
||||
continue
|
||||
|
||||
# Quote names with spaces or special chars (matching original format)
|
||||
needs_quote = " " in name or "(" in name or ")" in name
|
||||
name_str = f'"{name}"' if needs_quote else name
|
||||
rom_parts = [f"name {name_str}"]
|
||||
size = fe.get("size")
|
||||
parts = [f"name {_quote(self._rom_name(fe))}"]
|
||||
size = fe.size()
|
||||
if size:
|
||||
rom_parts.append(f"size {size}")
|
||||
crc = fe.get("crc32", "")
|
||||
parts.append(f"size {size}")
|
||||
crc = fe.hash("crc32")
|
||||
if crc:
|
||||
rom_parts.append(f"crc {str(crc).upper()}")
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = md5[0] if md5 else ""
|
||||
parts.append(f"crc {crc.upper()}")
|
||||
md5 = fe.hash("md5")
|
||||
if md5:
|
||||
rom_parts.append(f"md5 {md5}")
|
||||
sha1 = fe.get("sha1", "")
|
||||
if isinstance(sha1, list):
|
||||
sha1 = sha1[0] if sha1 else ""
|
||||
parts.append(f"md5 {md5}")
|
||||
sha1 = fe.hash("sha1")
|
||||
if sha1:
|
||||
rom_parts.append(f"sha1 {sha1}")
|
||||
parts.append(f"sha1 {sha1}")
|
||||
rendered.append(f"\trom ( {' '.join(parts)} )")
|
||||
|
||||
lines.append(f"\trom ( {' '.join(rom_parts)} )")
|
||||
if not rendered:
|
||||
continue
|
||||
lines.append("")
|
||||
# libretro's comment is the system name as the DAT spells it,
|
||||
# "Atari - 400-800". Prettifying it drops the separator.
|
||||
lines.append(f'\tcomment "{system.native_id}"')
|
||||
lines.extend(rendered)
|
||||
|
||||
lines.append(")")
|
||||
lines.append("")
|
||||
Path(output_path).write_text("\n".join(lines), encoding="utf-8")
|
||||
return {self.native_filename(): "\n".join(lines)}
|
||||
|
||||
def validate(self, truth_data: dict, output_path: str) -> list[str]:
|
||||
content = Path(output_path).read_text(encoding="utf-8")
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
content = produced[self.native_filename()]
|
||||
parsed = parse_dat(content)
|
||||
|
||||
exported_names: set[str] = set()
|
||||
for rom in parsed:
|
||||
exported_names.add(rom.name)
|
||||
exported = {rom.name for rom in parsed}
|
||||
|
||||
issues: list[str] = []
|
||||
for sys_data in truth_data.get("systems", {}).values():
|
||||
for fe in sys_data.get("files", []):
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
if not any(fe.hash(h) for h in ("crc32", "md5", "sha1")):
|
||||
continue
|
||||
if name not in exported_names:
|
||||
issues.append(f"missing: {name}")
|
||||
if self._rom_name(fe) not in exported:
|
||||
issues.append(f"absent from the DAT: {system.native_id}/{fe.name}")
|
||||
if not content.rstrip().endswith(")"):
|
||||
issues.append("the game block is not closed")
|
||||
return issues
|
||||
@@ -50,11 +50,6 @@ def parse_args(argv: list[str] | None = None) -> argparse.Namespace:
|
||||
default=None,
|
||||
help="hardware target filter",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--include-archived",
|
||||
action="store_true",
|
||||
help="include archived platforms with --all",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--platforms-dir",
|
||||
default=DEFAULT_PLATFORMS_DIR,
|
||||
@@ -87,7 +82,7 @@ def main(argv: list[str] | None = None) -> None:
|
||||
if args.all:
|
||||
platforms = list_registered_platforms(
|
||||
args.platforms_dir,
|
||||
include_archived=args.include_archived,
|
||||
include_archived=True,
|
||||
)
|
||||
else:
|
||||
platforms = [args.platform]
|
||||
|
||||
+38
-21
@@ -190,6 +190,20 @@ def check_consistency(verify_output: str, pack_output: str) -> bool:
|
||||
return all_ok
|
||||
|
||||
|
||||
class _Skipped:
|
||||
"""A step that did not run. Truthy, so it never fails the pipeline, and
|
||||
distinct, so the summary does not report it as work that was done."""
|
||||
|
||||
def __bool__(self) -> bool:
|
||||
return True
|
||||
|
||||
def __repr__(self) -> str:
|
||||
return "SKIPPED"
|
||||
|
||||
|
||||
SKIPPED = _Skipped()
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description="Run the full retrobios pipeline")
|
||||
parser.add_argument(
|
||||
@@ -296,7 +310,7 @@ def main():
|
||||
all_ok = all_ok and ok
|
||||
else:
|
||||
print("\n--- 2/8 refresh data directories: SKIPPED (--offline) ---")
|
||||
results["refresh_data"] = True
|
||||
results["refresh_data"] = SKIPPED
|
||||
|
||||
# Step 2a: Refresh MAME BIOS hashes
|
||||
if not args.offline:
|
||||
@@ -308,7 +322,7 @@ def main():
|
||||
all_ok = all_ok and ok
|
||||
else:
|
||||
print("\n--- 2a refresh MAME hashes: SKIPPED (--offline) ---")
|
||||
results["mame_hashes"] = True
|
||||
results["mame_hashes"] = SKIPPED
|
||||
|
||||
# Step 2a2: Refresh FBNeo BIOS hashes
|
||||
if not args.offline:
|
||||
@@ -320,7 +334,7 @@ def main():
|
||||
all_ok = all_ok and ok
|
||||
else:
|
||||
print("\n--- 2a2 refresh FBNeo hashes: SKIPPED (--offline) ---")
|
||||
results["fbneo_hashes"] = True
|
||||
results["fbneo_hashes"] = SKIPPED
|
||||
|
||||
# Step 2b: Check buildbot system directory (non-blocking)
|
||||
if args.check_buildbot and not args.offline:
|
||||
@@ -341,27 +355,23 @@ def main():
|
||||
"--output-dir",
|
||||
str(Path(args.output_dir) / "truth"),
|
||||
]
|
||||
if args.include_archived:
|
||||
truth_cmd.append("--include-archived")
|
||||
if args.target:
|
||||
truth_cmd.extend(["--target", args.target])
|
||||
ok, _ = run(truth_cmd, "2c generate truth")
|
||||
results["generate_truth"] = ok
|
||||
all_ok = all_ok and ok
|
||||
else:
|
||||
results["generate_truth"] = True
|
||||
results["generate_truth"] = SKIPPED
|
||||
|
||||
# Step 2d: Diff truth vs scraped
|
||||
if args.with_truth or args.with_export:
|
||||
diff_cmd = [sys.executable, "scripts/diff_truth.py", "--all"]
|
||||
if args.include_archived:
|
||||
diff_cmd.append("--include-archived")
|
||||
diff_cmd.extend(["--truth-dir", str(Path(args.output_dir) / "truth")])
|
||||
ok, _ = run(diff_cmd, "2d diff truth")
|
||||
results["diff_truth"] = ok
|
||||
all_ok = all_ok and ok
|
||||
else:
|
||||
results["diff_truth"] = True
|
||||
results["diff_truth"] = SKIPPED
|
||||
|
||||
# Step 2e: Export native formats
|
||||
if args.with_export:
|
||||
@@ -374,13 +384,16 @@ def main():
|
||||
"--truth-dir",
|
||||
str(Path(args.output_dir) / "truth"),
|
||||
]
|
||||
if args.include_archived:
|
||||
export_cmd.append("--include-archived")
|
||||
# Seven of the formats carry code, so the export patches the
|
||||
# platform's own file rather than regenerating it. Offline that
|
||||
# file has to be in the cache already.
|
||||
if not args.offline:
|
||||
export_cmd.append("--fetch")
|
||||
ok, _ = run(export_cmd, "2e export native")
|
||||
results["export_native"] = ok
|
||||
all_ok = all_ok and ok
|
||||
else:
|
||||
results["export_native"] = True
|
||||
results["export_native"] = SKIPPED
|
||||
|
||||
# Step 3: Verify
|
||||
verify_cmd = [sys.executable, "scripts/verify.py", "--all"]
|
||||
@@ -457,7 +470,7 @@ def main():
|
||||
all_ok = all_ok and ok
|
||||
else:
|
||||
print("\n--- 4/8 generate packs: SKIPPED (--skip-packs) ---")
|
||||
results["generate_packs"] = True
|
||||
results["generate_packs"] = SKIPPED
|
||||
|
||||
# Step 4b: Generate install manifests
|
||||
if not args.skip_packs:
|
||||
@@ -480,7 +493,7 @@ def main():
|
||||
all_ok = all_ok and ok
|
||||
else:
|
||||
print("\n--- 4b/8 generate install manifests: SKIPPED (--skip-packs) ---")
|
||||
results["generate_manifests"] = True
|
||||
results["generate_manifests"] = SKIPPED
|
||||
|
||||
# Step 4c: Generate target manifests
|
||||
if not args.skip_packs:
|
||||
@@ -496,7 +509,7 @@ def main():
|
||||
all_ok = all_ok and ok
|
||||
else:
|
||||
print("\n--- 4c/8 generate target manifests: SKIPPED (--skip-packs) ---")
|
||||
results["generate_target_manifests"] = True
|
||||
results["generate_target_manifests"] = SKIPPED
|
||||
|
||||
# Step 5: Consistency check
|
||||
if pack_output and verify_output:
|
||||
@@ -505,7 +518,7 @@ def main():
|
||||
all_ok = all_ok and ok
|
||||
else:
|
||||
print("\n--- 5/8 consistency check: SKIPPED ---")
|
||||
results["consistency"] = True
|
||||
results["consistency"] = SKIPPED
|
||||
|
||||
# Step 6: Pack integrity (extract + hash verification)
|
||||
if not args.skip_packs:
|
||||
@@ -524,7 +537,7 @@ def main():
|
||||
all_ok = all_ok and ok
|
||||
else:
|
||||
print("\n--- 6/8 pack integrity: SKIPPED (--skip-packs) ---")
|
||||
results["pack_integrity"] = True
|
||||
results["pack_integrity"] = SKIPPED
|
||||
|
||||
# Step 7: Generate README
|
||||
if not args.skip_docs:
|
||||
@@ -543,7 +556,7 @@ def main():
|
||||
all_ok = all_ok and ok
|
||||
else:
|
||||
print("\n--- 7/8 generate readme: SKIPPED (--skip-docs) ---")
|
||||
results["generate_readme"] = True
|
||||
results["generate_readme"] = SKIPPED
|
||||
|
||||
# Step 8: Generate site pages
|
||||
if not args.skip_docs:
|
||||
@@ -555,13 +568,17 @@ def main():
|
||||
all_ok = all_ok and ok
|
||||
else:
|
||||
print("\n--- 8/8 generate site: SKIPPED (--skip-docs) ---")
|
||||
results["generate_site"] = True
|
||||
results["generate_site"] = SKIPPED
|
||||
|
||||
# Summary
|
||||
total_elapsed = time.monotonic() - total_start
|
||||
print(f"\n{'=' * 60}")
|
||||
for step, ok in results.items():
|
||||
print(f" {step:.<40} {'OK' if ok else 'FAILED'}")
|
||||
for step, outcome in results.items():
|
||||
if outcome is SKIPPED:
|
||||
label = "SKIPPED"
|
||||
else:
|
||||
label = "OK" if outcome else "FAILED"
|
||||
print(f" {step:.<40} {label}")
|
||||
print(f" {'total':.<40} {total_elapsed:.1f}s")
|
||||
print(f"{'=' * 60}")
|
||||
print(f" Pipeline {'COMPLETE' if all_ok else 'FINISHED WITH ERRORS'}")
|
||||
|
||||
@@ -28,6 +28,15 @@ class BiosRequirement:
|
||||
required: bool = True
|
||||
zipped_file: str | None = None # If set, md5 is for this ROM inside the ZIP
|
||||
native_id: str | None = None # Original system name before normalization
|
||||
sha256: str | None = None
|
||||
alt_md5: str | None = None
|
||||
# How the platform itself writes the file reference. Its own spelling
|
||||
# is not always derivable from ours: libretro writes iplromco.dat bare
|
||||
# but ep128emu/roms/cpc464.rom with its directory.
|
||||
native_path: str | None = None
|
||||
# Fields the platform declares that have no equivalent in our model.
|
||||
# Kept verbatim so the native file can be written back unchanged.
|
||||
native: dict[str, object] = field(default_factory=dict)
|
||||
|
||||
|
||||
@dataclass
|
||||
@@ -58,6 +67,40 @@ class ChangeSet:
|
||||
MAX_RESPONSE_SIZE = 50 * 1024 * 1024 # 50 MB
|
||||
|
||||
|
||||
def requirement_entry(req: BiosRequirement) -> dict:
|
||||
"""Serialize a requirement to a platform YAML file entry.
|
||||
|
||||
Carries the platform's own system id and any field that has no place in
|
||||
our model. Without them the transcription is lossy in one direction:
|
||||
several native systems collapse onto one slug, and the exporter can no
|
||||
longer tell which of them a file came from.
|
||||
"""
|
||||
entry: dict = {
|
||||
"name": req.name,
|
||||
"destination": req.destination,
|
||||
"required": req.required,
|
||||
}
|
||||
for field_name in ("sha1", "md5", "sha256", "crc32"):
|
||||
value = getattr(req, field_name, None)
|
||||
if value:
|
||||
entry[field_name] = str(value).lower()
|
||||
if req.size:
|
||||
entry["size"] = req.size
|
||||
if req.zipped_file:
|
||||
entry["zipped_file"] = req.zipped_file
|
||||
if req.alt_md5:
|
||||
entry["alt_md5"] = str(req.alt_md5).lower()
|
||||
if req.native_id:
|
||||
entry["native_system"] = req.native_id
|
||||
if req.native_path and req.native_path != req.name:
|
||||
entry["native_path"] = req.native_path
|
||||
for key in sorted(req.native):
|
||||
value = req.native[key]
|
||||
if value not in (None, "", [], {}):
|
||||
entry[key] = value
|
||||
return entry
|
||||
|
||||
|
||||
def _read_limited(resp: object, max_bytes: int = MAX_RESPONSE_SIZE) -> bytes:
|
||||
"""Read an HTTP response with a size limit to prevent OOM."""
|
||||
chunks: list[bytes] = []
|
||||
|
||||
@@ -19,7 +19,7 @@ from pathlib import Path
|
||||
|
||||
from common import yaml_load
|
||||
|
||||
from .base_scraper import BaseScraper, BiosRequirement
|
||||
from .base_scraper import BaseScraper, BiosRequirement, requirement_entry
|
||||
|
||||
PLATFORM_NAME = "batocera"
|
||||
|
||||
@@ -309,7 +309,8 @@ class Scraper(BaseScraper):
|
||||
bios_files = sys_data.get("biosFiles", [])
|
||||
|
||||
for bios in bios_files:
|
||||
file_path = bios.get("file", "")
|
||||
declared_path = bios.get("file", "")
|
||||
file_path = declared_path
|
||||
md5 = _resolve_truncated_md5(bios.get("md5", ""), md5_index)
|
||||
zipped_file = bios.get("zippedFile", "")
|
||||
|
||||
@@ -324,9 +325,11 @@ class Scraper(BaseScraper):
|
||||
system=system_slug,
|
||||
md5=md5 or None,
|
||||
destination=file_path,
|
||||
native_path=declared_path,
|
||||
required=True,
|
||||
zipped_file=zipped_file or None,
|
||||
native_id=sys_key,
|
||||
native={"native_name": sys_data.get("name", "")},
|
||||
)
|
||||
)
|
||||
|
||||
@@ -365,17 +368,7 @@ class Scraper(BaseScraper):
|
||||
sys_entry["name"] = dname
|
||||
systems[req.system] = sys_entry
|
||||
|
||||
entry = {
|
||||
"name": req.name,
|
||||
"destination": req.destination,
|
||||
"required": req.required,
|
||||
}
|
||||
if req.md5:
|
||||
entry["md5"] = req.md5
|
||||
if req.zipped_file:
|
||||
entry["zipped_file"] = req.zipped_file
|
||||
|
||||
systems[req.system]["files"].append(entry)
|
||||
systems[req.system]["files"].append(requirement_entry(req))
|
||||
|
||||
batocera_version = ""
|
||||
if _STABLE_TAG != "master":
|
||||
|
||||
@@ -22,6 +22,7 @@ import re
|
||||
|
||||
try:
|
||||
from .base_scraper import (
|
||||
requirement_entry,
|
||||
BaseScraper,
|
||||
BiosRequirement,
|
||||
fetch_github_latest_version,
|
||||
@@ -29,6 +30,7 @@ try:
|
||||
)
|
||||
except ImportError:
|
||||
from base_scraper import (
|
||||
requirement_entry,
|
||||
BaseScraper,
|
||||
BiosRequirement,
|
||||
fetch_github_latest_version,
|
||||
@@ -352,6 +354,8 @@ class Scraper(BaseScraper):
|
||||
sha1=rec["sha1"],
|
||||
size=rec["size"] if rec["size"] else None,
|
||||
required=rec.get("status") != "Bad",
|
||||
destination=rec["name"],
|
||||
native_id=rec["system"],
|
||||
)
|
||||
requirements.append(req)
|
||||
|
||||
@@ -366,17 +370,7 @@ class Scraper(BaseScraper):
|
||||
if req.system not in systems:
|
||||
systems[req.system] = {"files": []}
|
||||
|
||||
entry: dict = {
|
||||
"name": req.name,
|
||||
"destination": req.name,
|
||||
"required": req.required,
|
||||
}
|
||||
if req.sha1:
|
||||
entry["sha1"] = req.sha1.lower()
|
||||
if req.size:
|
||||
entry["size"] = req.size
|
||||
|
||||
systems[req.system]["files"].append(entry)
|
||||
systems[req.system]["files"].append(requirement_entry(req))
|
||||
|
||||
version = _STABLE_TAG if _STABLE_TAG != "master" else ""
|
||||
|
||||
|
||||
@@ -18,9 +18,19 @@ import urllib.error
|
||||
import urllib.request
|
||||
|
||||
try:
|
||||
from .base_scraper import BaseScraper, BiosRequirement, fetch_github_latest_version
|
||||
from .base_scraper import (
|
||||
BaseScraper,
|
||||
BiosRequirement,
|
||||
fetch_github_latest_version,
|
||||
requirement_entry,
|
||||
)
|
||||
except ImportError:
|
||||
from base_scraper import BaseScraper, BiosRequirement, fetch_github_latest_version
|
||||
from base_scraper import (
|
||||
BaseScraper,
|
||||
BiosRequirement,
|
||||
fetch_github_latest_version,
|
||||
requirement_entry,
|
||||
)
|
||||
|
||||
PLATFORM_NAME = "emudeck"
|
||||
|
||||
@@ -54,9 +64,18 @@ HASH_ARRAY_MAP = {
|
||||
"SaturnBios": "sega-saturn",
|
||||
}
|
||||
|
||||
# Every check checkBIOS.sh defines, and the system it stands for. A function
|
||||
# missing here is a check whose hashes nothing would carry.
|
||||
FUNCTION_HASH_MAP = {
|
||||
"checkPS1BIOS": "sony-playstation",
|
||||
"checkPS2BIOS": "sony-playstation-2",
|
||||
"checkSegaCDBios": "sega-mega-cd",
|
||||
"checkSaturnBios": "sega-saturn",
|
||||
"checkDreamcastBios": "sega-dreamcast",
|
||||
"checkDSBios": "nintendo-ds",
|
||||
"checkCitronBios": "nintendo-switch",
|
||||
"checkRyujinxBios": "nintendo-switch",
|
||||
"checkYuzuBios": "nintendo-switch",
|
||||
}
|
||||
|
||||
SYSTEM_SLUG_MAP = {
|
||||
@@ -171,13 +190,17 @@ _RE_ARRAY = re.compile(
|
||||
re.MULTILINE,
|
||||
)
|
||||
|
||||
# checkBIOS.sh declares its checks as `checkPS1BIOS(){`, with no `function`
|
||||
# keyword and no settled casing for BIOS.
|
||||
_RE_FUNC = re.compile(
|
||||
r"function\s+(check\w+Bios)\s*\(\)",
|
||||
r"^[ \t]*(?:function\s+)?(check\w*(?:BIOS|Bios))\s*\(\)\s*\{",
|
||||
re.MULTILINE,
|
||||
)
|
||||
|
||||
# The hash list is named differently in each check (PSBios, hashes, ...), so
|
||||
# it is found by shape rather than by name.
|
||||
_RE_LOCAL_HASHES = re.compile(
|
||||
r"local\s+hashes=\(\s*((?:[0-9a-fA-F]+\s*)+)\)",
|
||||
r"(?:local\s+)?\w+=\(\s*((?:[0-9a-fA-F]{32}\s*)+)\)",
|
||||
re.MULTILINE,
|
||||
)
|
||||
|
||||
@@ -346,6 +369,7 @@ class Scraper(BaseScraper):
|
||||
system=system,
|
||||
destination=f.get("destination", f["name"]),
|
||||
required=True,
|
||||
native_id=system,
|
||||
)
|
||||
)
|
||||
|
||||
@@ -357,6 +381,7 @@ class Scraper(BaseScraper):
|
||||
md5=md5,
|
||||
destination="",
|
||||
required=True,
|
||||
native_id=system,
|
||||
)
|
||||
)
|
||||
|
||||
@@ -379,6 +404,7 @@ class Scraper(BaseScraper):
|
||||
system=system,
|
||||
destination=f.get("destination", f["name"]),
|
||||
required=True,
|
||||
native_id=system,
|
||||
)
|
||||
)
|
||||
|
||||
@@ -398,14 +424,7 @@ class Scraper(BaseScraper):
|
||||
if req.system not in systems:
|
||||
systems[req.system] = {"files": []}
|
||||
|
||||
entry: dict = {
|
||||
"name": req.name,
|
||||
"destination": req.destination,
|
||||
"required": req.required,
|
||||
}
|
||||
if req.md5:
|
||||
entry["md5"] = req.md5
|
||||
systems[req.system]["files"].append(entry)
|
||||
systems[req.system]["files"].append(requirement_entry(req))
|
||||
|
||||
version = ""
|
||||
try:
|
||||
|
||||
@@ -11,7 +11,12 @@ from __future__ import annotations
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
|
||||
from .base_scraper import BaseScraper, BiosRequirement, fetch_github_latest_version
|
||||
from .base_scraper import (
|
||||
BaseScraper,
|
||||
BiosRequirement,
|
||||
fetch_github_latest_version,
|
||||
requirement_entry,
|
||||
)
|
||||
from .dat_parser import parse_dat, parse_dat_metadata, validate_dat_format
|
||||
|
||||
PLATFORM_NAME = "libretro"
|
||||
@@ -132,6 +137,7 @@ class Scraper(BaseScraper):
|
||||
destination=destination,
|
||||
required=True,
|
||||
native_id=native_system,
|
||||
native_path=rom.name,
|
||||
)
|
||||
)
|
||||
|
||||
@@ -247,21 +253,7 @@ class Scraper(BaseScraper):
|
||||
system_entry["docs"] = cm["docs"]
|
||||
systems[req.system] = system_entry
|
||||
|
||||
entry = {
|
||||
"name": req.name,
|
||||
"destination": req.destination,
|
||||
"required": req.required,
|
||||
}
|
||||
if req.sha1:
|
||||
entry["sha1"] = req.sha1
|
||||
if req.md5:
|
||||
entry["md5"] = req.md5
|
||||
if req.crc32:
|
||||
entry["crc32"] = req.crc32
|
||||
if req.size:
|
||||
entry["size"] = req.size
|
||||
|
||||
systems[req.system]["files"].append(entry)
|
||||
systems[req.system]["files"].append(requirement_entry(req))
|
||||
|
||||
# Systems not in System.dat but needed for RetroArch -added via
|
||||
# shared groups in _shared.yml. The includes directive is resolved
|
||||
|
||||
@@ -29,9 +29,19 @@ import zipfile
|
||||
from datetime import datetime, timezone
|
||||
|
||||
try:
|
||||
from .base_scraper import BaseScraper, BiosRequirement, _read_limited
|
||||
from .base_scraper import (
|
||||
BaseScraper,
|
||||
BiosRequirement,
|
||||
_read_limited,
|
||||
requirement_entry,
|
||||
)
|
||||
except ImportError:
|
||||
from base_scraper import BaseScraper, BiosRequirement, _read_limited
|
||||
from base_scraper import (
|
||||
BaseScraper,
|
||||
BiosRequirement,
|
||||
_read_limited,
|
||||
requirement_entry,
|
||||
)
|
||||
|
||||
PLATFORM_NAME = "misterfpga"
|
||||
|
||||
@@ -193,15 +203,7 @@ class Scraper(BaseScraper):
|
||||
"docs": f"https://github.com/MiSTer-devel/{repo}" if repo else "",
|
||||
},
|
||||
)
|
||||
file_entry: dict = {
|
||||
"name": req.name,
|
||||
"destination": req.destination,
|
||||
"required": req.required,
|
||||
"md5": req.md5,
|
||||
}
|
||||
if req.size is not None:
|
||||
file_entry["size"] = req.size
|
||||
entry["files"].append(file_entry)
|
||||
entry["files"].append(requirement_entry(req))
|
||||
|
||||
for entry in systems.values():
|
||||
if not entry["docs"]:
|
||||
|
||||
@@ -15,11 +15,9 @@ Recalbox verification logic:
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
|
||||
from common import parse_untrusted_xml
|
||||
|
||||
from .base_scraper import BaseScraper, BiosRequirement
|
||||
from .base_scraper import BaseScraper, BiosRequirement, requirement_entry
|
||||
|
||||
PLATFORM_NAME = "recalbox"
|
||||
|
||||
@@ -136,6 +134,7 @@ class Scraper(BaseScraper):
|
||||
for system_elem in root.findall(".//system"):
|
||||
platform = system_elem.get("platform", "")
|
||||
system_slug = SYSTEM_SLUG_MAP.get(platform, platform)
|
||||
fullname = system_elem.get("fullname", "")
|
||||
|
||||
for bios_elem in system_elem.findall("bios"):
|
||||
paths_str = bios_elem.get("path", "")
|
||||
@@ -154,11 +153,30 @@ class Scraper(BaseScraper):
|
||||
md5_list = [m.strip() for m in md5_str.split(",") if m.strip()]
|
||||
all_md5 = ",".join(md5_list) if md5_list else None
|
||||
|
||||
dedup_key = primary_path
|
||||
dedup_key = (platform, primary_path)
|
||||
if dedup_key in seen:
|
||||
continue
|
||||
seen.add(dedup_key)
|
||||
|
||||
native: dict[str, object] = {}
|
||||
if fullname:
|
||||
native["native_name"] = fullname
|
||||
core = bios_elem.get("core", "").strip()
|
||||
if core:
|
||||
native["core"] = core
|
||||
note = bios_elem.get("note", "").strip()
|
||||
if note:
|
||||
native["note"] = note
|
||||
# Recalbox reads a missing attribute as true for both flags,
|
||||
# so only the explicit value carries information.
|
||||
hash_match = bios_elem.get("hashMatchMandatory")
|
||||
if hash_match is not None:
|
||||
native["hash_match_mandatory"] = hash_match != "false"
|
||||
if bios_elem.get("mandatory") is not None:
|
||||
native["mandatory_declared"] = mandatory
|
||||
if len(paths) > 1:
|
||||
native["alt_paths"] = paths[1:]
|
||||
|
||||
requirements.append(
|
||||
BiosRequirement(
|
||||
name=name,
|
||||
@@ -167,56 +185,12 @@ class Scraper(BaseScraper):
|
||||
destination=primary_path,
|
||||
required=mandatory,
|
||||
native_id=platform,
|
||||
native=native,
|
||||
)
|
||||
)
|
||||
|
||||
return requirements
|
||||
|
||||
def fetch_full_requirements(self) -> list[dict]:
|
||||
"""Parse es_bios.xml preserving all Recalbox-specific fields."""
|
||||
raw = self._fetch_raw()
|
||||
root = parse_untrusted_xml(raw, "es_bios.xml")
|
||||
requirements = []
|
||||
|
||||
for system_elem in root.findall(".//system"):
|
||||
platform = system_elem.get("platform", "")
|
||||
system_name = system_elem.get("name", platform)
|
||||
system_slug = SYSTEM_SLUG_MAP.get(platform, platform)
|
||||
|
||||
for bios_elem in system_elem.findall("bios"):
|
||||
paths_str = bios_elem.get("path", "")
|
||||
md5_str = bios_elem.get("md5", "")
|
||||
core = bios_elem.get("core", "")
|
||||
mandatory = bios_elem.get("mandatory", "true") != "false"
|
||||
hash_match_mandatory = (
|
||||
bios_elem.get("hashMatchMandatory", "true") != "false"
|
||||
)
|
||||
note = bios_elem.get("note", "")
|
||||
|
||||
paths = [p.strip() for p in paths_str.split("|") if p.strip()]
|
||||
md5_list = [m.strip() for m in md5_str.split(",") if m.strip()]
|
||||
|
||||
if not paths:
|
||||
continue
|
||||
|
||||
name = paths[0].split("/")[-1] if "/" in paths[0] else paths[0]
|
||||
|
||||
requirements.append(
|
||||
{
|
||||
"name": name,
|
||||
"system": system_slug,
|
||||
"system_name": system_name,
|
||||
"paths": paths,
|
||||
"md5_list": md5_list,
|
||||
"core": core,
|
||||
"mandatory": mandatory,
|
||||
"hash_match_mandatory": hash_match_mandatory,
|
||||
"note": note,
|
||||
}
|
||||
)
|
||||
|
||||
return requirements
|
||||
|
||||
def validate_format(self, raw_data: str) -> bool:
|
||||
"""Validate es_bios.xml format."""
|
||||
return "<biosList" in raw_data and "<system" in raw_data and "<bios" in raw_data
|
||||
@@ -225,7 +199,7 @@ class Scraper(BaseScraper):
|
||||
"""Generate a platform YAML config dict from scraped data."""
|
||||
requirements = self.fetch_requirements()
|
||||
|
||||
systems = {}
|
||||
systems: dict[str, dict] = {}
|
||||
for req in requirements:
|
||||
if req.system not in systems:
|
||||
sys_entry: dict = {"files": []}
|
||||
@@ -233,15 +207,7 @@ class Scraper(BaseScraper):
|
||||
sys_entry["native_id"] = req.native_id
|
||||
systems[req.system] = sys_entry
|
||||
|
||||
entry = {
|
||||
"name": req.name,
|
||||
"destination": req.destination,
|
||||
"required": req.required,
|
||||
}
|
||||
if req.md5:
|
||||
entry["md5"] = req.md5
|
||||
|
||||
systems[req.system]["files"].append(entry)
|
||||
systems[req.system]["files"].append(requirement_entry(req))
|
||||
|
||||
version = _STABLE_TAG if _STABLE_TAG != "master" else ""
|
||||
if not version:
|
||||
@@ -262,55 +228,9 @@ class Scraper(BaseScraper):
|
||||
|
||||
def main():
|
||||
"""CLI entry point."""
|
||||
import argparse
|
||||
import json
|
||||
from .base_scraper import scraper_cli
|
||||
|
||||
parser = argparse.ArgumentParser(description="Scrape Recalbox es_bios.xml")
|
||||
parser.add_argument("--dry-run", action="store_true")
|
||||
parser.add_argument("--json", action="store_true")
|
||||
parser.add_argument(
|
||||
"--full", action="store_true", help="Show full Recalbox-specific fields"
|
||||
)
|
||||
parser.add_argument("--output", "-o")
|
||||
args = parser.parse_args()
|
||||
|
||||
scraper = Scraper()
|
||||
|
||||
try:
|
||||
if args.full:
|
||||
reqs = scraper.fetch_full_requirements()
|
||||
print(json.dumps(reqs[:5], indent=2))
|
||||
print(f"\nTotal: {len(reqs)} BIOS entries")
|
||||
return
|
||||
reqs = scraper.fetch_requirements()
|
||||
except (ConnectionError, ValueError) as e:
|
||||
print(f"Error: {e}", file=sys.stderr)
|
||||
sys.exit(1)
|
||||
|
||||
if args.dry_run:
|
||||
from collections import defaultdict
|
||||
|
||||
by_system = defaultdict(list)
|
||||
for r in reqs:
|
||||
by_system[r.system].append(r)
|
||||
for sys_name, files in sorted(by_system.items()):
|
||||
print(f"\n{sys_name} ({len(files)} files):")
|
||||
for f in files[:5]:
|
||||
print(f" {f.name} (md5={f.md5[:12] if f.md5 else 'N/A'}...)")
|
||||
if len(files) > 5:
|
||||
print(f" ... +{len(files) - 5} more")
|
||||
print(f"\nTotal: {len(reqs)} BIOS files across {len(by_system)} systems")
|
||||
return
|
||||
|
||||
if args.json:
|
||||
config = scraper.generate_platform_yaml()
|
||||
print(json.dumps(config, indent=2))
|
||||
return
|
||||
|
||||
by_system = {}
|
||||
for r in reqs:
|
||||
by_system.setdefault(r.system, []).append(r)
|
||||
print(f"Scraped {len(reqs)} BIOS files across {len(by_system)} systems")
|
||||
scraper_cli(Scraper, "Scrape Recalbox es_bios.xml")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -11,9 +11,19 @@ from __future__ import annotations
|
||||
import json
|
||||
|
||||
try:
|
||||
from .base_scraper import BaseScraper, BiosRequirement, fetch_github_latest_version
|
||||
from .base_scraper import (
|
||||
BaseScraper,
|
||||
BiosRequirement,
|
||||
fetch_github_latest_version,
|
||||
requirement_entry,
|
||||
)
|
||||
except ImportError:
|
||||
from base_scraper import BaseScraper, BiosRequirement, fetch_github_latest_version
|
||||
from base_scraper import (
|
||||
BaseScraper,
|
||||
BiosRequirement,
|
||||
fetch_github_latest_version,
|
||||
requirement_entry,
|
||||
)
|
||||
|
||||
PLATFORM_NAME = "retrobat"
|
||||
|
||||
@@ -73,7 +83,8 @@ class Scraper(BaseScraper):
|
||||
if not isinstance(bios, dict):
|
||||
continue
|
||||
|
||||
file_path = bios.get("file", "")
|
||||
declared_path = bios.get("file", "")
|
||||
file_path = declared_path
|
||||
md5 = bios.get("md5", "")
|
||||
|
||||
if not file_path:
|
||||
@@ -85,6 +96,11 @@ class Scraper(BaseScraper):
|
||||
|
||||
name = file_path.split("/")[-1] if "/" in file_path else file_path
|
||||
|
||||
native: dict[str, object] = {}
|
||||
sys_name = sys_data.get("name", "") if isinstance(sys_data, dict) else ""
|
||||
if sys_name:
|
||||
native["native_name"] = sys_name
|
||||
|
||||
requirements.append(
|
||||
BiosRequirement(
|
||||
name=name,
|
||||
@@ -92,6 +108,9 @@ class Scraper(BaseScraper):
|
||||
md5=md5 or None,
|
||||
destination=file_path,
|
||||
required=True,
|
||||
native_id=sys_key,
|
||||
native_path=declared_path,
|
||||
native=native,
|
||||
)
|
||||
)
|
||||
|
||||
@@ -139,15 +158,7 @@ class Scraper(BaseScraper):
|
||||
sys_entry["name"] = dname
|
||||
systems[req.system] = sys_entry
|
||||
|
||||
entry = {
|
||||
"name": req.name,
|
||||
"destination": req.destination,
|
||||
"required": req.required,
|
||||
}
|
||||
if req.md5:
|
||||
entry["md5"] = req.md5
|
||||
|
||||
systems[req.system]["files"].append(entry)
|
||||
systems[req.system]["files"].append(requirement_entry(req))
|
||||
|
||||
version = ""
|
||||
tag = fetch_github_latest_version(GITHUB_REPO)
|
||||
|
||||
@@ -36,10 +36,10 @@ import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
try:
|
||||
from .base_scraper import BaseScraper, BiosRequirement
|
||||
from .base_scraper import BaseScraper, BiosRequirement, requirement_entry
|
||||
except ImportError:
|
||||
sys.path.insert(0, str(Path(__file__).parent.parent))
|
||||
from scraper.base_scraper import BaseScraper, BiosRequirement
|
||||
from scraper.base_scraper import BaseScraper, BiosRequirement, requirement_entry
|
||||
|
||||
PLATFORM_NAME = "retrodeck"
|
||||
COMPONENTS_REPO = "RetroDECK/components"
|
||||
@@ -385,13 +385,25 @@ class Scraper(BaseScraper):
|
||||
continue
|
||||
seen.add(key)
|
||||
|
||||
native: dict[str, object] = {"component": comp_key}
|
||||
description = str(entry.get("description", "")).strip()
|
||||
if description:
|
||||
native["description"] = description
|
||||
if required_raw not in (None, ""):
|
||||
native["required_label"] = str(required_raw)
|
||||
|
||||
sha256 = str(entry.get("sha256", "")).strip().lower()
|
||||
|
||||
requirements.append(
|
||||
BiosRequirement(
|
||||
name=filename,
|
||||
system=system,
|
||||
destination=destination,
|
||||
md5=md5,
|
||||
sha256=sha256 or None,
|
||||
required=required,
|
||||
native_id=str(raw_system),
|
||||
native=native,
|
||||
)
|
||||
)
|
||||
|
||||
@@ -413,14 +425,7 @@ class Scraper(BaseScraper):
|
||||
systems: dict[str, dict] = {}
|
||||
for req in reqs:
|
||||
sys_entry = systems.setdefault(req.system, {"files": []})
|
||||
file_entry: dict = {
|
||||
"name": req.name,
|
||||
"destination": req.destination,
|
||||
"required": req.required,
|
||||
}
|
||||
if req.md5:
|
||||
file_entry["md5"] = req.md5
|
||||
sys_entry["files"].append(file_entry)
|
||||
sys_entry["files"].append(requirement_entry(req))
|
||||
|
||||
try:
|
||||
from .base_scraper import fetch_github_latest_version
|
||||
|
||||
@@ -0,0 +1,138 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Scraper for RetroPie's package list.
|
||||
|
||||
Source: RetroPie/RetroPie-Setup -> scriptmodules/
|
||||
Format: one bash script per package, `rp_module_id` naming it
|
||||
|
||||
RetroPie publishes no BIOS list: platforms.cfg carries extensions and full
|
||||
names, and the files a package needs are named in prose inside its
|
||||
`rp_module_help`, without hashes. So there is nothing here to transcribe as
|
||||
requirements, and `fetch_requirements` returns none. What RetroPie does
|
||||
state precisely is which packages it ships, and that is what this reads.
|
||||
|
||||
It matters because the config said `cores: all_libretro`, inherited from
|
||||
RetroArch: RetroPie claimed every libretro core in existence while shipping
|
||||
96 of them, and claimed none of the standalone emulators it also packages
|
||||
(openmsx, pcsx2, xroar, sdltrs, amiberry and the rest). Both halves were
|
||||
wrong, in opposite directions.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import re
|
||||
import tarfile
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
|
||||
try:
|
||||
from .base_scraper import BaseScraper, BiosRequirement
|
||||
except ImportError:
|
||||
from base_scraper import BaseScraper, BiosRequirement
|
||||
|
||||
PLATFORM_NAME = "retropie"
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://codeload.github.com/RetroPie/RetroPie-Setup/tar.gz/refs/heads/master"
|
||||
)
|
||||
GITHUB_REPO = "RetroPie/RetroPie-Setup"
|
||||
MAX_ARCHIVE = 64 * 1024 * 1024
|
||||
|
||||
_MODULE_ID = re.compile(r'^rp_module_id="([^"]+)"', re.MULTILINE)
|
||||
# Sections RetroPie does not build as emulators: setup helpers, themes,
|
||||
# drivers and the like carry no core.
|
||||
_PACKAGE_DIRS = ("emulators", "libretrocores", "ports")
|
||||
|
||||
|
||||
class Scraper(BaseScraper):
|
||||
"""Scraper for the RetroPie package list."""
|
||||
|
||||
def __init__(self, url: str = SOURCE_URL):
|
||||
super().__init__(url=url)
|
||||
self._modules: list[str] | None = None
|
||||
|
||||
def _fetch_archive(self) -> bytes:
|
||||
request = urllib.request.Request(
|
||||
self.url, headers={"User-Agent": "retrobios-scraper/1.0"}
|
||||
)
|
||||
try:
|
||||
with urllib.request.urlopen(request, timeout=60) as response:
|
||||
payload = response.read(MAX_ARCHIVE + 1)
|
||||
except urllib.error.URLError as exc:
|
||||
raise ConnectionError(f"Failed to fetch {self.url}: {exc}") from exc
|
||||
if len(payload) > MAX_ARCHIVE:
|
||||
raise ValueError(f"{self.url}: response larger than {MAX_ARCHIVE} bytes")
|
||||
return payload
|
||||
|
||||
def module_ids(self, payload: bytes | None = None) -> list[str]:
|
||||
"""Every package RetroPie ships, by the id its script declares."""
|
||||
if self._modules is not None:
|
||||
return self._modules
|
||||
|
||||
raw = payload if payload is not None else self._fetch_archive()
|
||||
found: set[str] = set()
|
||||
with tarfile.open(fileobj=io.BytesIO(raw), mode="r:gz") as archive:
|
||||
for member in archive:
|
||||
if not member.isfile() or not member.name.endswith(".sh"):
|
||||
continue
|
||||
relative = member.name.split("/", 1)[-1]
|
||||
parts = relative.split("/")
|
||||
if len(parts) != 3 or parts[0] != "scriptmodules":
|
||||
continue
|
||||
if parts[1] not in _PACKAGE_DIRS:
|
||||
continue
|
||||
handle = archive.extractfile(member)
|
||||
if handle is None:
|
||||
continue
|
||||
text = handle.read().decode("utf-8", errors="replace")
|
||||
match = _MODULE_ID.search(text)
|
||||
if match:
|
||||
found.add(match.group(1))
|
||||
|
||||
self._modules = sorted(found)
|
||||
return self._modules
|
||||
|
||||
def fetch_requirements(self) -> list[BiosRequirement]:
|
||||
"""None: RetroPie names BIOS in prose, with no hash to transcribe.
|
||||
|
||||
The prose is read where it can be acted on, by the exporter that
|
||||
rewrites those sentences.
|
||||
"""
|
||||
self.module_ids()
|
||||
return []
|
||||
|
||||
def validate_format(self, raw_data: str) -> bool:
|
||||
return bool(self.module_ids())
|
||||
|
||||
def generate_platform_yaml(self) -> dict:
|
||||
"""Build the RetroPie platform configuration.
|
||||
|
||||
The systems stay inherited from RetroArch: RetroPie installs the
|
||||
libretro cores and reads the same files, at BIOS/ instead of
|
||||
system/. Only the core list is its own.
|
||||
"""
|
||||
# A libretro package is lr-<core>; a standalone package is named
|
||||
# after the emulator itself.
|
||||
cores = sorted({module.removeprefix("lr-") for module in self.module_ids()})
|
||||
|
||||
return {
|
||||
"inherits": "retroarch",
|
||||
"platform": "RetroPie",
|
||||
"homepage": "https://retropie.org.uk",
|
||||
"source": SOURCE_URL,
|
||||
"base_destination": "BIOS",
|
||||
"cores": cores,
|
||||
}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
try:
|
||||
from .base_scraper import scraper_cli
|
||||
except ImportError:
|
||||
from base_scraper import scraper_cli
|
||||
|
||||
scraper_cli(Scraper, "Scrape the RetroPie package list")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -18,9 +18,19 @@ from __future__ import annotations
|
||||
import re
|
||||
|
||||
try:
|
||||
from .base_scraper import BaseScraper, BiosRequirement, fetch_github_latest_version
|
||||
from .base_scraper import (
|
||||
BaseScraper,
|
||||
BiosRequirement,
|
||||
fetch_github_latest_version,
|
||||
requirement_entry,
|
||||
)
|
||||
except ImportError:
|
||||
from base_scraper import BaseScraper, BiosRequirement, fetch_github_latest_version
|
||||
from base_scraper import (
|
||||
BaseScraper,
|
||||
BiosRequirement,
|
||||
fetch_github_latest_version,
|
||||
requirement_entry,
|
||||
)
|
||||
|
||||
PLATFORM_NAME = "rocknix"
|
||||
|
||||
@@ -136,16 +146,15 @@ class Scraper(BaseScraper):
|
||||
# Declared paths are relative to the storage root: strip the
|
||||
# leading bios/ so destinations match base_destination
|
||||
destination = path[len("bios/"):] if path.startswith("bios/") else path
|
||||
md5_values = [bios.get("md5", ""), bios.get("altmd5", "")]
|
||||
md5 = ",".join(m for m in md5_values if m) or None
|
||||
|
||||
req = BiosRequirement(
|
||||
name=destination.rsplit("/", 1)[-1],
|
||||
system=system_slug,
|
||||
md5=md5,
|
||||
md5=bios.get("md5", "") or None,
|
||||
alt_md5=bios.get("altmd5", "") or None,
|
||||
destination=destination,
|
||||
required=True,
|
||||
native_id=key,
|
||||
native={"native_name": entry["name"]},
|
||||
)
|
||||
if bios.get("zippedFile"):
|
||||
req.zipped_file = bios["zippedFile"]
|
||||
@@ -168,16 +177,7 @@ class Scraper(BaseScraper):
|
||||
"name": names.get(req.native_id, req.system),
|
||||
},
|
||||
)
|
||||
file_entry: dict = {
|
||||
"name": req.name,
|
||||
"destination": req.destination,
|
||||
"required": req.required,
|
||||
}
|
||||
if req.md5:
|
||||
file_entry["md5"] = req.md5
|
||||
if getattr(req, "zipped_file", None):
|
||||
file_entry["zipped_file"] = req.zipped_file
|
||||
entry["files"].append(file_entry)
|
||||
entry["files"].append(requirement_entry(req))
|
||||
|
||||
return {
|
||||
"platform": "ROCKNIX",
|
||||
|
||||
@@ -26,9 +26,19 @@ import json
|
||||
import sys
|
||||
|
||||
try:
|
||||
from .base_scraper import BaseScraper, BiosRequirement, fetch_github_latest_version
|
||||
from .base_scraper import (
|
||||
BaseScraper,
|
||||
BiosRequirement,
|
||||
fetch_github_latest_version,
|
||||
requirement_entry,
|
||||
)
|
||||
except ImportError:
|
||||
from base_scraper import BaseScraper, BiosRequirement, fetch_github_latest_version
|
||||
from base_scraper import (
|
||||
BaseScraper,
|
||||
BiosRequirement,
|
||||
fetch_github_latest_version,
|
||||
requirement_entry,
|
||||
)
|
||||
|
||||
PLATFORM_NAME = "romm"
|
||||
|
||||
@@ -165,6 +175,7 @@ class Scraper(BaseScraper):
|
||||
size=size,
|
||||
destination=f"{slug}/{filename}",
|
||||
required=True,
|
||||
native_id=slug,
|
||||
)
|
||||
)
|
||||
|
||||
@@ -183,7 +194,6 @@ class Scraper(BaseScraper):
|
||||
for key in list(data.keys())[:5]:
|
||||
if ":" not in key:
|
||||
return False
|
||||
_, _entry = key.split(":", 1), data[key]
|
||||
if not isinstance(data[key], dict):
|
||||
return False
|
||||
if "md5" not in data[key] and "sha1" not in data[key]:
|
||||
@@ -200,21 +210,7 @@ class Scraper(BaseScraper):
|
||||
if req.system not in systems:
|
||||
systems[req.system] = {"files": []}
|
||||
|
||||
entry: dict = {
|
||||
"name": req.name,
|
||||
"destination": req.destination,
|
||||
"required": req.required,
|
||||
}
|
||||
if req.sha1:
|
||||
entry["sha1"] = req.sha1
|
||||
if req.md5:
|
||||
entry["md5"] = req.md5
|
||||
if req.crc32:
|
||||
entry["crc32"] = req.crc32
|
||||
if req.size:
|
||||
entry["size"] = req.size
|
||||
|
||||
systems[req.system]["files"].append(entry)
|
||||
systems[req.system]["files"].append(requirement_entry(req))
|
||||
|
||||
version = _STABLE_TAG if _STABLE_TAG != "master" else ""
|
||||
|
||||
|
||||
@@ -180,9 +180,24 @@ def _merge_file_into_system(
|
||||
"note",
|
||||
"aliases",
|
||||
"contents",
|
||||
"region",
|
||||
):
|
||||
if file_entry.get(field) is not None and existing.get(field) is None:
|
||||
existing[field] = file_entry[field]
|
||||
# Search order is a fact of the code, so it travels with the entry.
|
||||
# The best rank any core gives it is kept and the disagreement is
|
||||
# recorded beside it, the way slot.py already does: the rank says
|
||||
# how early to read the file, the conflict says the order cannot
|
||||
# decide which single file to keep.
|
||||
theirs = file_entry.get("priority")
|
||||
if theirs is not None:
|
||||
ours = existing.get("priority")
|
||||
if ours is None:
|
||||
existing["priority"] = theirs
|
||||
else:
|
||||
if ours != theirs:
|
||||
existing["priority_conflict"] = True
|
||||
existing["priority"] = min(ours, theirs)
|
||||
return
|
||||
|
||||
entry: dict = {"name": file_entry["name"]}
|
||||
@@ -204,6 +219,8 @@ def _merge_file_into_system(
|
||||
"max_size",
|
||||
"aliases",
|
||||
"contents",
|
||||
"priority",
|
||||
"region",
|
||||
):
|
||||
val = file_entry.get(field)
|
||||
if val is not None:
|
||||
|
||||
Reference in new issue
Block a user