feat: make the native export round-trip a platform file

This commit is contained in:
Abdessamad Derraz committed 2026-09-05 17:32:48 +02:00
1 parent 7b93285e1d
commit 691ccbfca7
65 files changed
+18417 -2115

No files matched your search

+141 -56
View File
@@ -1,81 +1,166 @@
"""Abstract base class for platform exporters."""
"""Contract shared by the platform exporters.
An exporter answers one question: what would this platform's own file look
like if it were corrected. It is handed the platform's file when we have it,
because several of these formats carry code, and a generator that emits only
the data block hands the maintainer something that no longer runs.
"""
from __future__ import annotations
from abc import ABC, abstractmethod
from .baseline import NativeFile, NativeSystem, Report
class BaseExporter(ABC):
"""Base class for exporting truth data to native platform formats."""
"""Base class for writing a platform's own BIOS file, corrected."""
@staticmethod
@abstractmethod
def platform_name() -> str:
"""Return the platform identifier this exporter targets."""
@staticmethod
@abstractmethod
def export(
def native_filename() -> str:
"""Return the name the platform gives this file."""
@staticmethod
def carries() -> frozenset[str]:
"""Fields this format has somewhere to put.
A correction the file cannot state is not a correction it delivers,
and counting it would announce a change the maintainer will not find
in the diff.
"""
return frozenset({"md5"})
@staticmethod
def native_sources() -> dict[str, str]:
"""Return {relative path: URL} of the originals this exporter patches.
An exporter that can build its file from nothing returns an empty
mapping. One whose format carries code cannot, and names the file it
needs so the caller can fetch it.
"""
return {}
@staticmethod
def needs_original() -> bool:
"""Whether a faithful export requires the platform's own file."""
return False
@staticmethod
def may_write_nothing() -> bool:
"""Whether producing no file is an outcome rather than a failure.
True only where the unit is a correction rather than a document:
RetroPie has no BIOS list to rewrite, so a run with nothing to
correct writes nothing and is right to.
"""
return False
@abstractmethod
def render(
self,
truth_data: dict,
output_path: str,
scraped_data: dict | None = None,
) -> None:
"""Export truth data to the native platform format."""
systems: dict[str, NativeSystem],
report: Report,
originals: dict[str, str],
scraped: dict | None = None,
) -> dict[str, str]:
"""Return {relative output path: file content}."""
@abstractmethod
def validate(self, truth_data: dict, output_path: str) -> list[str]:
"""Validate exported file against truth data, return list of issues."""
def validate(
self,
systems: dict[str, NativeSystem],
produced: dict[str, str],
) -> list[str]:
"""Check the produced files against what was asked of them."""
# Shared helpers
@classmethod
def exportable(
cls,
systems: dict[str, NativeSystem],
require: str = "",
) -> list[tuple[NativeSystem, list[NativeFile]]]:
"""Systems paired with the files this format can actually carry.
`require` names a hash the format cannot omit. An entry the format
has no way to express is dropped here and counted by the caller,
never written as a half entry the platform would read as a file that
can never verify.
"""
result: list[tuple[NativeSystem, list[NativeFile]]] = []
for system in systems.values():
files = [fe for fe in system.files if fe.name and cls.writable(fe, require)]
if files:
result.append((system, files))
return result
@staticmethod
def _is_pattern(name: str) -> bool:
"""Check if a filename is a placeholder pattern (not a real file)."""
return "<" in name or ">" in name or "*" in name
def requires() -> str:
"""The hash this format cannot write a new entry without."""
return ""
@staticmethod
def _dest(fe: dict) -> str:
"""Get destination path for a file entry, falling back to name."""
return fe.get("path") or fe.get("destination") or fe.get("name", "")
def can_add() -> bool:
"""Whether a file the platform does not declare can be stated at all.
False where an entry is a declaration in code rather than a row:
BizHawk wires each firmware into an option list and a status that
only its source expresses, and writing C# for one is not something
an exporter can do safely.
"""
return True
@classmethod
def writable(cls, fe: NativeFile, require: str = "") -> bool:
"""Whether this format can carry the entry.
An entry the platform already declares is always carried, hash or
no hash: it is in their file today, and an export that drops it
hands back a file poorer than the one it corrects. The requirement
only gates what we would be adding.
"""
if fe.platform is not None:
return True
if not cls.can_add():
return False
require = require or cls.requires()
return not require or bool(fe.hash(require))
def outcome(
self,
systems: dict[str, NativeSystem],
produced: dict[str, str],
) -> str | None:
"""What this export did, when counting file entries would not say it.
Most formats state one entry per file, so the caller's own count is
the answer. RetroPie states a sentence per package, and a count of
file entries would describe work it did not do.
"""
return None
@staticmethod
def _display_name(
sys_id: str,
scraped_sys: dict | None = None,
) -> str:
"""Get display name for a system from scraped data or slug."""
if scraped_sys:
name = scraped_sys.get("name")
if name:
return name
# Fallback: convert slug to display name with acronym handling
def display_name(system: NativeSystem) -> str:
"""The name the platform shows for a system."""
if system.name:
return system.name
for fe in system.files:
native = fe.native("native_name", "")
if native:
return str(native)
_UPPER = {
"3do",
"cdi",
"cpc",
"cps1",
"cps2",
"cps3",
"dos",
"gba",
"gbc",
"hle",
"msx",
"nes",
"nds",
"ngp",
"psp",
"psx",
"sms",
"snes",
"stv",
"tvc",
"vb",
"zx",
"3do", "cdi", "cpc", "cps1", "cps2", "cps3", "dos", "gba", "gbc",
"hle", "msx", "nes", "nds", "ngp", "psp", "psx", "sms", "snes",
"stv", "tvc", "vb", "zx",
}
parts = sys_id.replace("-", " ").split()
result = []
for p in parts:
if p.lower() in _UPPER:
result.append(p.upper())
else:
result.append(p.capitalize())
return " ".join(result)
parts = system.native_id.replace("-", " ").replace("_", " ").split()
return " ".join(
p.upper() if p.lower() in _UPPER else p.capitalize() for p in parts
)
+373
View File
@@ -0,0 +1,373 @@
"""Reconciliation of a platform's own declarations with the ground truth.
An export is not the truth rendered in a native syntax. It is the platform's
file, corrected: what the truth can prove is applied, what it says nothing
about is left alone, and what it knows and the platform lacks is added. A
platform that loses two thirds of its systems to an export cannot use it.
Systems are keyed by the identifier the platform itself uses. Several of our
slugs collapse onto one native id (Recalbox files pcengine, pcenginecd and
supergrafx under one slug) and one slug can carry several native ids, so the
grouping is rebuilt from the per-file native_system the scrapers record.
"""
from __future__ import annotations
import sys
from dataclasses import dataclass, field
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from common import _norm_system_id
HASH_FIELDS = ("sha1", "md5", "sha256", "crc32")
def _hash_values(entry: dict, field_name: str) -> list[str]:
"""Return a hash field as a list, whatever shape it was written in."""
raw = entry.get(field_name)
if not raw:
return []
if isinstance(raw, list):
return [str(v).strip().lower() for v in raw if str(v).strip()]
return [v.strip().lower() for v in str(raw).split(",") if v.strip()]
def _is_placeholder(name: str) -> bool:
return "<" in name or ">" in name or "*" in name
@dataclass
class NativeFile:
"""One file as the platform will read it, after correction."""
name: str
destination: str
native_system: str
platform: dict | None = None
truth: dict | None = None
corrections: list[str] = field(default_factory=list)
@property
def origin(self) -> str:
if self.platform is not None and self.truth is not None:
return "both"
return "platform" if self.platform is not None else "truth"
def hashes(self, field_name: str) -> list[str]:
"""Accepted values for a hash, truth first when it has an opinion.
The truth is read from the emulator's source; the platform list is a
secondary source. When both speak and disagree, the truth decides and
the divergence is recorded, never silently merged: an emulator that
rejects a file will reject it whatever the platform declares.
"""
truth_values = _hash_values(self.truth or {}, field_name)
platform_values = _hash_values(self.platform or {}, field_name)
if truth_values and platform_values and not set(truth_values) & set(
platform_values
):
return truth_values
if truth_values:
# Keep the platform's extra accepted revisions alongside ours.
merged = list(truth_values)
merged.extend(v for v in platform_values if v not in merged)
return merged
return platform_values
def hash(self, field_name: str) -> str:
values = self.hashes(field_name)
return values[0] if values else ""
@property
def required(self) -> bool:
if self.truth is not None and self.truth.get("required") is not None:
return bool(self.truth["required"])
if self.platform is not None and self.platform.get("required") is not None:
return bool(self.platform["required"])
return True
@property
def priority(self) -> int | None:
"""Where the code looks for this file, lowest first.
DuckStation's FindBIOSImageInDirectory keeps the image whose
priority is lower, and a search list a core walks in order maps
onto it as 1, 2, 3. Where cores disagree this is the best rank any
of them gives: nothing is dropped on it, so a file that is some
core's first choice is read early. None when nothing states one.
"""
for entry in (self.truth, self.platform):
if entry and entry.get("priority") is not None:
return int(entry["priority"])
return None
def size(self) -> int | None:
for entry in (self.truth, self.platform):
if entry and entry.get("size"):
return int(entry["size"])
return None
def native(self, key: str, default: object = "") -> object:
"""Read a field the platform declares and we only carry through."""
for entry in (self.platform, self.truth):
if entry and entry.get(key) not in (None, ""):
return entry[key]
return default
def cores(self) -> list[str]:
"""Cores that want this file, the platform's naming preferred."""
declared = self.native("core", "")
if declared:
return [c.strip() for c in str(declared).split(",") if c.strip()]
if self.truth:
return [f"libretro/{c}" for c in self.truth.get("_cores", [])]
return []
@dataclass
class NativeSystem:
"""One system as the platform names it."""
native_id: str
name: str = ""
files: list[NativeFile] = field(default_factory=list)
from_platform: bool = False
@property
def origin(self) -> str:
return "platform" if self.from_platform else "truth"
@dataclass
class Report:
"""What the reconciliation changed, so the caller can say it out loud."""
systems_kept: int = 0
systems_added: int = 0
files_kept: int = 0
files_added: int = 0
hashes_corrected: list[str] = field(default_factory=list)
required_corrected: list[str] = field(default_factory=list)
@property
def corrections(self) -> int:
return len(self.hashes_corrected) + len(self.required_corrected)
def _native_id_of(sys_key: str, sys_data: dict, file_entry: dict) -> str:
"""The platform's own id for the system a file belongs to."""
return (
file_entry.get("native_system")
or sys_data.get("native_id")
or sys_key
)
def _match_key(entry: dict) -> tuple[str, str]:
dest = str(entry.get("destination") or entry.get("path") or entry.get("name", ""))
return dest.casefold(), str(entry.get("name", "")).casefold()
def build_native_model(
truth: dict,
scraped: dict | None,
) -> tuple[dict[str, NativeSystem], Report]:
"""Rebuild the platform's systems, corrected by the truth.
Returns the systems keyed by native id, in the platform's own order
first and truth-only additions after, plus what changed.
"""
report = Report()
systems: dict[str, NativeSystem] = {}
scraped_systems = (scraped or {}).get("systems", {})
# Pass 1: the platform's own file, grouped as the platform groups it.
for sys_key, sys_data in scraped_systems.items():
for file_entry in sys_data.get("files", []):
native_id = _native_id_of(sys_key, sys_data, file_entry)
system = systems.get(native_id)
if system is None:
system = NativeSystem(
native_id=native_id,
name=str(file_entry.get("native_name") or sys_data.get("name", "")),
from_platform=True,
)
systems[native_id] = system
report.systems_kept += 1
name = str(file_entry.get("name", ""))
destination = str(file_entry.get("destination") or name)
system.files.append(
NativeFile(
name=name,
destination=destination,
native_system=native_id,
platform=file_entry,
)
)
report.files_kept += 1
# Which native ids a truth system may contribute to.
norm_to_scraped: dict[str, str] = {
_norm_system_id(key): key for key in scraped_systems
}
native_by_norm: dict[str, str] = {}
for system in systems.values():
native_by_norm.setdefault(_norm_system_id(system.native_id), system.native_id)
def _target_native_ids(truth_sid: str) -> list[str]:
scraped_key = truth_sid if truth_sid in scraped_systems else None
if scraped_key is None:
scraped_key = norm_to_scraped.get(_norm_system_id(truth_sid))
if scraped_key is not None:
sys_data = scraped_systems[scraped_key]
ids = {
_native_id_of(scraped_key, sys_data, fe)
for fe in sys_data.get("files", [])
}
if not ids:
return [sys_data.get("native_id") or scraped_key]
# A file the platform does not declare joins the system's primary
# id, not whichever of its native ids sorts first: Recalbox files
# pcengine, pcenginecd and supergrafx under one slug, and an
# addition belongs to the one the system is named for.
primary = sys_data.get("native_id")
ordered = sorted(ids)
if primary in ids:
ordered.remove(primary)
ordered.insert(0, primary)
return ordered
direct = native_by_norm.get(_norm_system_id(truth_sid))
return [direct] if direct else [truth_sid]
# Pass 2: apply the truth onto that grouping.
for truth_sid in sorted(truth.get("systems", {})):
truth_sys = truth["systems"][truth_sid]
truth_files = truth_sys.get("files", [])
if not truth_files:
continue
target_ids = _target_native_ids(truth_sid)
for truth_entry in truth_files:
name = str(truth_entry.get("name", ""))
if not name or name.startswith("_") or _is_placeholder(name):
continue
t_dest, t_name = _match_key(truth_entry)
t_hashes = {
value
for field_name in HASH_FIELDS
for value in _hash_values(truth_entry, field_name)
}
candidates = [
candidate
for native_id in target_ids
for candidate in systems.get(
native_id, NativeSystem(native_id)
).files
if candidate.truth is None
]
def by_destination(candidate: NativeFile) -> bool:
theirs = _match_key(candidate.platform or {})[0]
return bool(t_dest) and theirs == t_dest
def by_name(candidate: NativeFile) -> bool:
theirs = _match_key(candidate.platform or {})[1]
return bool(t_name) and theirs == t_name
def by_hash(candidate: NativeFile) -> bool:
if not t_hashes:
return False
theirs = {
value
for field_name in HASH_FIELDS
for value in _hash_values(candidate.platform or {}, field_name)
}
return bool(t_hashes & theirs)
# Tried in order across every candidate, not per candidate: with
# three IPL.bin under one system, separated only by their path, a
# first-match-wins scan would attach the truth to whichever came
# first and correct the wrong region's file.
matched: NativeFile | None = None
for test in (by_destination, by_name, by_hash):
matched = next((c for c in candidates if test(c)), None)
if matched is not None:
break
if matched is not None:
matched.truth = truth_entry
for field_name in HASH_FIELDS:
ours = set(_hash_values(truth_entry, field_name))
theirs = set(_hash_values(matched.platform or {}, field_name))
if ours and theirs and not ours & theirs:
matched.corrections.append(field_name)
report.hashes_corrected.append(
f"{matched.native_system}/{matched.name} {field_name}"
)
t_req = truth_entry.get("required")
p_req = (matched.platform or {}).get("required")
if (
t_req is not None
and p_req is not None
and bool(t_req) != bool(p_req)
):
matched.corrections.append("required")
report.required_corrected.append(
f"{matched.native_system}/{matched.name}"
)
continue
# The truth knows a file the platform does not declare.
native_id = target_ids[0]
system = systems.get(native_id)
if system is None:
system = NativeSystem(native_id=native_id, from_platform=False)
systems[native_id] = system
report.systems_added += 1
destination = str(
truth_entry.get("path") or truth_entry.get("destination") or name
)
system.files.append(
NativeFile(
name=name,
destination=destination,
native_system=native_id,
truth=truth_entry,
)
)
report.files_added += 1
# A reader takes the files in the order the list gives, so the one the
# code looks for first is named first. Only what we add is ordered: what
# the platform already wrote keeps the place the platform gave it.
for system in systems.values():
head = [fe for fe in system.files if fe.platform is not None]
tail = [fe for fe in system.files if fe.platform is None]
system.files = head + search_order(tail)
return systems, report
def search_order(files: list[NativeFile]) -> list[NativeFile]:
"""Order files the way the code looks for them, best first.
`priority:` is that order where the source states it, lowest first.
Where it does not, the order the entries were declared in is the order
the code walks, so it is left alone.
"""
ranked = [(fe.priority, position, fe) for position, fe in enumerate(files)]
return [
fe
for _, _, fe in sorted(
ranked,
key=lambda item: (
(0, item[0]) if item[0] is not None else (1, item[1])
),
)
]
+274 -82
View File
@@ -1,109 +1,301 @@
"""Exporter for Batocera batocera-systems format.
"""Exporter for Batocera's batocera-systems.
Produces a Python dict matching the exact format of
batocera-linux/batocera-scripts/scripts/batocera-systems.
batocera-systems is an executable script: the systems dict is one block
inside it, the rest is the checker Batocera runs. Only the block is
rewritten, entry by entry, so the comments the maintainers wrote between
entries and the code around them survive untouched.
"""
from __future__ import annotations
from pathlib import Path
import re
from .base_exporter import BaseExporter
from .baseline import NativeFile, NativeSystem, Report
SOURCE_URL = (
"https://raw.githubusercontent.com/batocera-linux/batocera.linux/master"
"/package/batocera/core/batocera-scripts/scripts/batocera-systems"
)
_ENTRY_START = re.compile(r'^(\s{4})"([^"]+)":\s*\{')
_INDENT = " " * 4
class Exporter(BaseExporter):
"""Export truth data to Batocera batocera-systems format."""
"""Write Batocera's batocera-systems, corrected."""
@staticmethod
def platform_name() -> str:
return "batocera"
def export(
self,
truth_data: dict,
output_path: str,
scraped_data: dict | None = None,
) -> None:
# Build native_id and display name maps from scraped data
native_map: dict[str, str] = {}
if scraped_data:
for sys_id, sys_data in scraped_data.get("systems", {}).items():
nid = sys_data.get("native_id")
if nid:
native_map[sys_id] = nid
@staticmethod
def native_filename() -> str:
return "batocera-systems"
lines: list[str] = ["systems = {", ""]
@staticmethod
def requires() -> str:
return "md5"
systems = truth_data.get("systems", {})
for sys_id in sorted(systems):
sys_data = systems[sys_id]
files = sys_data.get("files", [])
if not files:
@staticmethod
def carries() -> frozenset[str]:
return frozenset({"md5"})
@staticmethod
def native_sources() -> dict[str, str]:
return {"batocera-systems": SOURCE_URL}
@staticmethod
def needs_original() -> bool:
# The data block is a fraction of the file; the rest is the checker.
return True
def _bios_files(self, files: list[NativeFile]) -> str:
return ", ".join(self._bios_items(files))
def _bios_items(self, files: list[NativeFile]) -> list[str]:
parts: list[str] = []
for fe in files:
# The platform states an unhashed file as an empty md5 rather
# than leaving it out, and so do we.
item = [f'"md5": "{fe.hash("md5")}"']
alt = fe.native("alt_md5", "")
if alt:
item.append(f'"altmd5": "{alt}"')
declared = fe.native("native_path", "")
path = str(declared) if declared else f"bios/{fe.destination}"
item.append(f'"file": "{path}"')
zipped = fe.native("zipped_file", "")
if zipped:
item.append(f'"zippedFile": "{zipped}"')
parts.append("{ " + ", ".join(item) + " }")
return parts
def _entry_line(self, system: NativeSystem, files: list[NativeFile]) -> str:
return (
f'{_INDENT}"{system.native_id}": '
f'{{ "name": "{self.display_name(system)}", '
f'"biosFiles": [ {self._bios_files(files)} ] }},'
)
@staticmethod
def _item_notes(lines: list[str]) -> tuple[dict[str, str], dict[str, list[str]]]:
"""The notes a maintainer wrote about each file, keyed by its path.
Rewriting an entry as one line would drop them, and a note like
"ideally - 94bc50..." is the only record of why a hash is blank.
A comment on its own line belongs to the file below it; a comment at
the end of a line belongs to the file on it.
"""
trailing: dict[str, str] = {}
leading: dict[str, list[str]] = {}
pending: list[str] = []
for line in lines:
stripped = line.strip()
hash_at = line.find("#")
if stripped.startswith("#"):
pending.append(stripped)
continue
path = re.search(r'"file"\s*:\s*"([^"]+)"', line)
if not path:
continue
key = path.group(1)
if pending:
leading[key] = pending
pending = []
if 0 <= hash_at and hash_at > line.index(key):
trailing[key] = line[hash_at:].rstrip()
return trailing, leading
native_id = native_map.get(sys_id, sys_id)
scraped_sys = (
scraped_data.get("systems", {}).get(sys_id) if scraped_data else None
def _entry_block(
self,
system: NativeSystem,
files: list[NativeFile],
original: list[str],
) -> list[str]:
"""One entry, keeping the layout and the notes it was written with."""
trailing, leading = self._item_notes(original)
items = self._bios_items(files)
one_line = len(original) == 1 and not trailing and not leading
if one_line:
return [self._entry_line(system, files)]
head = (
f'{_INDENT}"{system.native_id}": '
f'{{ "name": "{self.display_name(system)}", "biosFiles": ['
)
pad = " " * (len(_INDENT) + 4)
lines = [head]
for index, item in enumerate(items):
key = self._item_path(item)
lines.extend(f"{pad}{note}" for note in leading.get(key, []))
separator = "," if index < len(items) - 1 else ""
note = trailing.get(key, "")
suffix = f" {note}" if note else ""
lines.append(f"{pad}{item}{separator}{suffix}")
lines.append(f"{_INDENT}] }},")
return lines
@staticmethod
def _item_path(item: str) -> str:
match = re.search(r'"file": "([^"]+)"', item)
return match.group(1) if match else ""
@staticmethod
def _split_original(original: str) -> tuple[list[str], list[str], list[str]]:
"""Cut the script into what precedes the dict, the dict, what follows."""
lines = original.split("\n")
start = next(
(i for i, line in enumerate(lines) if line.startswith("systems = {")),
None,
)
if start is None:
raise ValueError("batocera-systems: no systems dict found")
end = next(
(i for i in range(start + 1, len(lines)) if lines[i].startswith("}")),
None,
)
if end is None:
raise ValueError("batocera-systems: the systems dict is not closed")
return lines[: start + 1], lines[start + 1 : end], lines[end:]
@staticmethod
def _entry_spans(body: list[str]) -> dict[str, tuple[int, int]]:
"""Locate each top-level entry, which may wrap over several lines."""
spans: dict[str, tuple[int, int]] = {}
index = 0
while index < len(body):
match = _ENTRY_START.match(body[index])
if not match:
index += 1
continue
key = match.group(2)
depth = 0
end = index
for cursor in range(index, len(body)):
depth += body[cursor].count("{") + body[cursor].count("[")
depth -= body[cursor].count("}") + body[cursor].count("]")
if depth <= 0:
end = cursor
break
else:
end = len(body) - 1
spans[key] = (index, end)
index = end + 1
return spans
@staticmethod
def _parse_entry(lines: list[str]) -> dict | None:
"""Evaluate one entry so two spellings of the same data compare equal."""
text = "\n".join(lines).strip().rstrip(",")
namespace: dict[str, object] = {}
try:
exec(f"entry = {{{text}}}", {}, namespace) # noqa: S102
except (SyntaxError, ValueError, TypeError):
return None
entry = namespace.get("entry")
return entry if isinstance(entry, dict) else None
def render(
self,
systems: dict[str, NativeSystem],
report: Report,
originals: dict[str, str],
scraped: dict | None = None,
) -> dict[str, str]:
exportable = dict(
(system.native_id, (system, files))
for system, files in self.exportable(systems, require="md5")
)
original = originals.get(self.native_filename(), "")
if not original:
raise ValueError(
f"{self.native_filename()} cannot be written without the "
"platform's own file: the systems dict is a fraction of a "
"script, and the rest of it is the checker"
)
display_name = self._display_name(sys_id, scraped_sys)
# Build md5 lookup from scraped data for this system
scraped_md5: dict[str, str] = {}
if scraped_data:
s_sys = scraped_data.get("systems", {}).get(sys_id, {})
for sf in s_sys.get("files", []):
sname = sf.get("name", "").lower()
smd5 = sf.get("md5", "")
if sname and smd5:
scraped_md5[sname] = smd5
head, body, tail = self._split_original(original)
spans = self._entry_spans(body)
# Build biosFiles entries as compact single-line dicts
# Original format ALWAYS has md5 — use scraped md5 as fallback
bios_parts: list[str] = []
for fe in files:
name = fe.get("name", "")
if name.startswith("_") or self._is_pattern(name):
continue
dest = self._dest(fe)
md5 = fe.get("md5", "")
if isinstance(md5, list):
md5 = md5[0] if md5 else ""
if not md5:
md5 = scraped_md5.get(name.lower(), "")
rebuilt: list[str] = []
written: set[str] = set()
index = 0
for key, (start, end) in sorted(spans.items(), key=lambda kv: kv[1][0]):
rebuilt.extend(body[index:start])
pair = exportable.get(key)
index = end + 1
if pair is None:
# The truth has nothing to say and no file left to declare.
rebuilt.extend(body[start:end + 1])
continue
written.add(key)
original_lines = body[start:end + 1]
replacement = self._entry_line(*pair)
before = self._parse_entry(original_lines)
after = self._parse_entry([replacement])
if before is not None and before == after:
# Nothing changed: keep the maintainer's own lines, comments
# and alignment included, so the diff shows only corrections.
rebuilt.extend(original_lines)
else:
rebuilt.extend(self._entry_block(*pair, original_lines))
rebuilt.extend(body[index:])
# Original format requires md5 for every entry — skip without
if not md5:
continue
bios_parts.append(f'{{ "md5": "{md5}", "file": "bios/{dest}" }}')
added = [
self._entry_line(system, files)
for native_id, (system, files) in exportable.items()
if native_id not in written
]
if added:
while rebuilt and not rebuilt[-1].strip():
rebuilt.pop()
rebuilt.append("")
rebuilt.extend(sorted(added))
rebuilt.append("")
bios_str = ", ".join(bios_parts)
line = (
f' "{native_id}": '
f'{{ "name": "{display_name}", '
f'"biosFiles": [ {bios_str} ] }},'
)
lines.append(line)
return {self.native_filename(): "\n".join([*head, *rebuilt, *tail])}
lines.append("")
lines.append("}")
lines.append("")
Path(output_path).write_text("\n".join(lines), encoding="utf-8")
def validate(self, truth_data: dict, output_path: str) -> list[str]:
content = Path(output_path).read_text(encoding="utf-8")
def validate(
self,
systems: dict[str, NativeSystem],
produced: dict[str, str],
) -> list[str]:
content = produced[self.native_filename()]
issues: list[str] = []
for sys_data in truth_data.get("systems", {}).values():
for fe in sys_data.get("files", []):
name = fe.get("name", "")
if name.startswith("_") or self._is_pattern(name):
continue
# Skip entries without md5 (not exportable in this format)
md5 = fe.get("md5", "")
if isinstance(md5, list):
md5 = md5[0] if md5 else ""
if not md5:
continue
dest = self._dest(fe)
if dest not in content and name not in content:
issues.append(f"missing: {name}")
namespace: dict[str, object] = {}
block = content.split("\nsystems = {", 1)
if len(block) != 2:
return ["no systems dict in the output"]
end = block[1].find("\n}")
if end < 0:
return ["the systems dict is not closed"]
try:
exec("systems = {" + block[1][:end] + "\n}", {}, namespace) # noqa: S102
except SyntaxError as exc:
return [f"the systems dict does not parse: {exc}"]
exported = namespace.get("systems", {})
if not isinstance(exported, dict):
return ["the systems dict did not evaluate to a dict"]
for system, files in self.exportable(systems, require="md5"):
entry = exported.get(system.native_id)
if entry is None:
issues.append(f"system absent: {system.native_id}")
continue
declared = {bios.get("file", "") for bios in entry.get("biosFiles", [])}
for fe in files:
path = str(fe.native("native_path", "")) or f"bios/{fe.destination}"
if path not in declared:
issues.append(f"absent: {system.native_id}/{fe.name}")
for native_id, entry in exported.items():
if not entry.get("biosFiles"):
issues.append(f"empty entry: {native_id}")
if "def checkBios(" not in content:
issues.append("the checker the script exists for is missing")
return issues
+153
View File
@@ -0,0 +1,153 @@
"""Exporter for BizHawk's FirmwareDatabase.cs.
The database is C#: every firmware is a call whose arguments are the SHA1,
the size, the file name and a description, wired into option lists and
status flags that only the source expresses. The calls are rewritten in
place, and nothing else in the file is touched.
"""
from __future__ import annotations
import re
from .base_exporter import BaseExporter
from .baseline import NativeFile, NativeSystem, Report
SOURCE_URL = (
"https://raw.githubusercontent.com/TASEmulators/BizHawk/master"
"/src/BizHawk.Emulation.Common/Database/FirmwareDatabase.cs"
)
# File("<sha1>", <size>, "<name>", ...) and the same three arguments inside
# FirmwareAndOption(<sha1>, <size>, <system>, <id>, <name>, ...).
_FILE_CALL = re.compile(
r'(File\(\s*")([0-9A-Fa-f]{40})("\s*,\s*)(\d+)(\s*,\s*")([^"]+)(")'
)
_FIRMWARE_AND_OPTION = re.compile(
r'(FirmwareAndOption\(\s*")([0-9A-Fa-f]{40})("\s*,\s*)(\d+)'
r'(\s*,\s*"[^"]*"\s*,\s*"[^"]*"\s*,\s*")([^"]+)(")'
)
class Exporter(BaseExporter):
"""Write BizHawk's FirmwareDatabase.cs, corrected."""
@staticmethod
def platform_name() -> str:
return "bizhawk"
@staticmethod
def native_filename() -> str:
return "FirmwareDatabase.cs"
@staticmethod
def carries() -> frozenset[str]:
return frozenset({"sha1", "size"})
@staticmethod
def native_sources() -> dict[str, str]:
return {"FirmwareDatabase.cs": SOURCE_URL}
@staticmethod
def can_add() -> bool:
# A firmware is a call wired into an option list and a status; the
# exporter corrects the calls that exist, it does not write C#.
return False
@staticmethod
def needs_original() -> bool:
# The database is code: option lists, statuses and the systems they
# hang off exist nowhere else.
return True
@staticmethod
def _unambiguous(systems: dict[str, NativeSystem]) -> dict[str, NativeFile]:
"""Files whose name identifies exactly one entry with a SHA1.
BizHawk names a firmware by file name inside a system, and the same
name recurs across systems. Correcting on a name that resolves to
two different sets of bytes would corrupt the database, so only the
names that resolve to one are touched.
"""
seen: dict[str, list[NativeFile]] = {}
for system in systems.values():
for fe in system.files:
if fe.hash("sha1"):
seen.setdefault(fe.name.casefold(), []).append(fe)
resolved: dict[str, NativeFile] = {}
for name, entries in seen.items():
hashes = {fe.hash("sha1").lower() for fe in entries}
if len(hashes) == 1:
resolved[name] = entries[0]
return resolved
def render(
self,
systems: dict[str, NativeSystem],
report: Report,
originals: dict[str, str],
scraped: dict | None = None,
) -> dict[str, str]:
source = originals.get(self.native_filename(), "")
if not source:
raise ValueError(
"FirmwareDatabase.cs cannot be written without BizHawk's own "
"file: the database is C#, not data"
)
index = self._unambiguous(systems)
def commented_out(text: str, position: int) -> bool:
"""Whether the call sits on a line the compiler never sees.
BizHawk keeps disabled entries in place behind //, and a hash
written into one of those is a change to a comment.
"""
line_start = text.rfind("\n", 0, position) + 1
return text[line_start:position].lstrip().startswith("//")
def rewrite(match: re.Match[str], name_group: int) -> str:
if commented_out(match.string, match.start()):
return match.group(0)
name = match.group(name_group)
fe = index.get(name.casefold())
if fe is None:
return match.group(0)
sha1 = fe.hash("sha1").upper()
size = fe.size() or int(match.group(4))
groups = list(match.groups())
groups[1] = sha1
groups[3] = str(size)
return "".join(groups)
patched = _FILE_CALL.sub(lambda m: rewrite(m, 6), source)
patched = _FIRMWARE_AND_OPTION.sub(lambda m: rewrite(m, 6), patched)
return {self.native_filename(): patched}
def validate(
self,
systems: dict[str, NativeSystem],
produced: dict[str, str],
) -> list[str]:
content = produced[self.native_filename()]
issues: list[str] = []
if "FirmwareDatabase" not in content:
issues.append("the class the database lives in is missing")
if content.count("{") != content.count("}"):
issues.append("braces are unbalanced, the file would not compile")
index = self._unambiguous(systems)
declared: dict[str, str] = {}
for pattern in (_FILE_CALL, _FIRMWARE_AND_OPTION):
for match in pattern.finditer(content):
line_start = content.rfind("\n", 0, match.start()) + 1
if content[line_start : match.start()].lstrip().startswith("//"):
continue
declared[match.group(6).casefold()] = match.group(2).lower()
for name, fe in index.items():
written = declared.get(name)
if written is not None and written != fe.hash("sha1").lower():
issues.append(f"hash not applied: {fe.name}")
return issues
+135 -184
View File
@@ -1,215 +1,166 @@
"""Exporter for EmuDeck checkBIOS.sh format.
"""Exporter for EmuDeck's checkBIOS.sh.
Produces a bash script matching the exact pattern of EmuDeck's
functions/checkBIOS.sh: per-system check functions with MD5 arrays
inside the function body, iterating over $biosPath/* files.
Two patterns:
- MD5 pattern: systems with known hashes, loop $biosPath/*, md5sum each, match
- File-exists pattern: systems with specific paths, check -f
checkBIOS.sh is a shell library: each system is a function EmuDeck calls by
name, and the only data in it is the MD5 list each function matches against.
Writing the functions from a table would publish a file missing whichever
checks the table forgot, so the original is patched instead and every
function it defines keeps its shape, its scan directory and its output.
"""
from __future__ import annotations
import re
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from scraper.emudeck_scraper import FUNCTION_HASH_MAP, _RE_FUNC, _RE_LOCAL_HASHES
from .base_exporter import BaseExporter
from .baseline import NativeSystem, Report
# Map our system IDs to EmuDeck function naming conventions
_SYSTEM_CONFIG: dict[str, dict] = {
"sony-playstation": {
"func": "checkPS1BIOS",
"var": "PSXBIOS",
"array": "PSBios",
"pattern": "md5",
},
"sony-playstation-2": {
"func": "checkPS2BIOS",
"var": "PS2BIOS",
"array": "PS2Bios",
"pattern": "md5",
},
"sega-mega-cd": {
"func": "checkSegaCDBios",
"var": "SEGACDBIOS",
"array": "CDBios",
"pattern": "md5",
},
"sega-saturn": {
"func": "checkSaturnBios",
"var": "SATURNBIOS",
"array": "SaturnBios",
"pattern": "md5",
},
"sega-dreamcast": {
"func": "checkDreamcastBios",
"var": "BIOS",
"array": "hashes",
"pattern": "md5",
},
"nintendo-ds": {
"func": "checkDSBios",
"var": "BIOS",
"array": "hashes",
"pattern": "md5",
},
"nintendo-switch": {
"func": "checkCitronBios",
"pattern": "file-exists",
"firmware_path": "$biosPath/citron/firmware",
"keys_path": "$biosPath/citron/keys/prod.keys",
},
}
def _make_md5_function(cfg: dict, md5s: list[str]) -> list[str]:
"""Generate a MD5-checking function matching EmuDeck's exact pattern."""
func = cfg["func"]
var = cfg["var"]
array = cfg["array"]
md5_str = " ".join(md5s)
return [
f"{func}(){{",
"",
f'\t{var}="NULL"',
"",
'\tfor entry in "$biosPath/"*',
"\tdo",
'\t\tif [ -f "$entry" ]; then',
'\t\t\tmd5=($(md5sum "$entry"))',
f'\t\t\tif [[ "${var}" != true ]]; then',
f"\t\t\t\t{array}=({md5_str})",
f'\t\t\t\tfor i in "${{{array}[@]}}"',
"\t\t\t\tdo",
'\t\t\t\tif [[ "$md5" == *"${i}"* ]]; then',
f"\t\t\t\t\t{var}=true",
"\t\t\t\t\tbreak",
"\t\t\t\telse",
f"\t\t\t\t\t{var}=false",
"\t\t\t\tfi",
"\t\t\t\tdone",
"\t\t\tfi",
"\t\tfi",
"\tdone",
"",
"",
f"\tif [ ${var} == true ]; then",
'\t\techo "$entry true";',
"\telse",
'\t\techo "false";',
"\tfi",
"}",
]
def _make_file_exists_function(cfg: dict) -> list[str]:
"""Generate a file-exists function matching EmuDeck's pattern."""
func = cfg["func"]
firmware = cfg.get("firmware_path", "")
keys = cfg.get("keys_path", "")
return [
f"{func}(){{",
"",
f'\tlocal FIRMWARE="{firmware}"',
f'\tlocal KEYS="{keys}"',
'\tif [[ -f "$KEYS" ]] && [[ "$( ls -A "$FIRMWARE")" ]]; then',
'\t\t\techo "true";',
"\telse",
'\t\t\techo "false";',
"\tfi",
"}",
]
SOURCE_URL = (
"https://raw.githubusercontent.com/dragoonDorise/EmuDeck/main"
"/functions/checkBIOS.sh"
)
_MD5 = re.compile(r"^[0-9a-f]{32}$")
class Exporter(BaseExporter):
"""Export truth data to EmuDeck checkBIOS.sh format."""
"""Write EmuDeck's checkBIOS.sh, corrected."""
@staticmethod
def platform_name() -> str:
return "emudeck"
def export(
self,
truth_data: dict,
output_path: str,
scraped_data: dict | None = None,
) -> None:
lines: list[str] = ["#!/bin/bash"]
@staticmethod
def native_filename() -> str:
return "checkBIOS.sh"
systems = truth_data.get("systems", {})
@staticmethod
def requires() -> str:
return "md5"
for sys_id, cfg in sorted(_SYSTEM_CONFIG.items(), key=lambda x: x[1]["func"]):
sys_data = systems.get(sys_id)
if not sys_data:
@staticmethod
def carries() -> frozenset[str]:
return frozenset({"md5"})
@staticmethod
def native_sources() -> dict[str, str]:
return {"checkBIOS.sh": SOURCE_URL}
@staticmethod
def needs_original() -> bool:
# The checks are code, and EmuDeck calls them by name.
return True
@staticmethod
def can_add() -> bool:
"""A hash is corrected in place, never added to an array.
EmuDeck's frontend calls one check for several emulators
(EmulatorsDetailPage.jsx: 'ra' asks checkPS1BIOS, checkSegaCDBios,
checkSaturnBios, checkDSBios and checkDreamcastBios, and
'duckstation' asks checkPS1BIOS as well), and nothing in
checkBIOS.sh names them. Growing an array therefore changes answers
for consumers the array does not list: DuckStation boots from a PS2
image, the RetroArch PSX cores do not, so adding the ones
DuckStation accepts would report a BIOS to a card that has none.
Correcting a value in place changes no consumer's set.
"""
return False
@classmethod
def _md5s(cls, systems: dict[str, NativeSystem], system_id: str) -> list[str]:
"""Every MD5 the system accepts, in a stable order, deduplicated."""
seen: list[str] = []
for system in systems.values():
if system.native_id != system_id:
continue
for fe in system.files:
if not cls.writable(fe):
continue
for value in fe.hashes("md5"):
if _MD5.match(value) and value not in seen:
seen.append(value)
return seen
lines.append("")
def _function_spans(self, script: str) -> list[tuple[str, int, int]]:
"""Name and byte span of every check the script defines."""
matches = list(_RE_FUNC.finditer(script))
spans: list[tuple[str, int, int]] = []
for index, match in enumerate(matches):
end = (
matches[index + 1].start()
if index + 1 < len(matches)
else len(script)
)
spans.append((match.group(1), match.start(), end))
return spans
if cfg["pattern"] == "md5":
md5s: list[str] = []
for fe in sys_data.get("files", []):
name = fe.get("name", "")
if self._is_pattern(name) or name.startswith("_"):
continue
md5 = fe.get("md5", "")
if isinstance(md5, list):
md5s.extend(
m for m in md5 if m and re.fullmatch(r"[a-f0-9]{32}", m)
)
elif md5 and re.fullmatch(r"[a-f0-9]{32}", md5):
md5s.append(md5)
if md5s:
lines.extend(_make_md5_function(cfg, md5s))
elif cfg["pattern"] == "file-exists":
lines.extend(_make_file_exists_function(cfg))
def render(
self,
systems: dict[str, NativeSystem],
report: Report,
originals: dict[str, str],
scraped: dict | None = None,
) -> dict[str, str]:
script = originals.get(self.native_filename(), "")
if not script:
raise ValueError(
"checkBIOS.sh cannot be written without EmuDeck's own file: "
"the checks are code, not data"
)
lines.append("")
Path(output_path).write_text("\n".join(lines), encoding="utf-8")
pieces: list[str] = []
cursor = 0
for name, start, end in self._function_spans(script):
pieces.append(script[cursor:start])
body = script[start:end]
system_id = FUNCTION_HASH_MAP.get(name)
md5s = self._md5s(systems, system_id) if system_id else []
match = _RE_LOCAL_HASHES.search(body)
if md5s and match:
# An array compared by membership says nothing about order,
# so the same set is left as the maintainer wrote it.
if set(md5s) != set(match.group(1).split()):
body = (
body[: match.start(1)]
+ " ".join(md5s)
+ body[match.end(1) :]
)
pieces.append(body)
cursor = end
pieces.append(script[cursor:])
def validate(self, truth_data: dict, output_path: str) -> list[str]:
content = Path(output_path).read_text(encoding="utf-8")
return {self.native_filename(): "".join(pieces)}
def validate(
self,
systems: dict[str, NativeSystem],
produced: dict[str, str],
) -> list[str]:
content = produced[self.native_filename()]
issues: list[str] = []
systems = truth_data.get("systems", {})
for sys_id, cfg in _SYSTEM_CONFIG.items():
if cfg["pattern"] != "md5":
defined = {name for name, _, _ in self._function_spans(content)}
for name, system_id in FUNCTION_HASH_MAP.items():
if name not in defined:
issues.append(f"check absent from the output: {name}")
continue
sys_data = systems.get(sys_id)
if not sys_data:
md5s = self._md5s(systems, system_id)
if not md5s:
continue
for fe in sys_data.get("files", []):
# export skips placeholders and private entries, so looking
# for them here makes the exporter reject its own output.
name = fe.get("name", "")
if self._is_pattern(name) or name.startswith("_"):
continue
md5 = fe.get("md5", "")
if isinstance(md5, list):
md5 = md5[0] if md5 else ""
if md5 and re.fullmatch(r"[a-f0-9]{32}", md5) and md5 not in content:
issues.append(f"missing md5: {md5} ({name})")
for sys_id, cfg in _SYSTEM_CONFIG.items():
func = cfg["func"]
if func in content:
body = next(
content[start:end]
for fname, start, end in self._function_spans(content)
if fname == name
)
if not _RE_LOCAL_HASHES.search(body):
# A check with no hash list is a path check, not a hash check.
continue
sys_data = systems.get(sys_id)
if not sys_data or not sys_data.get("files"):
continue
# Only flag if the system has usable data for the function type
if cfg["pattern"] == "md5":
has_md5 = any(
fe.get("md5")
and isinstance(fe.get("md5"), str)
and re.fullmatch(r"[a-f0-9]{32}", fe["md5"])
for fe in sys_data["files"]
)
if has_md5:
issues.append(f"missing function: {func}")
elif cfg["pattern"] == "file-exists":
issues.append(f"missing function: {func}")
for md5 in md5s:
if md5 not in body:
issues.append(f"absent from {name}: {md5}")
return issues
+2 -6
View File
@@ -1,8 +1,4 @@
"""Exporter for Lakka (System.dat format, same as RetroArch).
Lakka inherits RetroArch cores and uses the same System.dat format.
Delegates to systemdat_exporter for export and validation.
"""
"""Exporter for Lakka, which reads RetroArch's System.dat unchanged."""
from __future__ import annotations
@@ -10,7 +6,7 @@ from .systemdat_exporter import Exporter as SystemDatExporter
class Exporter(SystemDatExporter):
"""Export truth data to Lakka System.dat format."""
"""Write Lakka's System.dat, corrected."""
@staticmethod
def platform_name() -> str:
+125
View File
@@ -0,0 +1,125 @@
"""Exporter for MiSTer's BiosDB (bios_db.json, shipped zipped).
A Downloader database entry carries the download URL, the install path and
the tag ids alongside the hash, and none of those are ours to invent. Only
the hash and the size are rewritten, in MiSTer's own database.
"""
from __future__ import annotations
import io
import json
import zipfile
from collections import OrderedDict
from .base_exporter import BaseExporter
from .baseline import NativeSystem, Report
SOURCE_URL = (
"https://raw.githubusercontent.com/ajgowans/BiosDB_MiSTer/db/bios_db.json.zip"
)
_DB_NAME = "bios_db.json"
class Exporter(BaseExporter):
"""Write MiSTer's bios_db.json, corrected."""
@staticmethod
def platform_name() -> str:
return "misterfpga"
@staticmethod
def native_filename() -> str:
return _DB_NAME
@staticmethod
def carries() -> frozenset[str]:
return frozenset({"md5", "size"})
@staticmethod
def native_sources() -> dict[str, str]:
return {"bios_db.json.zip": SOURCE_URL}
@staticmethod
def can_add() -> bool:
# Every entry carries the URL MiSTer installs it from, and that is
# not ours to invent.
return False
@staticmethod
def needs_original() -> bool:
# Entries carry a URL and a tag vocabulary the database owns.
return True
@staticmethod
def unpack(raw: bytes) -> dict[str, str]:
"""Read the database out of the archive MiSTer publishes."""
with zipfile.ZipFile(io.BytesIO(raw)) as archive:
return {_DB_NAME: archive.read(_DB_NAME).decode("utf-8")}
def _by_path(self, systems: dict[str, NativeSystem]) -> dict[str, object]:
indexed: dict[str, object] = {}
for system in systems.values():
for fe in system.files:
if fe.destination:
indexed[f"games/{fe.destination}"] = fe
return indexed
def render(
self,
systems: dict[str, NativeSystem],
report: Report,
originals: dict[str, str],
scraped: dict | None = None,
) -> dict[str, str]:
original = originals.get(_DB_NAME) or originals.get("bios_db.json.zip")
if not original:
raise ValueError(
"bios_db.json cannot be written without MiSTer's own database: "
"every entry carries a URL and tag ids that are not ours"
)
database = json.loads(original, object_pairs_hook=OrderedDict)
indexed = self._by_path(systems)
for path, entry in database.get("files", {}).items():
fe = indexed.get(path)
if fe is None:
continue
md5 = fe.hash("md5")
if md5:
entry["hash"] = md5
size = fe.size()
if size:
entry["size"] = size
return {_DB_NAME: json.dumps(database, indent=2, ensure_ascii=False) + "\n"}
def validate(
self,
systems: dict[str, NativeSystem],
produced: dict[str, str],
) -> list[str]:
try:
database = json.loads(produced[_DB_NAME])
except json.JSONDecodeError as exc:
return [f"the database does not parse: {exc}"]
issues: list[str] = []
if not database.get("db_id"):
issues.append("the database lost its db_id")
files = database.get("files", {})
if not files:
issues.append("the database has no files left")
for path, entry in files.items():
if not entry.get("hash"):
issues.append(f"entry without a hash: {path}")
if not entry.get("url"):
issues.append(f"entry without a URL, uninstallable: {path}")
indexed = self._by_path(systems)
for path, fe in indexed.items():
declared = files.get(path)
md5 = fe.hash("md5")
if declared is not None and md5 and declared.get("hash") != md5:
issues.append(f"hash not applied: {path}")
return issues
+134 -104
View File
@@ -1,135 +1,165 @@
"""Exporter for Recalbox es_bios.xml format.
"""Exporter for Recalbox's es_bios.xml.
Produces XML matching the exact format of recalbox's es_bios.xml:
- XML namespace declaration
- <system fullname="..." platform="...">
- <bios path="system/file" md5="..." core="..." /> with optional mandatory, hashMatchMandatory, note
- mandatory absent = true (only explicit when false)
- 2-space indentation
The file is validated by es_bios.xsd, which makes path, md5 and core
required on every bios element. An entry we cannot give all three to is not
written: Recalbox would reject the file whole.
mandatory and hashMatchMandatory are separate axes. Recalbox reads a missing
attribute as true for both, so each is written only when it is false, or
when the platform stated it explicitly.
"""
from __future__ import annotations
import sys
from pathlib import Path
from xml.etree.ElementTree import ParseError
from xml.sax.saxutils import quoteattr
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from common import parse_untrusted_xml
from .base_exporter import BaseExporter
from .baseline import NativeFile, NativeSystem, Report
SOURCE_URL = (
"https://gitlab.com/recalbox/recalbox/-/raw/master/board/recalbox/fsoverlay"
"/recalbox/share_init/system/.emulationstation/es_bios.xml"
)
SCHEMA_URL = (
"https://gitlab.com/recalbox/recalbox/-/raw/master/board/recalbox/fsoverlay"
"/recalbox/share_init/system/.emulationstation/es_bios.xsd"
)
class Exporter(BaseExporter):
"""Export truth data to Recalbox es_bios.xml format."""
"""Write Recalbox's es_bios.xml, corrected."""
@staticmethod
def platform_name() -> str:
return "recalbox"
def export(
self,
truth_data: dict,
output_path: str,
scraped_data: dict | None = None,
) -> None:
native_map: dict[str, str] = {}
if scraped_data:
for sys_id, sys_data in scraped_data.get("systems", {}).items():
nid = sys_data.get("native_id")
if nid:
native_map[sys_id] = nid
@staticmethod
def native_filename() -> str:
return "es_bios.xml"
lines: list[str] = [
@staticmethod
def carries() -> frozenset[str]:
return frozenset({"md5", "required"})
@staticmethod
def native_sources() -> dict[str, str]:
return {"es_bios.xml": SOURCE_URL, "es_bios.xsd": SCHEMA_URL}
def _path(self, fe: NativeFile, native_id: str) -> str:
"""The path Recalbox reads, pipe-joined when it accepts several.
A path Recalbox already states is reproduced exactly: several of its
entries sit at the BIOS root with no directory at all, and prefixing
them with the system would point the frontend somewhere else.
"""
if fe.platform is not None:
path = str(fe.platform.get("destination") or fe.name)
alternatives = fe.platform.get("alt_paths") or []
if alternatives:
return "|".join([path, *[str(a) for a in alternatives]])
return path
dest = fe.destination or fe.name
return dest if "/" in dest else f"{native_id}/{dest}"
def _bios_element(self, fe: NativeFile, native_id: str) -> str:
attrs = [f"path={quoteattr(self._path(fe, native_id))}"]
attrs.append(f'md5={quoteattr(",".join(fe.hashes("md5")))}')
attrs.append(f'core={quoteattr(",".join(fe.cores()))}')
if not fe.required:
attrs.append('mandatory="false"')
elif fe.native("mandatory_declared", None) is True:
attrs.append('mandatory="true"')
hash_match = fe.native("hash_match_mandatory", None)
if hash_match is False:
attrs.append('hashMatchMandatory="false"')
elif hash_match is True:
attrs.append('hashMatchMandatory="true"')
note = " ".join(str(fe.native("note", "")).split())
if note:
attrs.append(f"note={quoteattr(note)}")
return f" <bios {' '.join(attrs)} />"
@classmethod
def writable(cls, fe: NativeFile, require: str = "") -> bool:
"""es_bios.xsd makes md5 and core required; without them, no element.
An entry Recalbox already ships has both, so this only ever gates
what we would be adding.
"""
return bool(fe.hashes("md5")) and bool(fe.cores())
def render(
self,
systems: dict[str, NativeSystem],
report: Report,
originals: dict[str, str],
scraped: dict | None = None,
) -> dict[str, str]:
lines = [
'<?xml version="1.0" encoding="UTF-8"?>',
'<biosList xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"'
' xsi:noNamespaceSchemaLocation="es_bios.xsd">',
]
systems = truth_data.get("systems", {})
for sys_id in sorted(systems):
sys_data = systems[sys_id]
files = sys_data.get("files", [])
if not files:
for system in sorted(systems.values(), key=lambda s: s.native_id):
writable = [fe for fe in system.files if self.writable(fe)]
if not writable:
continue
native_id = native_map.get(sys_id, sys_id)
scraped_sys = (
scraped_data.get("systems", {}).get(sys_id) if scraped_data else None
)
display_name = self._display_name(sys_id, scraped_sys)
lines.append(f' <system fullname="{display_name}" platform="{native_id}">')
# Build path lookup from scraped data for this system
scraped_paths: dict[str, str] = {}
if scraped_data:
s_sys = scraped_data.get("systems", {}).get(sys_id, {})
for sf in s_sys.get("files", []):
sname = sf.get("name", "").lower()
spath = sf.get("destination", sf.get("name", ""))
if sname and spath:
scraped_paths[sname] = spath
for fe in files:
name = fe.get("name", "")
if name.startswith("_") or self._is_pattern(name):
continue
# Use scraped path when available (preserves original format)
path = scraped_paths.get(name.lower())
if not path:
dest = self._dest(fe)
path = f"{native_id}/{dest}" if "/" not in dest else dest
md5 = fe.get("md5", "")
if isinstance(md5, list):
md5 = ",".join(md5)
required = fe.get("required", True)
# Build cores string from _cores
cores_list = fe.get("_cores", [])
core_str = (
",".join(f"libretro/{c}" for c in cores_list) if cores_list else ""
)
attrs = [f'path="{path}"']
if md5:
attrs.append(f'md5="{md5}"')
if not required:
attrs.append('mandatory="false"')
if not required:
attrs.append('hashMatchMandatory="true"')
if core_str:
attrs.append(f'core="{core_str}"')
lines.append(f" <bios {' '.join(attrs)} />")
fullname = quoteattr(self.display_name(system))
platform = quoteattr(system.native_id)
lines.append(f" <system fullname={fullname} platform={platform}>")
for fe in writable:
lines.append(self._bios_element(fe, system.native_id))
lines.append(" </system>")
lines.append("</biosList>")
lines.append("")
Path(output_path).write_text("\n".join(lines), encoding="utf-8")
def validate(self, truth_data: dict, output_path: str) -> list[str]:
from xml.etree.ElementTree import parse as xml_parse
tree = xml_parse(output_path)
root = tree.getroot()
exported_paths: set[str] = set()
for bios_el in root.iter("bios"):
path = bios_el.get("path", "")
if path:
exported_paths.add(path.lower())
exported_paths.add(path.split("/")[-1].lower())
return {self.native_filename(): "\n".join(lines)}
def validate(
self,
systems: dict[str, NativeSystem],
produced: dict[str, str],
) -> list[str]:
content = produced[self.native_filename()]
issues: list[str] = []
for sys_data in truth_data.get("systems", {}).values():
for fe in sys_data.get("files", []):
name = fe.get("name", "")
if name.startswith("_") or self._is_pattern(name):
try:
root = parse_untrusted_xml(content, self.native_filename())
except (ParseError, ValueError) as exc:
return [f"the XML does not parse: {exc}"]
for element in root.iter("bios"):
for attribute in ("path", "md5", "core"):
if not element.get(attribute):
issues.append(
f"es_bios.xsd requires {attribute}: "
f"{element.get('path', '?')}"
)
for element in root.iter("system"):
if not list(element):
issues.append(f"empty system: {element.get('platform', '?')}")
for attribute in ("fullname", "platform"):
if not element.get(attribute):
issues.append(f"es_bios.xsd requires {attribute} on system")
exported = {
element.get("path", "").casefold() for element in root.iter("bios")
}
for system in systems.values():
for fe in system.files:
if not self.writable(fe):
continue
dest = self._dest(fe)
if (
name.lower() not in exported_paths
and dest.lower() not in exported_paths
):
issues.append(f"missing: {name}")
if self._path(fe, system.native_id).casefold() not in exported:
issues.append(f"absent: {system.native_id}/{fe.name}")
return issues
+90 -88
View File
@@ -1,116 +1,118 @@
"""Exporter for RetroBat batocera-systems.json format.
"""Exporter for RetroBat's batocera-systems.json.
Produces JSON matching the exact format of
RetroBat-Official/emulatorlauncher/batocera-systems/Resources/batocera-systems.json:
- System keys with "name" and "biosFiles" fields
- Each biosFile has "md5" before "file" (matching original key order)
Pure data: a system key carrying a name and a biosFiles array whose entries
state md5 then file, in that order.
"""
from __future__ import annotations
import json
from collections import OrderedDict
from pathlib import Path
from .base_exporter import BaseExporter
from .baseline import NativeSystem, Report
SOURCE_URL = (
"https://raw.githubusercontent.com/RetroBat-Official/emulatorlauncher/master"
"/batocera-systems/Resources/batocera-systems.json"
)
class Exporter(BaseExporter):
"""Export truth data to RetroBat batocera-systems.json format."""
"""Write RetroBat's batocera-systems.json, corrected."""
@staticmethod
def platform_name() -> str:
return "retrobat"
def export(
@staticmethod
def native_filename() -> str:
return "batocera-systems.json"
@staticmethod
def requires() -> str:
return "md5"
@staticmethod
def carries() -> frozenset[str]:
return frozenset({"md5"})
@staticmethod
def native_sources() -> dict[str, str]:
return {"batocera-systems.json": SOURCE_URL}
def render(
self,
truth_data: dict,
output_path: str,
scraped_data: dict | None = None,
) -> None:
native_map: dict[str, str] = {}
if scraped_data:
for sys_id, sys_data in scraped_data.get("systems", {}).items():
nid = sys_data.get("native_id")
if nid:
native_map[sys_id] = nid
systems: dict[str, NativeSystem],
report: Report,
originals: dict[str, str],
scraped: dict | None = None,
) -> dict[str, str]:
# Keep the platform's own key order when we have its file, so the
# diff a maintainer reads is the corrections and nothing else.
order: list[str] = []
original = originals.get(self.native_filename(), "")
if original:
try:
order = list(json.loads(original))
except json.JSONDecodeError:
order = []
output: OrderedDict[str, dict] = OrderedDict()
systems = truth_data.get("systems", {})
for sys_id in sorted(systems):
sys_data = systems[sys_id]
files = sys_data.get("files", [])
if not files:
continue
native_id = native_map.get(sys_id, sys_id)
scraped_sys = (
scraped_data.get("systems", {}).get(sys_id) if scraped_data else None
)
display_name = self._display_name(sys_id, scraped_sys)
bios_files: list[OrderedDict] = []
exportable = {
system.native_id: (system, files)
for system, files in self.exportable(systems, require="md5")
}
keys = [k for k in order if k in exportable]
keys.extend(sorted(k for k in exportable if k not in keys))
output: OrderedDict[str, object] = OrderedDict()
for key in keys:
system, files = exportable[key]
bios_files = []
for fe in files:
name = fe.get("name", "")
if name.startswith("_") or self._is_pattern(name):
continue
dest = self._dest(fe)
md5 = fe.get("md5", "")
if isinstance(md5, list):
md5 = md5[0] if md5 else ""
# Original format requires md5 for every entry
if not md5:
continue
entry: OrderedDict[str, str] = OrderedDict()
entry["md5"] = md5
entry["file"] = f"bios/{dest}"
entry["md5"] = fe.hash("md5")
declared = fe.native("native_path", "")
entry["file"] = str(declared) if declared else f"bios/{fe.destination}"
bios_files.append(entry)
system_entry: OrderedDict[str, object] = OrderedDict()
system_entry["name"] = self.display_name(system)
system_entry["biosFiles"] = bios_files
output[key] = system_entry
if bios_files:
if native_id in output:
existing_files = {
e.get("file") for e in output[native_id]["biosFiles"]
}
for entry in bios_files:
if entry.get("file") not in existing_files:
output[native_id]["biosFiles"].append(entry)
else:
sys_entry: OrderedDict[str, object] = OrderedDict()
sys_entry["name"] = display_name
sys_entry["biosFiles"] = bios_files
output[native_id] = sys_entry
text = json.dumps(output, indent=2, ensure_ascii=False) + "\n"
return {self.native_filename(): text}
Path(output_path).write_text(
json.dumps(output, indent=2, ensure_ascii=False) + "\n",
encoding="utf-8",
)
def validate(self, truth_data: dict, output_path: str) -> list[str]:
data = json.loads(Path(output_path).read_text(encoding="utf-8"))
exported_files: set[str] = set()
for sys_data in data.values():
for bf in sys_data.get("biosFiles", []):
path = bf.get("file", "")
stripped = path.removeprefix("bios/")
exported_files.add(stripped)
basename = path.split("/")[-1] if "/" in path else path
exported_files.add(basename)
def validate(
self,
systems: dict[str, NativeSystem],
produced: dict[str, str],
) -> list[str]:
try:
data = json.loads(produced[self.native_filename()])
except json.JSONDecodeError as exc:
return [f"the JSON does not parse: {exc}"]
issues: list[str] = []
for sys_data in truth_data.get("systems", {}).values():
for fe in sys_data.get("files", []):
name = fe.get("name", "")
if name.startswith("_") or self._is_pattern(name):
continue
md5 = fe.get("md5", "")
if isinstance(md5, list):
md5 = md5[0] if md5 else ""
if not md5:
continue
dest = self._dest(fe)
if name not in exported_files and dest not in exported_files:
issues.append(f"missing: {name}")
for key, entry in data.items():
if not entry.get("name"):
issues.append(f"system without a name: {key}")
if not entry.get("biosFiles"):
issues.append(f"empty entry: {key}")
for bios in entry.get("biosFiles", []):
# RetroBat states an unhashed file with an empty md5, so only
# a missing path makes an entry unusable.
if "md5" not in bios or not bios.get("file"):
issues.append(f"incomplete entry: {key}/{bios.get('file', '?')}")
for system, files in self.exportable(systems, require="md5"):
entry = data.get(system.native_id)
if entry is None:
issues.append(f"system absent: {system.native_id}")
continue
declared = {bios.get("file") for bios in entry.get("biosFiles", [])}
for fe in files:
path = str(fe.native("native_path", "")) or f"bios/{fe.destination}"
if path not in declared:
issues.append(f"absent: {system.native_id}/{fe.name}")
return issues
+152 -180
View File
@@ -1,210 +1,182 @@
"""Exporter for RetroDECK component_manifest.json format.
"""Exporter for RetroDECK's component manifests.
Produces a JSON file compatible with RetroDECK's component manifests.
Each system maps to a component with BIOS entries containing filename,
md5 (comma-separated if multiple), paths ($bios_path default), and
required status.
Path tokens: $bios_path for bios/, $roms_path for roms/.
Entries without an explicit path default to $bios_path.
RetroDECK has no single BIOS file. Each component carries its own
component_manifest.json, and the BIOS list sits inside it next to the
component's name, description and presets, at one of three keys. Only that
list is rewritten, in the component's own file, so everything else the
manifest drives is left alone.
"""
from __future__ import annotations
import json
import re
from collections import OrderedDict
from pathlib import Path
from .base_exporter import BaseExporter
from .baseline import NativeFile, NativeSystem, Report
# retrobios slug -> RetroDECK system ID (reverse of scraper SYSTEM_SLUG_MAP)
_REVERSE_SLUG: dict[str, str] = {
"nintendo-nes": "nes",
"nintendo-snes": "snes",
"nintendo-64": "n64",
"nintendo-64dd": "n64dd",
"nintendo-gamecube": "gc",
"nintendo-wii": "wii",
"nintendo-wii-u": "wiiu",
"nintendo-switch": "switch",
"nintendo-gb": "gb",
"nintendo-gbc": "gbc",
"nintendo-gba": "gba",
"nintendo-ds": "nds",
"nintendo-3ds": "3ds",
"nintendo-fds": "fds",
"nintendo-sgb": "sgb",
"nintendo-virtual-boy": "virtualboy",
"nintendo-pokemon-mini": "pokemini",
"sony-playstation": "psx",
"sony-playstation-2": "ps2",
"sony-playstation-3": "ps3",
"sony-psp": "psp",
"sony-psvita": "psvita",
"sega-mega-drive": "megadrive",
"sega-mega-cd": "megacd",
"sega-saturn": "saturn",
"sega-dreamcast": "dreamcast",
"sega-dreamcast-arcade": "naomi",
"sega-game-gear": "gamegear",
"sega-master-system": "mastersystem",
"nec-pc-engine": "pcengine",
"nec-pc-fx": "pcfx",
"nec-pc-98": "pc98",
"nec-pc-88": "pc88",
"3do": "3do",
"amstrad-cpc": "amstradcpc",
"arcade": "arcade",
"atari-400-800": "atari800",
"atari-5200": "atari5200",
"atari-7800": "atari7800",
"atari-jaguar": "atarijaguar",
"atari-lynx": "atarilynx",
"atari-st": "atarist",
"commodore-c64": "c64",
"commodore-amiga": "amiga",
"philips-cdi": "cdimono1",
"fairchild-channel-f": "channelf",
"coleco-colecovision": "colecovision",
"mattel-intellivision": "intellivision",
"microsoft-msx": "msx",
"microsoft-xbox": "xbox",
"doom": "doom",
"j2me": "j2me",
"apple-macintosh-ii": "macintosh",
"apple-ii": "apple2",
"apple-iigs": "apple2gs",
"enterprise-64-128": "enterprise",
"tiger-game-com": "gamecom",
"hartung-game-master": "gmaster",
"epoch-scv": "scv",
"watara-supervision": "supervision",
"bandai-wonderswan": "wonderswan",
"snk-neogeo-cd": "neogeocd",
"tandy-coco": "coco",
"tandy-trs-80": "trs80",
"dragon-32-64": "dragon",
"pico8": "pico8",
"wolfenstein-3d": "wolfenstein",
"sinclair-zx-spectrum": "zxspectrum",
}
def _dest_to_path_token(destination: str) -> str:
"""Convert a truth destination path to a RetroDECK path token."""
if destination.startswith("roms/"):
return "$roms_path/" + destination.removeprefix("roms/")
if destination.startswith("bios/"):
return "$bios_path/" + destination.removeprefix("bios/")
# Default: bios path
return "$bios_path/" + destination
COMPONENTS_REPO = "RetroDECK/components"
COMPONENTS_BRANCH = "main"
RAW_BASE = f"https://raw.githubusercontent.com/{COMPONENTS_REPO}/{COMPONENTS_BRANCH}"
MANIFEST = "component_manifest.json"
class Exporter(BaseExporter):
"""Export truth data to RetroDECK component_manifest.json format."""
"""Write RetroDECK's component manifests, corrected."""
@staticmethod
def platform_name() -> str:
return "retrodeck"
def export(
@staticmethod
def native_filename() -> str:
return MANIFEST
@staticmethod
def carries() -> frozenset[str]:
return frozenset({"md5", "sha256", "required"})
@staticmethod
def needs_original() -> bool:
# A manifest is mostly presets and launch configuration; rebuilding
# one from BIOS data alone would throw the component away.
return True
@staticmethod
def component_url(component: str) -> str:
return f"{RAW_BASE}/{component}/{MANIFEST}"
def components(self, systems: dict[str, NativeSystem]) -> list[str]:
"""Components the corrected data touches."""
found: set[str] = set()
for system in systems.values():
for fe in system.files:
component = str(fe.native("component", ""))
if component:
found.add(component)
return sorted(found)
@staticmethod
def _entry(fe: NativeFile) -> OrderedDict:
entry: OrderedDict[str, object] = OrderedDict()
entry["filename"] = fe.name
md5 = ",".join(fe.hashes("md5"))
if md5:
entry["md5"] = md5
sha256 = fe.hash("sha256")
if sha256:
entry["sha256"] = sha256
entry["system"] = fe.native_system
description = fe.native("description", "")
if description:
entry["description"] = str(description)
# RetroDECK words the requirement in prose ("Required", "At least one
# BIOS file required"), so the platform's own wording is kept and a
# boolean is only rendered when there is none to keep.
label = fe.native("required_label", "")
if label:
entry["required"] = str(label)
elif fe.required:
entry["required"] = "Required"
destination = fe.destination
if destination and destination not in (fe.name, f"bios/{fe.name}"):
directory = destination.rsplit("/", 1)[0]
entry["paths"] = "$bios_path/" + directory.removeprefix("bios/")
return entry
def _by_component(
self, systems: dict[str, NativeSystem]
) -> dict[str, list[NativeFile]]:
grouped: dict[str, list[NativeFile]] = {}
for system in systems.values():
for fe in system.files:
component = str(fe.native("component", ""))
if component:
grouped.setdefault(component, []).append(fe)
return grouped
@staticmethod
def _bios_holder(component_value: dict) -> tuple[dict, str] | None:
"""Where in a manifest the BIOS list lives, if it has one."""
if "bios" in component_value:
return component_value, "bios"
for key in ("preset_actions", "cores"):
nested = component_value.get(key)
if isinstance(nested, dict) and "bios" in nested:
return nested, "bios"
return None
def render(
self,
truth_data: dict,
output_path: str,
scraped_data: dict | None = None,
) -> None:
native_map: dict[str, str] = {}
if scraped_data:
for sys_id, sys_data in scraped_data.get("systems", {}).items():
nid = sys_data.get("native_id")
if nid:
native_map[sys_id] = nid
systems: dict[str, NativeSystem],
report: Report,
originals: dict[str, str],
scraped: dict | None = None,
) -> dict[str, str]:
grouped = self._by_component(systems)
produced: dict[str, str] = {}
manifest: OrderedDict[str, dict] = OrderedDict()
systems = truth_data.get("systems", {})
for sys_id in sorted(systems):
sys_data = systems[sys_id]
files = sys_data.get("files", [])
if not files:
for component, files in sorted(grouped.items()):
path = f"{component}/{MANIFEST}"
original = originals.get(path)
if not original:
continue
try:
manifest = json.loads(original, object_pairs_hook=OrderedDict)
except json.JSONDecodeError:
continue
native_id = native_map.get(sys_id, _REVERSE_SLUG.get(sys_id, sys_id))
bios_entries: list[OrderedDict] = []
for fe in files:
name = fe.get("name", "")
if name.startswith("_") or self._is_pattern(name):
entries = [self._entry(fe) for fe in files]
for component_value in manifest.values():
if not isinstance(component_value, dict):
continue
dest = self._dest(fe)
path_token = _dest_to_path_token(dest)
md5 = fe.get("md5", "")
if isinstance(md5, list):
md5 = ",".join(m for m in md5 if m)
required = fe.get("required", True)
entry: OrderedDict[str, object] = OrderedDict()
entry["filename"] = name
if md5:
# Validate MD5 entries
parts = [
m.strip().lower()
for m in str(md5).split(",")
if re.fullmatch(r"[0-9a-f]{32}", m.strip())
]
if parts:
entry["md5"] = ",".join(parts) if len(parts) > 1 else parts[0]
entry["paths"] = path_token
entry["required"] = required
system_val = native_id
entry["system"] = system_val
bios_entries.append(entry)
if bios_entries:
if native_id in manifest:
# Merge into existing component (multiple truth systems
# may map to the same native ID)
existing_names = {
e["filename"] for e in manifest[native_id]["bios"]
}
for entry in bios_entries:
if entry["filename"] not in existing_names:
manifest[native_id]["bios"].append(entry)
holder = self._bios_holder(component_value)
if holder is None:
component_value["bios"] = entries
else:
component = OrderedDict()
component["system"] = native_id
component["bios"] = bios_entries
manifest[native_id] = component
container, key = holder
container[key] = entries
break
Path(output_path).write_text(
json.dumps(manifest, indent=2, ensure_ascii=False) + "\n",
encoding="utf-8",
)
produced[path] = json.dumps(manifest, indent=2, ensure_ascii=False) + "\n"
def validate(self, truth_data: dict, output_path: str) -> list[str]:
data = json.loads(Path(output_path).read_text(encoding="utf-8"))
exported_names: set[str] = set()
for comp_data in data.values():
bios = comp_data.get("bios", [])
if isinstance(bios, list):
for entry in bios:
fn = entry.get("filename", "")
if fn:
exported_names.add(fn)
return produced
def validate(
self,
systems: dict[str, NativeSystem],
produced: dict[str, str],
) -> list[str]:
issues: list[str] = []
for sys_data in truth_data.get("systems", {}).values():
for fe in sys_data.get("files", []):
name = fe.get("name", "")
if name.startswith("_") or self._is_pattern(name):
grouped = self._by_component(systems)
for component, files in grouped.items():
path = f"{component}/{MANIFEST}"
if path not in produced:
issues.append(f"manifest not written: {path}")
continue
try:
manifest = json.loads(produced[path])
except json.JSONDecodeError as exc:
issues.append(f"{path} does not parse: {exc}")
continue
declared: set[str] = set()
for component_value in manifest.values():
if not isinstance(component_value, dict):
continue
if name not in exported_names:
issues.append(f"missing: {name}")
holder = self._bios_holder(component_value)
if holder is None:
continue
container, key = holder
for entry in container[key]:
declared.add(entry.get("filename", ""))
if not component_value.get("name") and not component_value.get(
"system"
):
issues.append(f"{path}: the component lost its identity")
for fe in files:
if fe.name not in declared:
issues.append(f"absent from {path}: {fe.name}")
return issues
+301 -6
View File
@@ -1,17 +1,312 @@
"""Exporter for RetroPie (System.dat format, same as RetroArch).
"""Exporter for RetroPie's scriptmodules.
RetroPie inherits RetroArch cores and uses the same System.dat format.
Delegates to systemdat_exporter for export and validation.
RetroPie ships no BIOS list. What it maintains is one shell script per
package, and the BIOS files a package needs are named in its
`rp_module_help` string, in prose a person reads before copying files.
platforms.cfg carries extensions and full names only.
So the correctable unit is that sentence, and the correction is a file name
missing from it. A name RetroPie already writes is never removed, and no
sentence is invented where a maintainer wrote none: a package we would have
to document from scratch is reported, not drafted.
"""
from __future__ import annotations
from .systemdat_exporter import Exporter as SystemDatExporter
import io
import re
import sys
import tarfile
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from common import load_emulator_profiles
from .base_exporter import BaseExporter
from .baseline import NativeFile, NativeSystem, Report, search_order
SOURCE_URL = (
"https://codeload.github.com/RetroPie/RetroPie-Setup/tar.gz/refs/heads/master"
)
ARCHIVE = "RetroPie-Setup.tar.gz"
# Resolved from the file rather than the working directory: the profiles
# are what say which files a package needs.
EMULATORS_DIR = Path(__file__).resolve().parents[2] / "emulators"
# The longest list RetroPie writes is six names (lr-atari800). Past that the
# sentence stops being something a reader uses, so the names are reported
# instead of appended.
MAX_NAMES = 6
_MODULE_ID = re.compile(r'rp_module_id="([^"]+)"')
_MODULE_HELP = re.compile(r'rp_module_help="((?:[^"\\]|\\.)*)"')
_FILENAME = re.compile(r"[A-Za-z0-9][\w.+-]*\.[A-Za-z0-9]{1,5}\b")
# RetroPie words the instruction several ways ("Copy the required BIOS files
# a.bin and b.bin to $biosdir", "The Sega CD requires the BIOS files a.bin,
# b.bin copied to $biosdir"), so the clause is found by the word BIOS rather
# than by a sentence template, and the names in it are the list to extend.
_BIOS_CLAUSE = re.compile(r"BIOS\b.*?(?=\\n|$)", re.DOTALL)
_BIOS_NOUN = re.compile(r"BIOS (files?)\b")
class Exporter(SystemDatExporter):
"""Export truth data to RetroPie System.dat format."""
class Exporter(BaseExporter):
"""Write RetroPie's scriptmodules, corrected."""
@staticmethod
def platform_name() -> str:
return "retropie"
@staticmethod
def native_filename() -> str:
return "scriptmodules"
@staticmethod
def carries() -> frozenset[str]:
# The help names files; it states no hash and no requirement flag.
return frozenset({"name"})
@staticmethod
def native_sources() -> dict[str, str]:
return {ARCHIVE: SOURCE_URL}
@staticmethod
def needs_original() -> bool:
# The declaration is a sentence inside a shell script.
return True
@staticmethod
def may_write_nothing() -> bool:
# Nothing to correct is a result, not a failure.
return True
@staticmethod
def unpack(raw: bytes) -> dict[str, str]:
"""Keep the scriptmodules that say anything about BIOS files."""
found: dict[str, str] = {}
with tarfile.open(fileobj=io.BytesIO(raw), mode="r:gz") as archive:
for member in archive:
if not member.isfile() or not member.name.endswith(".sh"):
continue
relative = member.name.split("/", 1)[-1]
if not relative.startswith("scriptmodules/"):
continue
handle = archive.extractfile(member)
if handle is None:
continue
text = handle.read().decode("utf-8", errors="replace")
if "BIOS" in text:
found[relative] = text
return found
def _core_index(self) -> dict[str, str]:
"""Module id (without its lr- prefix) to the profile it stands for."""
index: dict[str, str] = {}
for key, profile in load_emulator_profiles(str(EMULATORS_DIR)).items():
index[key.replace("-", "_").lower()] = key
for name in profile.get("cores", []) or []:
index[str(name).replace("-", "_").lower()] = key
return index
@staticmethod
def _files_by_core(systems: dict[str, NativeSystem]) -> dict[str, list[NativeFile]]:
"""Which files each core asks for, as the truth read its source."""
grouped: dict[str, list[NativeFile]] = {}
for system in systems.values():
for fe in system.files:
for core in (fe.truth or {}).get("_cores", []):
grouped.setdefault(str(core), []).append(fe)
return grouped
@staticmethod
def _names_in(text: str) -> list[re.Match[str]]:
"""File names in a fragment, skipping what only looks like one.
A token preceded by a dot is the tail of a ROM extension list
(.atr.gz), and one preceded by a separator is part of a path
($biosdir/Machines/COL/coleco.rom): neither is a name in a list.
"""
found: list[re.Match[str]] = []
for match in _FILENAME.finditer(text):
before = text[match.start() - 1 : match.start()]
if before in (".", "/", "\\"):
continue
found.append(match)
return found
@classmethod
def _listed(cls, help_text: str) -> set[str]:
"""File names the help already writes, wherever in the string."""
plain = help_text.replace("\\n", " ")
return {match.group(0).lower() for match in cls._names_in(plain)}
@classmethod
def _insertion_point(cls, help_text: str) -> int | None:
"""Where a name joins the list, or None when there is no list."""
clause = _BIOS_CLAUSE.search(help_text)
if clause is None:
return None
names = cls._names_in(clause.group(0))
return clause.start() + names[-1].end() if names else None
@staticmethod
def _in_search_order(candidates: list[NativeFile]) -> list[str]:
"""The names in the order the code looks for them, best first.
Someone reading the sentence copies the files in the order it
gives, so the one the emulator prefers is named first. The model
already holds them in that order; this only removes the duplicates
a name can pick up from several cores.
"""
names: list[str] = []
for fe in search_order(candidates):
if fe.name not in names:
names.append(fe.name)
return names
@staticmethod
def _join(names: list[str]) -> str:
"""RetroPie's own idiom: a, b and c."""
if len(names) == 1:
return names[0]
return ", ".join(names[:-1]) + " and " + names[-1]
def modules(self, originals: dict[str, str]) -> dict[str, tuple[str, str]]:
"""Module id and help string of every script that mentions BIOS."""
found: dict[str, tuple[str, str]] = {}
for relative, text in originals.items():
if not relative.startswith("scriptmodules/"):
continue
module = _MODULE_ID.search(text)
help_text = _MODULE_HELP.search(text)
if not module or not help_text or "BIOS" not in help_text.group(1):
continue
found[relative] = (module.group(1), help_text.group(1))
return found
def render(
self,
systems: dict[str, NativeSystem],
report: Report,
originals: dict[str, str],
scraped: dict | None = None,
) -> dict[str, str]:
modules = self.modules(originals)
if not modules:
raise ValueError(
"the scriptmodules cannot be written without RetroPie's own "
"repository: the BIOS list is a sentence inside a shell script"
)
index = self._core_index()
by_core = self._files_by_core(systems)
# Aliases count as names we know: RetroPie writes dc_flash.bin where
# flycast's profile files it as an alias of another primary.
known: set[str] = set()
for system in systems.values():
for fe in system.files:
known.add(fe.name.lower())
known.update(
str(a).lower() for a in (fe.native("aliases", []) or [])
)
self._skipped: dict[str, list[str]] = {}
produced: dict[str, str] = {}
def skip(reason: str, module: str) -> None:
self._skipped.setdefault(reason, []).append(module)
for relative, (module_id, help_text) in sorted(modules.items()):
core = index.get(module_id.removeprefix("lr-").replace("-", "_").lower())
files = by_core.get(core or "", [])
if not files:
skip("no core of ours packages it", module_id)
continue
listed = self._listed(help_text)
unknown = sorted(listed - known)
if unknown:
skip(
f"names a file we have never seen ({', '.join(unknown)})",
module_id,
)
# A file already listed under one of its other names is not
# missing: proposing the primary would name the same bytes twice.
candidates = [
fe
for fe in files
if fe.required
and not (
{fe.name.lower()}
| {str(a).lower() for a in (fe.native("aliases", []) or [])}
)
& listed
]
missing = self._in_search_order(candidates)
if not missing:
continue
insert_at = self._insertion_point(help_text)
if insert_at is None:
# No enumeration to extend, and a sentence we would have to
# write ourselves is a documentation change, not a correction.
skip("names no file to extend", module_id)
continue
if len(listed) + len(missing) > MAX_NAMES:
skip("more names than the help enumerates", module_id)
continue
new_help = (
help_text[:insert_at]
+ ", "
+ self._join(missing)
+ help_text[insert_at:]
)
if len(listed) + len(missing) > 1:
new_help = _BIOS_NOUN.sub("BIOS files", new_help, count=1)
produced[relative] = originals[relative].replace(
f'rp_module_help="{help_text}"',
f'rp_module_help="{new_help}"',
1,
)
return produced
def outcome(
self,
systems: dict[str, NativeSystem],
produced: dict[str, str],
) -> str:
"""Packages are the unit here, not file entries."""
skipped: dict[str, list[str]] = getattr(self, "_skipped", {})
parts = [f"{len(produced)} packages corrected"]
for reason, modules in sorted(skipped.items()):
shown = ", ".join(sorted(modules)[:4])
if len(modules) > 4:
shown += f" and {len(modules) - 4} more"
parts.append(f"{len(modules)} {reason} ({shown})")
return "; ".join(parts)
def validate(
self,
systems: dict[str, NativeSystem],
produced: dict[str, str],
) -> list[str]:
issues: list[str] = []
for relative, text in produced.items():
module = _MODULE_ID.search(text)
help_text = _MODULE_HELP.search(text)
if not module:
issues.append(f"{relative}: the module lost its id")
if not help_text:
issues.append(f"{relative}: the help string is not closed")
continue
if "BIOS" not in help_text.group(1):
issues.append(f"{relative}: the BIOS sentence is gone")
names = self._listed(help_text.group(1))
if len(names) > MAX_NAMES:
issues.append(
f"{relative}: {len(names)} names, past what the help enumerates"
)
return issues
+30
View File
@@ -0,0 +1,30 @@
"""Exporter for ROCKNIX's rocknix-systems.
Same shape as Batocera's script: a systems mapping inside a checker ROCKNIX
runs, so only the mapping is rewritten.
"""
from __future__ import annotations
from .batocera_exporter import Exporter as BatoceraExporter
SOURCE_URL = (
"https://raw.githubusercontent.com/ROCKNIX/distribution/next/projects/ROCKNIX"
"/packages/rocknix/sources/scripts/rocknix-systems"
)
class Exporter(BatoceraExporter):
"""Write ROCKNIX's rocknix-systems, corrected."""
@staticmethod
def platform_name() -> str:
return "rocknix"
@staticmethod
def native_filename() -> str:
return "rocknix-systems"
@staticmethod
def native_sources() -> dict[str, str]:
return {"rocknix-systems": SOURCE_URL}
+105 -132
View File
@@ -1,160 +1,133 @@
"""Exporter for RomM known_bios_files.json format.
"""Exporter for RomM's known_bios_files.json.
Produces JSON matching the exact format of
rommapp/romm/backend/models/fixtures/known_bios_files.json:
- Keys are "igdb_slug:filename"
- Values contain size, crc, md5, sha1 (all optional but at least one hash)
- Hashes are lowercase hex strings
- Size is an integer
Keys are "<igdb slug>:<filename>". RomM verifies a firmware file with
`file_size_bytes == int(entry.get("size", 0))` and then one hash among md5,
sha1 and crc, so an entry without a size can never match and an entry
without a hash can never match either. Neither is written.
"""
from __future__ import annotations
import json
import sys
from collections import OrderedDict
from pathlib import Path
from .base_exporter import BaseExporter
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
# retrobios slug -> IGDB slug (reverse of scraper SLUG_MAP)
_REVERSE_SLUG: dict[str, str] = {
"3do": "3do",
"nintendo-64dd": "64dd",
"amstrad-cpc": "acpc",
"commodore-amiga": "amiga",
"arcade": "arcade",
"atari-st": "atari-st",
"atari-5200": "atari5200",
"atari-7800": "atari7800",
"atari-400-800": "atari8bit",
"coleco-colecovision": "colecovision",
"sega-dreamcast": "dc",
"doom": "doom",
"enterprise-64-128": "enterprise",
"fairchild-channel-f": "fairchild-channel-f",
"nintendo-fds": "fds",
"sega-game-gear": "gamegear",
"nintendo-gb": "gb",
"nintendo-gba": "gba",
"nintendo-gbc": "gbc",
"sega-mega-drive": "genesis",
"mattel-intellivision": "intellivision",
"j2me": "j2me",
"atari-lynx": "lynx",
"apple-macintosh-ii": "mac",
"microsoft-msx": "msx",
"nintendo-ds": "nds",
"snk-neogeo-cd": "neo-geo-cd",
"nintendo-nes": "nes",
"nintendo-gamecube": "ngc",
"magnavox-odyssey2": "odyssey-2-slash-videopac-g7000",
"nec-pc-98": "pc-9800-series",
"nec-pc-fx": "pc-fx",
"nintendo-pokemon-mini": "pokemon-mini",
"sony-playstation-2": "ps2",
"sony-psp": "psp",
"sony-playstation": "psx",
"nintendo-satellaview": "satellaview",
"sega-saturn": "saturn",
"scummvm": "scummvm",
"sega-mega-cd": "segacd",
"sharp-x68000": "sharp-x68000",
"sega-master-system": "sms",
"nintendo-snes": "snes",
"nintendo-sufami-turbo": "sufami-turbo",
"nintendo-sgb": "super-gb",
"nec-pc-engine": "tg16",
"videoton-tvc": "tvc",
"philips-videopac": "videopac-g7400",
"wolfenstein-3d": "wolfenstein",
"sharp-x1": "x1",
"microsoft-xbox": "xbox",
"sinclair-zx-spectrum": "zxs",
}
from scraper.romm_scraper import SLUG_MAP
from .base_exporter import BaseExporter
from .baseline import NativeFile, NativeSystem, Report
SOURCE_URL = (
"https://raw.githubusercontent.com/rommapp/romm/master/backend/models"
"/fixtures/known_bios_files.json"
)
class Exporter(BaseExporter):
"""Export truth data to RomM known_bios_files.json format."""
"""Write RomM's known_bios_files.json, corrected."""
@staticmethod
def platform_name() -> str:
return "romm"
def export(
self,
truth_data: dict,
output_path: str,
scraped_data: dict | None = None,
) -> None:
native_map: dict[str, str] = {}
if scraped_data:
for sys_id, sys_data in scraped_data.get("systems", {}).items():
nid = sys_data.get("native_id")
if nid:
native_map[sys_id] = nid
@staticmethod
def native_filename() -> str:
return "known_bios_files.json"
output: OrderedDict[str, dict] = OrderedDict()
@staticmethod
def carries() -> frozenset[str]:
return frozenset({"size", "crc32", "md5", "sha1"})
systems = truth_data.get("systems", {})
for sys_id in sorted(systems):
sys_data = systems[sys_id]
files = sys_data.get("files", [])
if not files:
continue
@staticmethod
def native_sources() -> dict[str, str]:
return {"known_bios_files.json": SOURCE_URL}
igdb_slug = native_map.get(sys_id, _REVERSE_SLUG.get(sys_id, sys_id))
for fe in files:
name = fe.get("name", "")
if name.startswith("_") or self._is_pattern(name):
continue
key = f"{igdb_slug}:{name}"
entry: OrderedDict[str, object] = OrderedDict()
size = fe.get("size")
if size is not None:
entry["size"] = int(size)
crc = fe.get("crc32", "")
if crc:
entry["crc"] = str(crc).strip().lower()
md5 = fe.get("md5", "")
if isinstance(md5, list):
md5 = md5[0] if md5 else ""
if md5:
entry["md5"] = str(md5).strip().lower()
sha1 = fe.get("sha1", "")
if isinstance(sha1, list):
sha1 = sha1[0] if sha1 else ""
if sha1:
entry["sha1"] = str(sha1).strip().lower()
output[key] = entry
Path(output_path).write_text(
json.dumps(output, indent=2, ensure_ascii=False) + "\n",
encoding="utf-8",
@staticmethod
def _verifiable(fe: NativeFile) -> bool:
return bool(fe.size()) and any(
fe.hash(h) for h in ("md5", "sha1", "crc32")
)
def validate(self, truth_data: dict, output_path: str) -> list[str]:
data = json.loads(Path(output_path).read_text(encoding="utf-8"))
@staticmethod
def _known_platform(native_id: str) -> bool:
"""RomM keys by IGDB platform slug, and only looks up its own.
exported_names: set[str] = set()
for key in data:
if ":" in key:
_, filename = key.split(":", 1)
exported_names.add(filename)
A key spelled with one of our slugs (capcom-cps3, snk-neogeo-mvs)
matches nothing on their side, so it is reported rather than
written.
"""
return native_id in SLUG_MAP
@classmethod
def writable(cls, fe: NativeFile, require: str = "") -> bool:
"""What RomM already ships stays; the conditions gate additions.
An entry of theirs that could never verify is still theirs, and the
round trip is not the place to decide otherwise.
"""
if fe.platform is not None:
return True
return cls._verifiable(fe) and cls._known_platform(fe.native_system)
def render(
self,
systems: dict[str, NativeSystem],
report: Report,
originals: dict[str, str],
scraped: dict | None = None,
) -> dict[str, str]:
output: OrderedDict[str, dict] = OrderedDict()
for system in sorted(systems.values(), key=lambda s: s.native_id):
for fe in sorted(system.files, key=lambda f: f.name):
if not self.writable(fe):
continue
entry: OrderedDict[str, str] = OrderedDict()
# The fixture states every value as a string, size included.
entry["size"] = str(fe.size())
crc = fe.hash("crc32")
if crc:
entry["crc"] = crc
md5 = fe.hash("md5")
if md5:
entry["md5"] = md5
sha1 = fe.hash("sha1")
if sha1:
entry["sha1"] = sha1
output[f"{system.native_id}:{fe.name}"] = entry
text = json.dumps(output, indent=2, ensure_ascii=False) + "\n"
return {self.native_filename(): text}
def validate(
self,
systems: dict[str, NativeSystem],
produced: dict[str, str],
) -> list[str]:
try:
data = json.loads(produced[self.native_filename()])
except json.JSONDecodeError as exc:
return [f"the JSON does not parse: {exc}"]
issues: list[str] = []
for sys_data in truth_data.get("systems", {}).values():
for fe in sys_data.get("files", []):
name = fe.get("name", "")
if name.startswith("_") or self._is_pattern(name):
for key, entry in data.items():
slug = key.split(":", 1)[0] if ":" in key else ""
if not slug:
issues.append(f"key without a platform slug: {key}")
elif not self._known_platform(slug):
issues.append(f"platform slug RomM does not know: {slug}")
if not entry.get("size"):
issues.append(f"entry without a size, never verifiable: {key}")
if not any(entry.get(h) for h in ("md5", "sha1", "crc")):
issues.append(f"entry without a hash, never verifiable: {key}")
for system in systems.values():
for fe in system.files:
if not self.writable(fe):
continue
if name not in exported_names:
issues.append(f"missing: {name}")
if f"{system.native_id}:{fe.name}" not in data:
issues.append(f"absent: {system.native_id}/{fe.name}")
return issues
+106 -75
View File
@@ -1,7 +1,7 @@
"""Exporter for libretro System.dat (clrmamepro DAT format).
Produces a single 'game' block with all ROMs grouped by system,
matching the exact format of libretro-database/dat/System.dat.
One 'game' block, systems separated by a comment line carrying the name
libretro gives them, matching libretro-database/dat/System.dat.
"""
from __future__ import annotations
@@ -14,43 +14,71 @@ sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from scraper.dat_parser import parse_dat
from .base_exporter import BaseExporter
from .baseline import NativeSystem, Report
SOURCE_URL = (
"https://raw.githubusercontent.com/libretro/libretro-database/master/dat/System.dat"
)
def _slug_to_native(slug: str) -> str:
"""Convert a system slug to 'Manufacturer - Console' format."""
parts = slug.split("-", 1)
if len(parts) == 1:
return parts[0].title()
manufacturer = parts[0].replace("-", " ").title()
console = parts[1].replace("-", " ").title()
return f"{manufacturer} - {console}"
def _quote(name: str) -> str:
"""Quote a ROM name the way the original does: only when it must be."""
return f'"{name}"' if any(c in name for c in ' ()') else name
class Exporter(BaseExporter):
"""Export truth data to libretro System.dat format."""
"""Write libretro's System.dat, corrected."""
@staticmethod
def platform_name() -> str:
return "retroarch"
def export(
self,
truth_data: dict,
output_path: str,
scraped_data: dict | None = None,
) -> None:
native_map: dict[str, str] = {}
if scraped_data:
for sys_id, sys_data in scraped_data.get("systems", {}).items():
nid = sys_data.get("native_id")
if nid:
native_map[sys_id] = nid
@staticmethod
def native_filename() -> str:
return "System.dat"
@staticmethod
def carries() -> frozenset[str]:
return frozenset({"size", "crc32", "md5", "sha1"})
@staticmethod
def native_sources() -> dict[str, str]:
return {"System.dat": SOURCE_URL}
@classmethod
def writable(cls, fe, require: str = "") -> bool:
"""A clrmamepro rom line is a hash record, and the DAT has no other.
The original never states a rom it cannot hash, so neither do we.
"""
if fe.platform is not None:
return True
return any(fe.hash(h) for h in ("crc32", "md5", "sha1"))
@staticmethod
def _rom_name(fe) -> str:
"""The name the DAT gives a rom.
libretro writes the path for some entries (ep128emu/roms/cpc464.rom)
and the bare name for others whose destination has a directory all
the same (iplromco.dat, which lives under keropi/), so its own
spelling is recorded rather than derived.
"""
declared = fe.native("native_path", "")
return str(declared) if declared else fe.name
def _header(self, originals: dict[str, str], scraped: dict | None) -> list[str]:
"""Reuse the original header verbatim when we have the original."""
original = originals.get(self.native_filename(), "")
if original:
head, sep, _ = original.partition("\ngame (")
if sep:
return head.split("\n")
# Match exact header format of libretro-database/dat/System.dat
version = ""
if scraped_data:
version = scraped_data.get("dat_version", scraped_data.get("version", ""))
lines: list[str] = [
if scraped:
version = scraped.get("dat_version", scraped.get("version", ""))
lines = [
"clrmamepro (",
'\tname "System"',
'\tdescription "System"',
@@ -61,73 +89,76 @@ class Exporter(BaseExporter):
lines.extend(
[
'\tauthor "libretro"',
'\thomepage "https://github.com/libretro/libretro-database/blob/master/dat/System.dat"',
'\turl "https://raw.githubusercontent.com/libretro/libretro-database/master/dat/System.dat"',
'\thomepage "https://github.com/libretro/libretro-database/blob/master'
'/dat/System.dat"',
'\turl "https://raw.githubusercontent.com/libretro/libretro-database'
'/master/dat/System.dat"',
")",
"",
"game (",
'\tname "System"',
'\tcomment "System"',
]
)
return lines
systems = truth_data.get("systems", {})
for sys_id in sorted(systems):
sys_data = systems[sys_id]
files = sys_data.get("files", [])
if not files:
continue
native_name = native_map.get(sys_id, _slug_to_native(sys_id))
lines.append("")
lines.append(f'\tcomment "{native_name}"')
def render(
self,
systems: dict[str, NativeSystem],
report: Report,
originals: dict[str, str],
scraped: dict | None = None,
) -> dict[str, str]:
lines = self._header(originals, scraped)
lines.extend(["game (", '\tname "System"', '\tcomment "System"'])
for system, files in sorted(
self.exportable(systems), key=lambda pair: pair[0].native_id
):
rendered: list[str] = []
for fe in files:
name = fe.get("name", "")
if name.startswith("_") or self._is_pattern(name):
if not any(fe.hash(h) for h in ("crc32", "md5", "sha1")):
continue
# Quote names with spaces or special chars (matching original format)
needs_quote = " " in name or "(" in name or ")" in name
name_str = f'"{name}"' if needs_quote else name
rom_parts = [f"name {name_str}"]
size = fe.get("size")
parts = [f"name {_quote(self._rom_name(fe))}"]
size = fe.size()
if size:
rom_parts.append(f"size {size}")
crc = fe.get("crc32", "")
parts.append(f"size {size}")
crc = fe.hash("crc32")
if crc:
rom_parts.append(f"crc {str(crc).upper()}")
md5 = fe.get("md5", "")
if isinstance(md5, list):
md5 = md5[0] if md5 else ""
parts.append(f"crc {crc.upper()}")
md5 = fe.hash("md5")
if md5:
rom_parts.append(f"md5 {md5}")
sha1 = fe.get("sha1", "")
if isinstance(sha1, list):
sha1 = sha1[0] if sha1 else ""
parts.append(f"md5 {md5}")
sha1 = fe.hash("sha1")
if sha1:
rom_parts.append(f"sha1 {sha1}")
parts.append(f"sha1 {sha1}")
rendered.append(f"\trom ( {' '.join(parts)} )")
lines.append(f"\trom ( {' '.join(rom_parts)} )")
if not rendered:
continue
lines.append("")
# libretro's comment is the system name as the DAT spells it,
# "Atari - 400-800". Prettifying it drops the separator.
lines.append(f'\tcomment "{system.native_id}"')
lines.extend(rendered)
lines.append(")")
lines.append("")
Path(output_path).write_text("\n".join(lines), encoding="utf-8")
return {self.native_filename(): "\n".join(lines)}
def validate(self, truth_data: dict, output_path: str) -> list[str]:
content = Path(output_path).read_text(encoding="utf-8")
def validate(
self,
systems: dict[str, NativeSystem],
produced: dict[str, str],
) -> list[str]:
content = produced[self.native_filename()]
parsed = parse_dat(content)
exported_names: set[str] = set()
for rom in parsed:
exported_names.add(rom.name)
exported = {rom.name for rom in parsed}
issues: list[str] = []
for sys_data in truth_data.get("systems", {}).values():
for fe in sys_data.get("files", []):
name = fe.get("name", "")
if name.startswith("_") or self._is_pattern(name):
for system in systems.values():
for fe in system.files:
if not any(fe.hash(h) for h in ("crc32", "md5", "sha1")):
continue
if name not in exported_names:
issues.append(f"missing: {name}")
if self._rom_name(fe) not in exported:
issues.append(f"absent from the DAT: {system.native_id}/{fe.name}")
if not content.rstrip().endswith(")"):
issues.append("the game block is not closed")
return issues