mirror of
https://github.com/Abdess/retroarch_system.git
synced 2026-10-11 14:03:23 -05:00
feat: make the native export round-trip a platform file
This commit is contained in:
1 parent
7b93285e1d
commit
691ccbfca7
65 files changed
+18417
-2115
No files matched your search
@@ -1,81 +1,166 @@
|
||||
"""Abstract base class for platform exporters."""
|
||||
"""Contract shared by the platform exporters.
|
||||
|
||||
An exporter answers one question: what would this platform's own file look
|
||||
like if it were corrected. It is handed the platform's file when we have it,
|
||||
because several of these formats carry code, and a generator that emits only
|
||||
the data block hands the maintainer something that no longer runs.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from abc import ABC, abstractmethod
|
||||
|
||||
from .baseline import NativeFile, NativeSystem, Report
|
||||
|
||||
|
||||
class BaseExporter(ABC):
|
||||
"""Base class for exporting truth data to native platform formats."""
|
||||
"""Base class for writing a platform's own BIOS file, corrected."""
|
||||
|
||||
@staticmethod
|
||||
@abstractmethod
|
||||
def platform_name() -> str:
|
||||
"""Return the platform identifier this exporter targets."""
|
||||
|
||||
@staticmethod
|
||||
@abstractmethod
|
||||
def export(
|
||||
def native_filename() -> str:
|
||||
"""Return the name the platform gives this file."""
|
||||
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
"""Fields this format has somewhere to put.
|
||||
|
||||
A correction the file cannot state is not a correction it delivers,
|
||||
and counting it would announce a change the maintainer will not find
|
||||
in the diff.
|
||||
"""
|
||||
return frozenset({"md5"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
"""Return {relative path: URL} of the originals this exporter patches.
|
||||
|
||||
An exporter that can build its file from nothing returns an empty
|
||||
mapping. One whose format carries code cannot, and names the file it
|
||||
needs so the caller can fetch it.
|
||||
"""
|
||||
return {}
|
||||
|
||||
@staticmethod
|
||||
def needs_original() -> bool:
|
||||
"""Whether a faithful export requires the platform's own file."""
|
||||
return False
|
||||
|
||||
@staticmethod
|
||||
def may_write_nothing() -> bool:
|
||||
"""Whether producing no file is an outcome rather than a failure.
|
||||
|
||||
True only where the unit is a correction rather than a document:
|
||||
RetroPie has no BIOS list to rewrite, so a run with nothing to
|
||||
correct writes nothing and is right to.
|
||||
"""
|
||||
return False
|
||||
|
||||
@abstractmethod
|
||||
def render(
|
||||
self,
|
||||
truth_data: dict,
|
||||
output_path: str,
|
||||
scraped_data: dict | None = None,
|
||||
) -> None:
|
||||
"""Export truth data to the native platform format."""
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
"""Return {relative output path: file content}."""
|
||||
|
||||
@abstractmethod
|
||||
def validate(self, truth_data: dict, output_path: str) -> list[str]:
|
||||
"""Validate exported file against truth data, return list of issues."""
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
"""Check the produced files against what was asked of them."""
|
||||
|
||||
# Shared helpers
|
||||
|
||||
@classmethod
|
||||
def exportable(
|
||||
cls,
|
||||
systems: dict[str, NativeSystem],
|
||||
require: str = "",
|
||||
) -> list[tuple[NativeSystem, list[NativeFile]]]:
|
||||
"""Systems paired with the files this format can actually carry.
|
||||
|
||||
`require` names a hash the format cannot omit. An entry the format
|
||||
has no way to express is dropped here and counted by the caller,
|
||||
never written as a half entry the platform would read as a file that
|
||||
can never verify.
|
||||
"""
|
||||
result: list[tuple[NativeSystem, list[NativeFile]]] = []
|
||||
for system in systems.values():
|
||||
files = [fe for fe in system.files if fe.name and cls.writable(fe, require)]
|
||||
if files:
|
||||
result.append((system, files))
|
||||
return result
|
||||
|
||||
@staticmethod
|
||||
def _is_pattern(name: str) -> bool:
|
||||
"""Check if a filename is a placeholder pattern (not a real file)."""
|
||||
return "<" in name or ">" in name or "*" in name
|
||||
def requires() -> str:
|
||||
"""The hash this format cannot write a new entry without."""
|
||||
return ""
|
||||
|
||||
@staticmethod
|
||||
def _dest(fe: dict) -> str:
|
||||
"""Get destination path for a file entry, falling back to name."""
|
||||
return fe.get("path") or fe.get("destination") or fe.get("name", "")
|
||||
def can_add() -> bool:
|
||||
"""Whether a file the platform does not declare can be stated at all.
|
||||
|
||||
False where an entry is a declaration in code rather than a row:
|
||||
BizHawk wires each firmware into an option list and a status that
|
||||
only its source expresses, and writing C# for one is not something
|
||||
an exporter can do safely.
|
||||
"""
|
||||
return True
|
||||
|
||||
@classmethod
|
||||
def writable(cls, fe: NativeFile, require: str = "") -> bool:
|
||||
"""Whether this format can carry the entry.
|
||||
|
||||
An entry the platform already declares is always carried, hash or
|
||||
no hash: it is in their file today, and an export that drops it
|
||||
hands back a file poorer than the one it corrects. The requirement
|
||||
only gates what we would be adding.
|
||||
"""
|
||||
if fe.platform is not None:
|
||||
return True
|
||||
if not cls.can_add():
|
||||
return False
|
||||
require = require or cls.requires()
|
||||
return not require or bool(fe.hash(require))
|
||||
|
||||
def outcome(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> str | None:
|
||||
"""What this export did, when counting file entries would not say it.
|
||||
|
||||
Most formats state one entry per file, so the caller's own count is
|
||||
the answer. RetroPie states a sentence per package, and a count of
|
||||
file entries would describe work it did not do.
|
||||
"""
|
||||
return None
|
||||
|
||||
@staticmethod
|
||||
def _display_name(
|
||||
sys_id: str,
|
||||
scraped_sys: dict | None = None,
|
||||
) -> str:
|
||||
"""Get display name for a system from scraped data or slug."""
|
||||
if scraped_sys:
|
||||
name = scraped_sys.get("name")
|
||||
if name:
|
||||
return name
|
||||
# Fallback: convert slug to display name with acronym handling
|
||||
def display_name(system: NativeSystem) -> str:
|
||||
"""The name the platform shows for a system."""
|
||||
if system.name:
|
||||
return system.name
|
||||
for fe in system.files:
|
||||
native = fe.native("native_name", "")
|
||||
if native:
|
||||
return str(native)
|
||||
_UPPER = {
|
||||
"3do",
|
||||
"cdi",
|
||||
"cpc",
|
||||
"cps1",
|
||||
"cps2",
|
||||
"cps3",
|
||||
"dos",
|
||||
"gba",
|
||||
"gbc",
|
||||
"hle",
|
||||
"msx",
|
||||
"nes",
|
||||
"nds",
|
||||
"ngp",
|
||||
"psp",
|
||||
"psx",
|
||||
"sms",
|
||||
"snes",
|
||||
"stv",
|
||||
"tvc",
|
||||
"vb",
|
||||
"zx",
|
||||
"3do", "cdi", "cpc", "cps1", "cps2", "cps3", "dos", "gba", "gbc",
|
||||
"hle", "msx", "nes", "nds", "ngp", "psp", "psx", "sms", "snes",
|
||||
"stv", "tvc", "vb", "zx",
|
||||
}
|
||||
parts = sys_id.replace("-", " ").split()
|
||||
result = []
|
||||
for p in parts:
|
||||
if p.lower() in _UPPER:
|
||||
result.append(p.upper())
|
||||
else:
|
||||
result.append(p.capitalize())
|
||||
return " ".join(result)
|
||||
parts = system.native_id.replace("-", " ").replace("_", " ").split()
|
||||
return " ".join(
|
||||
p.upper() if p.lower() in _UPPER else p.capitalize() for p in parts
|
||||
)
|
||||
@@ -0,0 +1,373 @@
|
||||
"""Reconciliation of a platform's own declarations with the ground truth.
|
||||
|
||||
An export is not the truth rendered in a native syntax. It is the platform's
|
||||
file, corrected: what the truth can prove is applied, what it says nothing
|
||||
about is left alone, and what it knows and the platform lacks is added. A
|
||||
platform that loses two thirds of its systems to an export cannot use it.
|
||||
|
||||
Systems are keyed by the identifier the platform itself uses. Several of our
|
||||
slugs collapse onto one native id (Recalbox files pcengine, pcenginecd and
|
||||
supergrafx under one slug) and one slug can carry several native ids, so the
|
||||
grouping is rebuilt from the per-file native_system the scrapers record.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from dataclasses import dataclass, field
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from common import _norm_system_id
|
||||
|
||||
HASH_FIELDS = ("sha1", "md5", "sha256", "crc32")
|
||||
|
||||
|
||||
def _hash_values(entry: dict, field_name: str) -> list[str]:
|
||||
"""Return a hash field as a list, whatever shape it was written in."""
|
||||
raw = entry.get(field_name)
|
||||
if not raw:
|
||||
return []
|
||||
if isinstance(raw, list):
|
||||
return [str(v).strip().lower() for v in raw if str(v).strip()]
|
||||
return [v.strip().lower() for v in str(raw).split(",") if v.strip()]
|
||||
|
||||
|
||||
def _is_placeholder(name: str) -> bool:
|
||||
return "<" in name or ">" in name or "*" in name
|
||||
|
||||
|
||||
@dataclass
|
||||
class NativeFile:
|
||||
"""One file as the platform will read it, after correction."""
|
||||
|
||||
name: str
|
||||
destination: str
|
||||
native_system: str
|
||||
platform: dict | None = None
|
||||
truth: dict | None = None
|
||||
corrections: list[str] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def origin(self) -> str:
|
||||
if self.platform is not None and self.truth is not None:
|
||||
return "both"
|
||||
return "platform" if self.platform is not None else "truth"
|
||||
|
||||
def hashes(self, field_name: str) -> list[str]:
|
||||
"""Accepted values for a hash, truth first when it has an opinion.
|
||||
|
||||
The truth is read from the emulator's source; the platform list is a
|
||||
secondary source. When both speak and disagree, the truth decides and
|
||||
the divergence is recorded, never silently merged: an emulator that
|
||||
rejects a file will reject it whatever the platform declares.
|
||||
"""
|
||||
truth_values = _hash_values(self.truth or {}, field_name)
|
||||
platform_values = _hash_values(self.platform or {}, field_name)
|
||||
if truth_values and platform_values and not set(truth_values) & set(
|
||||
platform_values
|
||||
):
|
||||
return truth_values
|
||||
if truth_values:
|
||||
# Keep the platform's extra accepted revisions alongside ours.
|
||||
merged = list(truth_values)
|
||||
merged.extend(v for v in platform_values if v not in merged)
|
||||
return merged
|
||||
return platform_values
|
||||
|
||||
def hash(self, field_name: str) -> str:
|
||||
values = self.hashes(field_name)
|
||||
return values[0] if values else ""
|
||||
|
||||
@property
|
||||
def required(self) -> bool:
|
||||
if self.truth is not None and self.truth.get("required") is not None:
|
||||
return bool(self.truth["required"])
|
||||
if self.platform is not None and self.platform.get("required") is not None:
|
||||
return bool(self.platform["required"])
|
||||
return True
|
||||
|
||||
@property
|
||||
def priority(self) -> int | None:
|
||||
"""Where the code looks for this file, lowest first.
|
||||
|
||||
DuckStation's FindBIOSImageInDirectory keeps the image whose
|
||||
priority is lower, and a search list a core walks in order maps
|
||||
onto it as 1, 2, 3. Where cores disagree this is the best rank any
|
||||
of them gives: nothing is dropped on it, so a file that is some
|
||||
core's first choice is read early. None when nothing states one.
|
||||
"""
|
||||
for entry in (self.truth, self.platform):
|
||||
if entry and entry.get("priority") is not None:
|
||||
return int(entry["priority"])
|
||||
return None
|
||||
|
||||
def size(self) -> int | None:
|
||||
for entry in (self.truth, self.platform):
|
||||
if entry and entry.get("size"):
|
||||
return int(entry["size"])
|
||||
return None
|
||||
|
||||
def native(self, key: str, default: object = "") -> object:
|
||||
"""Read a field the platform declares and we only carry through."""
|
||||
for entry in (self.platform, self.truth):
|
||||
if entry and entry.get(key) not in (None, ""):
|
||||
return entry[key]
|
||||
return default
|
||||
|
||||
def cores(self) -> list[str]:
|
||||
"""Cores that want this file, the platform's naming preferred."""
|
||||
declared = self.native("core", "")
|
||||
if declared:
|
||||
return [c.strip() for c in str(declared).split(",") if c.strip()]
|
||||
if self.truth:
|
||||
return [f"libretro/{c}" for c in self.truth.get("_cores", [])]
|
||||
return []
|
||||
|
||||
|
||||
@dataclass
|
||||
class NativeSystem:
|
||||
"""One system as the platform names it."""
|
||||
|
||||
native_id: str
|
||||
name: str = ""
|
||||
files: list[NativeFile] = field(default_factory=list)
|
||||
from_platform: bool = False
|
||||
|
||||
@property
|
||||
def origin(self) -> str:
|
||||
return "platform" if self.from_platform else "truth"
|
||||
|
||||
|
||||
@dataclass
|
||||
class Report:
|
||||
"""What the reconciliation changed, so the caller can say it out loud."""
|
||||
|
||||
systems_kept: int = 0
|
||||
systems_added: int = 0
|
||||
files_kept: int = 0
|
||||
files_added: int = 0
|
||||
hashes_corrected: list[str] = field(default_factory=list)
|
||||
required_corrected: list[str] = field(default_factory=list)
|
||||
|
||||
@property
|
||||
def corrections(self) -> int:
|
||||
return len(self.hashes_corrected) + len(self.required_corrected)
|
||||
|
||||
|
||||
def _native_id_of(sys_key: str, sys_data: dict, file_entry: dict) -> str:
|
||||
"""The platform's own id for the system a file belongs to."""
|
||||
return (
|
||||
file_entry.get("native_system")
|
||||
or sys_data.get("native_id")
|
||||
or sys_key
|
||||
)
|
||||
|
||||
|
||||
def _match_key(entry: dict) -> tuple[str, str]:
|
||||
dest = str(entry.get("destination") or entry.get("path") or entry.get("name", ""))
|
||||
return dest.casefold(), str(entry.get("name", "")).casefold()
|
||||
|
||||
|
||||
def build_native_model(
|
||||
truth: dict,
|
||||
scraped: dict | None,
|
||||
) -> tuple[dict[str, NativeSystem], Report]:
|
||||
"""Rebuild the platform's systems, corrected by the truth.
|
||||
|
||||
Returns the systems keyed by native id, in the platform's own order
|
||||
first and truth-only additions after, plus what changed.
|
||||
"""
|
||||
report = Report()
|
||||
systems: dict[str, NativeSystem] = {}
|
||||
|
||||
scraped_systems = (scraped or {}).get("systems", {})
|
||||
|
||||
# Pass 1: the platform's own file, grouped as the platform groups it.
|
||||
for sys_key, sys_data in scraped_systems.items():
|
||||
for file_entry in sys_data.get("files", []):
|
||||
native_id = _native_id_of(sys_key, sys_data, file_entry)
|
||||
system = systems.get(native_id)
|
||||
if system is None:
|
||||
system = NativeSystem(
|
||||
native_id=native_id,
|
||||
name=str(file_entry.get("native_name") or sys_data.get("name", "")),
|
||||
from_platform=True,
|
||||
)
|
||||
systems[native_id] = system
|
||||
report.systems_kept += 1
|
||||
name = str(file_entry.get("name", ""))
|
||||
destination = str(file_entry.get("destination") or name)
|
||||
system.files.append(
|
||||
NativeFile(
|
||||
name=name,
|
||||
destination=destination,
|
||||
native_system=native_id,
|
||||
platform=file_entry,
|
||||
)
|
||||
)
|
||||
report.files_kept += 1
|
||||
|
||||
# Which native ids a truth system may contribute to.
|
||||
norm_to_scraped: dict[str, str] = {
|
||||
_norm_system_id(key): key for key in scraped_systems
|
||||
}
|
||||
native_by_norm: dict[str, str] = {}
|
||||
for system in systems.values():
|
||||
native_by_norm.setdefault(_norm_system_id(system.native_id), system.native_id)
|
||||
|
||||
def _target_native_ids(truth_sid: str) -> list[str]:
|
||||
scraped_key = truth_sid if truth_sid in scraped_systems else None
|
||||
if scraped_key is None:
|
||||
scraped_key = norm_to_scraped.get(_norm_system_id(truth_sid))
|
||||
if scraped_key is not None:
|
||||
sys_data = scraped_systems[scraped_key]
|
||||
ids = {
|
||||
_native_id_of(scraped_key, sys_data, fe)
|
||||
for fe in sys_data.get("files", [])
|
||||
}
|
||||
if not ids:
|
||||
return [sys_data.get("native_id") or scraped_key]
|
||||
# A file the platform does not declare joins the system's primary
|
||||
# id, not whichever of its native ids sorts first: Recalbox files
|
||||
# pcengine, pcenginecd and supergrafx under one slug, and an
|
||||
# addition belongs to the one the system is named for.
|
||||
primary = sys_data.get("native_id")
|
||||
ordered = sorted(ids)
|
||||
if primary in ids:
|
||||
ordered.remove(primary)
|
||||
ordered.insert(0, primary)
|
||||
return ordered
|
||||
direct = native_by_norm.get(_norm_system_id(truth_sid))
|
||||
return [direct] if direct else [truth_sid]
|
||||
|
||||
# Pass 2: apply the truth onto that grouping.
|
||||
for truth_sid in sorted(truth.get("systems", {})):
|
||||
truth_sys = truth["systems"][truth_sid]
|
||||
truth_files = truth_sys.get("files", [])
|
||||
if not truth_files:
|
||||
continue
|
||||
|
||||
target_ids = _target_native_ids(truth_sid)
|
||||
|
||||
for truth_entry in truth_files:
|
||||
name = str(truth_entry.get("name", ""))
|
||||
if not name or name.startswith("_") or _is_placeholder(name):
|
||||
continue
|
||||
|
||||
t_dest, t_name = _match_key(truth_entry)
|
||||
t_hashes = {
|
||||
value
|
||||
for field_name in HASH_FIELDS
|
||||
for value in _hash_values(truth_entry, field_name)
|
||||
}
|
||||
|
||||
candidates = [
|
||||
candidate
|
||||
for native_id in target_ids
|
||||
for candidate in systems.get(
|
||||
native_id, NativeSystem(native_id)
|
||||
).files
|
||||
if candidate.truth is None
|
||||
]
|
||||
|
||||
def by_destination(candidate: NativeFile) -> bool:
|
||||
theirs = _match_key(candidate.platform or {})[0]
|
||||
return bool(t_dest) and theirs == t_dest
|
||||
|
||||
def by_name(candidate: NativeFile) -> bool:
|
||||
theirs = _match_key(candidate.platform or {})[1]
|
||||
return bool(t_name) and theirs == t_name
|
||||
|
||||
def by_hash(candidate: NativeFile) -> bool:
|
||||
if not t_hashes:
|
||||
return False
|
||||
theirs = {
|
||||
value
|
||||
for field_name in HASH_FIELDS
|
||||
for value in _hash_values(candidate.platform or {}, field_name)
|
||||
}
|
||||
return bool(t_hashes & theirs)
|
||||
|
||||
# Tried in order across every candidate, not per candidate: with
|
||||
# three IPL.bin under one system, separated only by their path, a
|
||||
# first-match-wins scan would attach the truth to whichever came
|
||||
# first and correct the wrong region's file.
|
||||
matched: NativeFile | None = None
|
||||
for test in (by_destination, by_name, by_hash):
|
||||
matched = next((c for c in candidates if test(c)), None)
|
||||
if matched is not None:
|
||||
break
|
||||
|
||||
if matched is not None:
|
||||
matched.truth = truth_entry
|
||||
for field_name in HASH_FIELDS:
|
||||
ours = set(_hash_values(truth_entry, field_name))
|
||||
theirs = set(_hash_values(matched.platform or {}, field_name))
|
||||
if ours and theirs and not ours & theirs:
|
||||
matched.corrections.append(field_name)
|
||||
report.hashes_corrected.append(
|
||||
f"{matched.native_system}/{matched.name} {field_name}"
|
||||
)
|
||||
t_req = truth_entry.get("required")
|
||||
p_req = (matched.platform or {}).get("required")
|
||||
if (
|
||||
t_req is not None
|
||||
and p_req is not None
|
||||
and bool(t_req) != bool(p_req)
|
||||
):
|
||||
matched.corrections.append("required")
|
||||
report.required_corrected.append(
|
||||
f"{matched.native_system}/{matched.name}"
|
||||
)
|
||||
continue
|
||||
|
||||
# The truth knows a file the platform does not declare.
|
||||
native_id = target_ids[0]
|
||||
system = systems.get(native_id)
|
||||
if system is None:
|
||||
system = NativeSystem(native_id=native_id, from_platform=False)
|
||||
systems[native_id] = system
|
||||
report.systems_added += 1
|
||||
destination = str(
|
||||
truth_entry.get("path") or truth_entry.get("destination") or name
|
||||
)
|
||||
system.files.append(
|
||||
NativeFile(
|
||||
name=name,
|
||||
destination=destination,
|
||||
native_system=native_id,
|
||||
truth=truth_entry,
|
||||
)
|
||||
)
|
||||
report.files_added += 1
|
||||
|
||||
# A reader takes the files in the order the list gives, so the one the
|
||||
# code looks for first is named first. Only what we add is ordered: what
|
||||
# the platform already wrote keeps the place the platform gave it.
|
||||
for system in systems.values():
|
||||
head = [fe for fe in system.files if fe.platform is not None]
|
||||
tail = [fe for fe in system.files if fe.platform is None]
|
||||
system.files = head + search_order(tail)
|
||||
|
||||
return systems, report
|
||||
|
||||
|
||||
def search_order(files: list[NativeFile]) -> list[NativeFile]:
|
||||
"""Order files the way the code looks for them, best first.
|
||||
|
||||
`priority:` is that order where the source states it, lowest first.
|
||||
Where it does not, the order the entries were declared in is the order
|
||||
the code walks, so it is left alone.
|
||||
"""
|
||||
ranked = [(fe.priority, position, fe) for position, fe in enumerate(files)]
|
||||
return [
|
||||
fe
|
||||
for _, _, fe in sorted(
|
||||
ranked,
|
||||
key=lambda item: (
|
||||
(0, item[0]) if item[0] is not None else (1, item[1])
|
||||
),
|
||||
)
|
||||
]
|
||||
@@ -1,109 +1,301 @@
|
||||
"""Exporter for Batocera batocera-systems format.
|
||||
"""Exporter for Batocera's batocera-systems.
|
||||
|
||||
Produces a Python dict matching the exact format of
|
||||
batocera-linux/batocera-scripts/scripts/batocera-systems.
|
||||
batocera-systems is an executable script: the systems dict is one block
|
||||
inside it, the rest is the checker Batocera runs. Only the block is
|
||||
rewritten, entry by entry, so the comments the maintainers wrote between
|
||||
entries and the code around them survive untouched.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
import re
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeFile, NativeSystem, Report
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://raw.githubusercontent.com/batocera-linux/batocera.linux/master"
|
||||
"/package/batocera/core/batocera-scripts/scripts/batocera-systems"
|
||||
)
|
||||
|
||||
_ENTRY_START = re.compile(r'^(\s{4})"([^"]+)":\s*\{')
|
||||
_INDENT = " " * 4
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Export truth data to Batocera batocera-systems format."""
|
||||
"""Write Batocera's batocera-systems, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "batocera"
|
||||
|
||||
def export(
|
||||
self,
|
||||
truth_data: dict,
|
||||
output_path: str,
|
||||
scraped_data: dict | None = None,
|
||||
) -> None:
|
||||
# Build native_id and display name maps from scraped data
|
||||
native_map: dict[str, str] = {}
|
||||
if scraped_data:
|
||||
for sys_id, sys_data in scraped_data.get("systems", {}).items():
|
||||
nid = sys_data.get("native_id")
|
||||
if nid:
|
||||
native_map[sys_id] = nid
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "batocera-systems"
|
||||
|
||||
lines: list[str] = ["systems = {", ""]
|
||||
@staticmethod
|
||||
def requires() -> str:
|
||||
return "md5"
|
||||
|
||||
systems = truth_data.get("systems", {})
|
||||
for sys_id in sorted(systems):
|
||||
sys_data = systems[sys_id]
|
||||
files = sys_data.get("files", [])
|
||||
if not files:
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"md5"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"batocera-systems": SOURCE_URL}
|
||||
|
||||
@staticmethod
|
||||
def needs_original() -> bool:
|
||||
# The data block is a fraction of the file; the rest is the checker.
|
||||
return True
|
||||
|
||||
def _bios_files(self, files: list[NativeFile]) -> str:
|
||||
return ", ".join(self._bios_items(files))
|
||||
|
||||
def _bios_items(self, files: list[NativeFile]) -> list[str]:
|
||||
parts: list[str] = []
|
||||
for fe in files:
|
||||
# The platform states an unhashed file as an empty md5 rather
|
||||
# than leaving it out, and so do we.
|
||||
item = [f'"md5": "{fe.hash("md5")}"']
|
||||
alt = fe.native("alt_md5", "")
|
||||
if alt:
|
||||
item.append(f'"altmd5": "{alt}"')
|
||||
declared = fe.native("native_path", "")
|
||||
path = str(declared) if declared else f"bios/{fe.destination}"
|
||||
item.append(f'"file": "{path}"')
|
||||
zipped = fe.native("zipped_file", "")
|
||||
if zipped:
|
||||
item.append(f'"zippedFile": "{zipped}"')
|
||||
parts.append("{ " + ", ".join(item) + " }")
|
||||
return parts
|
||||
|
||||
def _entry_line(self, system: NativeSystem, files: list[NativeFile]) -> str:
|
||||
return (
|
||||
f'{_INDENT}"{system.native_id}": '
|
||||
f'{{ "name": "{self.display_name(system)}", '
|
||||
f'"biosFiles": [ {self._bios_files(files)} ] }},'
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _item_notes(lines: list[str]) -> tuple[dict[str, str], dict[str, list[str]]]:
|
||||
"""The notes a maintainer wrote about each file, keyed by its path.
|
||||
|
||||
Rewriting an entry as one line would drop them, and a note like
|
||||
"ideally - 94bc50..." is the only record of why a hash is blank.
|
||||
A comment on its own line belongs to the file below it; a comment at
|
||||
the end of a line belongs to the file on it.
|
||||
"""
|
||||
trailing: dict[str, str] = {}
|
||||
leading: dict[str, list[str]] = {}
|
||||
pending: list[str] = []
|
||||
for line in lines:
|
||||
stripped = line.strip()
|
||||
hash_at = line.find("#")
|
||||
if stripped.startswith("#"):
|
||||
pending.append(stripped)
|
||||
continue
|
||||
path = re.search(r'"file"\s*:\s*"([^"]+)"', line)
|
||||
if not path:
|
||||
continue
|
||||
key = path.group(1)
|
||||
if pending:
|
||||
leading[key] = pending
|
||||
pending = []
|
||||
if 0 <= hash_at and hash_at > line.index(key):
|
||||
trailing[key] = line[hash_at:].rstrip()
|
||||
return trailing, leading
|
||||
|
||||
native_id = native_map.get(sys_id, sys_id)
|
||||
scraped_sys = (
|
||||
scraped_data.get("systems", {}).get(sys_id) if scraped_data else None
|
||||
def _entry_block(
|
||||
self,
|
||||
system: NativeSystem,
|
||||
files: list[NativeFile],
|
||||
original: list[str],
|
||||
) -> list[str]:
|
||||
"""One entry, keeping the layout and the notes it was written with."""
|
||||
trailing, leading = self._item_notes(original)
|
||||
items = self._bios_items(files)
|
||||
one_line = len(original) == 1 and not trailing and not leading
|
||||
if one_line:
|
||||
return [self._entry_line(system, files)]
|
||||
|
||||
head = (
|
||||
f'{_INDENT}"{system.native_id}": '
|
||||
f'{{ "name": "{self.display_name(system)}", "biosFiles": ['
|
||||
)
|
||||
pad = " " * (len(_INDENT) + 4)
|
||||
lines = [head]
|
||||
for index, item in enumerate(items):
|
||||
key = self._item_path(item)
|
||||
lines.extend(f"{pad}{note}" for note in leading.get(key, []))
|
||||
separator = "," if index < len(items) - 1 else ""
|
||||
note = trailing.get(key, "")
|
||||
suffix = f" {note}" if note else ""
|
||||
lines.append(f"{pad}{item}{separator}{suffix}")
|
||||
lines.append(f"{_INDENT}] }},")
|
||||
return lines
|
||||
|
||||
@staticmethod
|
||||
def _item_path(item: str) -> str:
|
||||
match = re.search(r'"file": "([^"]+)"', item)
|
||||
return match.group(1) if match else ""
|
||||
|
||||
@staticmethod
|
||||
def _split_original(original: str) -> tuple[list[str], list[str], list[str]]:
|
||||
"""Cut the script into what precedes the dict, the dict, what follows."""
|
||||
lines = original.split("\n")
|
||||
start = next(
|
||||
(i for i, line in enumerate(lines) if line.startswith("systems = {")),
|
||||
None,
|
||||
)
|
||||
if start is None:
|
||||
raise ValueError("batocera-systems: no systems dict found")
|
||||
end = next(
|
||||
(i for i in range(start + 1, len(lines)) if lines[i].startswith("}")),
|
||||
None,
|
||||
)
|
||||
if end is None:
|
||||
raise ValueError("batocera-systems: the systems dict is not closed")
|
||||
return lines[: start + 1], lines[start + 1 : end], lines[end:]
|
||||
|
||||
@staticmethod
|
||||
def _entry_spans(body: list[str]) -> dict[str, tuple[int, int]]:
|
||||
"""Locate each top-level entry, which may wrap over several lines."""
|
||||
spans: dict[str, tuple[int, int]] = {}
|
||||
index = 0
|
||||
while index < len(body):
|
||||
match = _ENTRY_START.match(body[index])
|
||||
if not match:
|
||||
index += 1
|
||||
continue
|
||||
key = match.group(2)
|
||||
depth = 0
|
||||
end = index
|
||||
for cursor in range(index, len(body)):
|
||||
depth += body[cursor].count("{") + body[cursor].count("[")
|
||||
depth -= body[cursor].count("}") + body[cursor].count("]")
|
||||
if depth <= 0:
|
||||
end = cursor
|
||||
break
|
||||
else:
|
||||
end = len(body) - 1
|
||||
spans[key] = (index, end)
|
||||
index = end + 1
|
||||
return spans
|
||||
|
||||
@staticmethod
|
||||
def _parse_entry(lines: list[str]) -> dict | None:
|
||||
"""Evaluate one entry so two spellings of the same data compare equal."""
|
||||
text = "\n".join(lines).strip().rstrip(",")
|
||||
namespace: dict[str, object] = {}
|
||||
try:
|
||||
exec(f"entry = {{{text}}}", {}, namespace) # noqa: S102
|
||||
except (SyntaxError, ValueError, TypeError):
|
||||
return None
|
||||
entry = namespace.get("entry")
|
||||
return entry if isinstance(entry, dict) else None
|
||||
|
||||
def render(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
exportable = dict(
|
||||
(system.native_id, (system, files))
|
||||
for system, files in self.exportable(systems, require="md5")
|
||||
)
|
||||
|
||||
original = originals.get(self.native_filename(), "")
|
||||
if not original:
|
||||
raise ValueError(
|
||||
f"{self.native_filename()} cannot be written without the "
|
||||
"platform's own file: the systems dict is a fraction of a "
|
||||
"script, and the rest of it is the checker"
|
||||
)
|
||||
display_name = self._display_name(sys_id, scraped_sys)
|
||||
|
||||
# Build md5 lookup from scraped data for this system
|
||||
scraped_md5: dict[str, str] = {}
|
||||
if scraped_data:
|
||||
s_sys = scraped_data.get("systems", {}).get(sys_id, {})
|
||||
for sf in s_sys.get("files", []):
|
||||
sname = sf.get("name", "").lower()
|
||||
smd5 = sf.get("md5", "")
|
||||
if sname and smd5:
|
||||
scraped_md5[sname] = smd5
|
||||
head, body, tail = self._split_original(original)
|
||||
spans = self._entry_spans(body)
|
||||
|
||||
# Build biosFiles entries as compact single-line dicts
|
||||
# Original format ALWAYS has md5 — use scraped md5 as fallback
|
||||
bios_parts: list[str] = []
|
||||
for fe in files:
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
continue
|
||||
dest = self._dest(fe)
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = md5[0] if md5 else ""
|
||||
if not md5:
|
||||
md5 = scraped_md5.get(name.lower(), "")
|
||||
rebuilt: list[str] = []
|
||||
written: set[str] = set()
|
||||
index = 0
|
||||
for key, (start, end) in sorted(spans.items(), key=lambda kv: kv[1][0]):
|
||||
rebuilt.extend(body[index:start])
|
||||
pair = exportable.get(key)
|
||||
index = end + 1
|
||||
if pair is None:
|
||||
# The truth has nothing to say and no file left to declare.
|
||||
rebuilt.extend(body[start:end + 1])
|
||||
continue
|
||||
written.add(key)
|
||||
original_lines = body[start:end + 1]
|
||||
replacement = self._entry_line(*pair)
|
||||
before = self._parse_entry(original_lines)
|
||||
after = self._parse_entry([replacement])
|
||||
if before is not None and before == after:
|
||||
# Nothing changed: keep the maintainer's own lines, comments
|
||||
# and alignment included, so the diff shows only corrections.
|
||||
rebuilt.extend(original_lines)
|
||||
else:
|
||||
rebuilt.extend(self._entry_block(*pair, original_lines))
|
||||
rebuilt.extend(body[index:])
|
||||
|
||||
# Original format requires md5 for every entry — skip without
|
||||
if not md5:
|
||||
continue
|
||||
bios_parts.append(f'{{ "md5": "{md5}", "file": "bios/{dest}" }}')
|
||||
added = [
|
||||
self._entry_line(system, files)
|
||||
for native_id, (system, files) in exportable.items()
|
||||
if native_id not in written
|
||||
]
|
||||
if added:
|
||||
while rebuilt and not rebuilt[-1].strip():
|
||||
rebuilt.pop()
|
||||
rebuilt.append("")
|
||||
rebuilt.extend(sorted(added))
|
||||
rebuilt.append("")
|
||||
|
||||
bios_str = ", ".join(bios_parts)
|
||||
line = (
|
||||
f' "{native_id}": '
|
||||
f'{{ "name": "{display_name}", '
|
||||
f'"biosFiles": [ {bios_str} ] }},'
|
||||
)
|
||||
lines.append(line)
|
||||
return {self.native_filename(): "\n".join([*head, *rebuilt, *tail])}
|
||||
|
||||
lines.append("")
|
||||
lines.append("}")
|
||||
lines.append("")
|
||||
Path(output_path).write_text("\n".join(lines), encoding="utf-8")
|
||||
|
||||
def validate(self, truth_data: dict, output_path: str) -> list[str]:
|
||||
content = Path(output_path).read_text(encoding="utf-8")
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
content = produced[self.native_filename()]
|
||||
issues: list[str] = []
|
||||
for sys_data in truth_data.get("systems", {}).values():
|
||||
for fe in sys_data.get("files", []):
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
continue
|
||||
# Skip entries without md5 (not exportable in this format)
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = md5[0] if md5 else ""
|
||||
if not md5:
|
||||
continue
|
||||
dest = self._dest(fe)
|
||||
if dest not in content and name not in content:
|
||||
issues.append(f"missing: {name}")
|
||||
|
||||
namespace: dict[str, object] = {}
|
||||
block = content.split("\nsystems = {", 1)
|
||||
if len(block) != 2:
|
||||
return ["no systems dict in the output"]
|
||||
end = block[1].find("\n}")
|
||||
if end < 0:
|
||||
return ["the systems dict is not closed"]
|
||||
try:
|
||||
exec("systems = {" + block[1][:end] + "\n}", {}, namespace) # noqa: S102
|
||||
except SyntaxError as exc:
|
||||
return [f"the systems dict does not parse: {exc}"]
|
||||
|
||||
exported = namespace.get("systems", {})
|
||||
if not isinstance(exported, dict):
|
||||
return ["the systems dict did not evaluate to a dict"]
|
||||
|
||||
for system, files in self.exportable(systems, require="md5"):
|
||||
entry = exported.get(system.native_id)
|
||||
if entry is None:
|
||||
issues.append(f"system absent: {system.native_id}")
|
||||
continue
|
||||
declared = {bios.get("file", "") for bios in entry.get("biosFiles", [])}
|
||||
for fe in files:
|
||||
path = str(fe.native("native_path", "")) or f"bios/{fe.destination}"
|
||||
if path not in declared:
|
||||
issues.append(f"absent: {system.native_id}/{fe.name}")
|
||||
|
||||
for native_id, entry in exported.items():
|
||||
if not entry.get("biosFiles"):
|
||||
issues.append(f"empty entry: {native_id}")
|
||||
|
||||
if "def checkBios(" not in content:
|
||||
issues.append("the checker the script exists for is missing")
|
||||
return issues
|
||||
@@ -0,0 +1,153 @@
|
||||
"""Exporter for BizHawk's FirmwareDatabase.cs.
|
||||
|
||||
The database is C#: every firmware is a call whose arguments are the SHA1,
|
||||
the size, the file name and a description, wired into option lists and
|
||||
status flags that only the source expresses. The calls are rewritten in
|
||||
place, and nothing else in the file is touched.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeFile, NativeSystem, Report
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://raw.githubusercontent.com/TASEmulators/BizHawk/master"
|
||||
"/src/BizHawk.Emulation.Common/Database/FirmwareDatabase.cs"
|
||||
)
|
||||
|
||||
# File("<sha1>", <size>, "<name>", ...) and the same three arguments inside
|
||||
# FirmwareAndOption(<sha1>, <size>, <system>, <id>, <name>, ...).
|
||||
_FILE_CALL = re.compile(
|
||||
r'(File\(\s*")([0-9A-Fa-f]{40})("\s*,\s*)(\d+)(\s*,\s*")([^"]+)(")'
|
||||
)
|
||||
_FIRMWARE_AND_OPTION = re.compile(
|
||||
r'(FirmwareAndOption\(\s*")([0-9A-Fa-f]{40})("\s*,\s*)(\d+)'
|
||||
r'(\s*,\s*"[^"]*"\s*,\s*"[^"]*"\s*,\s*")([^"]+)(")'
|
||||
)
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Write BizHawk's FirmwareDatabase.cs, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "bizhawk"
|
||||
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "FirmwareDatabase.cs"
|
||||
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"sha1", "size"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"FirmwareDatabase.cs": SOURCE_URL}
|
||||
|
||||
@staticmethod
|
||||
def can_add() -> bool:
|
||||
# A firmware is a call wired into an option list and a status; the
|
||||
# exporter corrects the calls that exist, it does not write C#.
|
||||
return False
|
||||
|
||||
@staticmethod
|
||||
def needs_original() -> bool:
|
||||
# The database is code: option lists, statuses and the systems they
|
||||
# hang off exist nowhere else.
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def _unambiguous(systems: dict[str, NativeSystem]) -> dict[str, NativeFile]:
|
||||
"""Files whose name identifies exactly one entry with a SHA1.
|
||||
|
||||
BizHawk names a firmware by file name inside a system, and the same
|
||||
name recurs across systems. Correcting on a name that resolves to
|
||||
two different sets of bytes would corrupt the database, so only the
|
||||
names that resolve to one are touched.
|
||||
"""
|
||||
seen: dict[str, list[NativeFile]] = {}
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
if fe.hash("sha1"):
|
||||
seen.setdefault(fe.name.casefold(), []).append(fe)
|
||||
resolved: dict[str, NativeFile] = {}
|
||||
for name, entries in seen.items():
|
||||
hashes = {fe.hash("sha1").lower() for fe in entries}
|
||||
if len(hashes) == 1:
|
||||
resolved[name] = entries[0]
|
||||
return resolved
|
||||
|
||||
def render(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
source = originals.get(self.native_filename(), "")
|
||||
if not source:
|
||||
raise ValueError(
|
||||
"FirmwareDatabase.cs cannot be written without BizHawk's own "
|
||||
"file: the database is C#, not data"
|
||||
)
|
||||
|
||||
index = self._unambiguous(systems)
|
||||
|
||||
def commented_out(text: str, position: int) -> bool:
|
||||
"""Whether the call sits on a line the compiler never sees.
|
||||
|
||||
BizHawk keeps disabled entries in place behind //, and a hash
|
||||
written into one of those is a change to a comment.
|
||||
"""
|
||||
line_start = text.rfind("\n", 0, position) + 1
|
||||
return text[line_start:position].lstrip().startswith("//")
|
||||
|
||||
def rewrite(match: re.Match[str], name_group: int) -> str:
|
||||
if commented_out(match.string, match.start()):
|
||||
return match.group(0)
|
||||
name = match.group(name_group)
|
||||
fe = index.get(name.casefold())
|
||||
if fe is None:
|
||||
return match.group(0)
|
||||
sha1 = fe.hash("sha1").upper()
|
||||
size = fe.size() or int(match.group(4))
|
||||
groups = list(match.groups())
|
||||
groups[1] = sha1
|
||||
groups[3] = str(size)
|
||||
return "".join(groups)
|
||||
|
||||
patched = _FILE_CALL.sub(lambda m: rewrite(m, 6), source)
|
||||
patched = _FIRMWARE_AND_OPTION.sub(lambda m: rewrite(m, 6), patched)
|
||||
return {self.native_filename(): patched}
|
||||
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
content = produced[self.native_filename()]
|
||||
issues: list[str] = []
|
||||
|
||||
if "FirmwareDatabase" not in content:
|
||||
issues.append("the class the database lives in is missing")
|
||||
if content.count("{") != content.count("}"):
|
||||
issues.append("braces are unbalanced, the file would not compile")
|
||||
|
||||
index = self._unambiguous(systems)
|
||||
declared: dict[str, str] = {}
|
||||
for pattern in (_FILE_CALL, _FIRMWARE_AND_OPTION):
|
||||
for match in pattern.finditer(content):
|
||||
line_start = content.rfind("\n", 0, match.start()) + 1
|
||||
if content[line_start : match.start()].lstrip().startswith("//"):
|
||||
continue
|
||||
declared[match.group(6).casefold()] = match.group(2).lower()
|
||||
|
||||
for name, fe in index.items():
|
||||
written = declared.get(name)
|
||||
if written is not None and written != fe.hash("sha1").lower():
|
||||
issues.append(f"hash not applied: {fe.name}")
|
||||
return issues
|
||||
@@ -1,215 +1,166 @@
|
||||
"""Exporter for EmuDeck checkBIOS.sh format.
|
||||
"""Exporter for EmuDeck's checkBIOS.sh.
|
||||
|
||||
Produces a bash script matching the exact pattern of EmuDeck's
|
||||
functions/checkBIOS.sh: per-system check functions with MD5 arrays
|
||||
inside the function body, iterating over $biosPath/* files.
|
||||
|
||||
Two patterns:
|
||||
- MD5 pattern: systems with known hashes, loop $biosPath/*, md5sum each, match
|
||||
- File-exists pattern: systems with specific paths, check -f
|
||||
checkBIOS.sh is a shell library: each system is a function EmuDeck calls by
|
||||
name, and the only data in it is the MD5 list each function matches against.
|
||||
Writing the functions from a table would publish a file missing whichever
|
||||
checks the table forgot, so the original is patched instead and every
|
||||
function it defines keeps its shape, its scan directory and its output.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from scraper.emudeck_scraper import FUNCTION_HASH_MAP, _RE_FUNC, _RE_LOCAL_HASHES
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeSystem, Report
|
||||
|
||||
# Map our system IDs to EmuDeck function naming conventions
|
||||
_SYSTEM_CONFIG: dict[str, dict] = {
|
||||
"sony-playstation": {
|
||||
"func": "checkPS1BIOS",
|
||||
"var": "PSXBIOS",
|
||||
"array": "PSBios",
|
||||
"pattern": "md5",
|
||||
},
|
||||
"sony-playstation-2": {
|
||||
"func": "checkPS2BIOS",
|
||||
"var": "PS2BIOS",
|
||||
"array": "PS2Bios",
|
||||
"pattern": "md5",
|
||||
},
|
||||
"sega-mega-cd": {
|
||||
"func": "checkSegaCDBios",
|
||||
"var": "SEGACDBIOS",
|
||||
"array": "CDBios",
|
||||
"pattern": "md5",
|
||||
},
|
||||
"sega-saturn": {
|
||||
"func": "checkSaturnBios",
|
||||
"var": "SATURNBIOS",
|
||||
"array": "SaturnBios",
|
||||
"pattern": "md5",
|
||||
},
|
||||
"sega-dreamcast": {
|
||||
"func": "checkDreamcastBios",
|
||||
"var": "BIOS",
|
||||
"array": "hashes",
|
||||
"pattern": "md5",
|
||||
},
|
||||
"nintendo-ds": {
|
||||
"func": "checkDSBios",
|
||||
"var": "BIOS",
|
||||
"array": "hashes",
|
||||
"pattern": "md5",
|
||||
},
|
||||
"nintendo-switch": {
|
||||
"func": "checkCitronBios",
|
||||
"pattern": "file-exists",
|
||||
"firmware_path": "$biosPath/citron/firmware",
|
||||
"keys_path": "$biosPath/citron/keys/prod.keys",
|
||||
},
|
||||
}
|
||||
|
||||
|
||||
def _make_md5_function(cfg: dict, md5s: list[str]) -> list[str]:
|
||||
"""Generate a MD5-checking function matching EmuDeck's exact pattern."""
|
||||
func = cfg["func"]
|
||||
var = cfg["var"]
|
||||
array = cfg["array"]
|
||||
md5_str = " ".join(md5s)
|
||||
|
||||
return [
|
||||
f"{func}(){{",
|
||||
"",
|
||||
f'\t{var}="NULL"',
|
||||
"",
|
||||
'\tfor entry in "$biosPath/"*',
|
||||
"\tdo",
|
||||
'\t\tif [ -f "$entry" ]; then',
|
||||
'\t\t\tmd5=($(md5sum "$entry"))',
|
||||
f'\t\t\tif [[ "${var}" != true ]]; then',
|
||||
f"\t\t\t\t{array}=({md5_str})",
|
||||
f'\t\t\t\tfor i in "${{{array}[@]}}"',
|
||||
"\t\t\t\tdo",
|
||||
'\t\t\t\tif [[ "$md5" == *"${i}"* ]]; then',
|
||||
f"\t\t\t\t\t{var}=true",
|
||||
"\t\t\t\t\tbreak",
|
||||
"\t\t\t\telse",
|
||||
f"\t\t\t\t\t{var}=false",
|
||||
"\t\t\t\tfi",
|
||||
"\t\t\t\tdone",
|
||||
"\t\t\tfi",
|
||||
"\t\tfi",
|
||||
"\tdone",
|
||||
"",
|
||||
"",
|
||||
f"\tif [ ${var} == true ]; then",
|
||||
'\t\techo "$entry true";',
|
||||
"\telse",
|
||||
'\t\techo "false";',
|
||||
"\tfi",
|
||||
"}",
|
||||
]
|
||||
|
||||
|
||||
def _make_file_exists_function(cfg: dict) -> list[str]:
|
||||
"""Generate a file-exists function matching EmuDeck's pattern."""
|
||||
func = cfg["func"]
|
||||
firmware = cfg.get("firmware_path", "")
|
||||
keys = cfg.get("keys_path", "")
|
||||
|
||||
return [
|
||||
f"{func}(){{",
|
||||
"",
|
||||
f'\tlocal FIRMWARE="{firmware}"',
|
||||
f'\tlocal KEYS="{keys}"',
|
||||
'\tif [[ -f "$KEYS" ]] && [[ "$( ls -A "$FIRMWARE")" ]]; then',
|
||||
'\t\t\techo "true";',
|
||||
"\telse",
|
||||
'\t\t\techo "false";',
|
||||
"\tfi",
|
||||
"}",
|
||||
]
|
||||
SOURCE_URL = (
|
||||
"https://raw.githubusercontent.com/dragoonDorise/EmuDeck/main"
|
||||
"/functions/checkBIOS.sh"
|
||||
)
|
||||
_MD5 = re.compile(r"^[0-9a-f]{32}$")
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Export truth data to EmuDeck checkBIOS.sh format."""
|
||||
"""Write EmuDeck's checkBIOS.sh, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "emudeck"
|
||||
|
||||
def export(
|
||||
self,
|
||||
truth_data: dict,
|
||||
output_path: str,
|
||||
scraped_data: dict | None = None,
|
||||
) -> None:
|
||||
lines: list[str] = ["#!/bin/bash"]
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "checkBIOS.sh"
|
||||
|
||||
systems = truth_data.get("systems", {})
|
||||
@staticmethod
|
||||
def requires() -> str:
|
||||
return "md5"
|
||||
|
||||
for sys_id, cfg in sorted(_SYSTEM_CONFIG.items(), key=lambda x: x[1]["func"]):
|
||||
sys_data = systems.get(sys_id)
|
||||
if not sys_data:
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"md5"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"checkBIOS.sh": SOURCE_URL}
|
||||
|
||||
@staticmethod
|
||||
def needs_original() -> bool:
|
||||
# The checks are code, and EmuDeck calls them by name.
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def can_add() -> bool:
|
||||
"""A hash is corrected in place, never added to an array.
|
||||
|
||||
EmuDeck's frontend calls one check for several emulators
|
||||
(EmulatorsDetailPage.jsx: 'ra' asks checkPS1BIOS, checkSegaCDBios,
|
||||
checkSaturnBios, checkDSBios and checkDreamcastBios, and
|
||||
'duckstation' asks checkPS1BIOS as well), and nothing in
|
||||
checkBIOS.sh names them. Growing an array therefore changes answers
|
||||
for consumers the array does not list: DuckStation boots from a PS2
|
||||
image, the RetroArch PSX cores do not, so adding the ones
|
||||
DuckStation accepts would report a BIOS to a card that has none.
|
||||
Correcting a value in place changes no consumer's set.
|
||||
"""
|
||||
return False
|
||||
|
||||
@classmethod
|
||||
def _md5s(cls, systems: dict[str, NativeSystem], system_id: str) -> list[str]:
|
||||
"""Every MD5 the system accepts, in a stable order, deduplicated."""
|
||||
seen: list[str] = []
|
||||
for system in systems.values():
|
||||
if system.native_id != system_id:
|
||||
continue
|
||||
for fe in system.files:
|
||||
if not cls.writable(fe):
|
||||
continue
|
||||
for value in fe.hashes("md5"):
|
||||
if _MD5.match(value) and value not in seen:
|
||||
seen.append(value)
|
||||
return seen
|
||||
|
||||
lines.append("")
|
||||
def _function_spans(self, script: str) -> list[tuple[str, int, int]]:
|
||||
"""Name and byte span of every check the script defines."""
|
||||
matches = list(_RE_FUNC.finditer(script))
|
||||
spans: list[tuple[str, int, int]] = []
|
||||
for index, match in enumerate(matches):
|
||||
end = (
|
||||
matches[index + 1].start()
|
||||
if index + 1 < len(matches)
|
||||
else len(script)
|
||||
)
|
||||
spans.append((match.group(1), match.start(), end))
|
||||
return spans
|
||||
|
||||
if cfg["pattern"] == "md5":
|
||||
md5s: list[str] = []
|
||||
for fe in sys_data.get("files", []):
|
||||
name = fe.get("name", "")
|
||||
if self._is_pattern(name) or name.startswith("_"):
|
||||
continue
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5s.extend(
|
||||
m for m in md5 if m and re.fullmatch(r"[a-f0-9]{32}", m)
|
||||
)
|
||||
elif md5 and re.fullmatch(r"[a-f0-9]{32}", md5):
|
||||
md5s.append(md5)
|
||||
if md5s:
|
||||
lines.extend(_make_md5_function(cfg, md5s))
|
||||
elif cfg["pattern"] == "file-exists":
|
||||
lines.extend(_make_file_exists_function(cfg))
|
||||
def render(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
script = originals.get(self.native_filename(), "")
|
||||
if not script:
|
||||
raise ValueError(
|
||||
"checkBIOS.sh cannot be written without EmuDeck's own file: "
|
||||
"the checks are code, not data"
|
||||
)
|
||||
|
||||
lines.append("")
|
||||
Path(output_path).write_text("\n".join(lines), encoding="utf-8")
|
||||
pieces: list[str] = []
|
||||
cursor = 0
|
||||
for name, start, end in self._function_spans(script):
|
||||
pieces.append(script[cursor:start])
|
||||
body = script[start:end]
|
||||
system_id = FUNCTION_HASH_MAP.get(name)
|
||||
md5s = self._md5s(systems, system_id) if system_id else []
|
||||
match = _RE_LOCAL_HASHES.search(body)
|
||||
if md5s and match:
|
||||
# An array compared by membership says nothing about order,
|
||||
# so the same set is left as the maintainer wrote it.
|
||||
if set(md5s) != set(match.group(1).split()):
|
||||
body = (
|
||||
body[: match.start(1)]
|
||||
+ " ".join(md5s)
|
||||
+ body[match.end(1) :]
|
||||
)
|
||||
pieces.append(body)
|
||||
cursor = end
|
||||
pieces.append(script[cursor:])
|
||||
|
||||
def validate(self, truth_data: dict, output_path: str) -> list[str]:
|
||||
content = Path(output_path).read_text(encoding="utf-8")
|
||||
return {self.native_filename(): "".join(pieces)}
|
||||
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
content = produced[self.native_filename()]
|
||||
issues: list[str] = []
|
||||
|
||||
systems = truth_data.get("systems", {})
|
||||
for sys_id, cfg in _SYSTEM_CONFIG.items():
|
||||
if cfg["pattern"] != "md5":
|
||||
defined = {name for name, _, _ in self._function_spans(content)}
|
||||
for name, system_id in FUNCTION_HASH_MAP.items():
|
||||
if name not in defined:
|
||||
issues.append(f"check absent from the output: {name}")
|
||||
continue
|
||||
sys_data = systems.get(sys_id)
|
||||
if not sys_data:
|
||||
md5s = self._md5s(systems, system_id)
|
||||
if not md5s:
|
||||
continue
|
||||
for fe in sys_data.get("files", []):
|
||||
# export skips placeholders and private entries, so looking
|
||||
# for them here makes the exporter reject its own output.
|
||||
name = fe.get("name", "")
|
||||
if self._is_pattern(name) or name.startswith("_"):
|
||||
continue
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = md5[0] if md5 else ""
|
||||
if md5 and re.fullmatch(r"[a-f0-9]{32}", md5) and md5 not in content:
|
||||
issues.append(f"missing md5: {md5} ({name})")
|
||||
|
||||
for sys_id, cfg in _SYSTEM_CONFIG.items():
|
||||
func = cfg["func"]
|
||||
if func in content:
|
||||
body = next(
|
||||
content[start:end]
|
||||
for fname, start, end in self._function_spans(content)
|
||||
if fname == name
|
||||
)
|
||||
if not _RE_LOCAL_HASHES.search(body):
|
||||
# A check with no hash list is a path check, not a hash check.
|
||||
continue
|
||||
sys_data = systems.get(sys_id)
|
||||
if not sys_data or not sys_data.get("files"):
|
||||
continue
|
||||
# Only flag if the system has usable data for the function type
|
||||
if cfg["pattern"] == "md5":
|
||||
has_md5 = any(
|
||||
fe.get("md5")
|
||||
and isinstance(fe.get("md5"), str)
|
||||
and re.fullmatch(r"[a-f0-9]{32}", fe["md5"])
|
||||
for fe in sys_data["files"]
|
||||
)
|
||||
if has_md5:
|
||||
issues.append(f"missing function: {func}")
|
||||
elif cfg["pattern"] == "file-exists":
|
||||
issues.append(f"missing function: {func}")
|
||||
|
||||
for md5 in md5s:
|
||||
if md5 not in body:
|
||||
issues.append(f"absent from {name}: {md5}")
|
||||
return issues
|
||||
@@ -1,8 +1,4 @@
|
||||
"""Exporter for Lakka (System.dat format, same as RetroArch).
|
||||
|
||||
Lakka inherits RetroArch cores and uses the same System.dat format.
|
||||
Delegates to systemdat_exporter for export and validation.
|
||||
"""
|
||||
"""Exporter for Lakka, which reads RetroArch's System.dat unchanged."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -10,7 +6,7 @@ from .systemdat_exporter import Exporter as SystemDatExporter
|
||||
|
||||
|
||||
class Exporter(SystemDatExporter):
|
||||
"""Export truth data to Lakka System.dat format."""
|
||||
"""Write Lakka's System.dat, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
|
||||
@@ -0,0 +1,125 @@
|
||||
"""Exporter for MiSTer's BiosDB (bios_db.json, shipped zipped).
|
||||
|
||||
A Downloader database entry carries the download URL, the install path and
|
||||
the tag ids alongside the hash, and none of those are ours to invent. Only
|
||||
the hash and the size are rewritten, in MiSTer's own database.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import io
|
||||
import json
|
||||
import zipfile
|
||||
from collections import OrderedDict
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeSystem, Report
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://raw.githubusercontent.com/ajgowans/BiosDB_MiSTer/db/bios_db.json.zip"
|
||||
)
|
||||
_DB_NAME = "bios_db.json"
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Write MiSTer's bios_db.json, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "misterfpga"
|
||||
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return _DB_NAME
|
||||
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"md5", "size"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"bios_db.json.zip": SOURCE_URL}
|
||||
|
||||
@staticmethod
|
||||
def can_add() -> bool:
|
||||
# Every entry carries the URL MiSTer installs it from, and that is
|
||||
# not ours to invent.
|
||||
return False
|
||||
|
||||
@staticmethod
|
||||
def needs_original() -> bool:
|
||||
# Entries carry a URL and a tag vocabulary the database owns.
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def unpack(raw: bytes) -> dict[str, str]:
|
||||
"""Read the database out of the archive MiSTer publishes."""
|
||||
with zipfile.ZipFile(io.BytesIO(raw)) as archive:
|
||||
return {_DB_NAME: archive.read(_DB_NAME).decode("utf-8")}
|
||||
|
||||
def _by_path(self, systems: dict[str, NativeSystem]) -> dict[str, object]:
|
||||
indexed: dict[str, object] = {}
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
if fe.destination:
|
||||
indexed[f"games/{fe.destination}"] = fe
|
||||
return indexed
|
||||
|
||||
def render(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
original = originals.get(_DB_NAME) or originals.get("bios_db.json.zip")
|
||||
if not original:
|
||||
raise ValueError(
|
||||
"bios_db.json cannot be written without MiSTer's own database: "
|
||||
"every entry carries a URL and tag ids that are not ours"
|
||||
)
|
||||
database = json.loads(original, object_pairs_hook=OrderedDict)
|
||||
indexed = self._by_path(systems)
|
||||
|
||||
for path, entry in database.get("files", {}).items():
|
||||
fe = indexed.get(path)
|
||||
if fe is None:
|
||||
continue
|
||||
md5 = fe.hash("md5")
|
||||
if md5:
|
||||
entry["hash"] = md5
|
||||
size = fe.size()
|
||||
if size:
|
||||
entry["size"] = size
|
||||
|
||||
return {_DB_NAME: json.dumps(database, indent=2, ensure_ascii=False) + "\n"}
|
||||
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
try:
|
||||
database = json.loads(produced[_DB_NAME])
|
||||
except json.JSONDecodeError as exc:
|
||||
return [f"the database does not parse: {exc}"]
|
||||
|
||||
issues: list[str] = []
|
||||
if not database.get("db_id"):
|
||||
issues.append("the database lost its db_id")
|
||||
files = database.get("files", {})
|
||||
if not files:
|
||||
issues.append("the database has no files left")
|
||||
for path, entry in files.items():
|
||||
if not entry.get("hash"):
|
||||
issues.append(f"entry without a hash: {path}")
|
||||
if not entry.get("url"):
|
||||
issues.append(f"entry without a URL, uninstallable: {path}")
|
||||
|
||||
indexed = self._by_path(systems)
|
||||
for path, fe in indexed.items():
|
||||
declared = files.get(path)
|
||||
md5 = fe.hash("md5")
|
||||
if declared is not None and md5 and declared.get("hash") != md5:
|
||||
issues.append(f"hash not applied: {path}")
|
||||
return issues
|
||||
@@ -1,135 +1,165 @@
|
||||
"""Exporter for Recalbox es_bios.xml format.
|
||||
"""Exporter for Recalbox's es_bios.xml.
|
||||
|
||||
Produces XML matching the exact format of recalbox's es_bios.xml:
|
||||
- XML namespace declaration
|
||||
- <system fullname="..." platform="...">
|
||||
- <bios path="system/file" md5="..." core="..." /> with optional mandatory, hashMatchMandatory, note
|
||||
- mandatory absent = true (only explicit when false)
|
||||
- 2-space indentation
|
||||
The file is validated by es_bios.xsd, which makes path, md5 and core
|
||||
required on every bios element. An entry we cannot give all three to is not
|
||||
written: Recalbox would reject the file whole.
|
||||
|
||||
mandatory and hashMatchMandatory are separate axes. Recalbox reads a missing
|
||||
attribute as true for both, so each is written only when it is false, or
|
||||
when the platform stated it explicitly.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
from xml.etree.ElementTree import ParseError
|
||||
from xml.sax.saxutils import quoteattr
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from common import parse_untrusted_xml
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeFile, NativeSystem, Report
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://gitlab.com/recalbox/recalbox/-/raw/master/board/recalbox/fsoverlay"
|
||||
"/recalbox/share_init/system/.emulationstation/es_bios.xml"
|
||||
)
|
||||
SCHEMA_URL = (
|
||||
"https://gitlab.com/recalbox/recalbox/-/raw/master/board/recalbox/fsoverlay"
|
||||
"/recalbox/share_init/system/.emulationstation/es_bios.xsd"
|
||||
)
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Export truth data to Recalbox es_bios.xml format."""
|
||||
"""Write Recalbox's es_bios.xml, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "recalbox"
|
||||
|
||||
def export(
|
||||
self,
|
||||
truth_data: dict,
|
||||
output_path: str,
|
||||
scraped_data: dict | None = None,
|
||||
) -> None:
|
||||
native_map: dict[str, str] = {}
|
||||
if scraped_data:
|
||||
for sys_id, sys_data in scraped_data.get("systems", {}).items():
|
||||
nid = sys_data.get("native_id")
|
||||
if nid:
|
||||
native_map[sys_id] = nid
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "es_bios.xml"
|
||||
|
||||
lines: list[str] = [
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"md5", "required"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"es_bios.xml": SOURCE_URL, "es_bios.xsd": SCHEMA_URL}
|
||||
|
||||
def _path(self, fe: NativeFile, native_id: str) -> str:
|
||||
"""The path Recalbox reads, pipe-joined when it accepts several.
|
||||
|
||||
A path Recalbox already states is reproduced exactly: several of its
|
||||
entries sit at the BIOS root with no directory at all, and prefixing
|
||||
them with the system would point the frontend somewhere else.
|
||||
"""
|
||||
if fe.platform is not None:
|
||||
path = str(fe.platform.get("destination") or fe.name)
|
||||
alternatives = fe.platform.get("alt_paths") or []
|
||||
if alternatives:
|
||||
return "|".join([path, *[str(a) for a in alternatives]])
|
||||
return path
|
||||
dest = fe.destination or fe.name
|
||||
return dest if "/" in dest else f"{native_id}/{dest}"
|
||||
|
||||
def _bios_element(self, fe: NativeFile, native_id: str) -> str:
|
||||
attrs = [f"path={quoteattr(self._path(fe, native_id))}"]
|
||||
attrs.append(f'md5={quoteattr(",".join(fe.hashes("md5")))}')
|
||||
attrs.append(f'core={quoteattr(",".join(fe.cores()))}')
|
||||
|
||||
if not fe.required:
|
||||
attrs.append('mandatory="false"')
|
||||
elif fe.native("mandatory_declared", None) is True:
|
||||
attrs.append('mandatory="true"')
|
||||
|
||||
hash_match = fe.native("hash_match_mandatory", None)
|
||||
if hash_match is False:
|
||||
attrs.append('hashMatchMandatory="false"')
|
||||
elif hash_match is True:
|
||||
attrs.append('hashMatchMandatory="true"')
|
||||
|
||||
note = " ".join(str(fe.native("note", "")).split())
|
||||
if note:
|
||||
attrs.append(f"note={quoteattr(note)}")
|
||||
|
||||
return f" <bios {' '.join(attrs)} />"
|
||||
|
||||
@classmethod
|
||||
def writable(cls, fe: NativeFile, require: str = "") -> bool:
|
||||
"""es_bios.xsd makes md5 and core required; without them, no element.
|
||||
|
||||
An entry Recalbox already ships has both, so this only ever gates
|
||||
what we would be adding.
|
||||
"""
|
||||
return bool(fe.hashes("md5")) and bool(fe.cores())
|
||||
|
||||
def render(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
lines = [
|
||||
'<?xml version="1.0" encoding="UTF-8"?>',
|
||||
'<biosList xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"'
|
||||
' xsi:noNamespaceSchemaLocation="es_bios.xsd">',
|
||||
]
|
||||
|
||||
systems = truth_data.get("systems", {})
|
||||
for sys_id in sorted(systems):
|
||||
sys_data = systems[sys_id]
|
||||
files = sys_data.get("files", [])
|
||||
if not files:
|
||||
for system in sorted(systems.values(), key=lambda s: s.native_id):
|
||||
writable = [fe for fe in system.files if self.writable(fe)]
|
||||
if not writable:
|
||||
continue
|
||||
|
||||
native_id = native_map.get(sys_id, sys_id)
|
||||
scraped_sys = (
|
||||
scraped_data.get("systems", {}).get(sys_id) if scraped_data else None
|
||||
)
|
||||
display_name = self._display_name(sys_id, scraped_sys)
|
||||
|
||||
lines.append(f' <system fullname="{display_name}" platform="{native_id}">')
|
||||
|
||||
# Build path lookup from scraped data for this system
|
||||
scraped_paths: dict[str, str] = {}
|
||||
if scraped_data:
|
||||
s_sys = scraped_data.get("systems", {}).get(sys_id, {})
|
||||
for sf in s_sys.get("files", []):
|
||||
sname = sf.get("name", "").lower()
|
||||
spath = sf.get("destination", sf.get("name", ""))
|
||||
if sname and spath:
|
||||
scraped_paths[sname] = spath
|
||||
|
||||
for fe in files:
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
continue
|
||||
|
||||
# Use scraped path when available (preserves original format)
|
||||
path = scraped_paths.get(name.lower())
|
||||
if not path:
|
||||
dest = self._dest(fe)
|
||||
path = f"{native_id}/{dest}" if "/" not in dest else dest
|
||||
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = ",".join(md5)
|
||||
|
||||
required = fe.get("required", True)
|
||||
|
||||
# Build cores string from _cores
|
||||
cores_list = fe.get("_cores", [])
|
||||
core_str = (
|
||||
",".join(f"libretro/{c}" for c in cores_list) if cores_list else ""
|
||||
)
|
||||
|
||||
attrs = [f'path="{path}"']
|
||||
if md5:
|
||||
attrs.append(f'md5="{md5}"')
|
||||
if not required:
|
||||
attrs.append('mandatory="false"')
|
||||
if not required:
|
||||
attrs.append('hashMatchMandatory="true"')
|
||||
if core_str:
|
||||
attrs.append(f'core="{core_str}"')
|
||||
|
||||
lines.append(f" <bios {' '.join(attrs)} />")
|
||||
|
||||
fullname = quoteattr(self.display_name(system))
|
||||
platform = quoteattr(system.native_id)
|
||||
lines.append(f" <system fullname={fullname} platform={platform}>")
|
||||
for fe in writable:
|
||||
lines.append(self._bios_element(fe, system.native_id))
|
||||
lines.append(" </system>")
|
||||
|
||||
lines.append("</biosList>")
|
||||
lines.append("")
|
||||
Path(output_path).write_text("\n".join(lines), encoding="utf-8")
|
||||
|
||||
def validate(self, truth_data: dict, output_path: str) -> list[str]:
|
||||
from xml.etree.ElementTree import parse as xml_parse
|
||||
|
||||
tree = xml_parse(output_path)
|
||||
root = tree.getroot()
|
||||
|
||||
exported_paths: set[str] = set()
|
||||
for bios_el in root.iter("bios"):
|
||||
path = bios_el.get("path", "")
|
||||
if path:
|
||||
exported_paths.add(path.lower())
|
||||
exported_paths.add(path.split("/")[-1].lower())
|
||||
return {self.native_filename(): "\n".join(lines)}
|
||||
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
content = produced[self.native_filename()]
|
||||
issues: list[str] = []
|
||||
for sys_data in truth_data.get("systems", {}).values():
|
||||
for fe in sys_data.get("files", []):
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
try:
|
||||
root = parse_untrusted_xml(content, self.native_filename())
|
||||
except (ParseError, ValueError) as exc:
|
||||
return [f"the XML does not parse: {exc}"]
|
||||
|
||||
for element in root.iter("bios"):
|
||||
for attribute in ("path", "md5", "core"):
|
||||
if not element.get(attribute):
|
||||
issues.append(
|
||||
f"es_bios.xsd requires {attribute}: "
|
||||
f"{element.get('path', '?')}"
|
||||
)
|
||||
for element in root.iter("system"):
|
||||
if not list(element):
|
||||
issues.append(f"empty system: {element.get('platform', '?')}")
|
||||
for attribute in ("fullname", "platform"):
|
||||
if not element.get(attribute):
|
||||
issues.append(f"es_bios.xsd requires {attribute} on system")
|
||||
|
||||
exported = {
|
||||
element.get("path", "").casefold() for element in root.iter("bios")
|
||||
}
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
if not self.writable(fe):
|
||||
continue
|
||||
dest = self._dest(fe)
|
||||
if (
|
||||
name.lower() not in exported_paths
|
||||
and dest.lower() not in exported_paths
|
||||
):
|
||||
issues.append(f"missing: {name}")
|
||||
if self._path(fe, system.native_id).casefold() not in exported:
|
||||
issues.append(f"absent: {system.native_id}/{fe.name}")
|
||||
return issues
|
||||
@@ -1,116 +1,118 @@
|
||||
"""Exporter for RetroBat batocera-systems.json format.
|
||||
"""Exporter for RetroBat's batocera-systems.json.
|
||||
|
||||
Produces JSON matching the exact format of
|
||||
RetroBat-Official/emulatorlauncher/batocera-systems/Resources/batocera-systems.json:
|
||||
- System keys with "name" and "biosFiles" fields
|
||||
- Each biosFile has "md5" before "file" (matching original key order)
|
||||
Pure data: a system key carrying a name and a biosFiles array whose entries
|
||||
state md5 then file, in that order.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from collections import OrderedDict
|
||||
from pathlib import Path
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeSystem, Report
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://raw.githubusercontent.com/RetroBat-Official/emulatorlauncher/master"
|
||||
"/batocera-systems/Resources/batocera-systems.json"
|
||||
)
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Export truth data to RetroBat batocera-systems.json format."""
|
||||
"""Write RetroBat's batocera-systems.json, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "retrobat"
|
||||
|
||||
def export(
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "batocera-systems.json"
|
||||
|
||||
@staticmethod
|
||||
def requires() -> str:
|
||||
return "md5"
|
||||
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"md5"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"batocera-systems.json": SOURCE_URL}
|
||||
|
||||
def render(
|
||||
self,
|
||||
truth_data: dict,
|
||||
output_path: str,
|
||||
scraped_data: dict | None = None,
|
||||
) -> None:
|
||||
native_map: dict[str, str] = {}
|
||||
if scraped_data:
|
||||
for sys_id, sys_data in scraped_data.get("systems", {}).items():
|
||||
nid = sys_data.get("native_id")
|
||||
if nid:
|
||||
native_map[sys_id] = nid
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
# Keep the platform's own key order when we have its file, so the
|
||||
# diff a maintainer reads is the corrections and nothing else.
|
||||
order: list[str] = []
|
||||
original = originals.get(self.native_filename(), "")
|
||||
if original:
|
||||
try:
|
||||
order = list(json.loads(original))
|
||||
except json.JSONDecodeError:
|
||||
order = []
|
||||
|
||||
output: OrderedDict[str, dict] = OrderedDict()
|
||||
|
||||
systems = truth_data.get("systems", {})
|
||||
for sys_id in sorted(systems):
|
||||
sys_data = systems[sys_id]
|
||||
files = sys_data.get("files", [])
|
||||
if not files:
|
||||
continue
|
||||
|
||||
native_id = native_map.get(sys_id, sys_id)
|
||||
scraped_sys = (
|
||||
scraped_data.get("systems", {}).get(sys_id) if scraped_data else None
|
||||
)
|
||||
display_name = self._display_name(sys_id, scraped_sys)
|
||||
bios_files: list[OrderedDict] = []
|
||||
exportable = {
|
||||
system.native_id: (system, files)
|
||||
for system, files in self.exportable(systems, require="md5")
|
||||
}
|
||||
keys = [k for k in order if k in exportable]
|
||||
keys.extend(sorted(k for k in exportable if k not in keys))
|
||||
|
||||
output: OrderedDict[str, object] = OrderedDict()
|
||||
for key in keys:
|
||||
system, files = exportable[key]
|
||||
bios_files = []
|
||||
for fe in files:
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
continue
|
||||
dest = self._dest(fe)
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = md5[0] if md5 else ""
|
||||
|
||||
# Original format requires md5 for every entry
|
||||
if not md5:
|
||||
continue
|
||||
entry: OrderedDict[str, str] = OrderedDict()
|
||||
entry["md5"] = md5
|
||||
entry["file"] = f"bios/{dest}"
|
||||
entry["md5"] = fe.hash("md5")
|
||||
declared = fe.native("native_path", "")
|
||||
entry["file"] = str(declared) if declared else f"bios/{fe.destination}"
|
||||
bios_files.append(entry)
|
||||
system_entry: OrderedDict[str, object] = OrderedDict()
|
||||
system_entry["name"] = self.display_name(system)
|
||||
system_entry["biosFiles"] = bios_files
|
||||
output[key] = system_entry
|
||||
|
||||
if bios_files:
|
||||
if native_id in output:
|
||||
existing_files = {
|
||||
e.get("file") for e in output[native_id]["biosFiles"]
|
||||
}
|
||||
for entry in bios_files:
|
||||
if entry.get("file") not in existing_files:
|
||||
output[native_id]["biosFiles"].append(entry)
|
||||
else:
|
||||
sys_entry: OrderedDict[str, object] = OrderedDict()
|
||||
sys_entry["name"] = display_name
|
||||
sys_entry["biosFiles"] = bios_files
|
||||
output[native_id] = sys_entry
|
||||
text = json.dumps(output, indent=2, ensure_ascii=False) + "\n"
|
||||
return {self.native_filename(): text}
|
||||
|
||||
Path(output_path).write_text(
|
||||
json.dumps(output, indent=2, ensure_ascii=False) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
def validate(self, truth_data: dict, output_path: str) -> list[str]:
|
||||
data = json.loads(Path(output_path).read_text(encoding="utf-8"))
|
||||
|
||||
exported_files: set[str] = set()
|
||||
for sys_data in data.values():
|
||||
for bf in sys_data.get("biosFiles", []):
|
||||
path = bf.get("file", "")
|
||||
stripped = path.removeprefix("bios/")
|
||||
exported_files.add(stripped)
|
||||
basename = path.split("/")[-1] if "/" in path else path
|
||||
exported_files.add(basename)
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
try:
|
||||
data = json.loads(produced[self.native_filename()])
|
||||
except json.JSONDecodeError as exc:
|
||||
return [f"the JSON does not parse: {exc}"]
|
||||
|
||||
issues: list[str] = []
|
||||
for sys_data in truth_data.get("systems", {}).values():
|
||||
for fe in sys_data.get("files", []):
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
continue
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = md5[0] if md5 else ""
|
||||
if not md5:
|
||||
continue
|
||||
dest = self._dest(fe)
|
||||
if name not in exported_files and dest not in exported_files:
|
||||
issues.append(f"missing: {name}")
|
||||
for key, entry in data.items():
|
||||
if not entry.get("name"):
|
||||
issues.append(f"system without a name: {key}")
|
||||
if not entry.get("biosFiles"):
|
||||
issues.append(f"empty entry: {key}")
|
||||
for bios in entry.get("biosFiles", []):
|
||||
# RetroBat states an unhashed file with an empty md5, so only
|
||||
# a missing path makes an entry unusable.
|
||||
if "md5" not in bios or not bios.get("file"):
|
||||
issues.append(f"incomplete entry: {key}/{bios.get('file', '?')}")
|
||||
|
||||
for system, files in self.exportable(systems, require="md5"):
|
||||
entry = data.get(system.native_id)
|
||||
if entry is None:
|
||||
issues.append(f"system absent: {system.native_id}")
|
||||
continue
|
||||
declared = {bios.get("file") for bios in entry.get("biosFiles", [])}
|
||||
for fe in files:
|
||||
path = str(fe.native("native_path", "")) or f"bios/{fe.destination}"
|
||||
if path not in declared:
|
||||
issues.append(f"absent: {system.native_id}/{fe.name}")
|
||||
return issues
|
||||
@@ -1,210 +1,182 @@
|
||||
"""Exporter for RetroDECK component_manifest.json format.
|
||||
"""Exporter for RetroDECK's component manifests.
|
||||
|
||||
Produces a JSON file compatible with RetroDECK's component manifests.
|
||||
Each system maps to a component with BIOS entries containing filename,
|
||||
md5 (comma-separated if multiple), paths ($bios_path default), and
|
||||
required status.
|
||||
|
||||
Path tokens: $bios_path for bios/, $roms_path for roms/.
|
||||
Entries without an explicit path default to $bios_path.
|
||||
RetroDECK has no single BIOS file. Each component carries its own
|
||||
component_manifest.json, and the BIOS list sits inside it next to the
|
||||
component's name, description and presets, at one of three keys. Only that
|
||||
list is rewritten, in the component's own file, so everything else the
|
||||
manifest drives is left alone.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import re
|
||||
from collections import OrderedDict
|
||||
from pathlib import Path
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeFile, NativeSystem, Report
|
||||
|
||||
# retrobios slug -> RetroDECK system ID (reverse of scraper SYSTEM_SLUG_MAP)
|
||||
_REVERSE_SLUG: dict[str, str] = {
|
||||
"nintendo-nes": "nes",
|
||||
"nintendo-snes": "snes",
|
||||
"nintendo-64": "n64",
|
||||
"nintendo-64dd": "n64dd",
|
||||
"nintendo-gamecube": "gc",
|
||||
"nintendo-wii": "wii",
|
||||
"nintendo-wii-u": "wiiu",
|
||||
"nintendo-switch": "switch",
|
||||
"nintendo-gb": "gb",
|
||||
"nintendo-gbc": "gbc",
|
||||
"nintendo-gba": "gba",
|
||||
"nintendo-ds": "nds",
|
||||
"nintendo-3ds": "3ds",
|
||||
"nintendo-fds": "fds",
|
||||
"nintendo-sgb": "sgb",
|
||||
"nintendo-virtual-boy": "virtualboy",
|
||||
"nintendo-pokemon-mini": "pokemini",
|
||||
"sony-playstation": "psx",
|
||||
"sony-playstation-2": "ps2",
|
||||
"sony-playstation-3": "ps3",
|
||||
"sony-psp": "psp",
|
||||
"sony-psvita": "psvita",
|
||||
"sega-mega-drive": "megadrive",
|
||||
"sega-mega-cd": "megacd",
|
||||
"sega-saturn": "saturn",
|
||||
"sega-dreamcast": "dreamcast",
|
||||
"sega-dreamcast-arcade": "naomi",
|
||||
"sega-game-gear": "gamegear",
|
||||
"sega-master-system": "mastersystem",
|
||||
"nec-pc-engine": "pcengine",
|
||||
"nec-pc-fx": "pcfx",
|
||||
"nec-pc-98": "pc98",
|
||||
"nec-pc-88": "pc88",
|
||||
"3do": "3do",
|
||||
"amstrad-cpc": "amstradcpc",
|
||||
"arcade": "arcade",
|
||||
"atari-400-800": "atari800",
|
||||
"atari-5200": "atari5200",
|
||||
"atari-7800": "atari7800",
|
||||
"atari-jaguar": "atarijaguar",
|
||||
"atari-lynx": "atarilynx",
|
||||
"atari-st": "atarist",
|
||||
"commodore-c64": "c64",
|
||||
"commodore-amiga": "amiga",
|
||||
"philips-cdi": "cdimono1",
|
||||
"fairchild-channel-f": "channelf",
|
||||
"coleco-colecovision": "colecovision",
|
||||
"mattel-intellivision": "intellivision",
|
||||
"microsoft-msx": "msx",
|
||||
"microsoft-xbox": "xbox",
|
||||
"doom": "doom",
|
||||
"j2me": "j2me",
|
||||
"apple-macintosh-ii": "macintosh",
|
||||
"apple-ii": "apple2",
|
||||
"apple-iigs": "apple2gs",
|
||||
"enterprise-64-128": "enterprise",
|
||||
"tiger-game-com": "gamecom",
|
||||
"hartung-game-master": "gmaster",
|
||||
"epoch-scv": "scv",
|
||||
"watara-supervision": "supervision",
|
||||
"bandai-wonderswan": "wonderswan",
|
||||
"snk-neogeo-cd": "neogeocd",
|
||||
"tandy-coco": "coco",
|
||||
"tandy-trs-80": "trs80",
|
||||
"dragon-32-64": "dragon",
|
||||
"pico8": "pico8",
|
||||
"wolfenstein-3d": "wolfenstein",
|
||||
"sinclair-zx-spectrum": "zxspectrum",
|
||||
}
|
||||
|
||||
|
||||
def _dest_to_path_token(destination: str) -> str:
|
||||
"""Convert a truth destination path to a RetroDECK path token."""
|
||||
if destination.startswith("roms/"):
|
||||
return "$roms_path/" + destination.removeprefix("roms/")
|
||||
if destination.startswith("bios/"):
|
||||
return "$bios_path/" + destination.removeprefix("bios/")
|
||||
# Default: bios path
|
||||
return "$bios_path/" + destination
|
||||
COMPONENTS_REPO = "RetroDECK/components"
|
||||
COMPONENTS_BRANCH = "main"
|
||||
RAW_BASE = f"https://raw.githubusercontent.com/{COMPONENTS_REPO}/{COMPONENTS_BRANCH}"
|
||||
MANIFEST = "component_manifest.json"
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Export truth data to RetroDECK component_manifest.json format."""
|
||||
"""Write RetroDECK's component manifests, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "retrodeck"
|
||||
|
||||
def export(
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return MANIFEST
|
||||
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"md5", "sha256", "required"})
|
||||
|
||||
@staticmethod
|
||||
def needs_original() -> bool:
|
||||
# A manifest is mostly presets and launch configuration; rebuilding
|
||||
# one from BIOS data alone would throw the component away.
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def component_url(component: str) -> str:
|
||||
return f"{RAW_BASE}/{component}/{MANIFEST}"
|
||||
|
||||
def components(self, systems: dict[str, NativeSystem]) -> list[str]:
|
||||
"""Components the corrected data touches."""
|
||||
found: set[str] = set()
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
component = str(fe.native("component", ""))
|
||||
if component:
|
||||
found.add(component)
|
||||
return sorted(found)
|
||||
|
||||
@staticmethod
|
||||
def _entry(fe: NativeFile) -> OrderedDict:
|
||||
entry: OrderedDict[str, object] = OrderedDict()
|
||||
entry["filename"] = fe.name
|
||||
md5 = ",".join(fe.hashes("md5"))
|
||||
if md5:
|
||||
entry["md5"] = md5
|
||||
sha256 = fe.hash("sha256")
|
||||
if sha256:
|
||||
entry["sha256"] = sha256
|
||||
entry["system"] = fe.native_system
|
||||
description = fe.native("description", "")
|
||||
if description:
|
||||
entry["description"] = str(description)
|
||||
# RetroDECK words the requirement in prose ("Required", "At least one
|
||||
# BIOS file required"), so the platform's own wording is kept and a
|
||||
# boolean is only rendered when there is none to keep.
|
||||
label = fe.native("required_label", "")
|
||||
if label:
|
||||
entry["required"] = str(label)
|
||||
elif fe.required:
|
||||
entry["required"] = "Required"
|
||||
destination = fe.destination
|
||||
if destination and destination not in (fe.name, f"bios/{fe.name}"):
|
||||
directory = destination.rsplit("/", 1)[0]
|
||||
entry["paths"] = "$bios_path/" + directory.removeprefix("bios/")
|
||||
return entry
|
||||
|
||||
def _by_component(
|
||||
self, systems: dict[str, NativeSystem]
|
||||
) -> dict[str, list[NativeFile]]:
|
||||
grouped: dict[str, list[NativeFile]] = {}
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
component = str(fe.native("component", ""))
|
||||
if component:
|
||||
grouped.setdefault(component, []).append(fe)
|
||||
return grouped
|
||||
|
||||
@staticmethod
|
||||
def _bios_holder(component_value: dict) -> tuple[dict, str] | None:
|
||||
"""Where in a manifest the BIOS list lives, if it has one."""
|
||||
if "bios" in component_value:
|
||||
return component_value, "bios"
|
||||
for key in ("preset_actions", "cores"):
|
||||
nested = component_value.get(key)
|
||||
if isinstance(nested, dict) and "bios" in nested:
|
||||
return nested, "bios"
|
||||
return None
|
||||
|
||||
def render(
|
||||
self,
|
||||
truth_data: dict,
|
||||
output_path: str,
|
||||
scraped_data: dict | None = None,
|
||||
) -> None:
|
||||
native_map: dict[str, str] = {}
|
||||
if scraped_data:
|
||||
for sys_id, sys_data in scraped_data.get("systems", {}).items():
|
||||
nid = sys_data.get("native_id")
|
||||
if nid:
|
||||
native_map[sys_id] = nid
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
grouped = self._by_component(systems)
|
||||
produced: dict[str, str] = {}
|
||||
|
||||
manifest: OrderedDict[str, dict] = OrderedDict()
|
||||
|
||||
systems = truth_data.get("systems", {})
|
||||
for sys_id in sorted(systems):
|
||||
sys_data = systems[sys_id]
|
||||
files = sys_data.get("files", [])
|
||||
if not files:
|
||||
for component, files in sorted(grouped.items()):
|
||||
path = f"{component}/{MANIFEST}"
|
||||
original = originals.get(path)
|
||||
if not original:
|
||||
continue
|
||||
try:
|
||||
manifest = json.loads(original, object_pairs_hook=OrderedDict)
|
||||
except json.JSONDecodeError:
|
||||
continue
|
||||
|
||||
native_id = native_map.get(sys_id, _REVERSE_SLUG.get(sys_id, sys_id))
|
||||
|
||||
bios_entries: list[OrderedDict] = []
|
||||
for fe in files:
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
entries = [self._entry(fe) for fe in files]
|
||||
for component_value in manifest.values():
|
||||
if not isinstance(component_value, dict):
|
||||
continue
|
||||
|
||||
dest = self._dest(fe)
|
||||
path_token = _dest_to_path_token(dest)
|
||||
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = ",".join(m for m in md5 if m)
|
||||
|
||||
required = fe.get("required", True)
|
||||
|
||||
entry: OrderedDict[str, object] = OrderedDict()
|
||||
entry["filename"] = name
|
||||
if md5:
|
||||
# Validate MD5 entries
|
||||
parts = [
|
||||
m.strip().lower()
|
||||
for m in str(md5).split(",")
|
||||
if re.fullmatch(r"[0-9a-f]{32}", m.strip())
|
||||
]
|
||||
if parts:
|
||||
entry["md5"] = ",".join(parts) if len(parts) > 1 else parts[0]
|
||||
entry["paths"] = path_token
|
||||
entry["required"] = required
|
||||
|
||||
system_val = native_id
|
||||
entry["system"] = system_val
|
||||
|
||||
bios_entries.append(entry)
|
||||
|
||||
if bios_entries:
|
||||
if native_id in manifest:
|
||||
# Merge into existing component (multiple truth systems
|
||||
# may map to the same native ID)
|
||||
existing_names = {
|
||||
e["filename"] for e in manifest[native_id]["bios"]
|
||||
}
|
||||
for entry in bios_entries:
|
||||
if entry["filename"] not in existing_names:
|
||||
manifest[native_id]["bios"].append(entry)
|
||||
holder = self._bios_holder(component_value)
|
||||
if holder is None:
|
||||
component_value["bios"] = entries
|
||||
else:
|
||||
component = OrderedDict()
|
||||
component["system"] = native_id
|
||||
component["bios"] = bios_entries
|
||||
manifest[native_id] = component
|
||||
container, key = holder
|
||||
container[key] = entries
|
||||
break
|
||||
|
||||
Path(output_path).write_text(
|
||||
json.dumps(manifest, indent=2, ensure_ascii=False) + "\n",
|
||||
encoding="utf-8",
|
||||
)
|
||||
produced[path] = json.dumps(manifest, indent=2, ensure_ascii=False) + "\n"
|
||||
|
||||
def validate(self, truth_data: dict, output_path: str) -> list[str]:
|
||||
data = json.loads(Path(output_path).read_text(encoding="utf-8"))
|
||||
|
||||
exported_names: set[str] = set()
|
||||
for comp_data in data.values():
|
||||
bios = comp_data.get("bios", [])
|
||||
if isinstance(bios, list):
|
||||
for entry in bios:
|
||||
fn = entry.get("filename", "")
|
||||
if fn:
|
||||
exported_names.add(fn)
|
||||
return produced
|
||||
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
issues: list[str] = []
|
||||
for sys_data in truth_data.get("systems", {}).values():
|
||||
for fe in sys_data.get("files", []):
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
grouped = self._by_component(systems)
|
||||
|
||||
for component, files in grouped.items():
|
||||
path = f"{component}/{MANIFEST}"
|
||||
if path not in produced:
|
||||
issues.append(f"manifest not written: {path}")
|
||||
continue
|
||||
try:
|
||||
manifest = json.loads(produced[path])
|
||||
except json.JSONDecodeError as exc:
|
||||
issues.append(f"{path} does not parse: {exc}")
|
||||
continue
|
||||
|
||||
declared: set[str] = set()
|
||||
for component_value in manifest.values():
|
||||
if not isinstance(component_value, dict):
|
||||
continue
|
||||
if name not in exported_names:
|
||||
issues.append(f"missing: {name}")
|
||||
holder = self._bios_holder(component_value)
|
||||
if holder is None:
|
||||
continue
|
||||
container, key = holder
|
||||
for entry in container[key]:
|
||||
declared.add(entry.get("filename", ""))
|
||||
if not component_value.get("name") and not component_value.get(
|
||||
"system"
|
||||
):
|
||||
issues.append(f"{path}: the component lost its identity")
|
||||
|
||||
for fe in files:
|
||||
if fe.name not in declared:
|
||||
issues.append(f"absent from {path}: {fe.name}")
|
||||
return issues
|
||||
@@ -1,17 +1,312 @@
|
||||
"""Exporter for RetroPie (System.dat format, same as RetroArch).
|
||||
"""Exporter for RetroPie's scriptmodules.
|
||||
|
||||
RetroPie inherits RetroArch cores and uses the same System.dat format.
|
||||
Delegates to systemdat_exporter for export and validation.
|
||||
RetroPie ships no BIOS list. What it maintains is one shell script per
|
||||
package, and the BIOS files a package needs are named in its
|
||||
`rp_module_help` string, in prose a person reads before copying files.
|
||||
platforms.cfg carries extensions and full names only.
|
||||
|
||||
So the correctable unit is that sentence, and the correction is a file name
|
||||
missing from it. A name RetroPie already writes is never removed, and no
|
||||
sentence is invented where a maintainer wrote none: a package we would have
|
||||
to document from scratch is reported, not drafted.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from .systemdat_exporter import Exporter as SystemDatExporter
|
||||
import io
|
||||
import re
|
||||
import sys
|
||||
import tarfile
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
from common import load_emulator_profiles
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeFile, NativeSystem, Report, search_order
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://codeload.github.com/RetroPie/RetroPie-Setup/tar.gz/refs/heads/master"
|
||||
)
|
||||
ARCHIVE = "RetroPie-Setup.tar.gz"
|
||||
# Resolved from the file rather than the working directory: the profiles
|
||||
# are what say which files a package needs.
|
||||
EMULATORS_DIR = Path(__file__).resolve().parents[2] / "emulators"
|
||||
|
||||
# The longest list RetroPie writes is six names (lr-atari800). Past that the
|
||||
# sentence stops being something a reader uses, so the names are reported
|
||||
# instead of appended.
|
||||
MAX_NAMES = 6
|
||||
|
||||
_MODULE_ID = re.compile(r'rp_module_id="([^"]+)"')
|
||||
_MODULE_HELP = re.compile(r'rp_module_help="((?:[^"\\]|\\.)*)"')
|
||||
_FILENAME = re.compile(r"[A-Za-z0-9][\w.+-]*\.[A-Za-z0-9]{1,5}\b")
|
||||
# RetroPie words the instruction several ways ("Copy the required BIOS files
|
||||
# a.bin and b.bin to $biosdir", "The Sega CD requires the BIOS files a.bin,
|
||||
# b.bin copied to $biosdir"), so the clause is found by the word BIOS rather
|
||||
# than by a sentence template, and the names in it are the list to extend.
|
||||
_BIOS_CLAUSE = re.compile(r"BIOS\b.*?(?=\\n|$)", re.DOTALL)
|
||||
_BIOS_NOUN = re.compile(r"BIOS (files?)\b")
|
||||
|
||||
|
||||
class Exporter(SystemDatExporter):
|
||||
"""Export truth data to RetroPie System.dat format."""
|
||||
class Exporter(BaseExporter):
|
||||
"""Write RetroPie's scriptmodules, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "retropie"
|
||||
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "scriptmodules"
|
||||
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
# The help names files; it states no hash and no requirement flag.
|
||||
return frozenset({"name"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {ARCHIVE: SOURCE_URL}
|
||||
|
||||
@staticmethod
|
||||
def needs_original() -> bool:
|
||||
# The declaration is a sentence inside a shell script.
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def may_write_nothing() -> bool:
|
||||
# Nothing to correct is a result, not a failure.
|
||||
return True
|
||||
|
||||
@staticmethod
|
||||
def unpack(raw: bytes) -> dict[str, str]:
|
||||
"""Keep the scriptmodules that say anything about BIOS files."""
|
||||
found: dict[str, str] = {}
|
||||
with tarfile.open(fileobj=io.BytesIO(raw), mode="r:gz") as archive:
|
||||
for member in archive:
|
||||
if not member.isfile() or not member.name.endswith(".sh"):
|
||||
continue
|
||||
relative = member.name.split("/", 1)[-1]
|
||||
if not relative.startswith("scriptmodules/"):
|
||||
continue
|
||||
handle = archive.extractfile(member)
|
||||
if handle is None:
|
||||
continue
|
||||
text = handle.read().decode("utf-8", errors="replace")
|
||||
if "BIOS" in text:
|
||||
found[relative] = text
|
||||
return found
|
||||
|
||||
def _core_index(self) -> dict[str, str]:
|
||||
"""Module id (without its lr- prefix) to the profile it stands for."""
|
||||
index: dict[str, str] = {}
|
||||
for key, profile in load_emulator_profiles(str(EMULATORS_DIR)).items():
|
||||
index[key.replace("-", "_").lower()] = key
|
||||
for name in profile.get("cores", []) or []:
|
||||
index[str(name).replace("-", "_").lower()] = key
|
||||
return index
|
||||
|
||||
@staticmethod
|
||||
def _files_by_core(systems: dict[str, NativeSystem]) -> dict[str, list[NativeFile]]:
|
||||
"""Which files each core asks for, as the truth read its source."""
|
||||
grouped: dict[str, list[NativeFile]] = {}
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
for core in (fe.truth or {}).get("_cores", []):
|
||||
grouped.setdefault(str(core), []).append(fe)
|
||||
return grouped
|
||||
|
||||
@staticmethod
|
||||
def _names_in(text: str) -> list[re.Match[str]]:
|
||||
"""File names in a fragment, skipping what only looks like one.
|
||||
|
||||
A token preceded by a dot is the tail of a ROM extension list
|
||||
(.atr.gz), and one preceded by a separator is part of a path
|
||||
($biosdir/Machines/COL/coleco.rom): neither is a name in a list.
|
||||
"""
|
||||
found: list[re.Match[str]] = []
|
||||
for match in _FILENAME.finditer(text):
|
||||
before = text[match.start() - 1 : match.start()]
|
||||
if before in (".", "/", "\\"):
|
||||
continue
|
||||
found.append(match)
|
||||
return found
|
||||
|
||||
@classmethod
|
||||
def _listed(cls, help_text: str) -> set[str]:
|
||||
"""File names the help already writes, wherever in the string."""
|
||||
plain = help_text.replace("\\n", " ")
|
||||
return {match.group(0).lower() for match in cls._names_in(plain)}
|
||||
|
||||
@classmethod
|
||||
def _insertion_point(cls, help_text: str) -> int | None:
|
||||
"""Where a name joins the list, or None when there is no list."""
|
||||
clause = _BIOS_CLAUSE.search(help_text)
|
||||
if clause is None:
|
||||
return None
|
||||
names = cls._names_in(clause.group(0))
|
||||
return clause.start() + names[-1].end() if names else None
|
||||
|
||||
@staticmethod
|
||||
def _in_search_order(candidates: list[NativeFile]) -> list[str]:
|
||||
"""The names in the order the code looks for them, best first.
|
||||
|
||||
Someone reading the sentence copies the files in the order it
|
||||
gives, so the one the emulator prefers is named first. The model
|
||||
already holds them in that order; this only removes the duplicates
|
||||
a name can pick up from several cores.
|
||||
"""
|
||||
names: list[str] = []
|
||||
for fe in search_order(candidates):
|
||||
if fe.name not in names:
|
||||
names.append(fe.name)
|
||||
return names
|
||||
|
||||
@staticmethod
|
||||
def _join(names: list[str]) -> str:
|
||||
"""RetroPie's own idiom: a, b and c."""
|
||||
if len(names) == 1:
|
||||
return names[0]
|
||||
return ", ".join(names[:-1]) + " and " + names[-1]
|
||||
|
||||
def modules(self, originals: dict[str, str]) -> dict[str, tuple[str, str]]:
|
||||
"""Module id and help string of every script that mentions BIOS."""
|
||||
found: dict[str, tuple[str, str]] = {}
|
||||
for relative, text in originals.items():
|
||||
if not relative.startswith("scriptmodules/"):
|
||||
continue
|
||||
module = _MODULE_ID.search(text)
|
||||
help_text = _MODULE_HELP.search(text)
|
||||
if not module or not help_text or "BIOS" not in help_text.group(1):
|
||||
continue
|
||||
found[relative] = (module.group(1), help_text.group(1))
|
||||
return found
|
||||
|
||||
def render(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
modules = self.modules(originals)
|
||||
if not modules:
|
||||
raise ValueError(
|
||||
"the scriptmodules cannot be written without RetroPie's own "
|
||||
"repository: the BIOS list is a sentence inside a shell script"
|
||||
)
|
||||
|
||||
index = self._core_index()
|
||||
by_core = self._files_by_core(systems)
|
||||
# Aliases count as names we know: RetroPie writes dc_flash.bin where
|
||||
# flycast's profile files it as an alias of another primary.
|
||||
known: set[str] = set()
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
known.add(fe.name.lower())
|
||||
known.update(
|
||||
str(a).lower() for a in (fe.native("aliases", []) or [])
|
||||
)
|
||||
self._skipped: dict[str, list[str]] = {}
|
||||
produced: dict[str, str] = {}
|
||||
|
||||
def skip(reason: str, module: str) -> None:
|
||||
self._skipped.setdefault(reason, []).append(module)
|
||||
|
||||
for relative, (module_id, help_text) in sorted(modules.items()):
|
||||
core = index.get(module_id.removeprefix("lr-").replace("-", "_").lower())
|
||||
files = by_core.get(core or "", [])
|
||||
if not files:
|
||||
skip("no core of ours packages it", module_id)
|
||||
continue
|
||||
|
||||
listed = self._listed(help_text)
|
||||
unknown = sorted(listed - known)
|
||||
if unknown:
|
||||
skip(
|
||||
f"names a file we have never seen ({', '.join(unknown)})",
|
||||
module_id,
|
||||
)
|
||||
|
||||
# A file already listed under one of its other names is not
|
||||
# missing: proposing the primary would name the same bytes twice.
|
||||
candidates = [
|
||||
fe
|
||||
for fe in files
|
||||
if fe.required
|
||||
and not (
|
||||
{fe.name.lower()}
|
||||
| {str(a).lower() for a in (fe.native("aliases", []) or [])}
|
||||
)
|
||||
& listed
|
||||
]
|
||||
missing = self._in_search_order(candidates)
|
||||
if not missing:
|
||||
continue
|
||||
|
||||
insert_at = self._insertion_point(help_text)
|
||||
if insert_at is None:
|
||||
# No enumeration to extend, and a sentence we would have to
|
||||
# write ourselves is a documentation change, not a correction.
|
||||
skip("names no file to extend", module_id)
|
||||
continue
|
||||
|
||||
if len(listed) + len(missing) > MAX_NAMES:
|
||||
skip("more names than the help enumerates", module_id)
|
||||
continue
|
||||
|
||||
new_help = (
|
||||
help_text[:insert_at]
|
||||
+ ", "
|
||||
+ self._join(missing)
|
||||
+ help_text[insert_at:]
|
||||
)
|
||||
if len(listed) + len(missing) > 1:
|
||||
new_help = _BIOS_NOUN.sub("BIOS files", new_help, count=1)
|
||||
produced[relative] = originals[relative].replace(
|
||||
f'rp_module_help="{help_text}"',
|
||||
f'rp_module_help="{new_help}"',
|
||||
1,
|
||||
)
|
||||
|
||||
return produced
|
||||
|
||||
def outcome(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> str:
|
||||
"""Packages are the unit here, not file entries."""
|
||||
skipped: dict[str, list[str]] = getattr(self, "_skipped", {})
|
||||
parts = [f"{len(produced)} packages corrected"]
|
||||
for reason, modules in sorted(skipped.items()):
|
||||
shown = ", ".join(sorted(modules)[:4])
|
||||
if len(modules) > 4:
|
||||
shown += f" and {len(modules) - 4} more"
|
||||
parts.append(f"{len(modules)} {reason} ({shown})")
|
||||
return "; ".join(parts)
|
||||
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
issues: list[str] = []
|
||||
for relative, text in produced.items():
|
||||
module = _MODULE_ID.search(text)
|
||||
help_text = _MODULE_HELP.search(text)
|
||||
if not module:
|
||||
issues.append(f"{relative}: the module lost its id")
|
||||
if not help_text:
|
||||
issues.append(f"{relative}: the help string is not closed")
|
||||
continue
|
||||
if "BIOS" not in help_text.group(1):
|
||||
issues.append(f"{relative}: the BIOS sentence is gone")
|
||||
names = self._listed(help_text.group(1))
|
||||
if len(names) > MAX_NAMES:
|
||||
issues.append(
|
||||
f"{relative}: {len(names)} names, past what the help enumerates"
|
||||
)
|
||||
return issues
|
||||
@@ -0,0 +1,30 @@
|
||||
"""Exporter for ROCKNIX's rocknix-systems.
|
||||
|
||||
Same shape as Batocera's script: a systems mapping inside a checker ROCKNIX
|
||||
runs, so only the mapping is rewritten.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from .batocera_exporter import Exporter as BatoceraExporter
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://raw.githubusercontent.com/ROCKNIX/distribution/next/projects/ROCKNIX"
|
||||
"/packages/rocknix/sources/scripts/rocknix-systems"
|
||||
)
|
||||
|
||||
|
||||
class Exporter(BatoceraExporter):
|
||||
"""Write ROCKNIX's rocknix-systems, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "rocknix"
|
||||
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "rocknix-systems"
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"rocknix-systems": SOURCE_URL}
|
||||
+105
-132
@@ -1,160 +1,133 @@
|
||||
"""Exporter for RomM known_bios_files.json format.
|
||||
"""Exporter for RomM's known_bios_files.json.
|
||||
|
||||
Produces JSON matching the exact format of
|
||||
rommapp/romm/backend/models/fixtures/known_bios_files.json:
|
||||
- Keys are "igdb_slug:filename"
|
||||
- Values contain size, crc, md5, sha1 (all optional but at least one hash)
|
||||
- Hashes are lowercase hex strings
|
||||
- Size is an integer
|
||||
Keys are "<igdb slug>:<filename>". RomM verifies a firmware file with
|
||||
`file_size_bytes == int(entry.get("size", 0))` and then one hash among md5,
|
||||
sha1 and crc, so an entry without a size can never match and an entry
|
||||
without a hash can never match either. Neither is written.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import sys
|
||||
from collections import OrderedDict
|
||||
from pathlib import Path
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
|
||||
# retrobios slug -> IGDB slug (reverse of scraper SLUG_MAP)
|
||||
_REVERSE_SLUG: dict[str, str] = {
|
||||
"3do": "3do",
|
||||
"nintendo-64dd": "64dd",
|
||||
"amstrad-cpc": "acpc",
|
||||
"commodore-amiga": "amiga",
|
||||
"arcade": "arcade",
|
||||
"atari-st": "atari-st",
|
||||
"atari-5200": "atari5200",
|
||||
"atari-7800": "atari7800",
|
||||
"atari-400-800": "atari8bit",
|
||||
"coleco-colecovision": "colecovision",
|
||||
"sega-dreamcast": "dc",
|
||||
"doom": "doom",
|
||||
"enterprise-64-128": "enterprise",
|
||||
"fairchild-channel-f": "fairchild-channel-f",
|
||||
"nintendo-fds": "fds",
|
||||
"sega-game-gear": "gamegear",
|
||||
"nintendo-gb": "gb",
|
||||
"nintendo-gba": "gba",
|
||||
"nintendo-gbc": "gbc",
|
||||
"sega-mega-drive": "genesis",
|
||||
"mattel-intellivision": "intellivision",
|
||||
"j2me": "j2me",
|
||||
"atari-lynx": "lynx",
|
||||
"apple-macintosh-ii": "mac",
|
||||
"microsoft-msx": "msx",
|
||||
"nintendo-ds": "nds",
|
||||
"snk-neogeo-cd": "neo-geo-cd",
|
||||
"nintendo-nes": "nes",
|
||||
"nintendo-gamecube": "ngc",
|
||||
"magnavox-odyssey2": "odyssey-2-slash-videopac-g7000",
|
||||
"nec-pc-98": "pc-9800-series",
|
||||
"nec-pc-fx": "pc-fx",
|
||||
"nintendo-pokemon-mini": "pokemon-mini",
|
||||
"sony-playstation-2": "ps2",
|
||||
"sony-psp": "psp",
|
||||
"sony-playstation": "psx",
|
||||
"nintendo-satellaview": "satellaview",
|
||||
"sega-saturn": "saturn",
|
||||
"scummvm": "scummvm",
|
||||
"sega-mega-cd": "segacd",
|
||||
"sharp-x68000": "sharp-x68000",
|
||||
"sega-master-system": "sms",
|
||||
"nintendo-snes": "snes",
|
||||
"nintendo-sufami-turbo": "sufami-turbo",
|
||||
"nintendo-sgb": "super-gb",
|
||||
"nec-pc-engine": "tg16",
|
||||
"videoton-tvc": "tvc",
|
||||
"philips-videopac": "videopac-g7400",
|
||||
"wolfenstein-3d": "wolfenstein",
|
||||
"sharp-x1": "x1",
|
||||
"microsoft-xbox": "xbox",
|
||||
"sinclair-zx-spectrum": "zxs",
|
||||
}
|
||||
from scraper.romm_scraper import SLUG_MAP
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeFile, NativeSystem, Report
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://raw.githubusercontent.com/rommapp/romm/master/backend/models"
|
||||
"/fixtures/known_bios_files.json"
|
||||
)
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Export truth data to RomM known_bios_files.json format."""
|
||||
"""Write RomM's known_bios_files.json, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "romm"
|
||||
|
||||
def export(
|
||||
self,
|
||||
truth_data: dict,
|
||||
output_path: str,
|
||||
scraped_data: dict | None = None,
|
||||
) -> None:
|
||||
native_map: dict[str, str] = {}
|
||||
if scraped_data:
|
||||
for sys_id, sys_data in scraped_data.get("systems", {}).items():
|
||||
nid = sys_data.get("native_id")
|
||||
if nid:
|
||||
native_map[sys_id] = nid
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "known_bios_files.json"
|
||||
|
||||
output: OrderedDict[str, dict] = OrderedDict()
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"size", "crc32", "md5", "sha1"})
|
||||
|
||||
systems = truth_data.get("systems", {})
|
||||
for sys_id in sorted(systems):
|
||||
sys_data = systems[sys_id]
|
||||
files = sys_data.get("files", [])
|
||||
if not files:
|
||||
continue
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"known_bios_files.json": SOURCE_URL}
|
||||
|
||||
igdb_slug = native_map.get(sys_id, _REVERSE_SLUG.get(sys_id, sys_id))
|
||||
|
||||
for fe in files:
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
continue
|
||||
|
||||
key = f"{igdb_slug}:{name}"
|
||||
|
||||
entry: OrderedDict[str, object] = OrderedDict()
|
||||
|
||||
size = fe.get("size")
|
||||
if size is not None:
|
||||
entry["size"] = int(size)
|
||||
|
||||
crc = fe.get("crc32", "")
|
||||
if crc:
|
||||
entry["crc"] = str(crc).strip().lower()
|
||||
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = md5[0] if md5 else ""
|
||||
if md5:
|
||||
entry["md5"] = str(md5).strip().lower()
|
||||
|
||||
sha1 = fe.get("sha1", "")
|
||||
if isinstance(sha1, list):
|
||||
sha1 = sha1[0] if sha1 else ""
|
||||
if sha1:
|
||||
entry["sha1"] = str(sha1).strip().lower()
|
||||
|
||||
output[key] = entry
|
||||
|
||||
Path(output_path).write_text(
|
||||
json.dumps(output, indent=2, ensure_ascii=False) + "\n",
|
||||
encoding="utf-8",
|
||||
@staticmethod
|
||||
def _verifiable(fe: NativeFile) -> bool:
|
||||
return bool(fe.size()) and any(
|
||||
fe.hash(h) for h in ("md5", "sha1", "crc32")
|
||||
)
|
||||
|
||||
def validate(self, truth_data: dict, output_path: str) -> list[str]:
|
||||
data = json.loads(Path(output_path).read_text(encoding="utf-8"))
|
||||
@staticmethod
|
||||
def _known_platform(native_id: str) -> bool:
|
||||
"""RomM keys by IGDB platform slug, and only looks up its own.
|
||||
|
||||
exported_names: set[str] = set()
|
||||
for key in data:
|
||||
if ":" in key:
|
||||
_, filename = key.split(":", 1)
|
||||
exported_names.add(filename)
|
||||
A key spelled with one of our slugs (capcom-cps3, snk-neogeo-mvs)
|
||||
matches nothing on their side, so it is reported rather than
|
||||
written.
|
||||
"""
|
||||
return native_id in SLUG_MAP
|
||||
|
||||
@classmethod
|
||||
def writable(cls, fe: NativeFile, require: str = "") -> bool:
|
||||
"""What RomM already ships stays; the conditions gate additions.
|
||||
|
||||
An entry of theirs that could never verify is still theirs, and the
|
||||
round trip is not the place to decide otherwise.
|
||||
"""
|
||||
if fe.platform is not None:
|
||||
return True
|
||||
return cls._verifiable(fe) and cls._known_platform(fe.native_system)
|
||||
|
||||
def render(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
output: OrderedDict[str, dict] = OrderedDict()
|
||||
|
||||
for system in sorted(systems.values(), key=lambda s: s.native_id):
|
||||
for fe in sorted(system.files, key=lambda f: f.name):
|
||||
if not self.writable(fe):
|
||||
continue
|
||||
entry: OrderedDict[str, str] = OrderedDict()
|
||||
# The fixture states every value as a string, size included.
|
||||
entry["size"] = str(fe.size())
|
||||
crc = fe.hash("crc32")
|
||||
if crc:
|
||||
entry["crc"] = crc
|
||||
md5 = fe.hash("md5")
|
||||
if md5:
|
||||
entry["md5"] = md5
|
||||
sha1 = fe.hash("sha1")
|
||||
if sha1:
|
||||
entry["sha1"] = sha1
|
||||
output[f"{system.native_id}:{fe.name}"] = entry
|
||||
|
||||
text = json.dumps(output, indent=2, ensure_ascii=False) + "\n"
|
||||
return {self.native_filename(): text}
|
||||
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
try:
|
||||
data = json.loads(produced[self.native_filename()])
|
||||
except json.JSONDecodeError as exc:
|
||||
return [f"the JSON does not parse: {exc}"]
|
||||
|
||||
issues: list[str] = []
|
||||
for sys_data in truth_data.get("systems", {}).values():
|
||||
for fe in sys_data.get("files", []):
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
for key, entry in data.items():
|
||||
slug = key.split(":", 1)[0] if ":" in key else ""
|
||||
if not slug:
|
||||
issues.append(f"key without a platform slug: {key}")
|
||||
elif not self._known_platform(slug):
|
||||
issues.append(f"platform slug RomM does not know: {slug}")
|
||||
if not entry.get("size"):
|
||||
issues.append(f"entry without a size, never verifiable: {key}")
|
||||
if not any(entry.get(h) for h in ("md5", "sha1", "crc")):
|
||||
issues.append(f"entry without a hash, never verifiable: {key}")
|
||||
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
if not self.writable(fe):
|
||||
continue
|
||||
if name not in exported_names:
|
||||
issues.append(f"missing: {name}")
|
||||
if f"{system.native_id}:{fe.name}" not in data:
|
||||
issues.append(f"absent: {system.native_id}/{fe.name}")
|
||||
return issues
|
||||
@@ -1,7 +1,7 @@
|
||||
"""Exporter for libretro System.dat (clrmamepro DAT format).
|
||||
|
||||
Produces a single 'game' block with all ROMs grouped by system,
|
||||
matching the exact format of libretro-database/dat/System.dat.
|
||||
One 'game' block, systems separated by a comment line carrying the name
|
||||
libretro gives them, matching libretro-database/dat/System.dat.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -14,43 +14,71 @@ sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
from scraper.dat_parser import parse_dat
|
||||
|
||||
from .base_exporter import BaseExporter
|
||||
from .baseline import NativeSystem, Report
|
||||
|
||||
SOURCE_URL = (
|
||||
"https://raw.githubusercontent.com/libretro/libretro-database/master/dat/System.dat"
|
||||
)
|
||||
|
||||
|
||||
def _slug_to_native(slug: str) -> str:
|
||||
"""Convert a system slug to 'Manufacturer - Console' format."""
|
||||
parts = slug.split("-", 1)
|
||||
if len(parts) == 1:
|
||||
return parts[0].title()
|
||||
manufacturer = parts[0].replace("-", " ").title()
|
||||
console = parts[1].replace("-", " ").title()
|
||||
return f"{manufacturer} - {console}"
|
||||
def _quote(name: str) -> str:
|
||||
"""Quote a ROM name the way the original does: only when it must be."""
|
||||
return f'"{name}"' if any(c in name for c in ' ()') else name
|
||||
|
||||
|
||||
class Exporter(BaseExporter):
|
||||
"""Export truth data to libretro System.dat format."""
|
||||
"""Write libretro's System.dat, corrected."""
|
||||
|
||||
@staticmethod
|
||||
def platform_name() -> str:
|
||||
return "retroarch"
|
||||
|
||||
def export(
|
||||
self,
|
||||
truth_data: dict,
|
||||
output_path: str,
|
||||
scraped_data: dict | None = None,
|
||||
) -> None:
|
||||
native_map: dict[str, str] = {}
|
||||
if scraped_data:
|
||||
for sys_id, sys_data in scraped_data.get("systems", {}).items():
|
||||
nid = sys_data.get("native_id")
|
||||
if nid:
|
||||
native_map[sys_id] = nid
|
||||
@staticmethod
|
||||
def native_filename() -> str:
|
||||
return "System.dat"
|
||||
|
||||
@staticmethod
|
||||
def carries() -> frozenset[str]:
|
||||
return frozenset({"size", "crc32", "md5", "sha1"})
|
||||
|
||||
@staticmethod
|
||||
def native_sources() -> dict[str, str]:
|
||||
return {"System.dat": SOURCE_URL}
|
||||
|
||||
@classmethod
|
||||
def writable(cls, fe, require: str = "") -> bool:
|
||||
"""A clrmamepro rom line is a hash record, and the DAT has no other.
|
||||
|
||||
The original never states a rom it cannot hash, so neither do we.
|
||||
"""
|
||||
if fe.platform is not None:
|
||||
return True
|
||||
return any(fe.hash(h) for h in ("crc32", "md5", "sha1"))
|
||||
|
||||
@staticmethod
|
||||
def _rom_name(fe) -> str:
|
||||
"""The name the DAT gives a rom.
|
||||
|
||||
libretro writes the path for some entries (ep128emu/roms/cpc464.rom)
|
||||
and the bare name for others whose destination has a directory all
|
||||
the same (iplromco.dat, which lives under keropi/), so its own
|
||||
spelling is recorded rather than derived.
|
||||
"""
|
||||
declared = fe.native("native_path", "")
|
||||
return str(declared) if declared else fe.name
|
||||
|
||||
def _header(self, originals: dict[str, str], scraped: dict | None) -> list[str]:
|
||||
"""Reuse the original header verbatim when we have the original."""
|
||||
original = originals.get(self.native_filename(), "")
|
||||
if original:
|
||||
head, sep, _ = original.partition("\ngame (")
|
||||
if sep:
|
||||
return head.split("\n")
|
||||
|
||||
# Match exact header format of libretro-database/dat/System.dat
|
||||
version = ""
|
||||
if scraped_data:
|
||||
version = scraped_data.get("dat_version", scraped_data.get("version", ""))
|
||||
lines: list[str] = [
|
||||
if scraped:
|
||||
version = scraped.get("dat_version", scraped.get("version", ""))
|
||||
lines = [
|
||||
"clrmamepro (",
|
||||
'\tname "System"',
|
||||
'\tdescription "System"',
|
||||
@@ -61,73 +89,76 @@ class Exporter(BaseExporter):
|
||||
lines.extend(
|
||||
[
|
||||
'\tauthor "libretro"',
|
||||
'\thomepage "https://github.com/libretro/libretro-database/blob/master/dat/System.dat"',
|
||||
'\turl "https://raw.githubusercontent.com/libretro/libretro-database/master/dat/System.dat"',
|
||||
'\thomepage "https://github.com/libretro/libretro-database/blob/master'
|
||||
'/dat/System.dat"',
|
||||
'\turl "https://raw.githubusercontent.com/libretro/libretro-database'
|
||||
'/master/dat/System.dat"',
|
||||
")",
|
||||
"",
|
||||
"game (",
|
||||
'\tname "System"',
|
||||
'\tcomment "System"',
|
||||
]
|
||||
)
|
||||
return lines
|
||||
|
||||
systems = truth_data.get("systems", {})
|
||||
for sys_id in sorted(systems):
|
||||
sys_data = systems[sys_id]
|
||||
files = sys_data.get("files", [])
|
||||
if not files:
|
||||
continue
|
||||
|
||||
native_name = native_map.get(sys_id, _slug_to_native(sys_id))
|
||||
lines.append("")
|
||||
lines.append(f'\tcomment "{native_name}"')
|
||||
def render(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
report: Report,
|
||||
originals: dict[str, str],
|
||||
scraped: dict | None = None,
|
||||
) -> dict[str, str]:
|
||||
lines = self._header(originals, scraped)
|
||||
lines.extend(["game (", '\tname "System"', '\tcomment "System"'])
|
||||
|
||||
for system, files in sorted(
|
||||
self.exportable(systems), key=lambda pair: pair[0].native_id
|
||||
):
|
||||
rendered: list[str] = []
|
||||
for fe in files:
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
if not any(fe.hash(h) for h in ("crc32", "md5", "sha1")):
|
||||
continue
|
||||
|
||||
# Quote names with spaces or special chars (matching original format)
|
||||
needs_quote = " " in name or "(" in name or ")" in name
|
||||
name_str = f'"{name}"' if needs_quote else name
|
||||
rom_parts = [f"name {name_str}"]
|
||||
size = fe.get("size")
|
||||
parts = [f"name {_quote(self._rom_name(fe))}"]
|
||||
size = fe.size()
|
||||
if size:
|
||||
rom_parts.append(f"size {size}")
|
||||
crc = fe.get("crc32", "")
|
||||
parts.append(f"size {size}")
|
||||
crc = fe.hash("crc32")
|
||||
if crc:
|
||||
rom_parts.append(f"crc {str(crc).upper()}")
|
||||
md5 = fe.get("md5", "")
|
||||
if isinstance(md5, list):
|
||||
md5 = md5[0] if md5 else ""
|
||||
parts.append(f"crc {crc.upper()}")
|
||||
md5 = fe.hash("md5")
|
||||
if md5:
|
||||
rom_parts.append(f"md5 {md5}")
|
||||
sha1 = fe.get("sha1", "")
|
||||
if isinstance(sha1, list):
|
||||
sha1 = sha1[0] if sha1 else ""
|
||||
parts.append(f"md5 {md5}")
|
||||
sha1 = fe.hash("sha1")
|
||||
if sha1:
|
||||
rom_parts.append(f"sha1 {sha1}")
|
||||
parts.append(f"sha1 {sha1}")
|
||||
rendered.append(f"\trom ( {' '.join(parts)} )")
|
||||
|
||||
lines.append(f"\trom ( {' '.join(rom_parts)} )")
|
||||
if not rendered:
|
||||
continue
|
||||
lines.append("")
|
||||
# libretro's comment is the system name as the DAT spells it,
|
||||
# "Atari - 400-800". Prettifying it drops the separator.
|
||||
lines.append(f'\tcomment "{system.native_id}"')
|
||||
lines.extend(rendered)
|
||||
|
||||
lines.append(")")
|
||||
lines.append("")
|
||||
Path(output_path).write_text("\n".join(lines), encoding="utf-8")
|
||||
return {self.native_filename(): "\n".join(lines)}
|
||||
|
||||
def validate(self, truth_data: dict, output_path: str) -> list[str]:
|
||||
content = Path(output_path).read_text(encoding="utf-8")
|
||||
def validate(
|
||||
self,
|
||||
systems: dict[str, NativeSystem],
|
||||
produced: dict[str, str],
|
||||
) -> list[str]:
|
||||
content = produced[self.native_filename()]
|
||||
parsed = parse_dat(content)
|
||||
|
||||
exported_names: set[str] = set()
|
||||
for rom in parsed:
|
||||
exported_names.add(rom.name)
|
||||
exported = {rom.name for rom in parsed}
|
||||
|
||||
issues: list[str] = []
|
||||
for sys_data in truth_data.get("systems", {}).values():
|
||||
for fe in sys_data.get("files", []):
|
||||
name = fe.get("name", "")
|
||||
if name.startswith("_") or self._is_pattern(name):
|
||||
for system in systems.values():
|
||||
for fe in system.files:
|
||||
if not any(fe.hash(h) for h in ("crc32", "md5", "sha1")):
|
||||
continue
|
||||
if name not in exported_names:
|
||||
issues.append(f"missing: {name}")
|
||||
if self._rom_name(fe) not in exported:
|
||||
issues.append(f"absent from the DAT: {system.native_id}/{fe.name}")
|
||||
if not content.rstrip().endswith(")"):
|
||||
issues.append("the game block is not closed")
|
||||
return issues
|
||||
Reference in new issue
Block a user