fix: decide hash mismatch by native mode

A declared hash that the local dump contradicts is not one situation. An
existence platform never reads the bytes, so withholding the file lets an
upstream list error remove something the frontend would have loaded; a
hash platform would reject it, so shipping it is pointless.

The mode now decides, at every point that had an opinion: pack building,
core complement, emulator packs, manifests, conformance and
_intentional_hash_exclusion. verify.find_undeclared_files follows, since
verify and generate_pack must agree file for file.

Also here: resolution reports which evidence matched rather than a flat
"exact", a path or filename can no longer override a declared hash, and
safe_extract_zip treats a Windows backslash as the separator it is
instead of refusing the archive.
This commit is contained in:
Abdessamad Derraz committed 2026-08-10 13:36:03 +02:00
1 parent a5f549a1c6
commit e8ee8b0954
12 files changed
+1805 -269

No files matched your search

+298 -53
View File
@@ -10,6 +10,8 @@ import contextlib
import hashlib
import json
import os
import re
import stat
import tempfile
import urllib.error
import urllib.parse
@@ -421,6 +423,24 @@ def list_available_targets(
return result
HASH_EXACT_RESOLUTION_STATUSES = frozenset(
{
"sha1_exact",
"sha256_exact",
"crc32_exact",
"md5_exact",
"md5_composite_exact",
"zip_exact",
"data_dir_hash_exact",
}
)
def resolution_is_hash_exact(status: str) -> bool:
"""Whether *status* proves content identity with a declared hash."""
return status in HASH_EXACT_RESOLUTION_STATUSES
def resolve_local_file(
file_entry: dict,
db: dict,
@@ -439,8 +459,12 @@ def resolve_local_file(
disambiguate when multiple files share the same name. Matched against
the by_path_suffix index built from the repo's directory structure.
Returns (local_path, status) where status is one of:
exact, zip_exact, hash_mismatch, not_found.
Returns ``(local_path, status)``. Statuses describe the evidence used:
``sha1_exact``, ``sha256_exact``, ``crc32_exact``, ``md5_exact``,
``md5_composite_exact``, ``zip_exact``, ``path_exact``, ``name_exact``,
``hash_mismatch`` or ``not_found`` (plus documented fallback statuses).
A path or filename is never allowed to override a declared strong hash.
"""
sha1 = file_entry.get("sha1")
name = file_entry.get("name", "")
@@ -464,54 +488,100 @@ def resolve_local_file(
names_to_try.append(hint_base)
md5_list = parse_md5_list(file_entry.get("md5"))
sha1_candidates = [
str(value).strip().lower()
for value in (sha1 if isinstance(sha1, list) else [sha1] if sha1 else [])
if str(value).strip()
]
sha256_raw = file_entry.get("sha256")
sha256_values = sha256_raw if isinstance(sha256_raw, list) else [sha256_raw]
sha256_candidates = [
candidate.strip().lower()
for value in sha256_values
if value
for candidate in str(value).split(",")
if len(candidate.strip()) == 64
]
crc_raw = str(file_entry.get("crc32", "") or "").strip().lower()
declared_size = file_entry.get("size")
has_strong_hash = bool(
sha1_candidates or sha256_candidates or md5_list or crc_raw
)
files_db = db.get("files", {})
by_md5 = db.get("indexes", {}).get("by_md5", {})
by_name = db.get("indexes", {}).get("by_name", {})
by_path_suffix = db.get("indexes", {}).get("by_path_suffix", {})
# 0. Path suffix exact match (for regional variants with same filename)
if dest_hint and by_path_suffix:
for match_sha1 in by_path_suffix.get(dest_hint, []):
if match_sha1 in files_db:
path = files_db[match_sha1]["path"]
if os.path.exists(path):
return path, "exact"
def _record_match_status(match_sha1: str) -> str | None:
"""Return hash evidence when a DB record satisfies every declaration."""
entry = files_db.get(match_sha1)
if not entry:
return None
statuses: list[str] = []
if sha1_candidates:
if match_sha1.lower() not in sha1_candidates:
return None
statuses.append("sha1_exact")
if sha256_candidates:
if str(entry.get("sha256", "")).lower() not in sha256_candidates:
return None
statuses.append("sha256_exact")
# With zipped_file the MD5 identifies the member, not the container.
if md5_list and not zipped_file:
actual_md5 = str(entry.get("md5", "")).lower()
if not any(actual_md5.startswith(expected) for expected in md5_list):
return None
statuses.append("md5_exact")
if crc_raw:
if str(entry.get("crc32", "")).lower() != crc_raw:
return None
if declared_size is not None:
allowed_sizes = (
declared_size if isinstance(declared_size, list) else [declared_size]
)
if entry.get("size") not in allowed_sizes:
return None
statuses.append("crc32_exact")
return statuses[0] if statuses else None
# 1. SHA1 exact match (accept list-valued sha1 from profiles)
sha1_candidates = sha1 if isinstance(sha1, list) else [sha1] if sha1 else []
for cand in sha1_candidates:
if isinstance(cand, str) and cand in files_db:
if cand in files_db:
path = files_db[cand]["path"]
if os.path.exists(path):
return path, "exact"
status = _record_match_status(cand)
if status and os.path.exists(path):
return path, status
# 1b. SHA256 exact match (profiles hashed from sources that publish
# sha256, e.g. MesenCE). A full sha256 is a strong identifier.
sha256_raw = str(file_entry.get("sha256", "") or "")
if sha256_raw:
if sha256_candidates:
by_sha256 = db.get("indexes", {}).get("by_sha256", {})
for cand in sha256_raw.split(","):
cand = cand.strip().lower()
if len(cand) == 64:
match = by_sha256.get(cand)
if match and match in files_db:
path = files_db[match]["path"]
if os.path.exists(path):
return path, "exact"
for cand in sha256_candidates:
match = by_sha256.get(cand)
if match and match in files_db:
path = files_db[match]["path"]
status = _record_match_status(match)
if status and os.path.exists(path):
return path, status
# 1c. CRC32 + size exact match, only when no stronger hash is declared
# (crc-only profiles: FBNeo, Clock Signal). CRC32 alone is weak, so the
# declared size must confirm the match.
crc_raw = str(file_entry.get("crc32", "") or "").strip().lower()
declared_size = file_entry.get("size")
if crc_raw and declared_size and not zipped_file and not md5_list:
# 1c. CRC32 lookup, only when no stronger hash is declared. A declared
# size confirms it when available; a handful of emulators validate CRC32
# alone, so those entries retain the same (weaker) evidence as the core.
if (
crc_raw
and not zipped_file
and not md5_list
and not sha1_candidates
and not sha256_candidates
):
by_crc32 = db.get("indexes", {}).get("by_crc32", {})
match = by_crc32.get(crc_raw)
if match and match in files_db:
entry = files_db[match]
path = entry["path"]
if entry.get("size") == declared_size and os.path.exists(path):
return path, "exact"
status = _record_match_status(match)
if status and os.path.exists(path):
return path, status
# 2. MD5 direct lookup (skip for zipped_file: md5 is inner ROM, not container)
# Guard: only accept if the found file's name matches the requested name
@@ -534,15 +604,31 @@ def resolve_local_file(
# Full MD5 (32 chars) is a strong identifier: trust it
# without name guard. Truncated MD5 still needs name check
# to avoid cross-contamination.
if os.path.exists(path):
status = _record_match_status(sha1_match)
if status and os.path.exists(path):
if len(md5_candidate) >= 32 or _md5_name_ok(path):
return path, "md5_exact"
return path, status
if len(md5_candidate) < 32:
for db_md5, db_sha1 in by_md5.items():
if db_md5.startswith(md5_candidate) and db_sha1 in files_db:
path = files_db[db_sha1]["path"]
if os.path.exists(path) and _md5_name_ok(path):
return path, "md5_exact"
status = _record_match_status(db_sha1)
if status and os.path.exists(path) and _md5_name_ok(path):
return path, status
# 2b. Path suffix lookup is useful for same-named regional files, but it
# is identity evidence only when no content hash was declared. A stale
# or incorrect destination can therefore never mask a hash mismatch.
if dest_hint and by_path_suffix:
for match_sha1 in by_path_suffix.get(dest_hint, []):
if match_sha1 in files_db:
path = files_db[match_sha1]["path"]
if os.path.exists(path):
if not has_strong_hash:
return path, "path_exact"
status = _record_match_status(match_sha1)
if status:
return path, status
# 3. No MD5 = any file with that name or alias (existence check)
def _size_ok(match_sha1: str) -> bool:
@@ -550,7 +636,7 @@ def resolve_local_file(
file_entry, files_db.get(match_sha1, {}).get("size")
)
if not md5_list:
if not has_strong_hash:
candidates = []
for try_name in names_to_try:
for match_sha1 in by_name.get(try_name, []):
@@ -575,7 +661,7 @@ def resolve_local_file(
candidates = [p for p in candidates if ".zip" in os.path.basename(p)]
primary = [p for p in candidates if "/.variants/" not in p]
if primary or candidates:
return (primary[0] if primary else candidates[0]), "exact"
return (primary[0] if primary else candidates[0]), "name_exact"
# 4. Name + alias fallback with md5_composite + direct MD5 per candidate
md5_set = set(md5_list)
@@ -595,17 +681,17 @@ def resolve_local_file(
candidates = [
(p, m) for p, m in candidates if ".zip" in os.path.basename(p)
]
if md5_set:
if md5_set and not (sha1_candidates or sha256_candidates or crc_raw):
for path, db_md5 in candidates:
if ".zip" in os.path.basename(path):
try:
composite = md5_composite(path).lower()
if composite in md5_set:
return path, "exact"
return path, "md5_composite_exact"
except (zipfile.BadZipFile, OSError):
pass
if db_md5.lower() in md5_set:
return path, "exact"
return path, "md5_exact"
# When zipped_file is set, only accept candidates that contain it
if zipped_file:
valid = []
@@ -628,7 +714,12 @@ def resolve_local_file(
# 5. zipped_file content match via pre-built index (last resort:
# matches inner ROM MD5 across ALL ZIPs in the repo, so only use
# when name-based resolution failed entirely)
if zipped_file and md5_list and zip_contents:
if (
zipped_file
and md5_list
and zip_contents
and not (sha1_candidates or sha256_candidates or crc_raw)
):
for md5_candidate in md5_list:
if md5_candidate in zip_contents:
zip_sha1 = zip_contents[md5_candidate]
@@ -638,7 +729,7 @@ def resolve_local_file(
return path, "zip_exact"
# MAME clone fallback: if a file was deduped, resolve via canonical
if _depth < 3:
if _depth < 3 and not has_strong_hash:
clone_map = _get_mame_clone_map()
canonical = clone_map.get(name)
if canonical and canonical != name:
@@ -655,6 +746,42 @@ def resolve_local_file(
return result[0], "mame_clone"
# Data directory fallback: scan data/ caches for matching filename
def _unindexed_path_status(candidate: str) -> str:
"""Validate a data-directory candidate without trusting its filename."""
if not has_strong_hash:
return "data_dir"
algorithms: set[str] = set()
if sha1_candidates:
algorithms.add("sha1")
if sha256_candidates:
algorithms.add("sha256")
if md5_list and not zipped_file:
algorithms.add("md5")
if crc_raw:
algorithms.add("crc32")
actual = compute_hashes(candidate, frozenset(algorithms)) if algorithms else {}
if sha1_candidates and actual.get("sha1", "").lower() not in sha1_candidates:
return "hash_mismatch"
if sha256_candidates and actual.get("sha256", "").lower() not in sha256_candidates:
return "hash_mismatch"
if md5_list and not zipped_file and not any(
actual.get("md5", "").lower().startswith(expected) for expected in md5_list
):
return "hash_mismatch"
if crc_raw and actual.get("crc32", "").lower() != crc_raw:
return "hash_mismatch"
if crc_raw and declared_size is not None:
allowed_sizes = declared_size if isinstance(declared_size, list) else [declared_size]
if os.path.getsize(candidate) not in allowed_sizes:
return "hash_mismatch"
if zipped_file and md5_list and not any(
check_inside_zip(candidate, zipped_file, expected) == "ok"
for expected in md5_list
):
return "hash_mismatch"
return "data_dir_hash_exact"
data_dir_mismatch: str | None = None
if data_dir_registry:
for _dd_key, dd_entry in data_dir_registry.items():
cache_dir = dd_entry.get("local_cache", "")
@@ -664,7 +791,10 @@ def resolve_local_file(
# Exact relative path
candidate = os.path.join(cache_dir, try_name)
if os.path.isfile(candidate):
return candidate, "data_dir"
status = _unindexed_path_status(candidate)
if status != "hash_mismatch":
return candidate, status
data_dir_mismatch = data_dir_mismatch or candidate
# Basename walk: find file anywhere in cache tree (case-insensitive)
basename_targets = {
(n.rsplit("/", 1)[-1] if "/" in n else n).casefold()
@@ -673,11 +803,18 @@ def resolve_local_file(
for root, _dirs, fnames in os.walk(cache_dir):
for fn in fnames:
if fn.casefold() in basename_targets:
return os.path.join(root, fn), "data_dir"
candidate = os.path.join(root, fn)
status = _unindexed_path_status(candidate)
if status != "hash_mismatch":
return candidate, status
data_dir_mismatch = data_dir_mismatch or candidate
if data_dir_mismatch:
return data_dir_mismatch, "hash_mismatch"
# Agnostic fallback: for filename-agnostic files, find any DB file
# matching the system path prefix and size criteria
if file_entry.get("agnostic"):
if file_entry.get("agnostic") and not has_strong_hash:
agnostic_prefix = file_entry.get("agnostic_path_prefix", "")
min_size = file_entry.get("min_size", 0)
max_size = file_entry.get("max_size", float("inf"))
@@ -1228,8 +1365,10 @@ def fetch_large_file(
dest_dir: str = LARGE_FILES_CACHE,
expected_sha1: str = "",
expected_md5: str = "",
*,
offline: bool = False,
) -> str | None:
"""Download a large file from the 'large-files' GitHub release if not cached."""
"""Return a verified cached large file, downloading it only when allowed."""
cached = os.path.join(dest_dir, name)
if os.path.exists(cached):
if expected_sha1 or expected_md5:
@@ -1249,6 +1388,9 @@ def fetch_large_file(
else:
return cached
if offline:
return None
os.makedirs(dest_dir, exist_ok=True)
# A per-process scratch name: two runs fetching the same asset into one
# shared path interleave their writes into a full-size, corrupt file.
@@ -1303,15 +1445,118 @@ def fetch_large_file(
return cached
def safe_extract_zip(zip_path: str, dest_dir: str) -> None:
"""Extract a ZIP file safely, preventing zip-slip path traversal."""
MAX_ZIP_MEMBERS = 100_000
MAX_ZIP_MEMBER_SIZE = 8 * 1024 * 1024 * 1024
# The largest generated pack is already ~5 GB uncompressed and the collection
# only grows; this bounds a malicious archive without capping a real one.
MAX_ZIP_TOTAL_SIZE = 64 * 1024 * 1024 * 1024
# DEFLATE cannot exceed roughly 1,032:1, so this rejects a declared ratio no
# real DEFLATE member can reach. Methods with a higher ceiling (bzip2, LZMA)
# are exempt and bounded by the per-member and per-archive size limits alone.
MAX_ZIP_COMPRESSION_RATIO = 1_100
_BOUNDED_RATIO_METHODS = (zipfile.ZIP_STORED, zipfile.ZIP_DEFLATED)
def safe_extract_zip(
zip_path: str,
dest_dir: str,
*,
max_members: int = MAX_ZIP_MEMBERS,
max_member_size: int = MAX_ZIP_MEMBER_SIZE,
max_total_size: int = MAX_ZIP_TOTAL_SIZE,
max_compression_ratio: int = MAX_ZIP_COMPRESSION_RATIO,
) -> None:
"""Extract a ZIP with traversal, link and resource-limit protection.
Files are streamed to a temporary sibling and atomically installed only
after their declared length and CRC have been checked by ``zipfile``.
"""
dest = os.path.realpath(dest_dir)
os.makedirs(dest, exist_ok=True)
with zipfile.ZipFile(zip_path, "r") as zf:
for member in zf.infolist():
member_path = os.path.realpath(os.path.join(dest, member.filename))
if not member_path.startswith(dest + os.sep) and member_path != dest:
raise ValueError(f"Zip slip detected: {member.filename}")
zf.extract(member, dest)
members = zf.infolist()
if len(members) > max_members:
raise ValueError(
f"ZIP has {len(members)} members; limit is {max_members}"
)
declared_total = 0
seen: set[str] = set()
for member in members:
# Archives written on Windows store a backslash separator. It is a
# separator, not a filename character, so it is normalized before
# the component checks rather than rejected.
name = member.filename.replace("\\", "/")
if not name or "\x00" in name:
raise ValueError(f"Unsafe ZIP member name: {member.filename!r}")
if name.startswith("/") or re.match(r"^[A-Za-z]:", name):
raise ValueError(f"Absolute ZIP member path: {name}")
parts = [part for part in name.split("/") if part]
if any(part in (".", "..") for part in parts):
raise ValueError(f"ZIP traversal detected: {name}")
normalized = "/".join(parts)
if normalized in seen:
raise ValueError(f"Duplicate ZIP member path: {name}")
seen.add(normalized)
mode = (member.external_attr >> 16) & 0xFFFF
file_type = stat.S_IFMT(mode)
if file_type not in (0, stat.S_IFREG, stat.S_IFDIR):
raise ValueError(f"ZIP link or special file rejected: {name}")
if member.flag_bits & 0x1:
raise ValueError(f"Encrypted ZIP member rejected: {name}")
if member.file_size > max_member_size:
raise ValueError(
f"ZIP member {name} is {member.file_size} bytes; "
f"limit is {max_member_size}"
)
declared_total += member.file_size
if declared_total > max_total_size:
raise ValueError(
f"ZIP expands to {declared_total} bytes; limit is {max_total_size}"
)
if member.file_size and member.compress_type in _BOUNDED_RATIO_METHODS:
if member.compress_size == 0:
raise ValueError(f"Invalid compression size for ZIP member: {name}")
if member.file_size / member.compress_size > max_compression_ratio:
raise ValueError(f"Suspicious compression ratio for ZIP member: {name}")
target = os.path.realpath(os.path.join(dest, *parts))
if not target.startswith(dest + os.sep) and target != dest:
raise ValueError(f"ZIP traversal detected: {name}")
if member.is_dir() or name.endswith("/"):
os.makedirs(target, exist_ok=True)
continue
os.makedirs(os.path.dirname(target), exist_ok=True)
tmp_path = ""
try:
with tempfile.NamedTemporaryFile(
mode="wb", dir=os.path.dirname(target), delete=False
) as tmp_file:
tmp_path = tmp_file.name
actual_size = 0
with zf.open(member, "r") as source:
while True:
chunk = source.read(1024 * 1024)
if not chunk:
break
actual_size += len(chunk)
if actual_size > member.file_size or actual_size > max_member_size:
raise ValueError(
f"ZIP member exceeded declared or configured size: {name}"
)
tmp_file.write(chunk)
if actual_size != member.file_size:
raise ValueError(
f"ZIP member size mismatch for {name}: "
f"{actual_size} != {member.file_size}"
)
os.replace(tmp_path, target)
tmp_path = ""
finally:
if tmp_path and os.path.exists(tmp_path):
os.unlink(tmp_path)
def list_emulator_profiles(emulators_dir: str, skip_aliases: bool = True) -> None:
+30 -11
View File
@@ -230,17 +230,26 @@ def cross_reference(
gaps = []
covered = []
unsourceable_list: list[dict] = []
archive_gaps: dict[str, dict] = {}
seen_files: set[str] = set()
archive_gaps: dict[tuple, dict] = {}
seen_files: set[tuple] = set()
for f in emu_files:
fname = f.get("name", "")
if not fname or fname in seen_files:
effective_path = f.get("path") or fname
seen_key = (
fname,
f.get("archive"),
effective_path,
f.get("system"),
f.get("variant_group"),
f.get("mode", "both"),
)
if not fname or seen_key in seen_files:
continue
# Collect unsourceable files separately (documented, not a gap)
unsourceable_reason = f.get("unsourceable", "")
if unsourceable_reason:
seen_files.add(fname)
seen_files.add(seen_key)
unsourceable_list.append({
"name": fname,
"required": f.get("required", False),
@@ -279,25 +288,33 @@ def cross_reference(
in_platform = archive in platform_names
if in_platform:
seen_files.add(fname)
seen_files.add(seen_key)
covered.append({
"name": fname,
"path": effective_path,
"required": f.get("required", False),
"in_platform": True,
})
continue
seen_files.add(fname)
seen_files.add(seen_key)
# Group archived files by archive name
if archive:
if archive not in archive_gaps:
archive_key = (
archive,
f.get("system"),
f.get("variant_group"),
f.get("mode", "both"),
)
if archive_key not in archive_gaps:
source = _resolve_archive_source(
archive, by_name, by_name_lower, data_names,
by_path_suffix,
)
archive_gaps[archive] = {
archive_gaps[archive_key] = {
"name": archive,
"path": archive,
"required": False,
"note": "",
"source_ref": "",
@@ -308,7 +325,7 @@ def cross_reference(
"archive_file_count": 0,
"archive_required_count": 0,
}
entry = archive_gaps[archive]
entry = archive_gaps[archive_key]
entry["archive_file_count"] += 1
if f.get("required", False):
entry["archive_required_count"] += 1
@@ -351,8 +368,9 @@ def cross_reference(
break
# Try SHA1 hash match
if source is None:
sha1 = f.get("sha1", "")
if sha1 and sha1 in db_files:
raw_sha1 = f.get("sha1", "")
sha1_values = raw_sha1 if isinstance(raw_sha1, list) else [raw_sha1]
if any(value and value in db_files for value in sha1_values):
source = "bios"
# Try CRC32 hash match
if source is None:
@@ -366,6 +384,7 @@ def cross_reference(
entry = {
"name": fname,
"path": effective_path,
"required": f.get("required", False),
"note": f.get("note", ""),
"source_ref": f.get("source_ref", ""),
+1
View File
@@ -342,6 +342,7 @@ def main():
total_size = sum(entry["size"] for entry in files.values())
database = {
"schema_version": 1,
"generated_at": datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ"),
"total_files": len(files),
"total_size": total_size,
+467 -138
View File
File diff suppressed because it is too large. Load diff
+70 -11
View File
@@ -4,6 +4,7 @@
Steps:
1. generate_db.py --force (rebuild database.json from bios/)
1b. provenance_report.py (dump-catalog coverage from provenance/)
1c. romset_recipes.py (archive identification, reconstruction targets)
2. refresh_data_dirs.py (update Dolphin Sys, PPSSPP, etc.)
3. verify.py --all (check all platforms)
4. generate_pack.py --all (build ZIP packs)
@@ -24,6 +25,7 @@ Usage:
from __future__ import annotations
import argparse
import re
import subprocess
import sys
import time
@@ -57,8 +59,6 @@ def parse_verify_counts(output: str) -> dict[str, tuple[int, int]]:
Matches: "Label: X/Y OK ..." or "Label: X/Y present ..."
Returns {group_label: (ok, total)}.
"""
import re
counts = {}
for line in output.splitlines():
m = re.match(r"^(.+?):\s+(\d+)/(\d+)\s+(OK|present)", line)
@@ -75,14 +75,16 @@ def parse_pack_counts(output: str) -> dict[str, tuple[int, int]]:
Returns {pack_label: (ok, total)}.
"""
import re
counts = {}
current_label = ""
for line in output.splitlines():
m = re.match(r"Generating (?:shared )?pack for (.+)\.\.\.", line)
if m:
current_label = m.group(1)
# Labels carry execution metadata such as ``[source=full]``.
# It is not part of the platform identity used for consistency.
current_label = re.sub(
r"\s+\[source=[^\]]+\]$", "", m.group(1).strip()
)
continue
if "files packed" not in line:
continue
@@ -100,34 +102,77 @@ def parse_pack_counts(output: str) -> dict[str, tuple[int, int]]:
return counts
def parse_pack_exclusions(output: str) -> dict[str, int]:
"""Extract intentional unsafe-omission counts from pack output."""
exclusions: dict[str, int] = {}
current_label = ""
for line in output.splitlines():
label_match = re.match(r"Generating (?:shared )?pack for (.+)\.\.\.", line)
if label_match:
current_label = re.sub(
r"\s+\[source=[^\]]+\]$", "", label_match.group(1).strip()
)
continue
if "files packed" not in line:
continue
excluded_match = re.search(r"(\d+) unsafe excluded", line)
exclusions[current_label] = (
int(excluded_match.group(1)) if excluded_match else 0
)
return exclusions
def _match_key(label: str) -> set[str]:
"""Comparable identity for a platform label.
Display labels and registry ids differ in punctuation and spacing
(``MiSTer FPGA`` against ``misterfpga``), and grouped packs join their
members with either separator.
"""
return {
re.sub(r"[^a-z0-9]+", "", part.strip().lower())
for part in label.replace("+", "/").split("/")
} - {""}
def check_consistency(verify_output: str, pack_output: str) -> bool:
"""Verify that check counts match between verify and pack for each platform."""
v = parse_verify_counts(verify_output)
p = parse_pack_counts(pack_output)
excluded = parse_pack_exclusions(pack_output)
print("\n--- 5/8 consistency check ---")
all_ok = True
for v_label, (v_ok, v_total) in sorted(v.items()):
# Match by name overlap (handles "Lakka + RetroArch" vs "Lakka / RetroArch")
# Match by normalized name overlap. Platform display labels and
# registry IDs legitimately differ in punctuation and spacing
# (notably ``MiSTer FPGA`` vs ``misterfpga``).
p_match = None
v_names = _match_key(v_label)
for p_label in p:
v_names = {n.strip().lower() for n in v_label.split("/")}
p_names = {n.strip().lower() for n in p_label.replace("+", "/").split("/")}
if v_names & p_names:
if v_names & _match_key(p_label):
p_match = p_label
break
if p_match:
p_ok, p_total = p[p_match]
p_excluded = excluded.get(p_match, 0)
if v_total != p_total:
print(f" {v_label}: MISMATCH total verify {v_total} != pack {p_total}")
all_ok = False
elif p_ok < v_ok:
elif p_ok + p_excluded < v_ok:
print(
f" {v_label}: MISMATCH pack {p_ok} OK < verify {v_ok} OK (/{v_total})"
f" {v_label}: MISMATCH pack accounts for "
f"{p_ok} OK + {p_excluded} unsafe exclusions "
f"< verify {v_ok} OK (/{v_total})"
)
all_ok = False
elif p_ok < v_ok:
print(
f" {v_label}: verify {v_ok}/{v_total} native; pack {p_ok} safe, "
f"{p_excluded} unsafe excluded OK"
)
elif p_ok == v_ok:
print(
f" {v_label}: verify {v_ok}/{v_total} == pack {p_ok}/{p_total} OK"
@@ -138,6 +183,7 @@ def check_consistency(verify_output: str, pack_output: str) -> bool:
)
else:
print(f" {v_label}: {v_ok}/{v_total} (no separate pack)")
all_ok = False
status = "OK" if all_ok else "FAILED"
print(f"--- consistency check: {status} ---")
@@ -219,6 +265,16 @@ def main():
results["provenance"] = ok
all_ok = all_ok and ok
# Step 1c: Which emulator version each arcade archive corresponds to, and
# which pinned archive the collection could rebuild from ROMs it holds.
# Read-only: writing reconstructions is an explicit --write invocation.
ok, out = run(
[sys.executable, "scripts/romset_recipes.py"],
"1c romset recipes",
)
results["romset_recipes"] = ok
all_ok = all_ok and ok
# Step 2: Refresh data directories
if not args.offline:
ok, out = run(
@@ -226,6 +282,7 @@ def main():
"2/8 refresh data directories",
)
results["refresh_data"] = ok
all_ok = all_ok and ok
else:
print("\n--- 2/8 refresh data directories: SKIPPED (--offline) ---")
results["refresh_data"] = True
@@ -237,6 +294,7 @@ def main():
"2a refresh MAME hashes",
)
results["mame_hashes"] = ok
all_ok = all_ok and ok
else:
print("\n--- 2a refresh MAME hashes: SKIPPED (--offline) ---")
results["mame_hashes"] = True
@@ -248,6 +306,7 @@ def main():
"2a2 refresh FBNeo hashes",
)
results["fbneo_hashes"] = ok
all_ok = all_ok and ok
else:
print("\n--- 2a2 refresh FBNeo hashes: SKIPPED (--offline) ---")
results["fbneo_hashes"] = True
+47 -22
View File
@@ -128,9 +128,9 @@ def region_tag(requested: list[str]) -> str:
def build_region_index(profiles: dict) -> dict[str, dict]:
"""Build a region lookup from emulator profiles.
Keyed by the entry's path when present, by name otherwise. Entries sharing a
key have their regions unioned, so an ambiguous lookup keeps more files
rather than fewer.
Keyed by the entry's path when present, by name otherwise. Untagged
declarations are recorded as ambiguity evidence: if the same lookup key is
both tagged and untagged, filtering keeps it instead of inventing a region.
"""
index: dict[str, dict] = {}
for emu_name, profile in sorted(profiles.items()):
@@ -145,16 +145,19 @@ def build_region_index(profiles: dict) -> dict[str, dict]:
raise ValueError(
f"{emu_name}: {f.get('name', '?')}: {exc}"
) from exc
if not regions:
continue
name = f.get("name", "")
path = f.get("path") or ""
# Path-keyed so same-named entries stay separate (Dolphin declares
# three IPL.bin), name-keyed so a candidate identified by name alone
# sees the union and is never dropped on ambiguity.
for key in {path, name} - {""}:
entry = index.setdefault(key, {"regions": set(), "emulators": []})
entry = index.setdefault(
key,
{"regions": set(), "has_untagged": False, "emulators": []},
)
entry["regions"] |= regions
if not regions:
entry["has_untagged"] = True
if emu_name not in entry["emulators"]:
entry["emulators"].append(emu_name)
return index
@@ -170,14 +173,16 @@ def lookup_regions(index: dict[str, dict], destination: str, name: str) -> set[s
if destination:
entry = index.get(destination)
if entry:
return set(entry["regions"])
return set() if entry.get("has_untagged") else set(entry["regions"])
parts = destination.split("/")
for i in range(1, len(parts)):
entry = index.get("/".join(parts[i:]))
if entry:
return set(entry["regions"])
return set() if entry.get("has_untagged") else set(entry["regions"])
entry = index.get(name)
return set(entry["regions"]) if entry else set()
if not entry or entry.get("has_untagged"):
return set()
return set(entry["regions"])
def _competing_ranks(
@@ -206,8 +211,10 @@ def resolve_region_drops(
) -> set[str]:
"""Destinations to skip for a requested region priority list.
Per group, only the best rank actually present survives. A destination kept
by any group is kept overall.
Per group, an exact/parent regional match beats other regional candidates.
A world candidate beats unmatched regional fallbacks, while untagged files
always survive. If neither a requested nor a world candidate exists, all
regional candidates survive so filtering can never empty a group.
"""
if not requested:
return set()
@@ -215,17 +222,31 @@ def resolve_region_drops(
keep: set[str] = set()
drop: set[str] = set()
for members in groups.values():
ranked = _competing_ranks(members, index, requested)
competing = {dest for _r, dest in ranked}
keep |= {dest for dest, _name in members if dest not in competing}
if not ranked:
continue
best = min(r for r, _ in ranked)
for r, destination in ranked:
if r == best:
keep.add(destination)
regional: list[tuple[int, str]] = []
world: set[str] = set()
untagged: set[str] = set()
for destination, name in members:
regions = lookup_regions(index, destination, name)
if not regions:
untagged.add(destination)
elif WORLD in regions:
world.add(destination)
else:
drop.add(destination)
regional.append((rank(regions, requested), destination))
keep |= untagged | world
if not regional:
continue
matched = [(r, destination) for r, destination in regional if r < len(requested)]
if matched:
best = min(r for r, _destination in matched)
keep |= {destination for r, destination in matched if r == best}
drop |= {destination for _r, destination in regional if destination not in keep}
elif world:
drop |= {destination for _r, destination in regional}
else:
# Preserve every unmatched candidate as a visible fallback.
keep |= {destination for _r, destination in regional}
return drop - keep
@@ -240,6 +261,10 @@ def fallback_groups(
out: list[str] = []
for group_id, members in groups.items():
ranked = _competing_ranks(members, index, requested)
if ranked and min(r for r, _ in ranked) == len(requested):
has_world = any(
WORLD in lookup_regions(index, destination, name)
for destination, name in members
)
if ranked and not has_world and min(r for r, _ in ranked) == len(requested):
out.append(group_id)
return sorted(out)
+11 -7
View File
@@ -19,7 +19,7 @@ from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
import region
from common import load_emulator_profiles, load_database
from common import load_emulator_profiles, load_database, parse_md5_list
# No-Intro filename tokens: full English territory names.
NOINTRO = {
@@ -104,12 +104,16 @@ def resolve_sha1(file_entry: dict, db: dict) -> str | None:
"""Resolve a profile file entry to a repo SHA1, or None when ambiguous."""
files = db["files"]
indexes = db["indexes"]
declared = str(file_entry.get("sha1") or "").lower()
if declared and declared in files:
return declared
md5 = str(file_entry.get("md5") or "").lower()
if md5 and md5 in indexes["by_md5"]:
return indexes["by_md5"][md5]
raw_sha1 = file_entry.get("sha1")
declared = raw_sha1 if isinstance(raw_sha1, list) else [raw_sha1]
sha1_hits = [str(value).lower() for value in declared if value]
sha1_hits = [value for value in sha1_hits if value in files]
if len(sha1_hits) == 1:
return sha1_hits[0]
for md5 in parse_md5_list(file_entry.get("md5")):
hit = indexes["by_md5"].get(md5)
if hit:
return hit
hits = indexes["by_name"].get(file_entry.get("name", ""), [])
if isinstance(hits, str):
hits = [hits]
+46 -9
View File
@@ -370,9 +370,9 @@ def find_undeclared_files(
relevant = resolve_platform_cores(config, profiles, target_cores=target_cores)
standalone_set = set(str(c) for c in config.get("standalone_cores", []))
undeclared = []
seen_files: set[tuple[str, str | None]] = set()
seen_files: set[tuple] = set()
# Track archives: archive_name -> {in_repo, emulator, files: [...], ...}
archive_entries: dict[str, dict] = {}
archive_entries: dict[tuple, dict] = {}
for emu_name, profile in sorted(profiles.items()):
if profile.get("type") in ("launcher", "alias"):
@@ -391,9 +391,25 @@ def find_undeclared_files(
for f in profile.get("files", []):
fname = f.get("name", "")
# Dedup by (name, archive): a filename declared loose by one core
# and inside an archive by another are distinct packaging shapes
seen_key = (fname, f.get("archive"))
effective_path = (
f.get("standalone_path") if is_standalone else f.get("path")
) or fname
raw_regions = f.get("region") or []
region_key = tuple(
str(value) for value in (
raw_regions if isinstance(raw_regions, list) else [raw_regions]
)
)
# Same-name requirements at different paths, for different systems,
# or in distinct variant groups are not interchangeable.
seen_key = (
fname,
f.get("archive"),
effective_path,
f.get("system"),
f.get("variant_group"),
region_key,
)
if not fname or seen_key in seen_files:
continue
# Skip unsourceable files (documented reason, not a gap)
@@ -436,13 +452,25 @@ def find_undeclared_files(
# Archived files are grouped by archive
if archive:
if archive not in archive_entries:
archive_key = (
archive,
f.get("system"),
f.get("variant_group"),
region_key,
is_standalone,
)
if archive_key not in archive_entries:
in_repo = _name_in_index(
archive, by_name, by_path_suffix, data_names,
by_name_lower,
)
archive_entries[archive] = {
archive_entries[archive_key] = {
"profile": emu_name,
"emulator": profile.get("emulator", emu_name),
"systems": list(profile.get("systems", [])),
"system": f.get("system"),
"region": f.get("region"),
"variant_group": f.get("variant_group"),
"name": archive,
"archive": archive,
"path": archive,
@@ -457,7 +485,7 @@ def find_undeclared_files(
"archive_file_count": 0,
"archive_required_count": 0,
}
entry = archive_entries[archive]
entry = archive_entries[archive_key]
entry["archive_file_count"] += 1
if f.get("required", False):
entry["archive_required_count"] += 1
@@ -487,13 +515,22 @@ def find_undeclared_files(
if not in_repo:
# Hash fallback: the repo may hold the content under a
# different filename (exos21.rom vs exos21.bin)
# generate_pack ships a core extra whose local copy
# contradicts the declared hash and reports the divergence.
# verify must agree with the builder or the two reports
# disagree on the same file.
_lp, _st = resolve_local_file(f, db, dest_hint=dest)
in_repo = _st not in ("not_found",) and _lp is not None
in_repo = _st != "not_found" and _lp is not None
checks = _parse_validation(f.get("validation"))
undeclared.append(
{
"profile": emu_name,
"emulator": profile.get("emulator", emu_name),
"systems": list(profile.get("systems", [])),
"system": f.get("system"),
"region": f.get("region"),
"variant_group": f.get("variant_group"),
"name": fname,
"path": dest,
"required": f.get("required", False),
+704
View File
@@ -0,0 +1,704 @@
"""Regression tests for the holistic reliability and security audit."""
from __future__ import annotations
import contextlib
import hashlib
import io
import json
import os
import re
import stat
import sys
import tempfile
import unittest
import zipfile
from pathlib import Path
from unittest import mock
import yaml
ROOT = Path(__file__).resolve().parent.parent
TMP_ROOT = ROOT / "tmp" / "tests"
TMP_ROOT.mkdir(parents=True, exist_ok=True)
sys.path.insert(0, str(ROOT))
sys.path.insert(0, str(ROOT / "scripts"))
import install
from scripts import pipeline, region, region_audit
from scripts.common import resolve_local_file, safe_extract_zip
from scripts.generate_pack import (
_emulator_region_group,
generate_target_manifests,
verify_pack_against_platform,
)
class ReadmeRegressions(unittest.TestCase):
"""The README advertises commands; these must stay executable.
Wording is the maintainer's, so nothing here asserts prose. What is
asserted is that every flag and URL the README hands a reader is one the
shipped scripts actually accept.
"""
def _quick_install(self) -> str:
readme = (ROOT / "README.md").read_text(encoding="utf-8")
return readme.split("## Quick Install", 1)[1].split(
"## Download BIOS packs", 1
)[0]
def test_advertised_bootstrap_urls_are_the_shipped_ones(self):
quick_install = self._quick_install()
for script in ("install.sh", "install.ps1"):
url = f"https://raw.githubusercontent.com/Abdess/retrobios/main/{script}"
self.assertIn(url, quick_install, f"{script} bootstrap URL missing")
self.assertIn(
"https://raw.githubusercontent.com/Abdess/retrobios/main/install.py",
(ROOT / "install.sh").read_text(encoding="utf-8"),
"install.sh must fetch install.py from the ref the README advertises",
)
def test_advertised_installer_flags_exist(self):
quick_install = self._quick_install()
advertised = set(re.findall(r"(?<![\w-])--[a-z][a-z-]+", quick_install))
parser_source = (ROOT / "install.py").read_text(encoding="utf-8")
supported = set(re.findall(r'add_argument\(\s*"(--[a-z][a-z-]+)"', parser_source))
unknown = advertised - supported
self.assertEqual(unknown, set(), f"README advertises unknown flags: {unknown}")
class PipelineRegressions(unittest.TestCase):
def test_pack_parser_removes_source_metadata(self):
output = "\n".join(
[
"Generating pack for RetroArch [source=full]...",
" 3 files packed (2 baseline + 1 from cores), 2/2 files OK",
]
)
self.assertEqual(pipeline.parse_pack_counts(output), {"RetroArch": (2, 2)})
def test_missing_pack_is_a_consistency_failure(self):
verify = "RetroArch: 2/2 OK"
with contextlib.redirect_stdout(io.StringIO()):
self.assertFalse(pipeline.check_consistency(verify, ""))
def test_display_name_matches_compact_registry_id(self):
verify = "MiSTer FPGA: 65/65 OK [md5]"
pack = "\n".join(
[
"Generating pack for misterfpga [source=full]...",
" pack.zip: 65 files packed (65 baseline + 0 from cores), "
"65/65 files OK [md5]",
]
)
with contextlib.redirect_stdout(io.StringIO()):
self.assertTrue(pipeline.check_consistency(verify, pack))
def test_native_existence_and_strict_pack_exclusions_are_consistent(self):
verify = "RetroArch: 3/3 present [existence]"
pack = "\n".join(
[
"Generating pack for RetroArch [source=full]...",
" pack.zip: 2 files packed (2 baseline + 0 from cores), "
"2/3 files OK, 1 unsafe excluded [existence]",
]
)
with contextlib.redirect_stdout(io.StringIO()):
self.assertTrue(pipeline.check_consistency(verify, pack))
self.assertEqual(pipeline.parse_pack_exclusions(pack), {"RetroArch": 1})
def test_unaccounted_pack_omission_is_a_consistency_failure(self):
verify = "RetroArch: 3/3 present [existence]"
pack = "\n".join(
[
"Generating pack for RetroArch [source=full]...",
" pack.zip: 1 files packed (1 baseline + 0 from cores), "
"1/3 files OK, 1 unsafe excluded, 1 missing [existence]",
]
)
with contextlib.redirect_stdout(io.StringIO()):
self.assertFalse(pipeline.check_consistency(verify, pack))
def test_every_refresh_failure_reaches_pipeline_exit_status(self):
for failed_label in (
"2/8 refresh data directories",
"2a refresh MAME hashes",
"2a2 refresh FBNeo hashes",
):
with self.subTest(failed_label=failed_label):
def fake_run(_command, label):
return label != failed_label, "RetroArch: 1/1 OK\n"
argv = ["pipeline.py", "--skip-packs", "--skip-docs"]
with (
mock.patch.object(sys, "argv", argv),
mock.patch.object(pipeline, "run", side_effect=fake_run),
contextlib.redirect_stdout(io.StringIO()),
self.assertRaises(SystemExit) as raised,
):
pipeline.main()
self.assertEqual(raised.exception.code, 1)
class ResolverRegressions(unittest.TestCase):
def _database(self, entries: dict[str, Path], suffix: str | None = None) -> dict:
files = {}
by_name: dict[str, list[str]] = {}
by_suffix: dict[str, list[str]] = {}
for name, path in entries.items():
payload = path.read_bytes()
sha1 = hashlib.sha1(payload).hexdigest()
md5 = hashlib.md5(payload).hexdigest()
sha256 = hashlib.sha256(payload).hexdigest()
files[sha1] = {
"path": str(path),
"name": name,
"size": len(payload),
"md5": md5,
"sha256": sha256,
"crc32": "00000000",
}
by_name.setdefault(name, []).append(sha1)
if suffix and name == "firmware.bin":
by_suffix.setdefault(suffix, []).append(sha1)
return {
"files": files,
"indexes": {
"by_name": by_name,
"by_md5": {entry["md5"]: sha1 for sha1, entry in files.items()},
"by_sha256": {
entry["sha256"]: sha1 for sha1, entry in files.items()
},
"by_crc32": {},
"by_path_suffix": by_suffix,
},
}
def test_hash_identity_wins_over_wrong_destination_hint(self):
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
root = Path(directory)
wrong = root / "wrong" / "firmware.bin"
correct = root / "correct" / "firmware.bin"
wrong.parent.mkdir()
correct.parent.mkdir()
wrong.write_bytes(b"wrong")
correct.write_bytes(b"correct")
db = self._database(
{"firmware.bin": wrong, "correct-name.bin": correct},
suffix="Console/USA/firmware.bin",
)
expected = hashlib.sha1(b"correct").hexdigest()
path, status = resolve_local_file(
{"name": "firmware.bin", "sha1": expected},
db,
dest_hint="Console/USA/firmware.bin",
)
self.assertEqual(path, str(correct))
self.assertEqual(status, "sha1_exact")
def test_name_cannot_mask_a_declared_hash_mismatch(self):
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
path = Path(directory) / "firmware.bin"
path.write_bytes(b"wrong")
db = self._database({"firmware.bin": path})
resolved, status = resolve_local_file(
{"name": "firmware.bin", "sha1": "f" * 40}, db
)
self.assertEqual(resolved, str(path))
self.assertEqual(status, "hash_mismatch")
class PackExclusionRegressions(unittest.TestCase):
def test_slug_platform_core_requirement_uses_the_generated_destination(self):
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
root = Path(directory)
platforms = root / "platforms"
platforms.mkdir()
core_payload = root / "extra.bin"
core_payload.write_bytes(b"mapped core payload")
core_sha1 = hashlib.sha1(core_payload.read_bytes()).hexdigest()
core_md5 = hashlib.md5(core_payload.read_bytes()).hexdigest()
database = {
"files": {
core_sha1: {
"name": "extra.bin",
"path": str(core_payload),
"md5": core_md5,
"size": core_payload.stat().st_size,
}
},
"indexes": {
"by_name": {"extra.bin": [core_sha1]},
"by_md5": {core_md5: core_sha1},
"by_sha256": {},
"by_crc32": {},
"by_path_suffix": {},
},
}
config = {
"platform": "Slug platform",
"base_destination": "bios",
"verification_mode": "existence",
"cores": ["core_a"],
"systems": {
"console-a": {
"files": [
{
"name": "base-a.bin",
"destination": "slug-a/base-a.bin",
}
]
},
"console-b": {
"files": [
{
"name": "base-b.bin",
"destination": "slug-b/base-b.bin",
}
]
},
},
}
(platforms / "slug.yml").write_text(
yaml.safe_dump(config), encoding="utf-8"
)
profiles = {
"core_a": {
"emulator": "Core A",
"type": "libretro",
"systems": ["console-a"],
"files": [
{
"name": "extra.bin",
"path": "extra.bin",
"sha1": core_sha1,
}
],
}
}
pack = root / "pack.zip"
with zipfile.ZipFile(pack, "w", zipfile.ZIP_DEFLATED) as archive:
archive.writestr("slug-a/base-a.bin", b"baseline a")
archive.writestr("slug-b/base-b.bin", b"baseline b")
archive.writestr("slug-a/extra.bin", core_payload.read_bytes())
result = verify_pack_against_platform(
str(pack),
"slug",
str(platforms),
db=database,
emu_profiles=profiles,
)
self.assertTrue(result[0], result[3])
self.assertEqual(result[3], [])
self.assertEqual(result[6:8], (1, 1))
def _mismatch_fixture(self, root: Path, mode: str) -> tuple[dict, Path]:
"""A platform declaring a hash the only local payload contradicts."""
platforms = root / "platforms"
platforms.mkdir()
payload = root / "firmware.bin"
payload.write_bytes(b"wrong local variant")
sha1 = hashlib.sha1(payload.read_bytes()).hexdigest()
md5 = hashlib.md5(payload.read_bytes()).hexdigest()
database = {
"files": {
sha1: {
"name": "firmware.bin",
"path": str(payload),
"md5": md5,
"size": payload.stat().st_size,
}
},
"indexes": {
"by_name": {"firmware.bin": [sha1]},
"by_md5": {md5: sha1},
"by_sha256": {},
"by_crc32": {},
"by_path_suffix": {},
},
}
config = {
"platform": f"Platform {mode}",
"verification_mode": mode,
"systems": {
"console": {
"files": [
{
"name": "firmware.bin",
"destination": "firmware.bin",
"sha1": "f" * 40,
}
]
}
},
}
(platforms / "plat.yml").write_text(yaml.safe_dump(config), encoding="utf-8")
return database, platforms
def test_hash_platform_accounts_for_an_unsafe_exclusion(self):
"""A platform that reads the bytes would reject the local payload."""
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
root = Path(directory)
database, platforms = self._mismatch_fixture(root, "md5")
pack = root / "pack.zip"
with zipfile.ZipFile(pack, "w", zipfile.ZIP_DEFLATED) as archive:
archive.writestr("README.txt", "safe subset")
result = verify_pack_against_platform(
str(pack),
"plat",
str(platforms),
db=database,
emu_profiles={},
)
self.assertTrue(result[0], result[3])
self.assertEqual(result[3], [])
self.assertEqual(result[8], 1)
def test_existence_platform_never_withholds_over_a_declared_hash(self):
"""RetroArch and friends only look for the filename.
An upstream hash the local dump contradicts must not remove a file the
frontend would have loaded, so its absence stays a conformance error.
"""
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
root = Path(directory)
database, platforms = self._mismatch_fixture(root, "existence")
pack = root / "pack.zip"
with zipfile.ZipFile(pack, "w", zipfile.ZIP_DEFLATED) as archive:
archive.writestr("README.txt", "no firmware")
result = verify_pack_against_platform(
str(pack),
"plat",
str(platforms),
db=database,
emu_profiles={},
)
self.assertFalse(result[0])
self.assertTrue(any("baseline missing" in e for e in result[3]), result[3])
self.assertEqual(result[8], 0)
def test_unexplained_missing_file_still_fails_conformance(self):
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
root = Path(directory)
platforms = root / "platforms"
platforms.mkdir()
config = {
"platform": "Missing",
"verification_mode": "existence",
"systems": {
"console": {
"files": [
{"name": "absent.bin", "destination": "absent.bin"}
]
}
},
}
(platforms / "missing.yml").write_text(
yaml.safe_dump(config), encoding="utf-8"
)
pack = root / "pack.zip"
with zipfile.ZipFile(pack, "w", zipfile.ZIP_DEFLATED) as archive:
archive.writestr("README.txt", "incomplete")
database = {
"files": {},
"indexes": {
"by_name": {},
"by_md5": {},
"by_sha256": {},
"by_crc32": {},
"by_path_suffix": {},
},
}
result = verify_pack_against_platform(
str(pack),
"missing",
str(platforms),
db=database,
emu_profiles={},
)
self.assertFalse(result[0])
self.assertTrue(any("baseline missing" in error for error in result[3]))
class RegionRegressions(unittest.TestCase):
def test_tagged_and_untagged_same_path_is_preserved(self):
profiles = {
"tagged": {
"type": "libretro",
"files": [
{"name": "bios.bin", "path": "sys/bios.bin", "region": ["japan"]}
],
},
"untagged": {
"type": "libretro",
"files": [{"name": "bios.bin", "path": "sys/bios.bin"}],
},
}
index = region.build_region_index(profiles)
self.assertEqual(region.lookup_regions(index, "sys/bios.bin", "bios.bin"), set())
drops = region.resolve_region_drops(
{"system": [("sys/bios.bin", "bios.bin")]},
index,
["north-america"],
)
self.assertEqual(drops, set())
def test_world_candidate_beats_unmatched_regional_fallback(self):
profiles = {
"core": {
"type": "libretro",
"files": [
{"name": "ntsc.bin", "region": ["world"]},
{"name": "pal.bin", "region": ["europe"]},
],
}
}
index = region.build_region_index(profiles)
groups = {"system": [("ntsc.bin", "ntsc.bin"), ("pal.bin", "pal.bin")]}
self.assertEqual(
region.resolve_region_drops(groups, index, ["north-america"]),
{"pal.bin"},
)
self.assertEqual(region.resolve_region_drops(groups, index, ["europe"]), set())
def test_multi_system_emulator_uses_separate_region_groups(self):
profile = {"systems": ["odyssey2", "videopac"]}
north_america = _emulator_region_group(
"o2em", profile, {"name": "o2rom.bin", "system": "odyssey2"}
)
europe = _emulator_region_group(
"o2em", profile, {"name": "c52.bin", "system": "videopac"}
)
self.assertNotEqual(north_america, europe)
def test_region_audit_accepts_list_valued_md5(self):
sha1 = "a" * 40
md5 = "b" * 32
db = {
"files": {sha1: {}},
"indexes": {"by_md5": {md5: sha1}, "by_name": {}},
}
self.assertEqual(
region_audit.resolve_sha1({"name": "bios.bin", "md5": [md5]}, db),
sha1,
)
class ArchiveSecurityRegressions(unittest.TestCase):
def _zip_path(self, directory: Path, name: str = "archive.zip") -> Path:
return directory / name
def test_member_count_limit_is_enforced(self):
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
root = Path(directory)
archive = self._zip_path(root)
with zipfile.ZipFile(archive, "w") as handle:
handle.writestr("one.bin", b"1")
handle.writestr("two.bin", b"2")
with self.assertRaisesRegex(ValueError, "members"):
safe_extract_zip(str(archive), str(root / "out"), max_members=1)
def test_symlink_member_is_rejected(self):
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
root = Path(directory)
archive = self._zip_path(root)
link = zipfile.ZipInfo("link")
link.create_system = 3
link.external_attr = (stat.S_IFLNK | 0o777) << 16
with zipfile.ZipFile(archive, "w") as handle:
handle.writestr(link, "target")
with self.assertRaisesRegex(ValueError, "link or special"):
safe_extract_zip(str(archive), str(root / "out"))
def test_windows_separator_is_a_path_not_a_rejection(self):
"""Archives written on Windows store a backslash separator.
download.py feeds third-party archives to this function, so a legal
Windows path must extract into a subdirectory instead of failing.
"""
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
root = Path(directory)
archive = self._zip_path(root)
with zipfile.ZipFile(archive, "w") as handle:
handle.writestr("sub\\rom.bin", b"payload")
out = root / "out"
safe_extract_zip(str(archive), str(out))
self.assertEqual((out / "sub" / "rom.bin").read_bytes(), b"payload")
def test_windows_separator_cannot_smuggle_traversal(self):
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
root = Path(directory)
archive = self._zip_path(root)
with zipfile.ZipFile(archive, "w") as handle:
handle.writestr("..\\escaped.bin", b"payload")
with self.assertRaisesRegex(ValueError, "traversal"):
safe_extract_zip(str(archive), str(root / "out"))
def test_high_ratio_method_is_bounded_by_size_not_ratio(self):
"""bzip2 legitimately exceeds the DEFLATE ceiling."""
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
root = Path(directory)
archive = self._zip_path(root)
with zipfile.ZipFile(archive, "w", zipfile.ZIP_BZIP2) as handle:
handle.writestr("zeros.bin", bytes(4_000_000))
out = root / "out"
safe_extract_zip(str(archive), str(out))
self.assertEqual((out / "zeros.bin").stat().st_size, 4_000_000)
def test_compression_ratio_limit_is_enforced(self):
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
root = Path(directory)
archive = self._zip_path(root)
with zipfile.ZipFile(archive, "w", zipfile.ZIP_DEFLATED) as handle:
handle.writestr("zeros.bin", bytes(16_384))
with self.assertRaisesRegex(ValueError, "compression ratio"):
safe_extract_zip(
str(archive), str(root / "out"), max_compression_ratio=2
)
class InstallerBoundaryRegressions(unittest.TestCase):
def _manifest(self, dest: str) -> dict:
return {
"manifest_version": 2,
"platform": "retroarch",
"files": [
{
"dest": dest,
"size": 1,
"sha1": "a" * 40,
"sha256": "b" * 64,
"repo_path": "bios/test.bin",
"cores": None,
}
],
"standalone_copies": [],
}
def test_manifest_destination_traversal_is_rejected(self):
with self.assertRaisesRegex(ValueError, "unsafe"):
install._validate_manifest(self._manifest("../escape"), "retroarch")
def test_manifest_repo_source_is_confined_to_bios(self):
manifest = self._manifest("safe.bin")
manifest["files"][0]["repo_path"] = "scripts/pipeline.py"
with self.assertRaisesRegex(ValueError, "outside bios"):
install._validate_manifest(manifest, "retroarch")
def test_omitted_destination_cannot_overlap_a_download(self):
manifest = self._manifest("safe.bin")
manifest["omitted_files"] = [
{
"dest": "safe.bin",
"name": "safe.bin",
"system": "console",
"required": True,
"reason": "hash_mismatch",
"cores": None,
}
]
manifest["total_omitted"] = 1
with self.assertRaisesRegex(ValueError, "conflicting omitted"):
install._validate_manifest(manifest, "retroarch")
def test_omitted_destination_traversal_is_rejected(self):
manifest = self._manifest("safe.bin")
manifest["omitted_files"] = [
{
"dest": "../unsafe.bin",
"name": "unsafe.bin",
"system": "console",
"required": True,
"reason": "hash_mismatch",
"cores": None,
}
]
manifest["total_omitted"] = 1
with self.assertRaisesRegex(ValueError, "unsafe"):
install._validate_manifest(manifest, "retroarch")
def test_target_schema_accepts_the_null_the_generator_emits(self):
"""The schema and generate_target_manifests must agree on null.
A target with no core list is written as null; a schema that rejects
it turns a valid manifest into a CI failure.
"""
from jsonschema import Draft202012Validator
schema = json.loads(
(ROOT / "schemas" / "target-manifest.schema.json").read_text(
encoding="utf-8"
)
)
validator = Draft202012Validator(schema)
document = {"windows": None, "switch": ["a5200"]}
self.assertEqual(list(validator.iter_errors(document)), [])
def test_pack_manifests_are_read_from_inside_the_archive(self):
"""generate_pack writes manifest.json into the ZIP, not beside it.
A filesystem glob over dist/ matches nothing and reports success
without validating a single document.
"""
import validate_schemas
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
dist = Path(directory)
broken = {
"schema_version": 1,
"version": 1,
"generator": "retrobios generate_pack.py",
"generated": "2026-08-10T00:00:00Z",
"files": [{"path": "a.bin", "sha1": "nope", "md5": "nope",
"size": 1, "status": "verified", "name": "a.bin"}],
"summary": {"total_files": 1, "verified": 1, "untracked": 0,
"errors": 0},
"errors": [],
}
with zipfile.ZipFile(dist / "Pack.zip", "w") as archive:
archive.writestr("manifest.json", json.dumps(broken))
errors = validate_schemas._validate_pack_manifests(dist)
self.assertTrue(errors, "an invalid in-archive manifest must be reported")
self.assertTrue(any("sha1" in message for message in errors), errors)
def test_null_core_list_keeps_the_other_targets(self):
"""A target without a core inventory must not void the manifest.
generate_target_manifests emits null for a target that publishes no
core list; rejecting the document would silently disable --target for
every target on that platform.
"""
normalized = install._validate_targets({"windows": None, "switch": ["a5200"]})
self.assertIsNone(normalized["windows"]["cores"])
self.assertEqual(normalized["switch"]["cores"], ["a5200"])
def test_legacy_target_lists_are_normalized(self):
self.assertEqual(
install._validate_targets({"rpi4": ["core-a", "core-b"]}),
{"rpi4": {"cores": ["core-a", "core-b"]}},
)
class TargetManifestRegressions(unittest.TestCase):
def test_yaml_scalar_core_is_rejected_instead_of_leaking_to_json(self):
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
root = Path(directory)
source = root / "source"
output = root / "output"
source.mkdir()
(source / "platform.yml").write_text(
"targets:\n device:\n cores: [81, valid-core]\n",
encoding="utf-8",
)
with self.assertRaisesRegex(ValueError, "non-empty strings"):
generate_target_manifests(str(source), str(output))
if __name__ == "__main__":
unittest.main()
+48 -18
View File
@@ -754,7 +754,7 @@ class TestE2E(unittest.TestCase):
"sha1": self.files["present_req.bin"]["sha1"],
}
path, status = resolve_local_file(entry, self.db)
self.assertEqual(status, "exact")
self.assertEqual(status, "sha1_exact")
self.assertIn("present_req.bin", path)
def test_02_resolve_md5(self):
@@ -768,12 +768,12 @@ class TestE2E(unittest.TestCase):
def test_03_resolve_name_no_md5(self):
entry = {"name": "no_md5.bin"}
path, status = resolve_local_file(entry, self.db)
self.assertEqual(status, "exact")
self.assertEqual(status, "name_exact")
def test_04_resolve_alias(self):
entry = {"name": "alias_alt.bin", "aliases": []}
path, status = resolve_local_file(entry, self.db)
self.assertEqual(status, "exact")
self.assertEqual(status, "name_exact")
self.assertIn("alias_target.bin", path)
def test_05_resolve_truncated_md5(self):
@@ -1328,8 +1328,8 @@ class TestE2E(unittest.TestCase):
)
self.assertIsNotNone(usa_path)
self.assertIsNotNone(eur_path)
self.assertEqual(usa_status, "exact")
self.assertEqual(eur_status, "exact")
self.assertEqual(usa_status, "path_exact")
self.assertEqual(eur_status, "path_exact")
# Must be DIFFERENT files
self.assertNotEqual(usa_path, eur_path)
# Verify content
@@ -2125,6 +2125,7 @@ class TestE2E(unittest.TestCase):
self.db,
self.bios_dir,
output_dir,
offline=True,
)
self.assertIsNotNone(zip_path)
with zipfile.ZipFile(zip_path) as zf:
@@ -2894,6 +2895,7 @@ class TestE2E(unittest.TestCase):
self.bios_dir,
output_dir,
emulators_dir=self.emulators_dir,
offline=True,
)
self.assertIsNotNone(zip_path)
with zipfile.ZipFile(zip_path) as zf:
@@ -3381,15 +3383,18 @@ class TestE2E(unittest.TestCase):
self.bios_dir,
registry_path,
emulators_dir=self.emulators_dir,
offline=True,
)
self.assertEqual(manifest["manifest_version"], 1)
self.assertEqual(manifest["manifest_version"], 2)
self.assertEqual(manifest["regions"], [])
self.assertEqual(manifest["platform"], "test_existence")
self.assertEqual(manifest["display_name"], "TestExistence")
self.assertIn("generated", manifest)
self.assertIn("files", manifest)
self.assertIsInstance(manifest["files"], list)
self.assertEqual(manifest["total_files"], len(manifest["files"]))
self.assertEqual(manifest["total_omitted"], len(manifest["omitted_files"]))
self.assertGreater(len(manifest["files"]), 0)
self.assertEqual(manifest["base_destination"], "system")
self.assertEqual(
@@ -3437,9 +3442,18 @@ class TestE2E(unittest.TestCase):
self.bios_dir,
registry_path,
emulators_dir=self.emulators_dir,
offline=True,
)
self.assertGreater(len(manifest["files"]), 0)
self.assertGreater(len(manifest["omitted_files"]), 0)
self.assertTrue(
all(
entry["reason"] in {"hash_mismatch", "not_found"}
for entry in manifest["omitted_files"]
if entry["cores"] is None
)
)
for f in manifest["files"]:
if f.get("release_asset"):
continue
@@ -3474,6 +3488,7 @@ class TestE2E(unittest.TestCase):
self.bios_dir,
output_dir,
emulators_dir=self.emulators_dir,
offline=True,
)
self.assertIsNotNone(zip_path)
@@ -3495,6 +3510,7 @@ class TestE2E(unittest.TestCase):
self.bios_dir,
registry_path,
emulators_dir=self.emulators_dir,
offline=True,
)
# Detect flat vs nested ZIP to build expected paths
base = manifest.get("base_destination", "")
@@ -4351,7 +4367,7 @@ class TestE2E(unittest.TestCase):
exp = Exporter()
exp.export(truth, out, scraped_data=scraped)
content = open(out).read()
content = Path(out).read_text(encoding="utf-8")
self.assertIn('"psx"', content)
self.assertIn("scph5501.bin", content)
self.assertIn("b" * 32, content)
@@ -4387,7 +4403,7 @@ class TestE2E(unittest.TestCase):
exp = Exporter()
exp.export(truth, out, scraped_data=scraped)
content = open(out).read()
content = Path(out).read_text(encoding="utf-8")
self.assertIn("<biosList", content)
self.assertIn('platform="psx"', content)
self.assertIn("fullname=", content)
@@ -4429,7 +4445,7 @@ class TestE2E(unittest.TestCase):
exp = Exporter()
exp.export(truth, out, scraped_data=scraped)
data = _json.loads(open(out).read())
data = _json.loads(Path(out).read_text(encoding="utf-8"))
self.assertIn("psx", data)
self.assertTrue(
any("scph5501" in bf["file"] for bf in data["psx"]["biosFiles"])
@@ -4785,12 +4801,12 @@ struct BurnDriver BurnDrvneogeo = {
manifest_full = generate_manifest(
"test_existence", self.platforms_dir, self.db, self.bios_dir,
registry_path, emulators_dir=self.emulators_dir, emu_profiles=profiles,
source="full",
source="full", offline=True,
)
manifest_plat = generate_manifest(
"test_existence", self.platforms_dir, self.db, self.bios_dir,
registry_path, emulators_dir=self.emulators_dir, emu_profiles=profiles,
source="platform",
source="platform", offline=True,
)
self.assertLessEqual(manifest_plat["total_files"], manifest_full["total_files"])
self.assertEqual(manifest_plat.get("source"), "platform")
@@ -4992,7 +5008,11 @@ struct BurnDriver BurnDrvneogeo = {
sha256 = hl.sha256(b"dsp firmware bytes").hexdigest()
db = {
"files": {
"sha1x": {"name": "SNES_dsp1.rom", "path": path},
"sha1x": {
"name": "SNES_dsp1.rom",
"path": path,
"sha256": sha256,
},
},
"indexes": {
"by_md5": {},
@@ -5006,7 +5026,7 @@ struct BurnDriver BurnDrvneogeo = {
entry = {"name": "dsp1.rom", "sha256": sha256}
local, status = resolve_local_file(entry, db)
self.assertEqual(local, path)
self.assertEqual(status, "exact")
self.assertEqual(status, "sha256_exact")
# Multi-hash list also resolves
entry = {"name": "dsp1.rom", "sha256": f"{'0' * 64},{sha256}"}
local, status = resolve_local_file(entry, db)
@@ -5092,7 +5112,12 @@ struct BurnDriver BurnDrvneogeo = {
crc = f"{zlib.crc32(data) & 0xffffffff:08x}"
db = {
"files": {
"sha1e": {"name": "exos21.rom", "path": path, "size": len(data)},
"sha1e": {
"name": "exos21.rom",
"path": path,
"size": len(data),
"crc32": crc,
},
},
"indexes": {
"by_md5": {},
@@ -5106,7 +5131,7 @@ struct BurnDriver BurnDrvneogeo = {
{"name": "exos21.bin", "crc32": crc, "size": len(data)}, db
)
self.assertEqual(local, path)
self.assertEqual(status, "exact")
self.assertEqual(status, "crc32_exact")
# Wrong size rejects the crc match
local, status = resolve_local_file(
{"name": "exos21.bin", "crc32": crc, "size": 1}, db
@@ -5137,7 +5162,7 @@ struct BurnDriver BurnDrvneogeo = {
{"name": "VEC_MineStorm.vec"}, db
)
self.assertEqual(local, path)
self.assertEqual(status, "exact")
self.assertEqual(status, "name_exact")
def test_215_check_member_hash_inside_zip(self):
"""zipped_file entries verify the ROM inside the ZIP, not the ZIP."""
@@ -5421,10 +5446,11 @@ struct BurnDriver BurnDrvneogeo = {
self.assertNotIn("rgn_us.bin", names)
self.assertNotIn("rgn_jp.bin", names)
def test_region_filter_keeps_group_with_no_match(self):
def test_region_filter_uses_world_when_no_regional_match(self):
names = self._names_in(self._region_pack(regions=["oceania"]))
self.assertIn("rgn_world.bin", names)
for expected in ("rgn_us.bin", "rgn_jp.bin", "rgn_eu.bin"):
self.assertIn(expected, names)
self.assertNotIn(expected, names)
def test_region_filter_keeps_untagged_and_world(self):
names = self._names_in(self._region_pack(regions=["north-america"]))
@@ -5477,6 +5503,7 @@ struct BurnDriver BurnDrvneogeo = {
self.root,
emu_profiles=self._region_profiles(),
regions=["north-america"],
offline=True,
)
self.assertTrue(paths)
for p in paths:
@@ -5503,6 +5530,7 @@ struct BurnDriver BurnDrvneogeo = {
regions=["north-america"],
)
packed = {os.path.basename(f["dest"]) for f in manifest["files"]}
self.assertEqual(manifest["regions"], ["north-america"])
self.assertIn("rgn_us.bin", packed)
self.assertNotIn("rgn_jp.bin", packed)
self.assertNotIn("rgn_eu.bin", packed)
@@ -5522,6 +5550,7 @@ struct BurnDriver BurnDrvneogeo = {
self.bios_dir,
registry_path,
emu_profiles=self._region_profiles(),
offline=True,
)
packed = {os.path.basename(f["dest"]) for f in manifest["files"]}
for expected in ("rgn_us.bin", "rgn_jp.bin", "rgn_eu.bin"):
@@ -5545,6 +5574,7 @@ struct BurnDriver BurnDrvneogeo = {
regions=["north-america"],
)
self.assertIsNotNone(out)
self.assertIn("NorthAmerica", os.path.basename(out))
names = self._names_in(out)
self.assertIn("rgn_us.bin", names)
self.assertNotIn("rgn_jp.bin", names)
+30
View File
@@ -126,6 +126,36 @@ class LargeFileCacheTest(unittest.TestCase):
common.fetch_large_file("asset.bin", dest_dir=self.dir), str(cached)
)
def test_offline_cache_miss_never_opens_the_network(self):
def fail(req, timeout=None):
raise AssertionError("offline mode must not open the network")
common.urllib.request.urlopen = fail
self.assertIsNone(
common.fetch_large_file(
"asset.bin", dest_dir=self.dir, offline=True
)
)
self.assertEqual(os.listdir(self.dir), [])
def test_offline_mode_still_uses_a_verified_cache_hit(self):
cached = Path(self.dir) / "asset.bin"
cached.write_bytes(PAYLOAD_A)
def fail(req, timeout=None):
raise AssertionError("offline cache hit must not open the network")
common.urllib.request.urlopen = fail
self.assertEqual(
common.fetch_large_file(
"asset.bin",
dest_dir=self.dir,
expected_sha1=hashlib.sha1(PAYLOAD_A).hexdigest(),
offline=True,
),
str(cached),
)
if __name__ == "__main__":
unittest.main()
+53
View File
@@ -65,6 +65,59 @@ class TorrentZipTest(unittest.TestCase):
if not checked:
self.skipTest("no TorrentZip romset available")
def test_source_renamed_sets_stay_reproducible(self):
"""Archives whose member names we corrected must rebuild identically.
Renaming a ROM by hand produced archives that matched no upstream set
and could not be regenerated from their contents either. The corrected
primaries carry the MAME 0.289 names and TorrentZip bytes, so anyone
holding the ROMs reproduces them exactly.
"""
corrected = [
os.path.join(REPO_ROOT, "bios", "Arcade", "MAME", "manager.zip"),
os.path.join(REPO_ROOT, "bios", "Arcade", "Arcade", "sys573.zip"),
os.path.join(REPO_ROOT, "bios", "Arcade", "Arcade", "cedmag.zip"),
os.path.join(
REPO_ROOT, "bios", "Entex", "Adventure Vision", "advision.zip"
),
]
checked = 0
for path in corrected:
if not os.path.exists(path):
continue
self.assertTrue(is_torrentzip(path), path)
with open(path, "rb") as fh:
original = fh.read()
self.assertEqual(rebuild_torrentzip(path), original, path)
checked += 1
if not checked:
self.skipTest("corrected romsets not present in this checkout")
def test_source_renamed_sets_carry_the_source_names(self):
"""The member names come from MAME, not from a YAML round-trip."""
expected = {
os.path.join(REPO_ROOT, "bios", "Arcade", "MAME", "manager.zip"): {
"01", "23",
},
os.path.join(
REPO_ROOT, "bios", "Entex", "Adventure Vision", "advision.zip"
): {"ins8048-11kdp_n.u5", "cop411l-kcn_n.u8"},
}
checked = 0
for path, names in expected.items():
if not os.path.exists(path):
continue
with zipfile.ZipFile(path) as archive:
members = {
info.filename
for info in archive.infolist()
if not info.is_dir()
}
self.assertEqual(members, names, path)
checked += 1
if not checked:
self.skipTest("corrected romsets not present in this checkout")
def test_identify_romset_selects_matching_recipe(self):
atoms = {"3610a686": b"alpha", "9d0d1b46": b"beta"}
import zlib