Files
libretro/scripts/cross_reference.py
T
Abdessamad Derraz 8c018849ae fix: return instead of continue in the split profile
The cross-reference loop body became its own function, but one exit
stayed a continue and left the module unparseable, which took the whole
test suite down with it. The other ten continues are inside genuine
inner loops and stand.

Output verified identical to the version before the split.
2026-08-11 18:41:04 +02:00

558 lines
19 KiB
Python

#!/usr/bin/env python3
"""Cross-reference emulator profiles against platform configs.
Identifies BIOS files that emulators need but platforms don't declare,
providing gap analysis for extended coverage.
Usage:
python scripts/cross_reference.py
python scripts/cross_reference.py --emulator dolphin
python scripts/cross_reference.py --json
"""
from __future__ import annotations
import argparse
import json
import os
import sys
from pathlib import Path
sys.path.insert(0, os.path.dirname(__file__))
from common import (
list_registered_platforms,
load_database,
load_emulator_profiles,
load_platform_config,
name_match_size_ok,
parse_md5_list,
require_yaml,
)
yaml = require_yaml()
DEFAULT_EMULATORS_DIR = "emulators"
DEFAULT_PLATFORMS_DIR = "platforms"
DEFAULT_DB = "database.json"
def load_platform_files(
platforms_dir: str,
platforms: list[str] | None = None,
) -> tuple[dict[str, set[str]], dict[str, set[str]]]:
"""Collect declared filenames + data_directories per system.
Restricted to *platforms* when given, otherwise every registered platform.
"""
declared = {}
platform_data_dirs = {}
for platform_name in platforms or list_registered_platforms(
platforms_dir, include_archived=True
):
config = load_platform_config(platform_name, platforms_dir)
for sys_id, system in config.get("systems", {}).items():
for fe in system.get("files", []):
name = fe.get("name", "")
if name:
declared.setdefault(sys_id, set()).add(name)
for dd in system.get("data_directories", []):
ref = dd.get("ref", "")
if ref:
platform_data_dirs.setdefault(sys_id, set()).add(ref)
return declared, platform_data_dirs
def _build_supplemental_index(
data_root: str = "data", bios_root: str = "bios"
) -> set[str]:
"""Build a set of filenames and directory names in data/ and inside bios/ ZIPs."""
names: set[str] = set()
root_path = Path(data_root)
if root_path.is_dir():
for fpath in root_path.rglob("*"):
if fpath.name.startswith("."):
continue
names.add(fpath.name)
names.add(fpath.name.lower())
if fpath.is_dir():
# Also index relative path from data/subdir/ for directory entries
parts = fpath.relative_to(root_path).parts
if len(parts) > 1:
rel = "/".join(parts[1:])
names.add(rel)
names.add(rel + "/")
names.add(rel.lower())
names.add(rel.lower() + "/")
bios_path = Path(bios_root)
if bios_path.is_dir():
# Index directory names for directory-type entries (e.g., "nestopia/samples/moepro/")
for dpath in bios_path.rglob("*"):
if dpath.is_dir() and not dpath.name.startswith("."):
names.add(dpath.name)
names.add(dpath.name.lower())
names.add(dpath.name + "/")
names.add(dpath.name.lower() + "/")
import zipfile
for zpath in bios_path.rglob("*.zip"):
try:
with zipfile.ZipFile(zpath) as zf:
for member in zf.namelist():
if not member.endswith("/"):
basename = (
member.rsplit("/", 1)[-1] if "/" in member else member
)
names.add(basename)
names.add(basename.lower())
except (zipfile.BadZipFile, OSError):
pass
return names
def _resolve_source(
fname: str,
by_name: dict[str, list],
by_name_lower: dict[str, str],
data_names: set[str] | None = None,
by_path_suffix: dict | None = None,
file_entry: dict | None = None,
db_files: dict | None = None,
) -> str | None:
"""Return the source category for a file, or None if not found.
Returns ``"bios"`` (in database.json / bios/), ``"data"`` (in data/),
or ``None`` (not available anywhere).
A name hit whose size contradicts a size the emulator verifies is not the
file, so it does not count as available.
"""
def _name_hit(name: str) -> bool:
"""Whether the files indexed under *name* can be this entry."""
if not file_entry or db_files is None:
return True
sha1s = by_name.get(name) or []
if not isinstance(sha1s, list):
sha1s = [sha1s]
return any(
name_match_size_ok(file_entry, db_files.get(sha1, {}).get("size"))
for sha1 in sha1s
)
# bios/ via database.json by_name
if fname in by_name and _name_hit(fname):
return "bios"
stripped = fname.rstrip("/")
basename = stripped.rsplit("/", 1)[-1] if "/" in stripped else None
if basename and basename in by_name and _name_hit(basename):
return "bios"
key = fname.lower()
if key in by_name_lower and _name_hit(by_name_lower[key]):
return "bios"
if basename:
if basename.lower() in by_name_lower and _name_hit(
by_name_lower[basename.lower()]
):
return "bios"
# bios/ via by_path_suffix (regional variants)
if by_path_suffix and fname in by_path_suffix:
return "bios"
# data/ supplemental index
if data_names:
if fname in data_names or key in data_names:
return "data"
if basename and (basename in data_names or basename.lower() in data_names):
return "data"
return None
def _resolve_archive_source(
archive_name: str,
by_name: dict[str, list],
by_name_lower: dict[str, str],
data_names: set[str] | None = None,
by_path_suffix: dict | None = None,
) -> str:
"""Resolve source for an archive (ZIP) name, returning a source category string."""
result = _resolve_source(
archive_name, by_name, by_name_lower, data_names, by_path_suffix,
)
return result if result is not None else "missing"
def _cross_reference_profile(
emu_name: str,
profile: dict,
declared: dict[str, set[str]],
report: dict,
index: dict,
) -> None:
"""Compare one emulator profile against what the platforms declare.
Split out of cross_reference, where it was a 200-line loop body that
carried the whole function's branching: every file is resolved by name,
by path, by hash and by archive membership before it can be called a
gap. *index* carries the database lookups those steps share.
"""
by_name = index["by_name"]
by_name_lower = index["by_name_lower"]
by_md5 = index["by_md5"]
by_crc32 = index["by_crc32"]
by_path_suffix = index["by_path_suffix"]
db_files = index["db_files"]
data_names = index["data_names"]
all_declared = index["all_declared"]
emu_files = profile.get("files", [])
systems = profile.get("systems", [])
# Skip filename-agnostic profiles (BIOS detected without fixed names)
if profile.get("bios_mode") == "agnostic":
return
if all_declared is not None:
platform_names = all_declared
else:
platform_names = set()
for sys_id in systems:
platform_names.update(declared.get(sys_id, set()))
gaps = []
covered = []
unsourceable_list: list[dict] = []
archive_gaps: dict[tuple, dict] = {}
seen_files: set[tuple] = set()
for f in emu_files:
fname = f.get("name", "")
effective_path = f.get("path") or fname
seen_key = (
fname,
f.get("archive"),
effective_path,
f.get("system"),
f.get("variant_group"),
f.get("mode", "both"),
)
if not fname or seen_key in seen_files:
continue
# Collect unsourceable files separately (documented, not a gap)
unsourceable_reason = f.get("unsourceable", "")
if unsourceable_reason:
seen_files.add(seen_key)
unsourceable_list.append({
"name": fname,
"required": f.get("required", False),
"reason": unsourceable_reason,
"source_ref": f.get("source_ref", ""),
})
continue
# Skip pattern placeholders (e.g., <bios>.bin, <user-selected>.bin)
if "<" in fname or ">" in fname or "*" in fname:
continue
# Skip UI-imported files with explicit path: null (not resolvable by pack)
if "path" in f and f["path"] is None:
continue
# Skip standalone-only files
file_mode = f.get("mode", "both")
if file_mode == "standalone":
continue
# Skip files loaded from non-system directories (save_dir, content_dir)
load_from = f.get("load_from", "")
if load_from and load_from != "system_dir":
continue
# Skip filename-agnostic files (handled by agnostic scan)
if f.get("agnostic"):
continue
archive = f.get("archive")
# Check platform declaration (by name or archive)
in_platform = fname in platform_names
if not in_platform and archive:
in_platform = archive in platform_names
if in_platform:
seen_files.add(seen_key)
covered.append({
"name": fname,
"path": effective_path,
"required": f.get("required", False),
"in_platform": True,
})
continue
seen_files.add(seen_key)
# Group archived files by archive name
if archive:
archive_key = (
archive,
f.get("system"),
f.get("variant_group"),
f.get("mode", "both"),
)
if archive_key not in archive_gaps:
source = _resolve_archive_source(
archive, by_name, by_name_lower, data_names,
by_path_suffix,
)
archive_gaps[archive_key] = {
"name": archive,
"path": archive,
"required": False,
"note": "",
"source_ref": "",
"in_platform": False,
"in_repo": source != "missing",
"source": source,
"archive": archive,
"archive_file_count": 0,
"archive_required_count": 0,
}
entry = archive_gaps[archive_key]
entry["archive_file_count"] += 1
if f.get("required", False):
entry["archive_required_count"] += 1
entry["required"] = True
if not entry["source_ref"] and f.get("source_ref"):
entry["source_ref"] = f["source_ref"]
continue
# --- resolve source provenance ---
storage = f.get("storage", "")
if storage in ("release", "large_file"):
source = "large_file"
else:
source = _resolve_source(
fname, by_name, by_name_lower, data_names, by_path_suffix,
f, db_files,
)
if source is None:
path_field = f.get("path", "")
if path_field and path_field != fname:
source = _resolve_source(
path_field, by_name, by_name_lower,
data_names, by_path_suffix, f, db_files,
)
# Try the alternate names the emulator accepts, like
# resolve_local_file does
if source is None:
for alias in f.get("aliases") or []:
source = _resolve_source(
alias, by_name, by_name_lower,
data_names, by_path_suffix, f, db_files,
)
if source is not None:
break
# Try MD5 hash match
if source is None:
for md5_val in parse_md5_list(f.get("md5")):
if by_md5.get(md5_val):
source = "bios"
break
# Try SHA1 hash match
if source is None:
raw_sha1 = f.get("sha1", "")
sha1_values = raw_sha1 if isinstance(raw_sha1, list) else [raw_sha1]
if any(value and value in db_files for value in sha1_values):
source = "bios"
# Try CRC32 hash match
if source is None:
crc32 = str(f.get("crc32", "")).lower()
if crc32 and by_crc32.get(crc32):
source = "bios"
if source is None:
source = "missing"
in_repo = source != "missing"
entry = {
"name": fname,
"path": effective_path,
"required": f.get("required", False),
"note": f.get("note", ""),
"source_ref": f.get("source_ref", ""),
"in_platform": False,
"in_repo": in_repo,
"source": source,
}
gaps.append(entry)
# Append grouped archive gaps
for ag in sorted(archive_gaps.values(), key=lambda e: e["name"]):
gaps.append(ag)
report[emu_name] = {
"emulator": profile.get("emulator", emu_name),
"systems": systems,
"total_files": len(emu_files),
"platform_covered": len(covered),
"gaps": len(gaps),
"gap_in_repo": sum(1 for g in gaps if g["in_repo"]),
"gap_missing": sum(1 for g in gaps if g["source"] == "missing"),
"gap_bios": sum(1 for g in gaps if g["source"] == "bios"),
"gap_data": sum(1 for g in gaps if g["source"] == "data"),
"gap_large_file": sum(1 for g in gaps if g["source"] == "large_file"),
"gap_details": gaps,
"unsourceable": unsourceable_list,
}
def cross_reference(
profiles: dict[str, dict],
declared: dict[str, set[str]],
db: dict,
platform_data_dirs: dict[str, set[str]] | None = None,
data_names: set[str] | None = None,
all_declared: set[str] | None = None,
) -> dict:
"""Compare emulator profiles against platform declarations.
Returns a report with gaps (files emulators need but platforms don't list)
and coverage stats. Each gap entry carries a ``source`` field indicating
where the file is available: ``"bios"`` (bios/ via database.json),
``"data"`` (data/ directory), ``"large_file"`` (GitHub release asset),
or ``"missing"`` (not available anywhere).
The boolean ``in_repo`` is derived: ``source != "missing"``.
When *all_declared* is provided (flat set of every filename declared by
any platform for any system), it is used for the ``in_platform`` check
instead of the per-system lookup. This is appropriate for the global
gap analysis page where "undeclared" means "no platform declares it at all".
"""
platform_data_dirs = platform_data_dirs or {}
by_name = db.get("indexes", {}).get("by_name", {})
by_name_lower = {k.lower(): k for k in by_name}
by_md5 = db.get("indexes", {}).get("by_md5", {})
by_crc32 = db.get("indexes", {}).get("by_crc32", {})
by_path_suffix = db.get("indexes", {}).get("by_path_suffix", {})
db_files = db.get("files", {})
report = {}
index = {
"by_name": by_name,
"by_name_lower": by_name_lower,
"by_md5": by_md5,
"by_crc32": by_crc32,
"by_path_suffix": by_path_suffix,
"db_files": db_files,
"data_names": data_names,
"all_declared": all_declared,
}
for emu_name, profile in profiles.items():
_cross_reference_profile(
emu_name, profile, declared, report, index
)
return report
def print_report(report: dict) -> None:
"""Print a human-readable gap analysis report."""
print("Emulator vs Platform Gap Analysis")
print("=" * 60)
total_gaps = 0
totals: dict[str, int] = {"bios": 0, "data": 0, "large_file": 0, "missing": 0}
for emu_name, data in sorted(report.items()):
gaps = data["gaps"]
if gaps == 0:
continue
parts = []
for key in ("bios", "data", "large_file", "missing"):
count = data.get(f"gap_{key}", 0)
if count:
parts.append(f"{count} {key}")
status = ", ".join(parts) if parts else "OK"
print(f"\n{data['emulator']} ({', '.join(data['systems'])})")
print(
f" {data['total_files']} files in profile, "
f"{data['platform_covered']} declared by platforms, "
f"{gaps} undeclared"
)
print(f" Gaps: {status}")
for g in data["gap_details"]:
req = "*" if g["required"] else " "
src = g.get("source", "missing").upper()
note = f" -- {g['note']}" if g["note"] else ""
archive_info = ""
if g.get("archive"):
fc = g.get("archive_file_count", 0)
rc = g.get("archive_required_count", 0)
archive_info = f" ({fc} files, {rc} required)"
print(f" {req} {g['name']} [{src}]{archive_info}{note}")
total_gaps += gaps
for key in totals:
totals[key] += data.get(f"gap_{key}", 0)
print(f"\n{'=' * 60}")
print(f"Total: {total_gaps} undeclared files across all emulators")
available = totals["bios"] + totals["data"] + totals["large_file"]
print(f" {available} available (bios: {totals['bios']}, data: {totals['data']}, "
f"large_file: {totals['large_file']})")
print(f" {totals['missing']} missing (need to be sourced)")
def main():
parser = argparse.ArgumentParser(description="Emulator vs platform gap analysis")
parser.add_argument("--emulators-dir", default=DEFAULT_EMULATORS_DIR)
parser.add_argument("--platforms-dir", default=DEFAULT_PLATFORMS_DIR)
parser.add_argument("--db", default=DEFAULT_DB)
parser.add_argument("--emulator", "-e", help="Analyze single emulator")
parser.add_argument(
"--platform", "-p", help="Restrict analysis to one platform's cores"
)
parser.add_argument("--target", "-t", help="Hardware target (e.g., switch, rpi4)")
parser.add_argument("--json", action="store_true", help="JSON output")
args = parser.parse_args()
profiles = load_emulator_profiles(args.emulators_dir)
if args.emulator:
profiles = {k: v for k, v in profiles.items() if k == args.emulator}
if args.target and not args.platform:
parser.error("--target requires --platform")
if args.platform:
from common import load_target_config, resolve_platform_cores
target_cores = (
load_target_config(args.platform, args.target, args.platforms_dir)
if args.target
else None
)
config = load_platform_config(args.platform, args.platforms_dir)
relevant = resolve_platform_cores(config, profiles, target_cores=target_cores)
profiles = {k: v for k, v in profiles.items() if k in relevant}
if not profiles:
print("No emulator profiles found.", file=sys.stderr)
return
declared, plat_data_dirs = load_platform_files(
args.platforms_dir, [args.platform] if args.platform else None
)
db = load_database(args.db)
data_names = _build_supplemental_index()
report = cross_reference(profiles, declared, db, plat_data_dirs, data_names)
if args.json:
print(json.dumps(report, indent=2))
else:
print_report(report)
if __name__ == "__main__":
main()