feat: keep cached platform originals in step

This commit is contained in:
Abdessamad Derraz committed 2026-10-04 16:36:35 +02:00
1 parent d50af62cde
commit b81199ff26
8 files changed
+654 -68

No files matched your search

+89 -1
View File
@@ -12,6 +12,11 @@ into a throwaway copy, then diffed against the committed file. The diff
decides, not a version string: a scraper that pins a new tag but produces
the same entries is reported as a version change and nothing else.
The native export patches each platform's own file, kept in a local cache.
That cache is checked against the platform file it was transcribed with: an
original cached before the last rescrape, or under another pin, is no longer
the file our data describes.
Emulator profiles are covered by profile_sync, which is slow (one API pass
per profile). ``--profiles`` runs it here and folds its verdict in.
@@ -43,6 +48,7 @@ from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
import export_native # noqa: E402
import refresh_data_dirs # noqa: E402
import upstream # noqa: E402
from common import ( # noqa: E402
@@ -50,6 +56,8 @@ from common import ( # noqa: E402
load_emulator_profiles,
yaml_load,
)
from exporter import discover_exporters # noqa: E402
from exporter.baseline import build_native_model # noqa: E402
from scripts.scraper.redump_dat_scraper import fetch_snapshot # noqa: E402
from scripts.scraper.targets import discover_target_scrapers # noqa: E402
@@ -66,7 +74,7 @@ INSTALLERS = ("install.sh", "install.ps1")
SHOWN_ITEMS = 4 # items named in a diff summary before ", ..."
SHOWN_NAMES = 12 # core names named in an unresolved row before ", ..."
AREAS = ("platforms", "targets", "coreinfo", "data", "catalogs", "ci", "profiles")
AREAS = ("platforms", "native", "targets", "coreinfo", "data", "catalogs", "ci", "profiles")
OK, STALE, UNKNOWN, ERROR, SKIPPED = "OK", "STALE", "UNKNOWN", "ERROR", "SKIPPED"
# Every status a finding may carry, in the order the summary counts them.
STATUSES = (STALE, UNKNOWN, ERROR, SKIPPED, OK)
@@ -279,6 +287,80 @@ def check_platforms(platforms_dir: Path, workdir: Path, jobs: int) -> list[Findi
return sorted(findings, key=lambda f: f.subject)
# --- native originals --------------------------------------------------------
def native_cache_state(
wanted: dict[str, str],
root: Path,
recorded: dict[str, str],
index: Path,
transcribed_at: float,
) -> str:
"""Why a platform's cached originals cannot be patched, '' when they can.
The export corrects the file our data was transcribed from. A cached
original is that file only if it came from the URL the platform file
names today and was fetched no earlier than the platform file was last
rewritten: a branch URL keeps its name while its content moves.
"""
for relative, url in sorted(wanted.items()):
path = root / relative
if not path.is_file():
return f"{relative} not cached"
known = recorded.get(export_native.source_key(path, index))
if known is None:
return f"{relative} cached from an unrecorded URL"
if known != url:
return f"{relative} cached from another revision"
if path.stat().st_mtime < transcribed_at:
return f"{relative} cached before the platform file was rewritten"
return ""
def _transcribed_at(platforms_dir: Path, platform: str) -> float:
"""Last rewrite of the platform file or of any file it inherits."""
newest = 0.0
seen: set[str] = set()
name: str | None = platform
while name and name not in seen:
seen.add(name)
path = platforms_dir / f"{name}.yml"
if not path.is_file():
break
newest = max(newest, path.stat().st_mtime)
with path.open(encoding="utf-8") as fh:
name = (yaml_load(fh) or {}).get("inherits")
return newest
def check_native(platforms_dir: Path, cache_dir: Path, truth_dir: Path) -> list[Finding]:
exporters = discover_exporters()
index = cache_dir / export_native.SOURCES_INDEX
recorded = export_native.load_sources(cache_dir)
findings: list[Finding] = []
for name in list_registered_platforms(str(platforms_dir), include_archived=True):
exporter_class = exporters.get(name)
if not exporter_class:
continue
truth, scraped = export_native.load_inputs(name, truth_dir, str(platforms_dir))
systems, _ = build_native_model(truth or {}, scraped)
wanted = export_native.wanted_sources(exporter_class(), systems, scraped)
reason = native_cache_state(
wanted, cache_dir / name, recorded, index, _transcribed_at(platforms_dir, name)
)
findings.append(
Finding(
"native",
name,
STALE if reason else OK,
local=f"{len(wanted)} file(s)",
detail=f"{reason}, export_native.py --refresh-cache fetches it" if reason else "",
)
)
return sorted(findings, key=lambda f: f.subject)
# --- targets ----------------------------------------------------------------
@@ -959,6 +1041,8 @@ def main() -> int:
parser.add_argument("--offline", action="store_true", help="no network, local ages only")
parser.add_argument("--jobs", type=int, default=4, help="parallel scrapers")
parser.add_argument("--cache-dir", default=".cache/upstream")
parser.add_argument("--native-cache-dir", default=export_native.DEFAULT_CACHE)
parser.add_argument("--truth-dir", default="dist/truth")
parser.add_argument("--platforms-dir", default="platforms")
parser.add_argument("--emulators-dir", default="emulators")
args = parser.parse_args()
@@ -979,6 +1063,10 @@ def main() -> int:
if args.offline
else check_platforms(platforms_dir, workdir, args.jobs)
)
if "native" in areas:
findings.extend(
check_native(platforms_dir, Path(args.native_cache_dir), Path(args.truth_dir))
)
if "targets" in areas:
findings.extend(check_targets(platforms_dir, workdir, profiles, args.jobs, args.offline))
if "coreinfo" in areas:
+124 -36
View File
@@ -10,6 +10,7 @@ a platform ships keeps working.
Usage:
python scripts/export_native.py --all --fetch
python scripts/export_native.py --platform recalbox --upstream-dir up/
python scripts/export_native.py --platform retrobat --refresh-cache
"""
from __future__ import annotations
@@ -28,6 +29,7 @@ from exporter import discover_exporters
from exporter.baseline import build_native_model
DEFAULT_CACHE = ".cache/upstream-native"
SOURCES_INDEX = ".sources.json"
_USER_AGENT = "retrobios-exporter/1.0"
_MAX_BYTES = 64 * 1024 * 1024
@@ -41,7 +43,29 @@ def _load_sources(index: Path | None) -> dict[str, str]:
return {}
def fetch(url: str, destination: Path, index: Path | None = None) -> bytes:
def source_key(destination: Path, index: Path | None) -> str:
"""How the index names a cached file: its path below the cache root.
A key that repeated the cache directory as it was typed changed with the
working directory, and a file recorded from one place looked unrecorded
from another.
"""
if index is not None:
try:
return destination.resolve().relative_to(index.parent.resolve()).as_posix()
except ValueError:
pass
return str(destination)
def load_sources(upstream_dir: Path) -> dict[str, str]:
"""Cached file -> URL it was fetched from, for a whole cache directory."""
return _load_sources(upstream_dir / SOURCES_INDEX)
def fetch(
url: str, destination: Path, index: Path | None = None, refresh: bool = False
) -> bytes:
"""Download an original once, then read it from the cache.
The cache path carries the file's own name and nothing of the revision it
@@ -50,10 +74,15 @@ def fetch(url: str, destination: Path, index: Path | None = None) -> bytes:
URL that produced each cached file is recorded beside the cache, and a
different URL refetches. A cache written before this index existed keeps
being served: nothing recorded means nothing contradicted.
A branch URL names no revision, so the same URL serves new bytes after
the platform moves. refresh downloads again whatever the cache holds:
it is what a rescrape calls, so the original and its transcription
describe the same moment.
"""
recorded = _load_sources(index)
key = str(destination)
if destination.exists() and recorded.get(key, url) == url:
key = source_key(destination, index)
if not refresh and destination.exists() and recorded.get(key, url) == url:
return destination.read_bytes()
request = urllib.request.Request(url, headers={"User-Agent": _USER_AGENT})
with urllib.request.urlopen(request, timeout=60) as response:
@@ -94,14 +123,10 @@ def _raw_url(url: str) -> str:
return url
def collect_originals(
exporter: object,
systems: dict,
upstream_dir: Path,
allow_fetch: bool,
scraped: dict | None = None,
) -> tuple[dict[str, str], list[str]]:
"""Gather the platform's own files, from disk or from upstream."""
def wanted_sources(
exporter: object, systems: dict, scraped: dict | None = None
) -> dict[str, str]:
"""Cache-relative name -> URL of every file the exporter patches."""
wanted = dict(exporter.native_sources())
components = getattr(exporter, "components", None)
if callable(components):
@@ -113,21 +138,39 @@ def collect_originals(
base = pinned_base(wanted, scraped)
if base:
wanted = {relative: base + relative for relative in wanted}
return wanted
def collect_originals(
exporter: object,
systems: dict,
upstream_dir: Path,
allow_fetch: bool,
scraped: dict | None = None,
refresh: bool = False,
) -> tuple[dict[str, str], list[str]]:
"""Gather the platform's own files, from disk or from upstream.
With fetching on, an existing file still goes through fetch(): reading
it straight from disk skipped the recorded URL, so a file cached under
one pin kept being patched after the platform YAML moved to the next.
"""
wanted = wanted_sources(exporter, systems, scraped)
root = upstream_dir / exporter.platform_name()
index = upstream_dir / SOURCES_INDEX
originals: dict[str, str] = {}
missing: list[str] = []
for relative, url in wanted.items():
path = root / relative
payload: bytes | None = None
if path.exists():
payload = path.read_bytes()
elif allow_fetch:
if allow_fetch:
try:
payload = fetch(url, path, upstream_dir / ".sources.json")
payload = fetch(url, path, index, refresh)
except (urllib.error.URLError, urllib.error.HTTPError, OSError) as exc:
missing.append(f"{relative}: {exc}")
continue
elif path.exists():
payload = path.read_bytes()
else:
missing.append(f"{relative}: absent and fetching is off")
continue
@@ -141,6 +184,42 @@ def collect_originals(
return originals, missing
def load_inputs(
platform: str, truth_dir: Path, platforms_dir: str
) -> tuple[dict | None, dict | None]:
"""The truth and the scraped config of a platform, None where absent."""
truth_file = truth_dir / f"{platform}.yml"
truth: dict | None = None
if truth_file.exists():
with open(truth_file) as handle:
truth = yaml_load(handle) or {}
try:
scraped = load_platform_config(platform, platforms_dir)
except (FileNotFoundError, OSError):
scraped = None
return truth, scraped
def refresh_cache(
platform: str,
exporter_class: type,
truth_dir: Path,
platforms_dir: str,
upstream_dir: Path,
) -> tuple[bool, list[str]]:
"""Download a platform's own files again. Returns (ok, messages)."""
truth, scraped = load_inputs(platform, truth_dir, platforms_dir)
systems, _ = build_native_model(truth or {}, scraped)
exporter = exporter_class()
wanted = wanted_sources(exporter, systems, scraped)
_, missing = collect_originals(
exporter, systems, upstream_dir, True, scraped, refresh=True
)
if missing:
return False, missing
return True, [f"{len(wanted)} file(s) fetched"]
def export_platform(
platform: str,
exporter_class: type,
@@ -153,19 +232,11 @@ def export_platform(
"""Write one platform's corrected file. Returns (ok, messages)."""
messages: list[str] = []
truth_file = truth_dir / f"{platform}.yml"
truth: dict = {}
if truth_file.exists():
with open(truth_file) as handle:
truth = yaml_load(handle) or {}
else:
truth, scraped = load_inputs(platform, truth_dir, platforms_dir)
if truth is None:
truth = {}
messages.append(f"no truth for {platform}, only the platform's own data")
try:
scraped = load_platform_config(platform, platforms_dir)
except (FileNotFoundError, OSError):
scraped = None
systems, report = build_native_model(truth, scraped)
if not systems:
return False, ["nothing to write: neither the platform nor the truth has data"]
@@ -253,6 +324,7 @@ def run(
platforms_dir: str,
upstream_dir: str,
allow_fetch: bool,
refresh_only: bool = False,
) -> int:
exporters = discover_exporters()
failures = 0
@@ -265,15 +337,24 @@ def run(
print(f" SKIP {platform}: no exporter")
continue
ok, messages = export_platform(
platform,
exporter_class,
Path(truth_dir),
Path(output_dir),
platforms_dir,
Path(upstream_dir),
allow_fetch,
)
if refresh_only:
ok, messages = refresh_cache(
platform,
exporter_class,
Path(truth_dir),
platforms_dir,
Path(upstream_dir),
)
else:
ok, messages = export_platform(
platform,
exporter_class,
Path(truth_dir),
Path(output_dir),
platforms_dir,
Path(upstream_dir),
allow_fetch,
)
label = "OK " if ok else "FAIL"
print(f" {label} {platform}")
for message in messages:
@@ -306,7 +387,13 @@ def main() -> None:
parser.add_argument(
"--fetch",
action="store_true",
help="download a platform's file when it is not in the cache",
help="download a platform's file when it is not in the cache, "
"or when the cache came from another URL",
)
parser.add_argument(
"--refresh-cache",
action="store_true",
help="download every platform file again and write no export",
)
args = parser.parse_args()
@@ -326,6 +413,7 @@ def main() -> None:
args.platforms_dir,
args.upstream_dir,
args.fetch,
args.refresh_cache,
)
)
+61 -18
View File
@@ -10,6 +10,10 @@ committed.
Mapping (stale -> command):
platforms/<name> python -m scripts.scraper.<module>_scraper
-o platforms/<name>.yml
then the native refresh below: the original
the export patches must be the one just scraped
native/<name> python scripts/export_native.py --platform <name>
--refresh-cache
targets/<name> python -m scripts.scraper.targets.<module>
-o platforms/targets/<name>.yml
data/<key> python scripts/refresh_data_dirs.py --key <key>
@@ -47,7 +51,7 @@ LOG_DIR = REPO_ROOT / "tmp" / "refresh_stale"
CHECK_FRESHNESS = REPO_ROOT / "scripts" / "check_freshness.py"
PLATFORMS_REGISTRY = REPO_ROOT / "platforms" / "_registry.yml"
AUTO_AREAS = ("platforms", "targets", "data", "catalogs")
AUTO_AREAS = ("platforms", "native", "targets", "data", "catalogs")
MANUAL_AREAS = ("coreinfo", "ci", "profiles")
JOB_TIMEOUT = 1800 # 30 minutes per refresher; scrapers rarely exceed 10
@@ -61,6 +65,8 @@ class Job:
subject: str
command: list[str]
log_path: Path
# Commands run after `command`, in order, only while each one succeeds.
then: tuple[tuple[str, ...], ...] = ()
@dataclass(frozen=True)
@@ -122,6 +128,15 @@ def plan_jobs(
seen: set[tuple[str, str]] = set()
surfaced: list[dict] = []
ignored: list[dict] = []
# A platform about to be rescraped refreshes its cached original itself,
# after the scrape: a separate native job would race the scraper.
rescraped = {
str(f.get("subject") or "")
for f in findings
if f.get("status") == "STALE"
and f.get("area") == "platforms"
and _command_for("platforms", str(f.get("subject") or ""), registry)
}
for finding in findings:
status = finding.get("status")
@@ -135,6 +150,10 @@ def plan_jobs(
ignored.append(finding)
continue
if area == "native" and subject in rescraped:
ignored.append(finding)
continue
command = _command_for(area, subject, registry)
if command is None:
surfaced.append(finding)
@@ -148,15 +167,33 @@ def plan_jobs(
seen.add(key)
log = LOG_DIR / f"{area}__{_slug(subject)}.log"
jobs.append(Job(area=area, subject=subject, command=command, log_path=log))
then: tuple[tuple[str, ...], ...] = ()
if area == "platforms":
then = (tuple(_native_refresh(subject)),)
jobs.append(
Job(area=area, subject=subject, command=command, log_path=log, then=then)
)
return jobs, surfaced, ignored
def _native_refresh(platform: str) -> list[str]:
return [
sys.executable,
"scripts/export_native.py",
"--platform",
platform,
"--refresh-cache",
]
def _command_for(
area: str, subject: str, registry: dict[str, dict]
) -> list[str] | None:
"""Map a stale finding to its refresh command, or None if manual."""
if area == "native":
return _native_refresh(subject)
if area == "platforms":
entry = registry.get(subject) or {}
scraper = entry.get("scraper")
@@ -250,23 +287,27 @@ def run_job(job: Job, env: dict[str, str]) -> JobResult:
"""Execute one refresh command and collect its output."""
start = time.monotonic()
job.log_path.parent.mkdir(parents=True, exist_ok=True)
returncode = 0
with job.log_path.open("w", encoding="utf-8") as log:
log.write(f"$ {' '.join(job.command)}\n")
log.flush()
try:
proc = subprocess.run(
job.command,
cwd=REPO_ROOT,
stdout=log,
stderr=subprocess.STDOUT,
env=env,
timeout=JOB_TIMEOUT,
check=False,
)
returncode = proc.returncode
except subprocess.TimeoutExpired:
log.write(f"\nTIMEOUT after {JOB_TIMEOUT}s\n")
returncode = 124
for command in (job.command, *job.then):
log.write(f"$ {' '.join(command)}\n")
log.flush()
try:
proc = subprocess.run(
list(command),
cwd=REPO_ROOT,
stdout=log,
stderr=subprocess.STDOUT,
env=env,
timeout=JOB_TIMEOUT,
check=False,
)
returncode = proc.returncode
except subprocess.TimeoutExpired:
log.write(f"\nTIMEOUT after {JOB_TIMEOUT}s\n")
returncode = 124
if returncode != 0:
break
duration = time.monotonic() - start
tail = _log_tail(job.log_path)
return JobResult(job=job, returncode=returncode, duration=duration, tail=tail)
@@ -388,6 +429,8 @@ def main() -> int:
f"{len(ignored)} already clean/skipped")
for job in sorted(jobs, key=lambda j: (j.area, j.subject)):
print(f" {job.area:10} {job.subject:28} {' '.join(job.command)}")
for command in job.then:
print(f" {'':10} {'':28} then {' '.join(command)}")
if surfaced:
print("\n[manual review needed]")
for f in surfaced:
+68
View File
@@ -8,7 +8,9 @@ report is folded, since each of those decides whether a row says STALE.
from __future__ import annotations
import os
import sys
import tempfile
import unittest
from pathlib import Path
@@ -19,6 +21,72 @@ import check_freshness as cf # noqa: E402
from common import yaml_load # noqa: E402
class NativeCacheTests(unittest.TestCase):
"""A cached original is patched only if it is the file we transcribed."""
def setUp(self):
self._tmp = tempfile.TemporaryDirectory()
self.addCleanup(self._tmp.cleanup)
self.cache = Path(self._tmp.name)
self.index = self.cache / cf.export_native.SOURCES_INDEX
self.root = self.cache / "plat"
self.root.mkdir()
self.wanted = {"list.txt": "https://example.invalid/tag-2/list.txt"}
def _cached(self, mtime: float) -> None:
path = self.root / "list.txt"
path.write_text("x", encoding="utf-8")
os.utime(path, (mtime, mtime))
def _state(self, recorded: dict, transcribed_at: float) -> str:
return cf.native_cache_state(
self.wanted, self.root, recorded, self.index, transcribed_at
)
def test_a_file_fetched_after_the_scrape_from_the_named_url_is_fresh(self):
self._cached(200.0)
recorded = {"plat/list.txt": self.wanted["list.txt"]}
self.assertEqual(self._state(recorded, 100.0), "")
def test_an_absent_file_is_named(self):
self.assertEqual(self._state({}, 0.0), "list.txt not cached")
def test_a_file_with_no_recorded_url_proves_nothing(self):
self._cached(200.0)
self.assertEqual(self._state({}, 100.0), "list.txt cached from an unrecorded URL")
def test_a_file_from_the_previous_pin_is_stale(self):
self._cached(200.0)
recorded = {"plat/list.txt": "https://example.invalid/tag-1/list.txt"}
self.assertEqual(self._state(recorded, 100.0), "list.txt cached from another revision")
def test_a_file_older_than_the_rescrape_is_stale(self):
"""A branch URL does not change when its content does."""
self._cached(100.0)
recorded = {"plat/list.txt": self.wanted["list.txt"]}
self.assertEqual(
self._state(recorded, 200.0),
"list.txt cached before the platform file was rewritten",
)
def test_the_transcription_date_follows_inheritance(self):
platforms = self.cache / "platforms"
platforms.mkdir()
(platforms / "parent.yml").write_text("platform: P\n", encoding="utf-8")
(platforms / "child.yml").write_text("inherits: parent\n", encoding="utf-8")
os.utime(platforms / "parent.yml", (300.0, 300.0))
os.utime(platforms / "child.yml", (100.0, 100.0))
self.assertEqual(cf._transcribed_at(platforms, "child"), 300.0)
self.assertEqual(cf._transcribed_at(platforms, "parent"), 300.0)
def test_every_registered_platform_with_an_exporter_gets_a_row(self):
rows = cf.check_native(REPO_ROOT / "platforms", self.cache, self.cache / "truth")
self.assertTrue(rows)
self.assertEqual({r.area for r in rows}, {"native"})
self.assertEqual({r.status for r in rows}, {cf.STALE})
self.assertIn("batocera", {r.subject for r in rows})
class PlatformDiffTests(unittest.TestCase):
def _platform(self, version="1", files=None, cores=None):
return {
+82 -1
View File
@@ -15,6 +15,7 @@ and skip when it has not been populated (python scripts/export_native.py
from __future__ import annotations
import json
import os
import re
import subprocess
import sys
@@ -33,7 +34,13 @@ from common import ( # noqa: E402
)
from exporter import discover_exporters # noqa: E402
from exporter.base_exporter import BaseExporter # noqa: E402
from export_native import pinned_base # noqa: E402
from export_native import ( # noqa: E402
SOURCES_INDEX,
collect_originals,
fetch,
pinned_base,
source_key,
)
from exporter.baseline import ( # noqa: E402
NativeFile,
build_native_model,
@@ -437,6 +444,80 @@ class PinnedRevision(unittest.TestCase):
self.assertTrue(base.startswith("https://"), base)
class _OneFileExporter:
"""An exporter reduced to what collect_originals reads."""
def __init__(self, url: str):
self._url = url
def platform_name(self) -> str:
return "plat"
def native_sources(self) -> dict[str, str]:
return {"list.txt": self._url}
class CachedOriginal(unittest.TestCase):
"""The cached original is the file the platform data was read from."""
def setUp(self):
self._tmp = tempfile.TemporaryDirectory()
self.addCleanup(self._tmp.cleanup)
self.root = Path(self._tmp.name)
self.cache = self.root / "cache"
self.index = self.cache / SOURCES_INDEX
def _upstream(self, name: str, text: str) -> str:
path = self.root / name
path.write_text(text, encoding="utf-8")
return path.as_uri()
def test_the_same_url_is_served_from_the_cache(self):
url = self._upstream("v1.txt", "one")
target = self.cache / "plat" / "list.txt"
self.assertEqual(fetch(url, target, self.index), b"one")
self._upstream("v1.txt", "moved")
self.assertEqual(fetch(url, target, self.index), b"one")
def test_refresh_reads_a_branch_url_again(self):
"""A branch URL keeps its name while the platform moves."""
url = self._upstream("branch.txt", "one")
target = self.cache / "plat" / "list.txt"
fetch(url, target, self.index)
self._upstream("branch.txt", "two")
self.assertEqual(fetch(url, target, self.index, refresh=True), b"two")
self.assertEqual(target.read_bytes(), b"two")
def test_a_new_pin_refetches_an_existing_file(self):
"""Reading the file straight from disk served the old pin forever."""
old = self._upstream("tag-1.txt", "one")
new = self._upstream("tag-2.txt", "two")
originals, missing = collect_originals(
_OneFileExporter(old), {}, self.cache, True
)
self.assertEqual((originals, missing), ({"list.txt": "one"}, []))
originals, missing = collect_originals(
_OneFileExporter(new), {}, self.cache, True
)
self.assertEqual((originals, missing), ({"list.txt": "two"}, []))
def test_offline_serves_what_is_cached_and_names_what_is_not(self):
url = self._upstream("v1.txt", "one")
_, missing = collect_originals(_OneFileExporter(url), {}, self.cache, False)
self.assertEqual(missing, ["list.txt: absent and fetching is off"])
collect_originals(_OneFileExporter(url), {}, self.cache, True)
originals, missing = collect_originals(
_OneFileExporter("file:///nowhere/else.txt"), {}, self.cache, False
)
self.assertEqual((originals, missing), ({"list.txt": "one"}, []))
def test_the_index_names_a_file_the_same_from_any_directory(self):
target = self.cache / "plat" / "list.txt"
self.assertEqual(source_key(target, self.index), "plat/list.txt")
relative = Path(os.path.relpath(target, Path.cwd()))
self.assertEqual(source_key(relative, self.index), "plat/list.txt")
class RecalboxExport(unittest.TestCase):
def _render(self):
systems, report = model()
+145
View File
@@ -0,0 +1,145 @@
"""refresh_stale: which command answers which stale row, and in what order.
The network side is check_freshness and the scrapers themselves. What is
locked here is the planning: a stale row maps to one command, rows a human
must read are surfaced instead of run, and a platform rescrape is followed
by the refresh of the original its export patches.
"""
from __future__ import annotations
import sys
import tempfile
import unittest
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(REPO_ROOT / "scripts"))
import refresh_stale as rs # noqa: E402
REGISTRY = {
"retrobat": {"scraper": "retrobat", "target_scraper": None},
"batocera": {"scraper": "batocera", "target_scraper": "batocera_targets"},
"lakka": {},
}
def finding(area: str, subject: str, status: str = "STALE") -> dict:
return {"area": area, "subject": subject, "status": status, "detail": ""}
class Planning(unittest.TestCase):
def _plan(self, *findings: dict):
return rs.plan_jobs(list(findings), REGISTRY)
def test_a_stale_platform_is_rescraped_then_its_original_is_refreshed(self):
jobs, surfaced, _ = self._plan(finding("platforms", "retrobat"))
self.assertEqual(surfaced, [])
self.assertEqual(len(jobs), 1)
self.assertEqual(
jobs[0].command[1:],
["-m", "scripts.scraper.retrobat_scraper", "-o", "platforms/retrobat.yml"],
)
self.assertEqual(
[list(c[1:]) for c in jobs[0].then],
[["scripts/export_native.py", "--platform", "retrobat", "--refresh-cache"]],
)
def test_a_native_row_of_a_rescraped_platform_is_not_a_second_job(self):
"""Two jobs would race: the refresh must follow the scrape."""
jobs, _, ignored = self._plan(
finding("native", "retrobat"), finding("platforms", "retrobat")
)
self.assertEqual([j.area for j in jobs], ["platforms"])
self.assertEqual([f["area"] for f in ignored], ["native"])
def test_a_stale_original_alone_is_refreshed(self):
jobs, _, _ = self._plan(finding("native", "lakka"))
self.assertEqual(
jobs[0].command[1:],
["scripts/export_native.py", "--platform", "lakka", "--refresh-cache"],
)
self.assertEqual(jobs[0].then, ())
def test_a_platform_without_a_scraper_is_surfaced(self):
jobs, surfaced, _ = self._plan(finding("platforms", "lakka"))
self.assertEqual(jobs, [])
self.assertEqual([f["subject"] for f in surfaced], ["lakka"])
def test_unprofiled_cores_need_a_human(self):
jobs, surfaced, _ = self._plan(finding("targets", "batocera cores"))
self.assertEqual(jobs, [])
self.assertEqual(len(surfaced), 1)
def test_targets_data_and_recipes_map_to_their_refreshers(self):
jobs, surfaced, _ = self._plan(
finding("targets", "batocera"),
finding("data", "dolphin-sys"),
finding("catalogs", "fbneo recipes"),
finding("catalogs", "redump"),
)
commands = {j.subject: " ".join(j.command[1:]) for j in jobs}
self.assertEqual(
commands,
{
"batocera": "-m scripts.scraper.targets.batocera_targets_scraper"
" -o platforms/targets/batocera.yml",
"dolphin-sys": "scripts/refresh_data_dirs.py --key dolphin-sys --force",
"fbneo recipes": "-m scripts.scraper.romset_dat_importer"
" --source fbneo --fetch",
},
)
self.assertEqual([f["subject"] for f in surfaced], ["redump"])
def test_only_stale_rows_run_and_errors_are_shown(self):
jobs, surfaced, ignored = self._plan(
finding("data", "a", "OK"),
finding("data", "b", "SKIPPED"),
finding("platforms", "retrobat", "ERROR"),
)
self.assertEqual(jobs, [])
self.assertEqual([f["status"] for f in surfaced], ["ERROR"])
self.assertEqual(len(ignored), 2)
def test_pins_and_profiles_are_never_run(self):
jobs, surfaced, _ = self._plan(
finding("ci", "pypi:jsonschema"),
finding("coreinfo", "libretro-core-info"),
finding("profiles", "profile_sync"),
)
self.assertEqual(jobs, [])
self.assertEqual(len(surfaced), 3)
class Running(unittest.TestCase):
def _job(self, directory: str, command: list[str], then=()) -> rs.Job:
return rs.Job("platforms", "x", command, Path(directory) / "x.log", then)
def test_a_follow_up_runs_after_a_success(self):
with tempfile.TemporaryDirectory() as directory:
marker = Path(directory) / "ran"
job = self._job(
directory,
[sys.executable, "-c", "pass"],
((sys.executable, "-c", f"open({str(marker)!r}, 'w').close()"),),
)
result = rs.run_job(job, {})
self.assertEqual(result.returncode, 0)
self.assertTrue(marker.exists())
def test_a_failed_scrape_leaves_the_original_alone(self):
with tempfile.TemporaryDirectory() as directory:
marker = Path(directory) / "ran"
job = self._job(
directory,
[sys.executable, "-c", "raise SystemExit(3)"],
((sys.executable, "-c", f"open({str(marker)!r}, 'w').close()"),),
)
result = rs.run_job(job, {})
self.assertEqual(result.returncode, 3)
self.assertFalse(marker.exists())
if __name__ == "__main__":
unittest.main()
+1 -1
View File
@@ -227,7 +227,7 @@ python scripts/generate_pack.py --all --verify-packs --output-dir dist/
```
Integrated as pipeline step 6/8 (runs after consistency check, before
README generation). Requires packs in `dist/` — skip with `--skip-packs`.
README generation). Requires packs in `dist/`; skip with `--skip-packs`.
## Verification discipline
+84 -11
View File
@@ -236,14 +236,26 @@ python scripts/export_native.py --platform batocera --fetch
python scripts/export_native.py --all --fetch --output-dir dist/upstream/
```
`--fetch` downloads the platform's own file once into
`.cache/upstream-native/`, at the revision the platform YAML's `source:`
names. Formats that carry code are patched from it rather than
regenerated, and the export fails without it.
`--fetch` downloads the platform's own file into `.cache/upstream-native/`,
at the revision the platform YAML's `source:` names. Formats that carry
code are patched from it rather than regenerated, and the export fails
without it. The URL each cached file came from is recorded beside the
cache, so a platform file that moves to a new tag is fetched again instead
of patching the previous one.
A branch URL keeps its name while its content moves, and no URL comparison
can see that. `--refresh-cache` downloads every file again and writes no
export; it is what follows a rescrape, so the original and the platform
YAML describe the same moment.
```bash
python scripts/export_native.py --platform retrobat --refresh-cache
python scripts/export_native.py --all --refresh-cache
```
A list is written in the order the code looks: `priority:` first (lowest
wins), then the profile's own declaration order. What a format cannot
state is reported rather than written — EmuDeck's arrays are shared
state is reported rather than written: EmuDeck's arrays are shared
between emulators its file does not name, so a hash is corrected in place
and never added.
@@ -326,9 +338,11 @@ Per part of a ref: `ANCHORED` (same content, same lines), `SHIFTED` (same
content, moved), `RENAMED` (the source file moved), `CHANGED` (content
edited), `AMBIGUOUS` (several equally good candidates), `GONE` (nothing
left to anchor to), `EXTERNAL` (the ref names a project the profile does
not declare, so no revision can confirm it). An entry carries the worst
not declare, so no revision can confirm it), `UNCHECKED` (a pin equal to
HEAD is judged on self-consistency, and an entry with neither a hash nor
a name gives that check nothing to look for). An entry carries the worst
status of its parts. Only `CHANGED`, `GONE` and `AMBIGUOUS` count as
needing a re-read.
needing a re-read; a zero next to a row of `unchecked` is not a proof.
A single cited line is often not distinctive, so the anchor widens by
steps of ±3, ±6, ±12, ±25 and ±50 lines until it is unique. Three further
@@ -424,9 +438,10 @@ directories the refs point at.
Writes are explicit and mechanical only. `--backfill-commits` fills a
missing `source_commit`, `--rebase-refs` recales `SHIFTED` and `RENAMED`
line ranges, `--bump-commit` advances `source_commit` to HEAD only when
nothing needs a re-read. All three refuse to run on a dirty `emulators/`
without `--force`, and every write is verified by reparsing the document.
line ranges and moves `source_commit` to HEAD with them, all or nothing,
`--bump-commit` advances `source_commit` to HEAD only when nothing needs a
re-read. All three refuse to run on a dirty `emulators/` without `--force`,
and every write is verified by reparsing the document.
Uses `GITHUB_TOKEN` when set, which `--all` requires. Responses are cached
under `.cache/upstream/`, addressed by commit sha, so `--offline` replays a
@@ -457,6 +472,62 @@ python scripts/refresh_data_dirs.py --platform batocera # single platform onl
python scripts/refresh_data_dirs.py --registry path/to/_data_dirs.yml
```
### check_freshness.py
Ask every transcribed layer whether the local copy is the one upstream serves
today, and answer with one row per subject. Platform and target files are
re-scraped through the scrapers' own write path into a throwaway copy and
diffed against the committed file, so the diff decides rather than a version
string. Core-info is listed and compared with the profiles (a core with no
profile, a profile that calls standalone a core libretro now builds). Data
directories are checked against their upstream, dump catalogs and romset
snapshots against the catalog they were read from (Redump DAT versions, the
No-Intro daily rebuild, the newest TOSEC pack, the newest MAME release, the
git blobs of the FBNeo DATs), and the CI toolchain against PyPI and the
actions' latest releases.
The `native` area needs no network. It reads the cache `export_native.py`
patches from and reports an original that is absent, that came from another
URL than the one the platform YAML names, or that was fetched before the
platform YAML was last rewritten.
```bash
python scripts/check_freshness.py
python scripts/check_freshness.py --only platforms,targets
python scripts/check_freshness.py --json
python scripts/check_freshness.py --profiles # also runs profile_sync (slow)
python scripts/check_freshness.py --offline # local ages only
```
The exit code is non-zero when anything is stale. Emulator profiles are the
one layer left to `profile_sync.py`, which the `--profiles` flag folds in.
### refresh_stale.py
Run the refresher behind every stale row of `check_freshness.py`, several at
a time, and leave the result in the working tree for review. Nothing is
committed.
| Stale row | Command |
|-----------|---------|
| `platforms/<name>` | the platform scraper, then the `native` refresh for the same platform |
| `native/<name>` | `export_native.py --platform <name> --refresh-cache` |
| `targets/<name>` | the target scraper |
| `data/<key>` | `refresh_data_dirs.py --key <key> --force` |
| `catalogs/mame recipes`, `catalogs/fbneo recipes` | `scraper/romset_dat_importer.py --source <source> --fetch` |
Rows that need a decision are listed and never run: a buildbot core with no
profile, a new `.info` file, the Redump, No-Intro and TOSEC packs, the CI
pins, and `profile_sync`. A failed scraper leaves its cached original
untouched. Each job writes its output to `tmp/refresh_stale/`, and a GitHub
token is taken from `GITHUB_TOKEN` or from `gh auth token`.
```bash
python scripts/refresh_stale.py --dry-run
python scripts/refresh_stale.py
python scripts/refresh_stale.py --only platforms,targets --jobs 6
```
### Other tools
| Script | Purpose |
@@ -474,12 +545,14 @@ python scripts/refresh_data_dirs.py --registry path/to/_data_dirs.yml
| `generate_site.py` | Generate all MkDocs site pages (this documentation) |
| `validate_site.py` | Validate rendered metadata, headings, image alternatives, JSON-LD, local resources, links and fragments |
| `romset_recipes.py` | Identify which emulator version an arcade archive matches, and rebuild a pinned archive from ROMs already held |
| `scraper/romset_dat_importer.py` | Fetch and import per-set recipes from MAME `-listxml` or FBNeo DATs into `recipes/` |
| `scraper/romset_dat_importer.py` | Fetch and import per-set recipes from MAME `-listxml` or FBNeo DATs into `recipes/`; `--fetch` alone takes the newest MAME release, and FBNeo DATs are refreshed by git blob sha |
| `deterministic_zip.py` | Rebuild MAME BIOS ZIPs deterministically (same ROMs = same hash) |
| `torrentzip.py` | Build TorrentZip archives for MAME/FBNeo ROM sets (archive bytes depend only on contents) |
| `crypto_verify.py` | 3DS RSA signature and AES crypto verification |
| `sect233r1.py` | Pure Python ECDSA verification on sect233r1 curve (3DS OTP cert) |
| `check_buildbot_system.py` | Detect stale data directories by comparing with buildbot |
| `check_freshness.py` | One report on every upstream the repository transcribes (see above) |
| `refresh_stale.py` | Run the refresher behind every stale row of that report (see above) |
| `migrate.py` | Migrate flat bios structure to Manufacturer/Console/ hierarchy |
## Installation tools