diff --git a/scripts/check_freshness.py b/scripts/check_freshness.py index 072dd63f..741cd8f3 100644 --- a/scripts/check_freshness.py +++ b/scripts/check_freshness.py @@ -12,6 +12,11 @@ into a throwaway copy, then diffed against the committed file. The diff decides, not a version string: a scraper that pins a new tag but produces the same entries is reported as a version change and nothing else. +The native export patches each platform's own file, kept in a local cache. +That cache is checked against the platform file it was transcribed with: an +original cached before the last rescrape, or under another pin, is no longer +the file our data describes. + Emulator profiles are covered by profile_sync, which is slow (one API pass per profile). ``--profiles`` runs it here and folds its verdict in. @@ -43,6 +48,7 @@ from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent)) sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +import export_native # noqa: E402 import refresh_data_dirs # noqa: E402 import upstream # noqa: E402 from common import ( # noqa: E402 @@ -50,6 +56,8 @@ from common import ( # noqa: E402 load_emulator_profiles, yaml_load, ) +from exporter import discover_exporters # noqa: E402 +from exporter.baseline import build_native_model # noqa: E402 from scripts.scraper.redump_dat_scraper import fetch_snapshot # noqa: E402 from scripts.scraper.targets import discover_target_scrapers # noqa: E402 @@ -66,7 +74,7 @@ INSTALLERS = ("install.sh", "install.ps1") SHOWN_ITEMS = 4 # items named in a diff summary before ", ..." SHOWN_NAMES = 12 # core names named in an unresolved row before ", ..." -AREAS = ("platforms", "targets", "coreinfo", "data", "catalogs", "ci", "profiles") +AREAS = ("platforms", "native", "targets", "coreinfo", "data", "catalogs", "ci", "profiles") OK, STALE, UNKNOWN, ERROR, SKIPPED = "OK", "STALE", "UNKNOWN", "ERROR", "SKIPPED" # Every status a finding may carry, in the order the summary counts them. STATUSES = (STALE, UNKNOWN, ERROR, SKIPPED, OK) @@ -279,6 +287,80 @@ def check_platforms(platforms_dir: Path, workdir: Path, jobs: int) -> list[Findi return sorted(findings, key=lambda f: f.subject) +# --- native originals -------------------------------------------------------- + + +def native_cache_state( + wanted: dict[str, str], + root: Path, + recorded: dict[str, str], + index: Path, + transcribed_at: float, +) -> str: + """Why a platform's cached originals cannot be patched, '' when they can. + + The export corrects the file our data was transcribed from. A cached + original is that file only if it came from the URL the platform file + names today and was fetched no earlier than the platform file was last + rewritten: a branch URL keeps its name while its content moves. + """ + for relative, url in sorted(wanted.items()): + path = root / relative + if not path.is_file(): + return f"{relative} not cached" + known = recorded.get(export_native.source_key(path, index)) + if known is None: + return f"{relative} cached from an unrecorded URL" + if known != url: + return f"{relative} cached from another revision" + if path.stat().st_mtime < transcribed_at: + return f"{relative} cached before the platform file was rewritten" + return "" + + +def _transcribed_at(platforms_dir: Path, platform: str) -> float: + """Last rewrite of the platform file or of any file it inherits.""" + newest = 0.0 + seen: set[str] = set() + name: str | None = platform + while name and name not in seen: + seen.add(name) + path = platforms_dir / f"{name}.yml" + if not path.is_file(): + break + newest = max(newest, path.stat().st_mtime) + with path.open(encoding="utf-8") as fh: + name = (yaml_load(fh) or {}).get("inherits") + return newest + + +def check_native(platforms_dir: Path, cache_dir: Path, truth_dir: Path) -> list[Finding]: + exporters = discover_exporters() + index = cache_dir / export_native.SOURCES_INDEX + recorded = export_native.load_sources(cache_dir) + findings: list[Finding] = [] + for name in list_registered_platforms(str(platforms_dir), include_archived=True): + exporter_class = exporters.get(name) + if not exporter_class: + continue + truth, scraped = export_native.load_inputs(name, truth_dir, str(platforms_dir)) + systems, _ = build_native_model(truth or {}, scraped) + wanted = export_native.wanted_sources(exporter_class(), systems, scraped) + reason = native_cache_state( + wanted, cache_dir / name, recorded, index, _transcribed_at(platforms_dir, name) + ) + findings.append( + Finding( + "native", + name, + STALE if reason else OK, + local=f"{len(wanted)} file(s)", + detail=f"{reason}, export_native.py --refresh-cache fetches it" if reason else "", + ) + ) + return sorted(findings, key=lambda f: f.subject) + + # --- targets ---------------------------------------------------------------- @@ -959,6 +1041,8 @@ def main() -> int: parser.add_argument("--offline", action="store_true", help="no network, local ages only") parser.add_argument("--jobs", type=int, default=4, help="parallel scrapers") parser.add_argument("--cache-dir", default=".cache/upstream") + parser.add_argument("--native-cache-dir", default=export_native.DEFAULT_CACHE) + parser.add_argument("--truth-dir", default="dist/truth") parser.add_argument("--platforms-dir", default="platforms") parser.add_argument("--emulators-dir", default="emulators") args = parser.parse_args() @@ -979,6 +1063,10 @@ def main() -> int: if args.offline else check_platforms(platforms_dir, workdir, args.jobs) ) + if "native" in areas: + findings.extend( + check_native(platforms_dir, Path(args.native_cache_dir), Path(args.truth_dir)) + ) if "targets" in areas: findings.extend(check_targets(platforms_dir, workdir, profiles, args.jobs, args.offline)) if "coreinfo" in areas: diff --git a/scripts/export_native.py b/scripts/export_native.py index a3eb0741..a687a949 100644 --- a/scripts/export_native.py +++ b/scripts/export_native.py @@ -10,6 +10,7 @@ a platform ships keeps working. Usage: python scripts/export_native.py --all --fetch python scripts/export_native.py --platform recalbox --upstream-dir up/ + python scripts/export_native.py --platform retrobat --refresh-cache """ from __future__ import annotations @@ -28,6 +29,7 @@ from exporter import discover_exporters from exporter.baseline import build_native_model DEFAULT_CACHE = ".cache/upstream-native" +SOURCES_INDEX = ".sources.json" _USER_AGENT = "retrobios-exporter/1.0" _MAX_BYTES = 64 * 1024 * 1024 @@ -41,7 +43,29 @@ def _load_sources(index: Path | None) -> dict[str, str]: return {} -def fetch(url: str, destination: Path, index: Path | None = None) -> bytes: +def source_key(destination: Path, index: Path | None) -> str: + """How the index names a cached file: its path below the cache root. + + A key that repeated the cache directory as it was typed changed with the + working directory, and a file recorded from one place looked unrecorded + from another. + """ + if index is not None: + try: + return destination.resolve().relative_to(index.parent.resolve()).as_posix() + except ValueError: + pass + return str(destination) + + +def load_sources(upstream_dir: Path) -> dict[str, str]: + """Cached file -> URL it was fetched from, for a whole cache directory.""" + return _load_sources(upstream_dir / SOURCES_INDEX) + + +def fetch( + url: str, destination: Path, index: Path | None = None, refresh: bool = False +) -> bytes: """Download an original once, then read it from the cache. The cache path carries the file's own name and nothing of the revision it @@ -50,10 +74,15 @@ def fetch(url: str, destination: Path, index: Path | None = None) -> bytes: URL that produced each cached file is recorded beside the cache, and a different URL refetches. A cache written before this index existed keeps being served: nothing recorded means nothing contradicted. + + A branch URL names no revision, so the same URL serves new bytes after + the platform moves. refresh downloads again whatever the cache holds: + it is what a rescrape calls, so the original and its transcription + describe the same moment. """ recorded = _load_sources(index) - key = str(destination) - if destination.exists() and recorded.get(key, url) == url: + key = source_key(destination, index) + if not refresh and destination.exists() and recorded.get(key, url) == url: return destination.read_bytes() request = urllib.request.Request(url, headers={"User-Agent": _USER_AGENT}) with urllib.request.urlopen(request, timeout=60) as response: @@ -94,14 +123,10 @@ def _raw_url(url: str) -> str: return url -def collect_originals( - exporter: object, - systems: dict, - upstream_dir: Path, - allow_fetch: bool, - scraped: dict | None = None, -) -> tuple[dict[str, str], list[str]]: - """Gather the platform's own files, from disk or from upstream.""" +def wanted_sources( + exporter: object, systems: dict, scraped: dict | None = None +) -> dict[str, str]: + """Cache-relative name -> URL of every file the exporter patches.""" wanted = dict(exporter.native_sources()) components = getattr(exporter, "components", None) if callable(components): @@ -113,21 +138,39 @@ def collect_originals( base = pinned_base(wanted, scraped) if base: wanted = {relative: base + relative for relative in wanted} + return wanted + +def collect_originals( + exporter: object, + systems: dict, + upstream_dir: Path, + allow_fetch: bool, + scraped: dict | None = None, + refresh: bool = False, +) -> tuple[dict[str, str], list[str]]: + """Gather the platform's own files, from disk or from upstream. + + With fetching on, an existing file still goes through fetch(): reading + it straight from disk skipped the recorded URL, so a file cached under + one pin kept being patched after the platform YAML moved to the next. + """ + wanted = wanted_sources(exporter, systems, scraped) root = upstream_dir / exporter.platform_name() + index = upstream_dir / SOURCES_INDEX originals: dict[str, str] = {} missing: list[str] = [] for relative, url in wanted.items(): path = root / relative payload: bytes | None = None - if path.exists(): - payload = path.read_bytes() - elif allow_fetch: + if allow_fetch: try: - payload = fetch(url, path, upstream_dir / ".sources.json") + payload = fetch(url, path, index, refresh) except (urllib.error.URLError, urllib.error.HTTPError, OSError) as exc: missing.append(f"{relative}: {exc}") continue + elif path.exists(): + payload = path.read_bytes() else: missing.append(f"{relative}: absent and fetching is off") continue @@ -141,6 +184,42 @@ def collect_originals( return originals, missing +def load_inputs( + platform: str, truth_dir: Path, platforms_dir: str +) -> tuple[dict | None, dict | None]: + """The truth and the scraped config of a platform, None where absent.""" + truth_file = truth_dir / f"{platform}.yml" + truth: dict | None = None + if truth_file.exists(): + with open(truth_file) as handle: + truth = yaml_load(handle) or {} + try: + scraped = load_platform_config(platform, platforms_dir) + except (FileNotFoundError, OSError): + scraped = None + return truth, scraped + + +def refresh_cache( + platform: str, + exporter_class: type, + truth_dir: Path, + platforms_dir: str, + upstream_dir: Path, +) -> tuple[bool, list[str]]: + """Download a platform's own files again. Returns (ok, messages).""" + truth, scraped = load_inputs(platform, truth_dir, platforms_dir) + systems, _ = build_native_model(truth or {}, scraped) + exporter = exporter_class() + wanted = wanted_sources(exporter, systems, scraped) + _, missing = collect_originals( + exporter, systems, upstream_dir, True, scraped, refresh=True + ) + if missing: + return False, missing + return True, [f"{len(wanted)} file(s) fetched"] + + def export_platform( platform: str, exporter_class: type, @@ -153,19 +232,11 @@ def export_platform( """Write one platform's corrected file. Returns (ok, messages).""" messages: list[str] = [] - truth_file = truth_dir / f"{platform}.yml" - truth: dict = {} - if truth_file.exists(): - with open(truth_file) as handle: - truth = yaml_load(handle) or {} - else: + truth, scraped = load_inputs(platform, truth_dir, platforms_dir) + if truth is None: + truth = {} messages.append(f"no truth for {platform}, only the platform's own data") - try: - scraped = load_platform_config(platform, platforms_dir) - except (FileNotFoundError, OSError): - scraped = None - systems, report = build_native_model(truth, scraped) if not systems: return False, ["nothing to write: neither the platform nor the truth has data"] @@ -253,6 +324,7 @@ def run( platforms_dir: str, upstream_dir: str, allow_fetch: bool, + refresh_only: bool = False, ) -> int: exporters = discover_exporters() failures = 0 @@ -265,15 +337,24 @@ def run( print(f" SKIP {platform}: no exporter") continue - ok, messages = export_platform( - platform, - exporter_class, - Path(truth_dir), - Path(output_dir), - platforms_dir, - Path(upstream_dir), - allow_fetch, - ) + if refresh_only: + ok, messages = refresh_cache( + platform, + exporter_class, + Path(truth_dir), + platforms_dir, + Path(upstream_dir), + ) + else: + ok, messages = export_platform( + platform, + exporter_class, + Path(truth_dir), + Path(output_dir), + platforms_dir, + Path(upstream_dir), + allow_fetch, + ) label = "OK " if ok else "FAIL" print(f" {label} {platform}") for message in messages: @@ -306,7 +387,13 @@ def main() -> None: parser.add_argument( "--fetch", action="store_true", - help="download a platform's file when it is not in the cache", + help="download a platform's file when it is not in the cache, " + "or when the cache came from another URL", + ) + parser.add_argument( + "--refresh-cache", + action="store_true", + help="download every platform file again and write no export", ) args = parser.parse_args() @@ -326,6 +413,7 @@ def main() -> None: args.platforms_dir, args.upstream_dir, args.fetch, + args.refresh_cache, ) ) diff --git a/scripts/refresh_stale.py b/scripts/refresh_stale.py index 8b096dac..f5f438fa 100755 --- a/scripts/refresh_stale.py +++ b/scripts/refresh_stale.py @@ -10,6 +10,10 @@ committed. Mapping (stale -> command): platforms/ python -m scripts.scraper._scraper -o platforms/.yml + then the native refresh below: the original + the export patches must be the one just scraped + native/ python scripts/export_native.py --platform + --refresh-cache targets/ python -m scripts.scraper.targets. -o platforms/targets/.yml data/ python scripts/refresh_data_dirs.py --key @@ -47,7 +51,7 @@ LOG_DIR = REPO_ROOT / "tmp" / "refresh_stale" CHECK_FRESHNESS = REPO_ROOT / "scripts" / "check_freshness.py" PLATFORMS_REGISTRY = REPO_ROOT / "platforms" / "_registry.yml" -AUTO_AREAS = ("platforms", "targets", "data", "catalogs") +AUTO_AREAS = ("platforms", "native", "targets", "data", "catalogs") MANUAL_AREAS = ("coreinfo", "ci", "profiles") JOB_TIMEOUT = 1800 # 30 minutes per refresher; scrapers rarely exceed 10 @@ -61,6 +65,8 @@ class Job: subject: str command: list[str] log_path: Path + # Commands run after `command`, in order, only while each one succeeds. + then: tuple[tuple[str, ...], ...] = () @dataclass(frozen=True) @@ -122,6 +128,15 @@ def plan_jobs( seen: set[tuple[str, str]] = set() surfaced: list[dict] = [] ignored: list[dict] = [] + # A platform about to be rescraped refreshes its cached original itself, + # after the scrape: a separate native job would race the scraper. + rescraped = { + str(f.get("subject") or "") + for f in findings + if f.get("status") == "STALE" + and f.get("area") == "platforms" + and _command_for("platforms", str(f.get("subject") or ""), registry) + } for finding in findings: status = finding.get("status") @@ -135,6 +150,10 @@ def plan_jobs( ignored.append(finding) continue + if area == "native" and subject in rescraped: + ignored.append(finding) + continue + command = _command_for(area, subject, registry) if command is None: surfaced.append(finding) @@ -148,15 +167,33 @@ def plan_jobs( seen.add(key) log = LOG_DIR / f"{area}__{_slug(subject)}.log" - jobs.append(Job(area=area, subject=subject, command=command, log_path=log)) + then: tuple[tuple[str, ...], ...] = () + if area == "platforms": + then = (tuple(_native_refresh(subject)),) + jobs.append( + Job(area=area, subject=subject, command=command, log_path=log, then=then) + ) return jobs, surfaced, ignored +def _native_refresh(platform: str) -> list[str]: + return [ + sys.executable, + "scripts/export_native.py", + "--platform", + platform, + "--refresh-cache", + ] + + def _command_for( area: str, subject: str, registry: dict[str, dict] ) -> list[str] | None: """Map a stale finding to its refresh command, or None if manual.""" + if area == "native": + return _native_refresh(subject) + if area == "platforms": entry = registry.get(subject) or {} scraper = entry.get("scraper") @@ -250,23 +287,27 @@ def run_job(job: Job, env: dict[str, str]) -> JobResult: """Execute one refresh command and collect its output.""" start = time.monotonic() job.log_path.parent.mkdir(parents=True, exist_ok=True) + returncode = 0 with job.log_path.open("w", encoding="utf-8") as log: - log.write(f"$ {' '.join(job.command)}\n") - log.flush() - try: - proc = subprocess.run( - job.command, - cwd=REPO_ROOT, - stdout=log, - stderr=subprocess.STDOUT, - env=env, - timeout=JOB_TIMEOUT, - check=False, - ) - returncode = proc.returncode - except subprocess.TimeoutExpired: - log.write(f"\nTIMEOUT after {JOB_TIMEOUT}s\n") - returncode = 124 + for command in (job.command, *job.then): + log.write(f"$ {' '.join(command)}\n") + log.flush() + try: + proc = subprocess.run( + list(command), + cwd=REPO_ROOT, + stdout=log, + stderr=subprocess.STDOUT, + env=env, + timeout=JOB_TIMEOUT, + check=False, + ) + returncode = proc.returncode + except subprocess.TimeoutExpired: + log.write(f"\nTIMEOUT after {JOB_TIMEOUT}s\n") + returncode = 124 + if returncode != 0: + break duration = time.monotonic() - start tail = _log_tail(job.log_path) return JobResult(job=job, returncode=returncode, duration=duration, tail=tail) @@ -388,6 +429,8 @@ def main() -> int: f"{len(ignored)} already clean/skipped") for job in sorted(jobs, key=lambda j: (j.area, j.subject)): print(f" {job.area:10} {job.subject:28} {' '.join(job.command)}") + for command in job.then: + print(f" {'':10} {'':28} then {' '.join(command)}") if surfaced: print("\n[manual review needed]") for f in surfaced: diff --git a/tests/test_check_freshness.py b/tests/test_check_freshness.py index d17b15cf..05e7c331 100644 --- a/tests/test_check_freshness.py +++ b/tests/test_check_freshness.py @@ -8,7 +8,9 @@ report is folded, since each of those decides whether a row says STALE. from __future__ import annotations +import os import sys +import tempfile import unittest from pathlib import Path @@ -19,6 +21,72 @@ import check_freshness as cf # noqa: E402 from common import yaml_load # noqa: E402 +class NativeCacheTests(unittest.TestCase): + """A cached original is patched only if it is the file we transcribed.""" + + def setUp(self): + self._tmp = tempfile.TemporaryDirectory() + self.addCleanup(self._tmp.cleanup) + self.cache = Path(self._tmp.name) + self.index = self.cache / cf.export_native.SOURCES_INDEX + self.root = self.cache / "plat" + self.root.mkdir() + self.wanted = {"list.txt": "https://example.invalid/tag-2/list.txt"} + + def _cached(self, mtime: float) -> None: + path = self.root / "list.txt" + path.write_text("x", encoding="utf-8") + os.utime(path, (mtime, mtime)) + + def _state(self, recorded: dict, transcribed_at: float) -> str: + return cf.native_cache_state( + self.wanted, self.root, recorded, self.index, transcribed_at + ) + + def test_a_file_fetched_after_the_scrape_from_the_named_url_is_fresh(self): + self._cached(200.0) + recorded = {"plat/list.txt": self.wanted["list.txt"]} + self.assertEqual(self._state(recorded, 100.0), "") + + def test_an_absent_file_is_named(self): + self.assertEqual(self._state({}, 0.0), "list.txt not cached") + + def test_a_file_with_no_recorded_url_proves_nothing(self): + self._cached(200.0) + self.assertEqual(self._state({}, 100.0), "list.txt cached from an unrecorded URL") + + def test_a_file_from_the_previous_pin_is_stale(self): + self._cached(200.0) + recorded = {"plat/list.txt": "https://example.invalid/tag-1/list.txt"} + self.assertEqual(self._state(recorded, 100.0), "list.txt cached from another revision") + + def test_a_file_older_than_the_rescrape_is_stale(self): + """A branch URL does not change when its content does.""" + self._cached(100.0) + recorded = {"plat/list.txt": self.wanted["list.txt"]} + self.assertEqual( + self._state(recorded, 200.0), + "list.txt cached before the platform file was rewritten", + ) + + def test_the_transcription_date_follows_inheritance(self): + platforms = self.cache / "platforms" + platforms.mkdir() + (platforms / "parent.yml").write_text("platform: P\n", encoding="utf-8") + (platforms / "child.yml").write_text("inherits: parent\n", encoding="utf-8") + os.utime(platforms / "parent.yml", (300.0, 300.0)) + os.utime(platforms / "child.yml", (100.0, 100.0)) + self.assertEqual(cf._transcribed_at(platforms, "child"), 300.0) + self.assertEqual(cf._transcribed_at(platforms, "parent"), 300.0) + + def test_every_registered_platform_with_an_exporter_gets_a_row(self): + rows = cf.check_native(REPO_ROOT / "platforms", self.cache, self.cache / "truth") + self.assertTrue(rows) + self.assertEqual({r.area for r in rows}, {"native"}) + self.assertEqual({r.status for r in rows}, {cf.STALE}) + self.assertIn("batocera", {r.subject for r in rows}) + + class PlatformDiffTests(unittest.TestCase): def _platform(self, version="1", files=None, cores=None): return { diff --git a/tests/test_exporters.py b/tests/test_exporters.py index 92de7cd6..82a6c09f 100644 --- a/tests/test_exporters.py +++ b/tests/test_exporters.py @@ -15,6 +15,7 @@ and skip when it has not been populated (python scripts/export_native.py from __future__ import annotations import json +import os import re import subprocess import sys @@ -33,7 +34,13 @@ from common import ( # noqa: E402 ) from exporter import discover_exporters # noqa: E402 from exporter.base_exporter import BaseExporter # noqa: E402 -from export_native import pinned_base # noqa: E402 +from export_native import ( # noqa: E402 + SOURCES_INDEX, + collect_originals, + fetch, + pinned_base, + source_key, +) from exporter.baseline import ( # noqa: E402 NativeFile, build_native_model, @@ -437,6 +444,80 @@ class PinnedRevision(unittest.TestCase): self.assertTrue(base.startswith("https://"), base) +class _OneFileExporter: + """An exporter reduced to what collect_originals reads.""" + + def __init__(self, url: str): + self._url = url + + def platform_name(self) -> str: + return "plat" + + def native_sources(self) -> dict[str, str]: + return {"list.txt": self._url} + + +class CachedOriginal(unittest.TestCase): + """The cached original is the file the platform data was read from.""" + + def setUp(self): + self._tmp = tempfile.TemporaryDirectory() + self.addCleanup(self._tmp.cleanup) + self.root = Path(self._tmp.name) + self.cache = self.root / "cache" + self.index = self.cache / SOURCES_INDEX + + def _upstream(self, name: str, text: str) -> str: + path = self.root / name + path.write_text(text, encoding="utf-8") + return path.as_uri() + + def test_the_same_url_is_served_from_the_cache(self): + url = self._upstream("v1.txt", "one") + target = self.cache / "plat" / "list.txt" + self.assertEqual(fetch(url, target, self.index), b"one") + self._upstream("v1.txt", "moved") + self.assertEqual(fetch(url, target, self.index), b"one") + + def test_refresh_reads_a_branch_url_again(self): + """A branch URL keeps its name while the platform moves.""" + url = self._upstream("branch.txt", "one") + target = self.cache / "plat" / "list.txt" + fetch(url, target, self.index) + self._upstream("branch.txt", "two") + self.assertEqual(fetch(url, target, self.index, refresh=True), b"two") + self.assertEqual(target.read_bytes(), b"two") + + def test_a_new_pin_refetches_an_existing_file(self): + """Reading the file straight from disk served the old pin forever.""" + old = self._upstream("tag-1.txt", "one") + new = self._upstream("tag-2.txt", "two") + originals, missing = collect_originals( + _OneFileExporter(old), {}, self.cache, True + ) + self.assertEqual((originals, missing), ({"list.txt": "one"}, [])) + originals, missing = collect_originals( + _OneFileExporter(new), {}, self.cache, True + ) + self.assertEqual((originals, missing), ({"list.txt": "two"}, [])) + + def test_offline_serves_what_is_cached_and_names_what_is_not(self): + url = self._upstream("v1.txt", "one") + _, missing = collect_originals(_OneFileExporter(url), {}, self.cache, False) + self.assertEqual(missing, ["list.txt: absent and fetching is off"]) + collect_originals(_OneFileExporter(url), {}, self.cache, True) + originals, missing = collect_originals( + _OneFileExporter("file:///nowhere/else.txt"), {}, self.cache, False + ) + self.assertEqual((originals, missing), ({"list.txt": "one"}, [])) + + def test_the_index_names_a_file_the_same_from_any_directory(self): + target = self.cache / "plat" / "list.txt" + self.assertEqual(source_key(target, self.index), "plat/list.txt") + relative = Path(os.path.relpath(target, Path.cwd())) + self.assertEqual(source_key(relative, self.index), "plat/list.txt") + + class RecalboxExport(unittest.TestCase): def _render(self): systems, report = model() diff --git a/tests/test_refresh_stale.py b/tests/test_refresh_stale.py new file mode 100644 index 00000000..e647088e --- /dev/null +++ b/tests/test_refresh_stale.py @@ -0,0 +1,145 @@ +"""refresh_stale: which command answers which stale row, and in what order. + +The network side is check_freshness and the scrapers themselves. What is +locked here is the planning: a stale row maps to one command, rows a human +must read are surfaced instead of run, and a platform rescrape is followed +by the refresh of the original its export patches. +""" + +from __future__ import annotations + +import sys +import tempfile +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(REPO_ROOT / "scripts")) + +import refresh_stale as rs # noqa: E402 + +REGISTRY = { + "retrobat": {"scraper": "retrobat", "target_scraper": None}, + "batocera": {"scraper": "batocera", "target_scraper": "batocera_targets"}, + "lakka": {}, +} + + +def finding(area: str, subject: str, status: str = "STALE") -> dict: + return {"area": area, "subject": subject, "status": status, "detail": ""} + + +class Planning(unittest.TestCase): + def _plan(self, *findings: dict): + return rs.plan_jobs(list(findings), REGISTRY) + + def test_a_stale_platform_is_rescraped_then_its_original_is_refreshed(self): + jobs, surfaced, _ = self._plan(finding("platforms", "retrobat")) + self.assertEqual(surfaced, []) + self.assertEqual(len(jobs), 1) + self.assertEqual( + jobs[0].command[1:], + ["-m", "scripts.scraper.retrobat_scraper", "-o", "platforms/retrobat.yml"], + ) + self.assertEqual( + [list(c[1:]) for c in jobs[0].then], + [["scripts/export_native.py", "--platform", "retrobat", "--refresh-cache"]], + ) + + def test_a_native_row_of_a_rescraped_platform_is_not_a_second_job(self): + """Two jobs would race: the refresh must follow the scrape.""" + jobs, _, ignored = self._plan( + finding("native", "retrobat"), finding("platforms", "retrobat") + ) + self.assertEqual([j.area for j in jobs], ["platforms"]) + self.assertEqual([f["area"] for f in ignored], ["native"]) + + def test_a_stale_original_alone_is_refreshed(self): + jobs, _, _ = self._plan(finding("native", "lakka")) + self.assertEqual( + jobs[0].command[1:], + ["scripts/export_native.py", "--platform", "lakka", "--refresh-cache"], + ) + self.assertEqual(jobs[0].then, ()) + + def test_a_platform_without_a_scraper_is_surfaced(self): + jobs, surfaced, _ = self._plan(finding("platforms", "lakka")) + self.assertEqual(jobs, []) + self.assertEqual([f["subject"] for f in surfaced], ["lakka"]) + + def test_unprofiled_cores_need_a_human(self): + jobs, surfaced, _ = self._plan(finding("targets", "batocera cores")) + self.assertEqual(jobs, []) + self.assertEqual(len(surfaced), 1) + + def test_targets_data_and_recipes_map_to_their_refreshers(self): + jobs, surfaced, _ = self._plan( + finding("targets", "batocera"), + finding("data", "dolphin-sys"), + finding("catalogs", "fbneo recipes"), + finding("catalogs", "redump"), + ) + commands = {j.subject: " ".join(j.command[1:]) for j in jobs} + self.assertEqual( + commands, + { + "batocera": "-m scripts.scraper.targets.batocera_targets_scraper" + " -o platforms/targets/batocera.yml", + "dolphin-sys": "scripts/refresh_data_dirs.py --key dolphin-sys --force", + "fbneo recipes": "-m scripts.scraper.romset_dat_importer" + " --source fbneo --fetch", + }, + ) + self.assertEqual([f["subject"] for f in surfaced], ["redump"]) + + def test_only_stale_rows_run_and_errors_are_shown(self): + jobs, surfaced, ignored = self._plan( + finding("data", "a", "OK"), + finding("data", "b", "SKIPPED"), + finding("platforms", "retrobat", "ERROR"), + ) + self.assertEqual(jobs, []) + self.assertEqual([f["status"] for f in surfaced], ["ERROR"]) + self.assertEqual(len(ignored), 2) + + def test_pins_and_profiles_are_never_run(self): + jobs, surfaced, _ = self._plan( + finding("ci", "pypi:jsonschema"), + finding("coreinfo", "libretro-core-info"), + finding("profiles", "profile_sync"), + ) + self.assertEqual(jobs, []) + self.assertEqual(len(surfaced), 3) + + +class Running(unittest.TestCase): + def _job(self, directory: str, command: list[str], then=()) -> rs.Job: + return rs.Job("platforms", "x", command, Path(directory) / "x.log", then) + + def test_a_follow_up_runs_after_a_success(self): + with tempfile.TemporaryDirectory() as directory: + marker = Path(directory) / "ran" + job = self._job( + directory, + [sys.executable, "-c", "pass"], + ((sys.executable, "-c", f"open({str(marker)!r}, 'w').close()"),), + ) + result = rs.run_job(job, {}) + self.assertEqual(result.returncode, 0) + self.assertTrue(marker.exists()) + + def test_a_failed_scrape_leaves_the_original_alone(self): + with tempfile.TemporaryDirectory() as directory: + marker = Path(directory) / "ran" + job = self._job( + directory, + [sys.executable, "-c", "raise SystemExit(3)"], + ((sys.executable, "-c", f"open({str(marker)!r}, 'w').close()"),), + ) + result = rs.run_job(job, {}) + self.assertEqual(result.returncode, 3) + self.assertFalse(marker.exists()) + + +if __name__ == "__main__": + unittest.main() diff --git a/wiki/testing-guide.md b/wiki/testing-guide.md index 94a16fbc..818b4ad7 100644 --- a/wiki/testing-guide.md +++ b/wiki/testing-guide.md @@ -227,7 +227,7 @@ python scripts/generate_pack.py --all --verify-packs --output-dir dist/ ``` Integrated as pipeline step 6/8 (runs after consistency check, before -README generation). Requires packs in `dist/` — skip with `--skip-packs`. +README generation). Requires packs in `dist/`; skip with `--skip-packs`. ## Verification discipline diff --git a/wiki/tools.md b/wiki/tools.md index b78634b3..dd385d0a 100644 --- a/wiki/tools.md +++ b/wiki/tools.md @@ -236,14 +236,26 @@ python scripts/export_native.py --platform batocera --fetch python scripts/export_native.py --all --fetch --output-dir dist/upstream/ ``` -`--fetch` downloads the platform's own file once into -`.cache/upstream-native/`, at the revision the platform YAML's `source:` -names. Formats that carry code are patched from it rather than -regenerated, and the export fails without it. +`--fetch` downloads the platform's own file into `.cache/upstream-native/`, +at the revision the platform YAML's `source:` names. Formats that carry +code are patched from it rather than regenerated, and the export fails +without it. The URL each cached file came from is recorded beside the +cache, so a platform file that moves to a new tag is fetched again instead +of patching the previous one. + +A branch URL keeps its name while its content moves, and no URL comparison +can see that. `--refresh-cache` downloads every file again and writes no +export; it is what follows a rescrape, so the original and the platform +YAML describe the same moment. + +```bash +python scripts/export_native.py --platform retrobat --refresh-cache +python scripts/export_native.py --all --refresh-cache +``` A list is written in the order the code looks: `priority:` first (lowest wins), then the profile's own declaration order. What a format cannot -state is reported rather than written — EmuDeck's arrays are shared +state is reported rather than written: EmuDeck's arrays are shared between emulators its file does not name, so a hash is corrected in place and never added. @@ -326,9 +338,11 @@ Per part of a ref: `ANCHORED` (same content, same lines), `SHIFTED` (same content, moved), `RENAMED` (the source file moved), `CHANGED` (content edited), `AMBIGUOUS` (several equally good candidates), `GONE` (nothing left to anchor to), `EXTERNAL` (the ref names a project the profile does -not declare, so no revision can confirm it). An entry carries the worst +not declare, so no revision can confirm it), `UNCHECKED` (a pin equal to +HEAD is judged on self-consistency, and an entry with neither a hash nor +a name gives that check nothing to look for). An entry carries the worst status of its parts. Only `CHANGED`, `GONE` and `AMBIGUOUS` count as -needing a re-read. +needing a re-read; a zero next to a row of `unchecked` is not a proof. A single cited line is often not distinctive, so the anchor widens by steps of ±3, ±6, ±12, ±25 and ±50 lines until it is unique. Three further @@ -424,9 +438,10 @@ directories the refs point at. Writes are explicit and mechanical only. `--backfill-commits` fills a missing `source_commit`, `--rebase-refs` recales `SHIFTED` and `RENAMED` -line ranges, `--bump-commit` advances `source_commit` to HEAD only when -nothing needs a re-read. All three refuse to run on a dirty `emulators/` -without `--force`, and every write is verified by reparsing the document. +line ranges and moves `source_commit` to HEAD with them, all or nothing, +`--bump-commit` advances `source_commit` to HEAD only when nothing needs a +re-read. All three refuse to run on a dirty `emulators/` without `--force`, +and every write is verified by reparsing the document. Uses `GITHUB_TOKEN` when set, which `--all` requires. Responses are cached under `.cache/upstream/`, addressed by commit sha, so `--offline` replays a @@ -457,6 +472,62 @@ python scripts/refresh_data_dirs.py --platform batocera # single platform onl python scripts/refresh_data_dirs.py --registry path/to/_data_dirs.yml ``` +### check_freshness.py + +Ask every transcribed layer whether the local copy is the one upstream serves +today, and answer with one row per subject. Platform and target files are +re-scraped through the scrapers' own write path into a throwaway copy and +diffed against the committed file, so the diff decides rather than a version +string. Core-info is listed and compared with the profiles (a core with no +profile, a profile that calls standalone a core libretro now builds). Data +directories are checked against their upstream, dump catalogs and romset +snapshots against the catalog they were read from (Redump DAT versions, the +No-Intro daily rebuild, the newest TOSEC pack, the newest MAME release, the +git blobs of the FBNeo DATs), and the CI toolchain against PyPI and the +actions' latest releases. + +The `native` area needs no network. It reads the cache `export_native.py` +patches from and reports an original that is absent, that came from another +URL than the one the platform YAML names, or that was fetched before the +platform YAML was last rewritten. + +```bash +python scripts/check_freshness.py +python scripts/check_freshness.py --only platforms,targets +python scripts/check_freshness.py --json +python scripts/check_freshness.py --profiles # also runs profile_sync (slow) +python scripts/check_freshness.py --offline # local ages only +``` + +The exit code is non-zero when anything is stale. Emulator profiles are the +one layer left to `profile_sync.py`, which the `--profiles` flag folds in. + +### refresh_stale.py + +Run the refresher behind every stale row of `check_freshness.py`, several at +a time, and leave the result in the working tree for review. Nothing is +committed. + +| Stale row | Command | +|-----------|---------| +| `platforms/` | the platform scraper, then the `native` refresh for the same platform | +| `native/` | `export_native.py --platform --refresh-cache` | +| `targets/` | the target scraper | +| `data/` | `refresh_data_dirs.py --key --force` | +| `catalogs/mame recipes`, `catalogs/fbneo recipes` | `scraper/romset_dat_importer.py --source --fetch` | + +Rows that need a decision are listed and never run: a buildbot core with no +profile, a new `.info` file, the Redump, No-Intro and TOSEC packs, the CI +pins, and `profile_sync`. A failed scraper leaves its cached original +untouched. Each job writes its output to `tmp/refresh_stale/`, and a GitHub +token is taken from `GITHUB_TOKEN` or from `gh auth token`. + +```bash +python scripts/refresh_stale.py --dry-run +python scripts/refresh_stale.py +python scripts/refresh_stale.py --only platforms,targets --jobs 6 +``` + ### Other tools | Script | Purpose | @@ -474,12 +545,14 @@ python scripts/refresh_data_dirs.py --registry path/to/_data_dirs.yml | `generate_site.py` | Generate all MkDocs site pages (this documentation) | | `validate_site.py` | Validate rendered metadata, headings, image alternatives, JSON-LD, local resources, links and fragments | | `romset_recipes.py` | Identify which emulator version an arcade archive matches, and rebuild a pinned archive from ROMs already held | -| `scraper/romset_dat_importer.py` | Fetch and import per-set recipes from MAME `-listxml` or FBNeo DATs into `recipes/` | +| `scraper/romset_dat_importer.py` | Fetch and import per-set recipes from MAME `-listxml` or FBNeo DATs into `recipes/`; `--fetch` alone takes the newest MAME release, and FBNeo DATs are refreshed by git blob sha | | `deterministic_zip.py` | Rebuild MAME BIOS ZIPs deterministically (same ROMs = same hash) | | `torrentzip.py` | Build TorrentZip archives for MAME/FBNeo ROM sets (archive bytes depend only on contents) | | `crypto_verify.py` | 3DS RSA signature and AES crypto verification | | `sect233r1.py` | Pure Python ECDSA verification on sect233r1 curve (3DS OTP cert) | | `check_buildbot_system.py` | Detect stale data directories by comparing with buildbot | +| `check_freshness.py` | One report on every upstream the repository transcribes (see above) | +| `refresh_stale.py` | Run the refresher behind every stale row of that report (see above) | | `migrate.py` | Migrate flat bios structure to Manufacturer/Console/ hierarchy | ## Installation tools