feat: keep cached platform originals in step

This commit is contained in:
Abdessamad Derraz committed 2026-10-04 16:36:35 +02:00
1 parent d50af62cde
commit b81199ff26
8 files changed
+654 -68

No files matched your search

+89 -1
View File
@@ -12,6 +12,11 @@ into a throwaway copy, then diffed against the committed file. The diff
decides, not a version string: a scraper that pins a new tag but produces decides, not a version string: a scraper that pins a new tag but produces
the same entries is reported as a version change and nothing else. the same entries is reported as a version change and nothing else.
The native export patches each platform's own file, kept in a local cache.
That cache is checked against the platform file it was transcribed with: an
original cached before the last rescrape, or under another pin, is no longer
the file our data describes.
Emulator profiles are covered by profile_sync, which is slow (one API pass Emulator profiles are covered by profile_sync, which is slow (one API pass
per profile). ``--profiles`` runs it here and folds its verdict in. per profile). ``--profiles`` runs it here and folds its verdict in.
@@ -43,6 +48,7 @@ from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent)) sys.path.insert(0, str(Path(__file__).resolve().parent))
sys.path.insert(0, str(Path(__file__).resolve().parents[1])) sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
import export_native # noqa: E402
import refresh_data_dirs # noqa: E402 import refresh_data_dirs # noqa: E402
import upstream # noqa: E402 import upstream # noqa: E402
from common import ( # noqa: E402 from common import ( # noqa: E402
@@ -50,6 +56,8 @@ from common import ( # noqa: E402
load_emulator_profiles, load_emulator_profiles,
yaml_load, yaml_load,
) )
from exporter import discover_exporters # noqa: E402
from exporter.baseline import build_native_model # noqa: E402
from scripts.scraper.redump_dat_scraper import fetch_snapshot # noqa: E402 from scripts.scraper.redump_dat_scraper import fetch_snapshot # noqa: E402
from scripts.scraper.targets import discover_target_scrapers # noqa: E402 from scripts.scraper.targets import discover_target_scrapers # noqa: E402
@@ -66,7 +74,7 @@ INSTALLERS = ("install.sh", "install.ps1")
SHOWN_ITEMS = 4 # items named in a diff summary before ", ..." SHOWN_ITEMS = 4 # items named in a diff summary before ", ..."
SHOWN_NAMES = 12 # core names named in an unresolved row before ", ..." SHOWN_NAMES = 12 # core names named in an unresolved row before ", ..."
AREAS = ("platforms", "targets", "coreinfo", "data", "catalogs", "ci", "profiles") AREAS = ("platforms", "native", "targets", "coreinfo", "data", "catalogs", "ci", "profiles")
OK, STALE, UNKNOWN, ERROR, SKIPPED = "OK", "STALE", "UNKNOWN", "ERROR", "SKIPPED" OK, STALE, UNKNOWN, ERROR, SKIPPED = "OK", "STALE", "UNKNOWN", "ERROR", "SKIPPED"
# Every status a finding may carry, in the order the summary counts them. # Every status a finding may carry, in the order the summary counts them.
STATUSES = (STALE, UNKNOWN, ERROR, SKIPPED, OK) STATUSES = (STALE, UNKNOWN, ERROR, SKIPPED, OK)
@@ -279,6 +287,80 @@ def check_platforms(platforms_dir: Path, workdir: Path, jobs: int) -> list[Findi
return sorted(findings, key=lambda f: f.subject) return sorted(findings, key=lambda f: f.subject)
# --- native originals --------------------------------------------------------
def native_cache_state(
wanted: dict[str, str],
root: Path,
recorded: dict[str, str],
index: Path,
transcribed_at: float,
) -> str:
"""Why a platform's cached originals cannot be patched, '' when they can.
The export corrects the file our data was transcribed from. A cached
original is that file only if it came from the URL the platform file
names today and was fetched no earlier than the platform file was last
rewritten: a branch URL keeps its name while its content moves.
"""
for relative, url in sorted(wanted.items()):
path = root / relative
if not path.is_file():
return f"{relative} not cached"
known = recorded.get(export_native.source_key(path, index))
if known is None:
return f"{relative} cached from an unrecorded URL"
if known != url:
return f"{relative} cached from another revision"
if path.stat().st_mtime < transcribed_at:
return f"{relative} cached before the platform file was rewritten"
return ""
def _transcribed_at(platforms_dir: Path, platform: str) -> float:
"""Last rewrite of the platform file or of any file it inherits."""
newest = 0.0
seen: set[str] = set()
name: str | None = platform
while name and name not in seen:
seen.add(name)
path = platforms_dir / f"{name}.yml"
if not path.is_file():
break
newest = max(newest, path.stat().st_mtime)
with path.open(encoding="utf-8") as fh:
name = (yaml_load(fh) or {}).get("inherits")
return newest
def check_native(platforms_dir: Path, cache_dir: Path, truth_dir: Path) -> list[Finding]:
exporters = discover_exporters()
index = cache_dir / export_native.SOURCES_INDEX
recorded = export_native.load_sources(cache_dir)
findings: list[Finding] = []
for name in list_registered_platforms(str(platforms_dir), include_archived=True):
exporter_class = exporters.get(name)
if not exporter_class:
continue
truth, scraped = export_native.load_inputs(name, truth_dir, str(platforms_dir))
systems, _ = build_native_model(truth or {}, scraped)
wanted = export_native.wanted_sources(exporter_class(), systems, scraped)
reason = native_cache_state(
wanted, cache_dir / name, recorded, index, _transcribed_at(platforms_dir, name)
)
findings.append(
Finding(
"native",
name,
STALE if reason else OK,
local=f"{len(wanted)} file(s)",
detail=f"{reason}, export_native.py --refresh-cache fetches it" if reason else "",
)
)
return sorted(findings, key=lambda f: f.subject)
# --- targets ---------------------------------------------------------------- # --- targets ----------------------------------------------------------------
@@ -959,6 +1041,8 @@ def main() -> int:
parser.add_argument("--offline", action="store_true", help="no network, local ages only") parser.add_argument("--offline", action="store_true", help="no network, local ages only")
parser.add_argument("--jobs", type=int, default=4, help="parallel scrapers") parser.add_argument("--jobs", type=int, default=4, help="parallel scrapers")
parser.add_argument("--cache-dir", default=".cache/upstream") parser.add_argument("--cache-dir", default=".cache/upstream")
parser.add_argument("--native-cache-dir", default=export_native.DEFAULT_CACHE)
parser.add_argument("--truth-dir", default="dist/truth")
parser.add_argument("--platforms-dir", default="platforms") parser.add_argument("--platforms-dir", default="platforms")
parser.add_argument("--emulators-dir", default="emulators") parser.add_argument("--emulators-dir", default="emulators")
args = parser.parse_args() args = parser.parse_args()
@@ -979,6 +1063,10 @@ def main() -> int:
if args.offline if args.offline
else check_platforms(platforms_dir, workdir, args.jobs) else check_platforms(platforms_dir, workdir, args.jobs)
) )
if "native" in areas:
findings.extend(
check_native(platforms_dir, Path(args.native_cache_dir), Path(args.truth_dir))
)
if "targets" in areas: if "targets" in areas:
findings.extend(check_targets(platforms_dir, workdir, profiles, args.jobs, args.offline)) findings.extend(check_targets(platforms_dir, workdir, profiles, args.jobs, args.offline))
if "coreinfo" in areas: if "coreinfo" in areas:
+124 -36
View File
@@ -10,6 +10,7 @@ a platform ships keeps working.
Usage: Usage:
python scripts/export_native.py --all --fetch python scripts/export_native.py --all --fetch
python scripts/export_native.py --platform recalbox --upstream-dir up/ python scripts/export_native.py --platform recalbox --upstream-dir up/
python scripts/export_native.py --platform retrobat --refresh-cache
""" """
from __future__ import annotations from __future__ import annotations
@@ -28,6 +29,7 @@ from exporter import discover_exporters
from exporter.baseline import build_native_model from exporter.baseline import build_native_model
DEFAULT_CACHE = ".cache/upstream-native" DEFAULT_CACHE = ".cache/upstream-native"
SOURCES_INDEX = ".sources.json"
_USER_AGENT = "retrobios-exporter/1.0" _USER_AGENT = "retrobios-exporter/1.0"
_MAX_BYTES = 64 * 1024 * 1024 _MAX_BYTES = 64 * 1024 * 1024
@@ -41,7 +43,29 @@ def _load_sources(index: Path | None) -> dict[str, str]:
return {} return {}
def fetch(url: str, destination: Path, index: Path | None = None) -> bytes: def source_key(destination: Path, index: Path | None) -> str:
"""How the index names a cached file: its path below the cache root.
A key that repeated the cache directory as it was typed changed with the
working directory, and a file recorded from one place looked unrecorded
from another.
"""
if index is not None:
try:
return destination.resolve().relative_to(index.parent.resolve()).as_posix()
except ValueError:
pass
return str(destination)
def load_sources(upstream_dir: Path) -> dict[str, str]:
"""Cached file -> URL it was fetched from, for a whole cache directory."""
return _load_sources(upstream_dir / SOURCES_INDEX)
def fetch(
url: str, destination: Path, index: Path | None = None, refresh: bool = False
) -> bytes:
"""Download an original once, then read it from the cache. """Download an original once, then read it from the cache.
The cache path carries the file's own name and nothing of the revision it The cache path carries the file's own name and nothing of the revision it
@@ -50,10 +74,15 @@ def fetch(url: str, destination: Path, index: Path | None = None) -> bytes:
URL that produced each cached file is recorded beside the cache, and a URL that produced each cached file is recorded beside the cache, and a
different URL refetches. A cache written before this index existed keeps different URL refetches. A cache written before this index existed keeps
being served: nothing recorded means nothing contradicted. being served: nothing recorded means nothing contradicted.
A branch URL names no revision, so the same URL serves new bytes after
the platform moves. refresh downloads again whatever the cache holds:
it is what a rescrape calls, so the original and its transcription
describe the same moment.
""" """
recorded = _load_sources(index) recorded = _load_sources(index)
key = str(destination) key = source_key(destination, index)
if destination.exists() and recorded.get(key, url) == url: if not refresh and destination.exists() and recorded.get(key, url) == url:
return destination.read_bytes() return destination.read_bytes()
request = urllib.request.Request(url, headers={"User-Agent": _USER_AGENT}) request = urllib.request.Request(url, headers={"User-Agent": _USER_AGENT})
with urllib.request.urlopen(request, timeout=60) as response: with urllib.request.urlopen(request, timeout=60) as response:
@@ -94,14 +123,10 @@ def _raw_url(url: str) -> str:
return url return url
def collect_originals( def wanted_sources(
exporter: object, exporter: object, systems: dict, scraped: dict | None = None
systems: dict, ) -> dict[str, str]:
upstream_dir: Path, """Cache-relative name -> URL of every file the exporter patches."""
allow_fetch: bool,
scraped: dict | None = None,
) -> tuple[dict[str, str], list[str]]:
"""Gather the platform's own files, from disk or from upstream."""
wanted = dict(exporter.native_sources()) wanted = dict(exporter.native_sources())
components = getattr(exporter, "components", None) components = getattr(exporter, "components", None)
if callable(components): if callable(components):
@@ -113,21 +138,39 @@ def collect_originals(
base = pinned_base(wanted, scraped) base = pinned_base(wanted, scraped)
if base: if base:
wanted = {relative: base + relative for relative in wanted} wanted = {relative: base + relative for relative in wanted}
return wanted
def collect_originals(
exporter: object,
systems: dict,
upstream_dir: Path,
allow_fetch: bool,
scraped: dict | None = None,
refresh: bool = False,
) -> tuple[dict[str, str], list[str]]:
"""Gather the platform's own files, from disk or from upstream.
With fetching on, an existing file still goes through fetch(): reading
it straight from disk skipped the recorded URL, so a file cached under
one pin kept being patched after the platform YAML moved to the next.
"""
wanted = wanted_sources(exporter, systems, scraped)
root = upstream_dir / exporter.platform_name() root = upstream_dir / exporter.platform_name()
index = upstream_dir / SOURCES_INDEX
originals: dict[str, str] = {} originals: dict[str, str] = {}
missing: list[str] = [] missing: list[str] = []
for relative, url in wanted.items(): for relative, url in wanted.items():
path = root / relative path = root / relative
payload: bytes | None = None payload: bytes | None = None
if path.exists(): if allow_fetch:
payload = path.read_bytes()
elif allow_fetch:
try: try:
payload = fetch(url, path, upstream_dir / ".sources.json") payload = fetch(url, path, index, refresh)
except (urllib.error.URLError, urllib.error.HTTPError, OSError) as exc: except (urllib.error.URLError, urllib.error.HTTPError, OSError) as exc:
missing.append(f"{relative}: {exc}") missing.append(f"{relative}: {exc}")
continue continue
elif path.exists():
payload = path.read_bytes()
else: else:
missing.append(f"{relative}: absent and fetching is off") missing.append(f"{relative}: absent and fetching is off")
continue continue
@@ -141,6 +184,42 @@ def collect_originals(
return originals, missing return originals, missing
def load_inputs(
platform: str, truth_dir: Path, platforms_dir: str
) -> tuple[dict | None, dict | None]:
"""The truth and the scraped config of a platform, None where absent."""
truth_file = truth_dir / f"{platform}.yml"
truth: dict | None = None
if truth_file.exists():
with open(truth_file) as handle:
truth = yaml_load(handle) or {}
try:
scraped = load_platform_config(platform, platforms_dir)
except (FileNotFoundError, OSError):
scraped = None
return truth, scraped
def refresh_cache(
platform: str,
exporter_class: type,
truth_dir: Path,
platforms_dir: str,
upstream_dir: Path,
) -> tuple[bool, list[str]]:
"""Download a platform's own files again. Returns (ok, messages)."""
truth, scraped = load_inputs(platform, truth_dir, platforms_dir)
systems, _ = build_native_model(truth or {}, scraped)
exporter = exporter_class()
wanted = wanted_sources(exporter, systems, scraped)
_, missing = collect_originals(
exporter, systems, upstream_dir, True, scraped, refresh=True
)
if missing:
return False, missing
return True, [f"{len(wanted)} file(s) fetched"]
def export_platform( def export_platform(
platform: str, platform: str,
exporter_class: type, exporter_class: type,
@@ -153,19 +232,11 @@ def export_platform(
"""Write one platform's corrected file. Returns (ok, messages).""" """Write one platform's corrected file. Returns (ok, messages)."""
messages: list[str] = [] messages: list[str] = []
truth_file = truth_dir / f"{platform}.yml" truth, scraped = load_inputs(platform, truth_dir, platforms_dir)
truth: dict = {} if truth is None:
if truth_file.exists(): truth = {}
with open(truth_file) as handle:
truth = yaml_load(handle) or {}
else:
messages.append(f"no truth for {platform}, only the platform's own data") messages.append(f"no truth for {platform}, only the platform's own data")
try:
scraped = load_platform_config(platform, platforms_dir)
except (FileNotFoundError, OSError):
scraped = None
systems, report = build_native_model(truth, scraped) systems, report = build_native_model(truth, scraped)
if not systems: if not systems:
return False, ["nothing to write: neither the platform nor the truth has data"] return False, ["nothing to write: neither the platform nor the truth has data"]
@@ -253,6 +324,7 @@ def run(
platforms_dir: str, platforms_dir: str,
upstream_dir: str, upstream_dir: str,
allow_fetch: bool, allow_fetch: bool,
refresh_only: bool = False,
) -> int: ) -> int:
exporters = discover_exporters() exporters = discover_exporters()
failures = 0 failures = 0
@@ -265,15 +337,24 @@ def run(
print(f" SKIP {platform}: no exporter") print(f" SKIP {platform}: no exporter")
continue continue
ok, messages = export_platform( if refresh_only:
platform, ok, messages = refresh_cache(
exporter_class, platform,
Path(truth_dir), exporter_class,
Path(output_dir), Path(truth_dir),
platforms_dir, platforms_dir,
Path(upstream_dir), Path(upstream_dir),
allow_fetch, )
) else:
ok, messages = export_platform(
platform,
exporter_class,
Path(truth_dir),
Path(output_dir),
platforms_dir,
Path(upstream_dir),
allow_fetch,
)
label = "OK " if ok else "FAIL" label = "OK " if ok else "FAIL"
print(f" {label} {platform}") print(f" {label} {platform}")
for message in messages: for message in messages:
@@ -306,7 +387,13 @@ def main() -> None:
parser.add_argument( parser.add_argument(
"--fetch", "--fetch",
action="store_true", action="store_true",
help="download a platform's file when it is not in the cache", help="download a platform's file when it is not in the cache, "
"or when the cache came from another URL",
)
parser.add_argument(
"--refresh-cache",
action="store_true",
help="download every platform file again and write no export",
) )
args = parser.parse_args() args = parser.parse_args()
@@ -326,6 +413,7 @@ def main() -> None:
args.platforms_dir, args.platforms_dir,
args.upstream_dir, args.upstream_dir,
args.fetch, args.fetch,
args.refresh_cache,
) )
) )
+61 -18
View File
@@ -10,6 +10,10 @@ committed.
Mapping (stale -> command): Mapping (stale -> command):
platforms/<name> python -m scripts.scraper.<module>_scraper platforms/<name> python -m scripts.scraper.<module>_scraper
-o platforms/<name>.yml -o platforms/<name>.yml
then the native refresh below: the original
the export patches must be the one just scraped
native/<name> python scripts/export_native.py --platform <name>
--refresh-cache
targets/<name> python -m scripts.scraper.targets.<module> targets/<name> python -m scripts.scraper.targets.<module>
-o platforms/targets/<name>.yml -o platforms/targets/<name>.yml
data/<key> python scripts/refresh_data_dirs.py --key <key> data/<key> python scripts/refresh_data_dirs.py --key <key>
@@ -47,7 +51,7 @@ LOG_DIR = REPO_ROOT / "tmp" / "refresh_stale"
CHECK_FRESHNESS = REPO_ROOT / "scripts" / "check_freshness.py" CHECK_FRESHNESS = REPO_ROOT / "scripts" / "check_freshness.py"
PLATFORMS_REGISTRY = REPO_ROOT / "platforms" / "_registry.yml" PLATFORMS_REGISTRY = REPO_ROOT / "platforms" / "_registry.yml"
AUTO_AREAS = ("platforms", "targets", "data", "catalogs") AUTO_AREAS = ("platforms", "native", "targets", "data", "catalogs")
MANUAL_AREAS = ("coreinfo", "ci", "profiles") MANUAL_AREAS = ("coreinfo", "ci", "profiles")
JOB_TIMEOUT = 1800 # 30 minutes per refresher; scrapers rarely exceed 10 JOB_TIMEOUT = 1800 # 30 minutes per refresher; scrapers rarely exceed 10
@@ -61,6 +65,8 @@ class Job:
subject: str subject: str
command: list[str] command: list[str]
log_path: Path log_path: Path
# Commands run after `command`, in order, only while each one succeeds.
then: tuple[tuple[str, ...], ...] = ()
@dataclass(frozen=True) @dataclass(frozen=True)
@@ -122,6 +128,15 @@ def plan_jobs(
seen: set[tuple[str, str]] = set() seen: set[tuple[str, str]] = set()
surfaced: list[dict] = [] surfaced: list[dict] = []
ignored: list[dict] = [] ignored: list[dict] = []
# A platform about to be rescraped refreshes its cached original itself,
# after the scrape: a separate native job would race the scraper.
rescraped = {
str(f.get("subject") or "")
for f in findings
if f.get("status") == "STALE"
and f.get("area") == "platforms"
and _command_for("platforms", str(f.get("subject") or ""), registry)
}
for finding in findings: for finding in findings:
status = finding.get("status") status = finding.get("status")
@@ -135,6 +150,10 @@ def plan_jobs(
ignored.append(finding) ignored.append(finding)
continue continue
if area == "native" and subject in rescraped:
ignored.append(finding)
continue
command = _command_for(area, subject, registry) command = _command_for(area, subject, registry)
if command is None: if command is None:
surfaced.append(finding) surfaced.append(finding)
@@ -148,15 +167,33 @@ def plan_jobs(
seen.add(key) seen.add(key)
log = LOG_DIR / f"{area}__{_slug(subject)}.log" log = LOG_DIR / f"{area}__{_slug(subject)}.log"
jobs.append(Job(area=area, subject=subject, command=command, log_path=log)) then: tuple[tuple[str, ...], ...] = ()
if area == "platforms":
then = (tuple(_native_refresh(subject)),)
jobs.append(
Job(area=area, subject=subject, command=command, log_path=log, then=then)
)
return jobs, surfaced, ignored return jobs, surfaced, ignored
def _native_refresh(platform: str) -> list[str]:
return [
sys.executable,
"scripts/export_native.py",
"--platform",
platform,
"--refresh-cache",
]
def _command_for( def _command_for(
area: str, subject: str, registry: dict[str, dict] area: str, subject: str, registry: dict[str, dict]
) -> list[str] | None: ) -> list[str] | None:
"""Map a stale finding to its refresh command, or None if manual.""" """Map a stale finding to its refresh command, or None if manual."""
if area == "native":
return _native_refresh(subject)
if area == "platforms": if area == "platforms":
entry = registry.get(subject) or {} entry = registry.get(subject) or {}
scraper = entry.get("scraper") scraper = entry.get("scraper")
@@ -250,23 +287,27 @@ def run_job(job: Job, env: dict[str, str]) -> JobResult:
"""Execute one refresh command and collect its output.""" """Execute one refresh command and collect its output."""
start = time.monotonic() start = time.monotonic()
job.log_path.parent.mkdir(parents=True, exist_ok=True) job.log_path.parent.mkdir(parents=True, exist_ok=True)
returncode = 0
with job.log_path.open("w", encoding="utf-8") as log: with job.log_path.open("w", encoding="utf-8") as log:
log.write(f"$ {' '.join(job.command)}\n") for command in (job.command, *job.then):
log.flush() log.write(f"$ {' '.join(command)}\n")
try: log.flush()
proc = subprocess.run( try:
job.command, proc = subprocess.run(
cwd=REPO_ROOT, list(command),
stdout=log, cwd=REPO_ROOT,
stderr=subprocess.STDOUT, stdout=log,
env=env, stderr=subprocess.STDOUT,
timeout=JOB_TIMEOUT, env=env,
check=False, timeout=JOB_TIMEOUT,
) check=False,
returncode = proc.returncode )
except subprocess.TimeoutExpired: returncode = proc.returncode
log.write(f"\nTIMEOUT after {JOB_TIMEOUT}s\n") except subprocess.TimeoutExpired:
returncode = 124 log.write(f"\nTIMEOUT after {JOB_TIMEOUT}s\n")
returncode = 124
if returncode != 0:
break
duration = time.monotonic() - start duration = time.monotonic() - start
tail = _log_tail(job.log_path) tail = _log_tail(job.log_path)
return JobResult(job=job, returncode=returncode, duration=duration, tail=tail) return JobResult(job=job, returncode=returncode, duration=duration, tail=tail)
@@ -388,6 +429,8 @@ def main() -> int:
f"{len(ignored)} already clean/skipped") f"{len(ignored)} already clean/skipped")
for job in sorted(jobs, key=lambda j: (j.area, j.subject)): for job in sorted(jobs, key=lambda j: (j.area, j.subject)):
print(f" {job.area:10} {job.subject:28} {' '.join(job.command)}") print(f" {job.area:10} {job.subject:28} {' '.join(job.command)}")
for command in job.then:
print(f" {'':10} {'':28} then {' '.join(command)}")
if surfaced: if surfaced:
print("\n[manual review needed]") print("\n[manual review needed]")
for f in surfaced: for f in surfaced:
+68
View File
@@ -8,7 +8,9 @@ report is folded, since each of those decides whether a row says STALE.
from __future__ import annotations from __future__ import annotations
import os
import sys import sys
import tempfile
import unittest import unittest
from pathlib import Path from pathlib import Path
@@ -19,6 +21,72 @@ import check_freshness as cf # noqa: E402
from common import yaml_load # noqa: E402 from common import yaml_load # noqa: E402
class NativeCacheTests(unittest.TestCase):
"""A cached original is patched only if it is the file we transcribed."""
def setUp(self):
self._tmp = tempfile.TemporaryDirectory()
self.addCleanup(self._tmp.cleanup)
self.cache = Path(self._tmp.name)
self.index = self.cache / cf.export_native.SOURCES_INDEX
self.root = self.cache / "plat"
self.root.mkdir()
self.wanted = {"list.txt": "https://example.invalid/tag-2/list.txt"}
def _cached(self, mtime: float) -> None:
path = self.root / "list.txt"
path.write_text("x", encoding="utf-8")
os.utime(path, (mtime, mtime))
def _state(self, recorded: dict, transcribed_at: float) -> str:
return cf.native_cache_state(
self.wanted, self.root, recorded, self.index, transcribed_at
)
def test_a_file_fetched_after_the_scrape_from_the_named_url_is_fresh(self):
self._cached(200.0)
recorded = {"plat/list.txt": self.wanted["list.txt"]}
self.assertEqual(self._state(recorded, 100.0), "")
def test_an_absent_file_is_named(self):
self.assertEqual(self._state({}, 0.0), "list.txt not cached")
def test_a_file_with_no_recorded_url_proves_nothing(self):
self._cached(200.0)
self.assertEqual(self._state({}, 100.0), "list.txt cached from an unrecorded URL")
def test_a_file_from_the_previous_pin_is_stale(self):
self._cached(200.0)
recorded = {"plat/list.txt": "https://example.invalid/tag-1/list.txt"}
self.assertEqual(self._state(recorded, 100.0), "list.txt cached from another revision")
def test_a_file_older_than_the_rescrape_is_stale(self):
"""A branch URL does not change when its content does."""
self._cached(100.0)
recorded = {"plat/list.txt": self.wanted["list.txt"]}
self.assertEqual(
self._state(recorded, 200.0),
"list.txt cached before the platform file was rewritten",
)
def test_the_transcription_date_follows_inheritance(self):
platforms = self.cache / "platforms"
platforms.mkdir()
(platforms / "parent.yml").write_text("platform: P\n", encoding="utf-8")
(platforms / "child.yml").write_text("inherits: parent\n", encoding="utf-8")
os.utime(platforms / "parent.yml", (300.0, 300.0))
os.utime(platforms / "child.yml", (100.0, 100.0))
self.assertEqual(cf._transcribed_at(platforms, "child"), 300.0)
self.assertEqual(cf._transcribed_at(platforms, "parent"), 300.0)
def test_every_registered_platform_with_an_exporter_gets_a_row(self):
rows = cf.check_native(REPO_ROOT / "platforms", self.cache, self.cache / "truth")
self.assertTrue(rows)
self.assertEqual({r.area for r in rows}, {"native"})
self.assertEqual({r.status for r in rows}, {cf.STALE})
self.assertIn("batocera", {r.subject for r in rows})
class PlatformDiffTests(unittest.TestCase): class PlatformDiffTests(unittest.TestCase):
def _platform(self, version="1", files=None, cores=None): def _platform(self, version="1", files=None, cores=None):
return { return {
+82 -1
View File
@@ -15,6 +15,7 @@ and skip when it has not been populated (python scripts/export_native.py
from __future__ import annotations from __future__ import annotations
import json import json
import os
import re import re
import subprocess import subprocess
import sys import sys
@@ -33,7 +34,13 @@ from common import ( # noqa: E402
) )
from exporter import discover_exporters # noqa: E402 from exporter import discover_exporters # noqa: E402
from exporter.base_exporter import BaseExporter # noqa: E402 from exporter.base_exporter import BaseExporter # noqa: E402
from export_native import pinned_base # noqa: E402 from export_native import ( # noqa: E402
SOURCES_INDEX,
collect_originals,
fetch,
pinned_base,
source_key,
)
from exporter.baseline import ( # noqa: E402 from exporter.baseline import ( # noqa: E402
NativeFile, NativeFile,
build_native_model, build_native_model,
@@ -437,6 +444,80 @@ class PinnedRevision(unittest.TestCase):
self.assertTrue(base.startswith("https://"), base) self.assertTrue(base.startswith("https://"), base)
class _OneFileExporter:
"""An exporter reduced to what collect_originals reads."""
def __init__(self, url: str):
self._url = url
def platform_name(self) -> str:
return "plat"
def native_sources(self) -> dict[str, str]:
return {"list.txt": self._url}
class CachedOriginal(unittest.TestCase):
"""The cached original is the file the platform data was read from."""
def setUp(self):
self._tmp = tempfile.TemporaryDirectory()
self.addCleanup(self._tmp.cleanup)
self.root = Path(self._tmp.name)
self.cache = self.root / "cache"
self.index = self.cache / SOURCES_INDEX
def _upstream(self, name: str, text: str) -> str:
path = self.root / name
path.write_text(text, encoding="utf-8")
return path.as_uri()
def test_the_same_url_is_served_from_the_cache(self):
url = self._upstream("v1.txt", "one")
target = self.cache / "plat" / "list.txt"
self.assertEqual(fetch(url, target, self.index), b"one")
self._upstream("v1.txt", "moved")
self.assertEqual(fetch(url, target, self.index), b"one")
def test_refresh_reads_a_branch_url_again(self):
"""A branch URL keeps its name while the platform moves."""
url = self._upstream("branch.txt", "one")
target = self.cache / "plat" / "list.txt"
fetch(url, target, self.index)
self._upstream("branch.txt", "two")
self.assertEqual(fetch(url, target, self.index, refresh=True), b"two")
self.assertEqual(target.read_bytes(), b"two")
def test_a_new_pin_refetches_an_existing_file(self):
"""Reading the file straight from disk served the old pin forever."""
old = self._upstream("tag-1.txt", "one")
new = self._upstream("tag-2.txt", "two")
originals, missing = collect_originals(
_OneFileExporter(old), {}, self.cache, True
)
self.assertEqual((originals, missing), ({"list.txt": "one"}, []))
originals, missing = collect_originals(
_OneFileExporter(new), {}, self.cache, True
)
self.assertEqual((originals, missing), ({"list.txt": "two"}, []))
def test_offline_serves_what_is_cached_and_names_what_is_not(self):
url = self._upstream("v1.txt", "one")
_, missing = collect_originals(_OneFileExporter(url), {}, self.cache, False)
self.assertEqual(missing, ["list.txt: absent and fetching is off"])
collect_originals(_OneFileExporter(url), {}, self.cache, True)
originals, missing = collect_originals(
_OneFileExporter("file:///nowhere/else.txt"), {}, self.cache, False
)
self.assertEqual((originals, missing), ({"list.txt": "one"}, []))
def test_the_index_names_a_file_the_same_from_any_directory(self):
target = self.cache / "plat" / "list.txt"
self.assertEqual(source_key(target, self.index), "plat/list.txt")
relative = Path(os.path.relpath(target, Path.cwd()))
self.assertEqual(source_key(relative, self.index), "plat/list.txt")
class RecalboxExport(unittest.TestCase): class RecalboxExport(unittest.TestCase):
def _render(self): def _render(self):
systems, report = model() systems, report = model()
+145
View File
@@ -0,0 +1,145 @@
"""refresh_stale: which command answers which stale row, and in what order.
The network side is check_freshness and the scrapers themselves. What is
locked here is the planning: a stale row maps to one command, rows a human
must read are surfaced instead of run, and a platform rescrape is followed
by the refresh of the original its export patches.
"""
from __future__ import annotations
import sys
import tempfile
import unittest
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parents[1]
sys.path.insert(0, str(REPO_ROOT / "scripts"))
import refresh_stale as rs # noqa: E402
REGISTRY = {
"retrobat": {"scraper": "retrobat", "target_scraper": None},
"batocera": {"scraper": "batocera", "target_scraper": "batocera_targets"},
"lakka": {},
}
def finding(area: str, subject: str, status: str = "STALE") -> dict:
return {"area": area, "subject": subject, "status": status, "detail": ""}
class Planning(unittest.TestCase):
def _plan(self, *findings: dict):
return rs.plan_jobs(list(findings), REGISTRY)
def test_a_stale_platform_is_rescraped_then_its_original_is_refreshed(self):
jobs, surfaced, _ = self._plan(finding("platforms", "retrobat"))
self.assertEqual(surfaced, [])
self.assertEqual(len(jobs), 1)
self.assertEqual(
jobs[0].command[1:],
["-m", "scripts.scraper.retrobat_scraper", "-o", "platforms/retrobat.yml"],
)
self.assertEqual(
[list(c[1:]) for c in jobs[0].then],
[["scripts/export_native.py", "--platform", "retrobat", "--refresh-cache"]],
)
def test_a_native_row_of_a_rescraped_platform_is_not_a_second_job(self):
"""Two jobs would race: the refresh must follow the scrape."""
jobs, _, ignored = self._plan(
finding("native", "retrobat"), finding("platforms", "retrobat")
)
self.assertEqual([j.area for j in jobs], ["platforms"])
self.assertEqual([f["area"] for f in ignored], ["native"])
def test_a_stale_original_alone_is_refreshed(self):
jobs, _, _ = self._plan(finding("native", "lakka"))
self.assertEqual(
jobs[0].command[1:],
["scripts/export_native.py", "--platform", "lakka", "--refresh-cache"],
)
self.assertEqual(jobs[0].then, ())
def test_a_platform_without_a_scraper_is_surfaced(self):
jobs, surfaced, _ = self._plan(finding("platforms", "lakka"))
self.assertEqual(jobs, [])
self.assertEqual([f["subject"] for f in surfaced], ["lakka"])
def test_unprofiled_cores_need_a_human(self):
jobs, surfaced, _ = self._plan(finding("targets", "batocera cores"))
self.assertEqual(jobs, [])
self.assertEqual(len(surfaced), 1)
def test_targets_data_and_recipes_map_to_their_refreshers(self):
jobs, surfaced, _ = self._plan(
finding("targets", "batocera"),
finding("data", "dolphin-sys"),
finding("catalogs", "fbneo recipes"),
finding("catalogs", "redump"),
)
commands = {j.subject: " ".join(j.command[1:]) for j in jobs}
self.assertEqual(
commands,
{
"batocera": "-m scripts.scraper.targets.batocera_targets_scraper"
" -o platforms/targets/batocera.yml",
"dolphin-sys": "scripts/refresh_data_dirs.py --key dolphin-sys --force",
"fbneo recipes": "-m scripts.scraper.romset_dat_importer"
" --source fbneo --fetch",
},
)
self.assertEqual([f["subject"] for f in surfaced], ["redump"])
def test_only_stale_rows_run_and_errors_are_shown(self):
jobs, surfaced, ignored = self._plan(
finding("data", "a", "OK"),
finding("data", "b", "SKIPPED"),
finding("platforms", "retrobat", "ERROR"),
)
self.assertEqual(jobs, [])
self.assertEqual([f["status"] for f in surfaced], ["ERROR"])
self.assertEqual(len(ignored), 2)
def test_pins_and_profiles_are_never_run(self):
jobs, surfaced, _ = self._plan(
finding("ci", "pypi:jsonschema"),
finding("coreinfo", "libretro-core-info"),
finding("profiles", "profile_sync"),
)
self.assertEqual(jobs, [])
self.assertEqual(len(surfaced), 3)
class Running(unittest.TestCase):
def _job(self, directory: str, command: list[str], then=()) -> rs.Job:
return rs.Job("platforms", "x", command, Path(directory) / "x.log", then)
def test_a_follow_up_runs_after_a_success(self):
with tempfile.TemporaryDirectory() as directory:
marker = Path(directory) / "ran"
job = self._job(
directory,
[sys.executable, "-c", "pass"],
((sys.executable, "-c", f"open({str(marker)!r}, 'w').close()"),),
)
result = rs.run_job(job, {})
self.assertEqual(result.returncode, 0)
self.assertTrue(marker.exists())
def test_a_failed_scrape_leaves_the_original_alone(self):
with tempfile.TemporaryDirectory() as directory:
marker = Path(directory) / "ran"
job = self._job(
directory,
[sys.executable, "-c", "raise SystemExit(3)"],
((sys.executable, "-c", f"open({str(marker)!r}, 'w').close()"),),
)
result = rs.run_job(job, {})
self.assertEqual(result.returncode, 3)
self.assertFalse(marker.exists())
if __name__ == "__main__":
unittest.main()
+1 -1
View File
@@ -227,7 +227,7 @@ python scripts/generate_pack.py --all --verify-packs --output-dir dist/
``` ```
Integrated as pipeline step 6/8 (runs after consistency check, before Integrated as pipeline step 6/8 (runs after consistency check, before
README generation). Requires packs in `dist/` — skip with `--skip-packs`. README generation). Requires packs in `dist/`; skip with `--skip-packs`.
## Verification discipline ## Verification discipline
+84 -11
View File
@@ -236,14 +236,26 @@ python scripts/export_native.py --platform batocera --fetch
python scripts/export_native.py --all --fetch --output-dir dist/upstream/ python scripts/export_native.py --all --fetch --output-dir dist/upstream/
``` ```
`--fetch` downloads the platform's own file once into `--fetch` downloads the platform's own file into `.cache/upstream-native/`,
`.cache/upstream-native/`, at the revision the platform YAML's `source:` at the revision the platform YAML's `source:` names. Formats that carry
names. Formats that carry code are patched from it rather than code are patched from it rather than regenerated, and the export fails
regenerated, and the export fails without it. without it. The URL each cached file came from is recorded beside the
cache, so a platform file that moves to a new tag is fetched again instead
of patching the previous one.
A branch URL keeps its name while its content moves, and no URL comparison
can see that. `--refresh-cache` downloads every file again and writes no
export; it is what follows a rescrape, so the original and the platform
YAML describe the same moment.
```bash
python scripts/export_native.py --platform retrobat --refresh-cache
python scripts/export_native.py --all --refresh-cache
```
A list is written in the order the code looks: `priority:` first (lowest A list is written in the order the code looks: `priority:` first (lowest
wins), then the profile's own declaration order. What a format cannot wins), then the profile's own declaration order. What a format cannot
state is reported rather than written — EmuDeck's arrays are shared state is reported rather than written: EmuDeck's arrays are shared
between emulators its file does not name, so a hash is corrected in place between emulators its file does not name, so a hash is corrected in place
and never added. and never added.
@@ -326,9 +338,11 @@ Per part of a ref: `ANCHORED` (same content, same lines), `SHIFTED` (same
content, moved), `RENAMED` (the source file moved), `CHANGED` (content content, moved), `RENAMED` (the source file moved), `CHANGED` (content
edited), `AMBIGUOUS` (several equally good candidates), `GONE` (nothing edited), `AMBIGUOUS` (several equally good candidates), `GONE` (nothing
left to anchor to), `EXTERNAL` (the ref names a project the profile does left to anchor to), `EXTERNAL` (the ref names a project the profile does
not declare, so no revision can confirm it). An entry carries the worst not declare, so no revision can confirm it), `UNCHECKED` (a pin equal to
HEAD is judged on self-consistency, and an entry with neither a hash nor
a name gives that check nothing to look for). An entry carries the worst
status of its parts. Only `CHANGED`, `GONE` and `AMBIGUOUS` count as status of its parts. Only `CHANGED`, `GONE` and `AMBIGUOUS` count as
needing a re-read. needing a re-read; a zero next to a row of `unchecked` is not a proof.
A single cited line is often not distinctive, so the anchor widens by A single cited line is often not distinctive, so the anchor widens by
steps of ±3, ±6, ±12, ±25 and ±50 lines until it is unique. Three further steps of ±3, ±6, ±12, ±25 and ±50 lines until it is unique. Three further
@@ -424,9 +438,10 @@ directories the refs point at.
Writes are explicit and mechanical only. `--backfill-commits` fills a Writes are explicit and mechanical only. `--backfill-commits` fills a
missing `source_commit`, `--rebase-refs` recales `SHIFTED` and `RENAMED` missing `source_commit`, `--rebase-refs` recales `SHIFTED` and `RENAMED`
line ranges, `--bump-commit` advances `source_commit` to HEAD only when line ranges and moves `source_commit` to HEAD with them, all or nothing,
nothing needs a re-read. All three refuse to run on a dirty `emulators/` `--bump-commit` advances `source_commit` to HEAD only when nothing needs a
without `--force`, and every write is verified by reparsing the document. re-read. All three refuse to run on a dirty `emulators/` without `--force`,
and every write is verified by reparsing the document.
Uses `GITHUB_TOKEN` when set, which `--all` requires. Responses are cached Uses `GITHUB_TOKEN` when set, which `--all` requires. Responses are cached
under `.cache/upstream/`, addressed by commit sha, so `--offline` replays a under `.cache/upstream/`, addressed by commit sha, so `--offline` replays a
@@ -457,6 +472,62 @@ python scripts/refresh_data_dirs.py --platform batocera # single platform onl
python scripts/refresh_data_dirs.py --registry path/to/_data_dirs.yml python scripts/refresh_data_dirs.py --registry path/to/_data_dirs.yml
``` ```
### check_freshness.py
Ask every transcribed layer whether the local copy is the one upstream serves
today, and answer with one row per subject. Platform and target files are
re-scraped through the scrapers' own write path into a throwaway copy and
diffed against the committed file, so the diff decides rather than a version
string. Core-info is listed and compared with the profiles (a core with no
profile, a profile that calls standalone a core libretro now builds). Data
directories are checked against their upstream, dump catalogs and romset
snapshots against the catalog they were read from (Redump DAT versions, the
No-Intro daily rebuild, the newest TOSEC pack, the newest MAME release, the
git blobs of the FBNeo DATs), and the CI toolchain against PyPI and the
actions' latest releases.
The `native` area needs no network. It reads the cache `export_native.py`
patches from and reports an original that is absent, that came from another
URL than the one the platform YAML names, or that was fetched before the
platform YAML was last rewritten.
```bash
python scripts/check_freshness.py
python scripts/check_freshness.py --only platforms,targets
python scripts/check_freshness.py --json
python scripts/check_freshness.py --profiles # also runs profile_sync (slow)
python scripts/check_freshness.py --offline # local ages only
```
The exit code is non-zero when anything is stale. Emulator profiles are the
one layer left to `profile_sync.py`, which the `--profiles` flag folds in.
### refresh_stale.py
Run the refresher behind every stale row of `check_freshness.py`, several at
a time, and leave the result in the working tree for review. Nothing is
committed.
| Stale row | Command |
|-----------|---------|
| `platforms/<name>` | the platform scraper, then the `native` refresh for the same platform |
| `native/<name>` | `export_native.py --platform <name> --refresh-cache` |
| `targets/<name>` | the target scraper |
| `data/<key>` | `refresh_data_dirs.py --key <key> --force` |
| `catalogs/mame recipes`, `catalogs/fbneo recipes` | `scraper/romset_dat_importer.py --source <source> --fetch` |
Rows that need a decision are listed and never run: a buildbot core with no
profile, a new `.info` file, the Redump, No-Intro and TOSEC packs, the CI
pins, and `profile_sync`. A failed scraper leaves its cached original
untouched. Each job writes its output to `tmp/refresh_stale/`, and a GitHub
token is taken from `GITHUB_TOKEN` or from `gh auth token`.
```bash
python scripts/refresh_stale.py --dry-run
python scripts/refresh_stale.py
python scripts/refresh_stale.py --only platforms,targets --jobs 6
```
### Other tools ### Other tools
| Script | Purpose | | Script | Purpose |
@@ -474,12 +545,14 @@ python scripts/refresh_data_dirs.py --registry path/to/_data_dirs.yml
| `generate_site.py` | Generate all MkDocs site pages (this documentation) | | `generate_site.py` | Generate all MkDocs site pages (this documentation) |
| `validate_site.py` | Validate rendered metadata, headings, image alternatives, JSON-LD, local resources, links and fragments | | `validate_site.py` | Validate rendered metadata, headings, image alternatives, JSON-LD, local resources, links and fragments |
| `romset_recipes.py` | Identify which emulator version an arcade archive matches, and rebuild a pinned archive from ROMs already held | | `romset_recipes.py` | Identify which emulator version an arcade archive matches, and rebuild a pinned archive from ROMs already held |
| `scraper/romset_dat_importer.py` | Fetch and import per-set recipes from MAME `-listxml` or FBNeo DATs into `recipes/` | | `scraper/romset_dat_importer.py` | Fetch and import per-set recipes from MAME `-listxml` or FBNeo DATs into `recipes/`; `--fetch` alone takes the newest MAME release, and FBNeo DATs are refreshed by git blob sha |
| `deterministic_zip.py` | Rebuild MAME BIOS ZIPs deterministically (same ROMs = same hash) | | `deterministic_zip.py` | Rebuild MAME BIOS ZIPs deterministically (same ROMs = same hash) |
| `torrentzip.py` | Build TorrentZip archives for MAME/FBNeo ROM sets (archive bytes depend only on contents) | | `torrentzip.py` | Build TorrentZip archives for MAME/FBNeo ROM sets (archive bytes depend only on contents) |
| `crypto_verify.py` | 3DS RSA signature and AES crypto verification | | `crypto_verify.py` | 3DS RSA signature and AES crypto verification |
| `sect233r1.py` | Pure Python ECDSA verification on sect233r1 curve (3DS OTP cert) | | `sect233r1.py` | Pure Python ECDSA verification on sect233r1 curve (3DS OTP cert) |
| `check_buildbot_system.py` | Detect stale data directories by comparing with buildbot | | `check_buildbot_system.py` | Detect stale data directories by comparing with buildbot |
| `check_freshness.py` | One report on every upstream the repository transcribes (see above) |
| `refresh_stale.py` | Run the refresher behind every stale row of that report (see above) |
| `migrate.py` | Migrate flat bios structure to Manufacturer/Console/ hierarchy | | `migrate.py` | Migrate flat bios structure to Manufacturer/Console/ hierarchy |
## Installation tools ## Installation tools