From 8a2a9c6ffa4f22c558eced21ae25e7f845309225 Mon Sep 17 00:00:00 2001 From: Abdessamad Derraz <3028866+Abdess@users.noreply.github.com> Date: Sun, 4 Oct 2026 14:54:11 +0200 Subject: [PATCH] feat: add freshness tooling and override defaults --- platforms/targets/_overrides.yml | 43 ++ scripts/check_freshness.py | 1003 ++++++++++++++++++++++++++++++ scripts/common.py | 30 + scripts/refresh_stale.py | 430 +++++++++++++ tests/test_check_freshness.py | 218 +++++++ 5 files changed, 1724 insertions(+) create mode 100644 scripts/check_freshness.py create mode 100755 scripts/refresh_stale.py create mode 100644 tests/test_check_freshness.py diff --git a/platforms/targets/_overrides.yml b/platforms/targets/_overrides.yml index 409a16b6..e16d2264 100644 --- a/platforms/targets/_overrides.yml +++ b/platforms/targets/_overrides.yml @@ -5,6 +5,8 @@ # Format: # platform_name: # targets: +# _default: # applies to every target +# remove_cores: [...] # target-name: # aliases: [alias1, alias2] # add_cores: [core_to_add] @@ -15,6 +17,27 @@ # doesn't resolve. These are confirmed available on x86_64 via Config.in. batocera: targets: + _default: + # Amiga machine variants (A500, CD32, etc.) are selectors passed to + # fsuae/amiberry, not emulators of their own; batocera-es-system + # groups them under the emulator's node in es_systems.yml. + # odcommander is OpenDingux's file manager; flatpak is a runtime + # launcher. Neither emulates anything. + remove_cores: + - A500 + - "A500+" + - A600 + - A1000 + - A1200 + - A3000 + - A4000 + - CD32 + - CDTV + - flatpak + - odcommander + - sh + - steam + - wine-tkg x86_64: add_cores: [citron, demul, model2, xenia] zen3: @@ -37,3 +60,23 @@ retroarch: aliases: [ps2] playstation-psp: aliases: [psp] + +retrodeck: + targets: + _default: + # RetroDECK ships RetroArch itself as a component; its target list + # includes the frontend name alongside the actual cores. + remove_cores: [retroarch] + +emudeck: + targets: + _default: + # EmuDeck's EmuScripts tree carries a template.sh placeholder used + # to scaffold new installers; it is not an emulator. + remove_cores: [template] + +retropie: + targets: + _default: + # Homebrew game scriptmodule, not an emulator. + remove_cores: [superflappybirds] diff --git a/scripts/check_freshness.py b/scripts/check_freshness.py new file mode 100644 index 00000000..072dd63f --- /dev/null +++ b/scripts/check_freshness.py @@ -0,0 +1,1003 @@ +#!/usr/bin/env python3 +"""Confront every transcribed layer with what its upstream serves today. + +Each layer the repository transcribes moves on its own schedule: platform +lists, hardware targets, core-info, buildbot system assets, dump catalogs, +romset DATs, the CI toolchain. This script asks each of them the same +question, is the local copy the one upstream serves, and answers with one +row per subject. + +Platform and target files are re-scraped through the scrapers' own write path +into a throwaway copy, then diffed against the committed file. The diff +decides, not a version string: a scraper that pins a new tag but produces +the same entries is reported as a version change and nothing else. + +Emulator profiles are covered by profile_sync, which is slow (one API pass +per profile). ``--profiles`` runs it here and folds its verdict in. + +Usage: + python scripts/check_freshness.py + python scripts/check_freshness.py --only platforms,targets + python scripts/check_freshness.py --json + python scripts/check_freshness.py --profiles + python scripts/check_freshness.py --offline # local age checks only +""" + +from __future__ import annotations + +import argparse +import contextlib +import hashlib +import io +import json +import re +import shutil +import subprocess +import sys +import tempfile +from concurrent.futures import ThreadPoolExecutor +from dataclasses import asdict, dataclass, field +from datetime import datetime, timezone +from pathlib import Path + +sys.path.insert(0, str(Path(__file__).resolve().parent)) +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) + +import refresh_data_dirs # noqa: E402 +import upstream # noqa: E402 +from common import ( # noqa: E402 + list_registered_platforms, + load_emulator_profiles, + yaml_load, +) +from scripts.scraper.redump_dat_scraper import fetch_snapshot # noqa: E402 +from scripts.scraper.targets import discover_target_scrapers # noqa: E402 + +REPO_ROOT = Path(__file__).resolve().parents[1] +CORE_INFO_REPO = "https://github.com/libretro/libretro-core-info" +NO_INTRO_MIRROR = "https://github.com/hugo19941994/auto-datfile-generator" +NO_INTRO_RELEASE = "Daily_Rebuild" +NO_INTRO_ASSET = "no-intro.zip" +TOSEC_DOWNLOADS = "https://www.tosecdev.org/downloads" +MAME_REPO = "https://github.com/mamedev/mame" +FBNEO_REPO = "https://github.com/libretro/FBNeo" +PYPI_URL = "https://pypi.org/pypi/{name}/json" +INSTALLERS = ("install.sh", "install.ps1") + +SHOWN_ITEMS = 4 # items named in a diff summary before ", ..." +SHOWN_NAMES = 12 # core names named in an unresolved row before ", ..." +AREAS = ("platforms", "targets", "coreinfo", "data", "catalogs", "ci", "profiles") +OK, STALE, UNKNOWN, ERROR, SKIPPED = "OK", "STALE", "UNKNOWN", "ERROR", "SKIPPED" +# Every status a finding may carry, in the order the summary counts them. +STATUSES = (STALE, UNKNOWN, ERROR, SKIPPED, OK) + + +@dataclass(frozen=True) +class Finding: + area: str + subject: str + status: str + local: str = "" + remote: str = "" + detail: str = "" + + +@dataclass +class PlatformDiff: + """What a fresh scrape changes in a platform or target file.""" + + version: tuple[str, str] | None = None + systems_added: list[str] = field(default_factory=list) + systems_removed: list[str] = field(default_factory=list) + files_added: list[str] = field(default_factory=list) + files_removed: list[str] = field(default_factory=list) + files_changed: list[str] = field(default_factory=list) + cores_added: list[str] = field(default_factory=list) + cores_removed: list[str] = field(default_factory=list) + + @property + def content_changed(self) -> bool: + return any( + ( + self.systems_added, + self.systems_removed, + self.files_added, + self.files_removed, + self.files_changed, + self.cores_added, + self.cores_removed, + ) + ) + + @property + def changed(self) -> bool: + return self.content_changed or self.version is not None + + def summary(self) -> str: + parts: list[str] = [] + if self.version: + parts.append(f"version {self.version[0]} -> {self.version[1]}") + for label, items in ( + ("+systems", self.systems_added), + ("-systems", self.systems_removed), + ("+files", self.files_added), + ("-files", self.files_removed), + ("~files", self.files_changed), + ("+cores", self.cores_added), + ("-cores", self.cores_removed), + ): + if items: + shown = ", ".join(items[:SHOWN_ITEMS]) + (", ..." if len(items) > SHOWN_ITEMS else "") + parts.append(f"{label} {len(items)} ({shown})") + return "; ".join(parts) if parts else "identical" + + +# --- platforms -------------------------------------------------------------- + + +def _file_key(entry: dict) -> str: + return str(entry.get("destination") or entry.get("name") or "") + + +def _core_names(config: dict) -> set[str]: + cores = config.get("cores") + names = set(map(str, cores)) if isinstance(cores, list) else set() + standalone = config.get("standalone_cores") + if isinstance(standalone, list): + names.update(f"standalone:{c}" for c in standalone) + return names + + +def diff_platform(old: dict, new: dict) -> PlatformDiff: + """Compare two platform files the way a reviewer reads the diff. + + Files are keyed by system and destination, so a hash or flag change on + an existing destination is a change, not a removal plus an addition. + """ + diff = PlatformDiff() + old_version, new_version = str(old.get("version", "")), str(new.get("version", "")) + if old_version != new_version: + diff.version = (old_version, new_version) + old_systems = old.get("systems") or {} + new_systems = new.get("systems") or {} + diff.systems_added = sorted(set(new_systems) - set(old_systems)) + diff.systems_removed = sorted(set(old_systems) - set(new_systems)) + for system in sorted(set(old_systems) & set(new_systems)): + old_files = { + _file_key(f): f for f in (old_systems[system] or {}).get("files") or [] + } + new_files = { + _file_key(f): f for f in (new_systems[system] or {}).get("files") or [] + } + diff.files_added.extend( + f"{system}/{k}" for k in sorted(set(new_files) - set(old_files)) + ) + diff.files_removed.extend( + f"{system}/{k}" for k in sorted(set(old_files) - set(new_files)) + ) + diff.files_changed.extend( + f"{system}/{k}" + for k in sorted(set(old_files) & set(new_files)) + if old_files[k] != new_files[k] + ) + old_cores, new_cores = _core_names(old), _core_names(new) + diff.cores_added = sorted(new_cores - old_cores) + diff.cores_removed = sorted(old_cores - new_cores) + return diff + + +def diff_targets(old: dict, new: dict) -> PlatformDiff: + """Compare two target files. The scrape stamp is not a change.""" + diff = PlatformDiff() + old_targets = old.get("targets") or {} + new_targets = new.get("targets") or {} + diff.systems_added = sorted(set(new_targets) - set(old_targets)) + diff.systems_removed = sorted(set(old_targets) - set(new_targets)) + for name in sorted(set(old_targets) & set(new_targets)): + before = set(map(str, (old_targets[name] or {}).get("cores") or [])) + after = set(map(str, (new_targets[name] or {}).get("cores") or [])) + diff.cores_added.extend(f"{name}/{c}" for c in sorted(after - before)) + diff.cores_removed.extend(f"{name}/{c}" for c in sorted(before - after)) + return diff + + +def _scrape_into_copy( + module: str, source: Path, workdir: Path, timeout: int = 900 +) -> tuple[Path, str | None]: + """Run a scraper's own write path on a throwaway copy of *source*. + + The CLI merges into the file it writes, so the copy carries every field + the scraper preserves and the diff shows only what the scrape changes. + """ + target = workdir / source.name + shutil.copyfile(source, target) + proc = subprocess.run( + [sys.executable, "-m", module, "-o", str(target)], + cwd=REPO_ROOT, + capture_output=True, + text=True, + timeout=timeout, + check=False, + ) + if proc.returncode != 0: + tail = (proc.stderr or proc.stdout).strip().splitlines()[-1:] + return target, tail[0] if tail else f"exit {proc.returncode}" + return target, None + + +def _scrapable_platforms(platforms_dir: Path) -> list[tuple[str, str, Path]]: + """(platform, scraper module, file) for every registered platform. + + A platform that only inherits another and declares no source of its own + (Lakka) has nothing to scrape: its scraper writes the parent's file. + """ + with (platforms_dir / "_registry.yml").open(encoding="utf-8") as fh: + registry = (yaml_load(fh) or {}).get("platforms") or {} + rows = [] + for name in list_registered_platforms(str(platforms_dir), include_archived=True): + path = platforms_dir / f"{name}.yml" + if not path.is_file(): + continue + with path.open(encoding="utf-8") as fh: + config = yaml_load(fh) or {} + scraper = (registry.get(name) or {}).get("scraper") + if not scraper or (config.get("inherits") and not config.get("source")): + continue + rows.append((name, f"scripts.scraper.{scraper}_scraper", path)) + return rows + + +def check_platforms(platforms_dir: Path, workdir: Path, jobs: int) -> list[Finding]: + rows = _scrapable_platforms(platforms_dir) + findings: list[Finding] = [] + + def run(row: tuple[str, str, Path]) -> Finding: + name, module, path = row + try: + fresh, error = _scrape_into_copy(module, path, workdir) + except subprocess.TimeoutExpired: + return Finding("platforms", name, ERROR, detail="scraper timed out") + if error: + return Finding("platforms", name, ERROR, detail=error) + with path.open(encoding="utf-8") as fh: + old = yaml_load(fh) or {} + with fresh.open(encoding="utf-8") as fh: + new = yaml_load(fh) or {} + diff = diff_platform(old, new) + status = STALE if diff.changed else OK + return Finding( + "platforms", + name, + status, + local=str(old.get("version", "")), + remote=str(new.get("version", "")), + detail=diff.summary(), + ) + + with ThreadPoolExecutor(max_workers=jobs) as pool: + findings.extend(pool.map(run, rows)) + return sorted(findings, key=lambda f: f.subject) + + +# --- targets ---------------------------------------------------------------- + + +def profile_name_index(profiles: dict[str, dict]) -> dict[str, str]: + """Upstream core name -> profile key, the index target filtering uses.""" + index: dict[str, str] = {} + for key, profile in profiles.items(): + index.setdefault(key, key) + for alias in profile.get("cores") or []: + index.setdefault(str(alias), key) + return index + + +def _removed_cores(overrides: dict, platform: str) -> dict[str, set[str]]: + targets = ((overrides.get(platform) or {}).get("targets")) or {} + default = set(map(str, (targets.get("_default") or {}).get("remove_cores") or [])) + return { + name: default | set(map(str, (entry or {}).get("remove_cores") or [])) + for name, entry in targets.items() + if name != "_default" + } | {"_default": default} + + +def unresolved_target_cores( + targets: dict, index: dict[str, str], removed: dict[str, set[str]] +) -> dict[str, list[str]]: + """Core names a target lists that no profile claims and no override drops.""" + unresolved: dict[str, list[str]] = {} + default_drop = removed.get("_default", set()) + for name, entry in (targets.get("targets") or {}).items(): + dropped = removed.get(name, default_drop) | default_drop + for core in map(str, (entry or {}).get("cores") or []): + if core not in index and core not in dropped: + unresolved.setdefault(core, []).append(name) + return {core: sorted(names) for core, names in sorted(unresolved.items())} + + +def _target_scrapers() -> dict[str, str]: + """platform -> scraper module for every target scraper on disk.""" + return { + platform: cls.__module__ + for platform, cls in discover_target_scrapers().items() + } + + +def check_targets( + platforms_dir: Path, workdir: Path, profiles: dict[str, dict], jobs: int, + offline: bool, +) -> list[Finding]: + targets_dir = platforms_dir / "targets" + overrides_path = targets_dir / "_overrides.yml" + overrides = {} + if overrides_path.is_file(): + with overrides_path.open(encoding="utf-8") as fh: + overrides = yaml_load(fh) or {} + index = profile_name_index(profiles) + scrapers = {} if offline else _target_scrapers() + findings: list[Finding] = [] + today = datetime.now(timezone.utc) + + def run(platform: str) -> list[Finding]: + path = targets_dir / f"{platform}.yml" + if not path.is_file(): + return [Finding("targets", platform, ERROR, detail="no target file")] + with path.open(encoding="utf-8") as fh: + old = yaml_load(fh) or {} + module = scrapers.get(platform) + if module is None: + # No scraper: a hand-written file has no upstream to diff against, + # so only its age and its core names can be reported. + stamp = str(old.get("scraped_at", ""))[:10] + age = _age_days(stamp, today) + out = [ + Finding( + "targets", + platform, + OK if age is not None else UNKNOWN, + local=f"static {stamp}", + detail=f"{age} days old" if age is not None else "no scrape stamp", + ) + ] + missing = unresolved_target_cores(old, index, _removed_cores(overrides, platform)) + out.extend(_unresolved_findings(platform, missing)) + return out + try: + fresh, error = _scrape_into_copy(module, path, workdir) + except subprocess.TimeoutExpired: + return [Finding("targets", platform, ERROR, detail="scraper timed out")] + if error: + return [Finding("targets", platform, ERROR, detail=error)] + with fresh.open(encoding="utf-8") as fh: + new = yaml_load(fh) or {} + diff = diff_targets(old, new) + out = [ + Finding( + "targets", + platform, + STALE if diff.content_changed else OK, + local=str(old.get("scraped_at", ""))[:10], + remote=str(new.get("scraped_at", ""))[:10], + detail=diff.summary(), + ) + ] + missing = unresolved_target_cores(new, index, _removed_cores(overrides, platform)) + out.extend(_unresolved_findings(platform, missing)) + return out + + names = sorted(p.stem for p in targets_dir.glob("*.yml") if not p.name.startswith("_")) + with ThreadPoolExecutor(max_workers=jobs) as pool: + for rows in pool.map(run, names): + findings.extend(rows) + return findings + + +def _unresolved_findings(platform: str, missing: dict[str, list[str]]) -> list[Finding]: + """One row per platform: the names its targets list that no profile claims. + + A buildbot name without a profile is a core the target filter cannot + see; a platform's own emulator id without one is an emulator nobody has + profiled yet. Both are actionable, so both are listed, but as one row so + the backlog does not bury the day's changes. + """ + if not missing: + return [] + names = sorted(missing) + shown = ", ".join(names[:SHOWN_NAMES]) + (", ..." if len(names) > SHOWN_NAMES else "") + return [ + Finding( + "targets", + f"{platform} cores", + STALE, + local=f"{len(names)} without a profile", + detail=f"cores: alias or _overrides remove_cores needed for {shown}", + ) + ] + + +def _age_days(stamp: str, now: datetime) -> int | None: + try: + then = datetime.strptime(stamp[:10], "%Y-%m-%d").replace(tzinfo=timezone.utc) + except ValueError: + return None + return (now - then).days + + +# --- core-info -------------------------------------------------------------- + + +def core_info_names(cache_dir: str, offline: bool) -> list[str] | None: + """Every core the libretro core-info repository describes.""" + repo = upstream.parse_repo(CORE_INFO_REPO) + if repo is None: + return None + payload = upstream._api( # noqa: SLF001 + f"{repo.api_base}/repos/{repo.slug}/contents?per_page=100", + cache_dir, + offline, + upstream.MOVING_TTL, + ) + if not isinstance(payload, list): + return None + names = [] + for entry in payload: + name = entry.get("name", "") if isinstance(entry, dict) else "" + if name.endswith("_libretro.info"): + names.append(name[: -len("_libretro.info")]) + return sorted(names) + + +def coreinfo_gaps( + names: list[str], profiles: dict[str, dict] +) -> tuple[list[str], list[tuple[str, str]]]: + """Cores core-info knows that the profiles do not, and profiles that + call standalone a core libretro now builds. + + Matching folds case: the buildbot serves lower-case names while a few + .info files keep the project's own casing. + """ + index = {k.casefold(): v for k, v in profile_name_index(profiles).items()} + unprofiled = [n for n in names if n.casefold() not in index] + standalone = [ + (n, index[n.casefold()]) + for n in names + if n.casefold() in index + and str(profiles[index[n.casefold()]].get("type", "")).strip() == "standalone" + ] + return unprofiled, standalone + + +def check_coreinfo(profiles: dict[str, dict], cache_dir: str, offline: bool) -> list[Finding]: + names = core_info_names(cache_dir, offline) + if names is None: + return [Finding("coreinfo", "libretro-core-info", UNKNOWN, detail="listing unavailable")] + unprofiled, standalone = coreinfo_gaps(names, profiles) + findings = [ + Finding( + "coreinfo", + "libretro-core-info", + STALE if unprofiled or standalone else OK, + local=f"{len(names) - len(unprofiled)} profiled", + remote=f"{len(names)} .info files", + ) + ] + findings.extend( + Finding("coreinfo", name, STALE, detail="no profile claims this core name") + for name in unprofiled + ) + findings.extend( + Finding( + "coreinfo", + name, + STALE, + local=f"{key}: type standalone", + detail="core-info describes a libretro build of it", + ) + for name, key in standalone + ) + return findings + + +# --- data directories ------------------------------------------------------- + + +def check_data(offline: bool) -> list[Finding]: + if offline: + return [Finding("data", "_data_dirs.yml", SKIPPED, detail="offline")] + registry = refresh_data_dirs.load_registry(str(REPO_ROOT / "platforms" / "_data_dirs.yml")) + sink = io.StringIO() + with contextlib.redirect_stdout(sink): + results = refresh_data_dirs.refresh_all(registry, dry_run=True) + findings = [] + for key, result in sorted(results.items()): + if result is None: + findings.append(Finding("data", key, UNKNOWN, detail="remote unreachable")) + elif result: + findings.append(Finding("data", key, STALE, detail="upstream changed, refresh_data_dirs.py fetches it")) + else: + findings.append(Finding("data", key, OK)) + return findings + + +# --- catalogs and recipes --------------------------------------------------- + + +def _snapshot(path: Path) -> dict: + if not path.is_file(): + return {} + with path.open(encoding="utf-8") as fh: + return json.load(fh) + + +def check_catalogs(cache_dir: str, offline: bool) -> list[Finding]: + findings = [] + findings.append(_check_redump(offline)) + findings.append(_check_no_intro(cache_dir, offline)) + findings.append(_check_tosec(offline)) + findings.append(_check_mame(cache_dir, offline)) + findings.append(_check_fbneo(cache_dir, offline)) + return findings + + +def _check_redump(offline: bool) -> Finding: + local = _snapshot(REPO_ROOT / "provenance" / "redump.json") + dats = local.get("dats") or {} + stamp = ", ".join(sorted(set(dats.values()))) or "none" + if offline: + return Finding("catalogs", "redump", SKIPPED, local=stamp, detail="offline") + try: + remote_dats, _ = fetch_snapshot() + except (ConnectionError, ValueError) as exc: + return Finding("catalogs", "redump", UNKNOWN, local=stamp, detail=str(exc)) + remote_stamp = ", ".join(sorted(set(remote_dats.values()))) + changed = sorted( + name for name in set(dats) | set(remote_dats) if dats.get(name) != remote_dats.get(name) + ) + return Finding( + "catalogs", + "redump", + STALE if changed else OK, + local=stamp, + remote=remote_stamp, + detail=("changed: " + ", ".join(changed)) if changed else f"{len(dats)} DATs", + ) + + +def _check_no_intro(cache_dir: str, offline: bool) -> Finding: + local = _snapshot(REPO_ROOT / "provenance" / "no-intro.json") + imported = str(local.get("imported_at", "")) + if offline: + return Finding("catalogs", "no-intro", SKIPPED, local=imported, detail="offline") + repo = upstream.parse_repo(NO_INTRO_MIRROR) + payload = upstream._api( # noqa: SLF001 + f"{repo.api_base}/repos/{repo.slug}/releases/tags/{NO_INTRO_RELEASE}", + cache_dir, + offline, + upstream.MOVING_TTL, + ) + assets = payload.get("assets") if isinstance(payload, dict) else None + updated = "" + for asset in assets or []: + if isinstance(asset, dict) and asset.get("name") == NO_INTRO_ASSET: + updated = str(asset.get("updated_at", ""))[:10] + if not updated: + return Finding("catalogs", "no-intro", UNKNOWN, local=imported, detail="asset not listed") + return Finding( + "catalogs", + "no-intro", + STALE if updated > imported else OK, + local=imported, + remote=updated, + detail=f"{NO_INTRO_ASSET} rebuilt {updated}", + ) + + +def tosec_latest_pack(html: str) -> str | None: + """Date of the newest TOSEC release the downloads page lists.""" + dates = re.findall(r'/downloads/category/\d+-(\d{4}-\d{2}-\d{2})"', html) + return max(dates) if dates else None + + +def _check_tosec(offline: bool) -> Finding: + local = _snapshot(REPO_ROOT / "provenance" / "tosec.json") + imported = str(local.get("imported_at", "")) + if offline: + return Finding("catalogs", "tosec", SKIPPED, local=imported, detail="offline") + try: + html = upstream._http_text(TOSEC_DOWNLOADS) # noqa: SLF001 + except upstream.UpstreamError as exc: + return Finding("catalogs", "tosec", UNKNOWN, local=imported, detail=str(exc)) + latest = tosec_latest_pack(html or "") + if latest is None: + return Finding("catalogs", "tosec", UNKNOWN, local=imported, detail="no release listed") + return Finding( + "catalogs", + "tosec", + STALE if latest > imported else OK, + local=imported, + remote=latest, + detail=f"newest pack TOSEC-v{latest}", + ) + + +def mame_versions(recipes: dict) -> list[str]: + """MAME version tags a recipe snapshot has absorbed, oldest first.""" + tags = set() + for label in (recipes.get("dats") or {}): + match = re.search(r"\b(mame\d{4})\b", str(label)) + if match: + tags.add(match.group(1)) + return sorted(tags) + + +def _check_mame(cache_dir: str, offline: bool) -> Finding: + local = _snapshot(REPO_ROOT / "recipes" / "mame.json") + versions = mame_versions(local) + newest = versions[-1] if versions else "none" + if offline: + return Finding("catalogs", "mame recipes", SKIPPED, local=newest, detail="offline") + release = upstream.latest_release(upstream.parse_repo(MAME_REPO), cache_dir, offline) + if release is None: + return Finding("catalogs", "mame recipes", UNKNOWN, local=newest, detail="no release seen") + return Finding( + "catalogs", + "mame recipes", + OK if release.tag in versions else STALE, + local=newest, + remote=f"{release.tag} ({release.date})", + detail=f"{len(versions)} versions imported", + ) + + +def fbneo_blob_drift(snapshot: dict, listing: object) -> tuple[list[str], bool]: + """DAT files whose git blob differs from the one the snapshot was read + from. The second value says whether the snapshot records blobs at all: + one written before that field existed can only be compared by date. + """ + recorded = (snapshot.get("upstream") or {}).get("blobs") or {} + if not recorded: + return [], False + remote = { + item["name"]: item.get("sha", "") + for item in (listing if isinstance(listing, list) else []) + if isinstance(item, dict) + and item.get("type") == "file" + and str(item.get("name", "")).lower().endswith(".dat") + } + drifted = sorted( + name for name in set(recorded) | set(remote) if recorded.get(name) != remote.get(name) + ) + return drifted, True + + +def _check_fbneo(cache_dir: str, offline: bool) -> Finding: + local = _snapshot(REPO_ROOT / "recipes" / "fbneo.json") + imported = str(local.get("imported_at", "")) + if offline: + return Finding("catalogs", "fbneo recipes", SKIPPED, local=imported, detail="offline") + repo = upstream.parse_repo(FBNEO_REPO) + listing = upstream._api( # noqa: SLF001 + f"{repo.api_base}/repos/{repo.slug}/contents/dats", + cache_dir, + offline, + upstream.MOVING_TTL, + ) + if not isinstance(listing, list): + return Finding("catalogs", "fbneo recipes", UNKNOWN, local=imported, detail="dats/ listing unavailable") + drifted, tracked = fbneo_blob_drift(local, listing) + if not tracked: + return Finding( + "catalogs", + "fbneo recipes", + STALE, + local=imported, + detail="snapshot records no upstream blobs: re-import with --fetch", + ) + shown = ", ".join(n.removeprefix("FinalBurn Neo (ClrMame Pro XML, ").removesuffix(" only).dat") for n in drifted[:4]) + return Finding( + "catalogs", + "fbneo recipes", + STALE if drifted else OK, + local=imported, + remote=f"{len(listing)} DATs listed", + detail=(f"{len(drifted)} DAT(s) changed upstream: {shown}" if drifted else "every DAT blob matches"), + ) + + +# --- CI toolchain ----------------------------------------------------------- + +_PIN_RE = re.compile(r'"?([A-Za-z0-9_.\-]+)((?:[<>=!~]=?[^,"\s]+)(?:,[<>=!~]=?[^,"\s]+)*)?"?') +_USES_RE = re.compile(r"uses:\s*([\w.\-]+/[\w.\-]+)@([0-9a-f]{40})\s*(?:#\s*(\S+))?") + + +def parse_pip_pins(text: str) -> dict[str, str]: + """Package -> specifier for every ``pip install`` line of a workflow.""" + pins: dict[str, str] = {} + for line in text.splitlines(): + stripped = line.strip() + if "pip install" not in stripped: + continue + args = stripped.split("pip install", 1)[1] + for raw in re.findall(r'"[^"]+"|\S+', args): + token = raw.strip('"') + if token.startswith("-") or not token: + continue + match = _PIN_RE.fullmatch(token) + if match: + pins.setdefault(match.group(1).lower(), match.group(2) or "") + return pins + + +def parse_action_pins(text: str) -> dict[str, tuple[str, str]]: + """Action -> (pinned sha, commented tag) for every ``uses:`` line.""" + return { + repo: (sha, tag or "") for repo, sha, tag in _USES_RE.findall(text) + } + + +def _version_tuple(version: str) -> tuple[int, ...]: + return tuple(int(p) for p in re.findall(r"\d+", version)) + + +_SPEC_CLAUSE = re.compile(r"([<>=!~]=?)\s*(.+)") +_SPEC_TESTS = { + "==": lambda target, bound: target == bound, + "!=": lambda target, bound: target != bound, + ">=": lambda target, bound: target >= bound, + ">": lambda target, bound: target > bound, + "<=": lambda target, bound: target <= bound, + "<": lambda target, bound: target < bound, +} + + +def specifier_allows(spec: str, version: str) -> bool: + """Whether a pip specifier admits a version. Pre-releases are compared + on their numeric parts, which is enough for the operators pip uses here.""" + target = _version_tuple(version) + for clause in filter(None, (c.strip() for c in spec.split(","))): + match = _SPEC_CLAUSE.fullmatch(clause) + test = _SPEC_TESTS.get(match.group(1)) if match else None + if test is None or not test(target, _version_tuple(match.group(2))): + return False + return True + + +def pypi_latest(name: str) -> str | None: + payload = upstream._http_json(PYPI_URL.format(name=name)) # noqa: SLF001 + if not isinstance(payload, dict): + return None + info = payload.get("info") or {} + return str(info.get("version") or "") or None + + +def check_ci(cache_dir: str, offline: bool) -> list[Finding]: + findings = [_check_installer_pins()] + workflows = sorted((REPO_ROOT / ".github" / "workflows").glob("*.yml")) + pins: dict[str, str] = {} + actions: dict[str, tuple[str, str]] = {} + for path in workflows: + text = path.read_text(encoding="utf-8") + for name, spec in parse_pip_pins(text).items(): + pins.setdefault(name, spec) + actions.update(parse_action_pins(text)) + if offline: + findings.append(Finding("ci", "workflow pins", SKIPPED, detail="offline")) + return findings + for name, spec in sorted(pins.items()): + try: + latest = pypi_latest(name) + except upstream.UpstreamError as exc: + findings.append(Finding("ci", f"pypi:{name}", UNKNOWN, local=spec, detail=str(exc))) + continue + if latest is None: + findings.append(Finding("ci", f"pypi:{name}", UNKNOWN, local=spec, detail="not on PyPI")) + continue + allowed = specifier_allows(spec, latest) + findings.append( + Finding( + "ci", + f"pypi:{name}", + OK if allowed else STALE, + local=spec or "unpinned", + remote=latest, + detail="latest admitted" if allowed else "latest release outside the pin", + ) + ) + for action, (sha, tag) in sorted(actions.items()): + repo = upstream.parse_repo(f"https://github.com/{action}") + release = upstream.latest_release(repo, cache_dir, offline) + if release is None: + findings.append(Finding("ci", f"action:{action}", UNKNOWN, local=f"{sha[:8]} {tag}", detail="no release seen")) + continue + latest_sha = upstream.tag_commit(repo, release.tag, cache_dir, offline) or "" + findings.append( + Finding( + "ci", + f"action:{action}", + OK if latest_sha == sha else STALE, + local=f"{sha[:8]} {tag}".strip(), + remote=f"{latest_sha[:8]} {release.tag}", + detail="" if latest_sha == sha else f"{release.tag} released {release.date}", + ) + ) + return findings + + +def _check_installer_pins() -> Finding: + """The bootstraps pin install.py by SHA-256; a stale pin refuses every install.""" + digest = hashlib.sha256((REPO_ROOT / "install.py").read_bytes()).hexdigest() + stale = [] + for name in INSTALLERS: + text = (REPO_ROOT / name).read_text(encoding="utf-8", errors="replace") + if digest not in text: + stale.append(name) + return Finding( + "ci", + "install.py pin", + STALE if stale else OK, + local=digest[:12], + detail=("pin outdated in " + ", ".join(stale)) if stale else "install.sh and install.ps1 match", + ) + + +# --- profiles --------------------------------------------------------------- + + +def check_profiles(run: bool, offline: bool) -> list[Finding]: + if not run: + return [ + Finding( + "profiles", + "profile_sync", + SKIPPED, + detail="pass --profiles, or run profile_sync.py --all --triage", + ) + ] + cmd = [ + sys.executable, + str(REPO_ROOT / "scripts" / "profile_sync.py"), + "--all", + "--json", + "--check-version", + "--detect-new-files", + "--watch-hashes", + ] + if offline: + cmd.append("--offline") + proc = subprocess.run(cmd, cwd=REPO_ROOT, capture_output=True, text=True, check=False) + if proc.returncode not in (0, 1): + tail = proc.stderr.strip().splitlines()[-1:] + return [Finding("profiles", "profile_sync", ERROR, detail=tail[0] if tail else f"exit {proc.returncode}")] + try: + report = json.loads(proc.stdout) + except json.JSONDecodeError as exc: + return [Finding("profiles", "profile_sync", ERROR, detail=f"unreadable report: {exc}")] + return summarize_profile_sync(report) + + +def summarize_profile_sync(report: object) -> list[Finding]: + """One row per profile with something to look at, from profile_sync's JSON. + + The JSON carries the ref verdicts and the pin against the tip; a profile + whose refs need review is stale, one the forge could not serve is unknown, + and one whose upstream moved while every ref still anchors is reported as + such without being counted as stale. + """ + if not isinstance(report, list): + return [Finding("profiles", "profile_sync", ERROR, detail="report is not a list")] + findings = [] + clean = moved = 0 + for row in report: + if not isinstance(row, dict): + continue + name = str(row.get("name") or "?") + if row.get("skipped"): + findings.append(Finding("profiles", name, UNKNOWN, detail=str(row["skipped"]))) + continue + review = row.get("needs_review") or 0 + pin, head = str(row.get("pin") or ""), str(row.get("head") or "") + if review: + findings.append( + Finding( + "profiles", + name, + STALE, + local=pin[:8], + remote=head[:8], + detail=f"{review} refs to review", + ) + ) + elif pin and head and pin != head: + moved += 1 + else: + clean += 1 + stale = sum(1 for f in findings if f.status == STALE) + findings.insert( + 0, + Finding( + "profiles", + "profile_sync", + STALE if stale else OK, + local=f"{clean} at tip, {moved} anchored past the tip", + remote=f"{stale} to review", + ), + ) + return findings + + +# --- reporting -------------------------------------------------------------- + + +def render(findings: list[Finding]) -> str: + lines = [] + width = max((len(f.subject) for f in findings), default=10) + width = min(width, 44) + for area in AREAS: + rows = [f for f in findings if f.area == area] + if not rows: + continue + lines.append(f"\n[{area}]") + for f in rows: + cols = f" {f.status:7} {f.subject[:width]:{width}}" + if f.local or f.remote: + cols += f" {f.local[:32]:32} -> {f.remote[:32]}" + if f.detail: + cols += f" {f.detail}" + lines.append(cols.rstrip()) + counts = {status: sum(1 for f in findings if f.status == status) for status in STATUSES} + summary = ", ".join(f"{n} {status.lower()}" for status, n in counts.items() if n) + lines.append(f"\nFRESHNESS: {summary}") + return "\n".join(lines) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) + parser.add_argument("--only", help="comma-separated areas: " + ",".join(AREAS)) + parser.add_argument("--json", action="store_true", help="machine-readable output") + parser.add_argument("--profiles", action="store_true", help="also run profile_sync (slow)") + parser.add_argument("--offline", action="store_true", help="no network, local ages only") + parser.add_argument("--jobs", type=int, default=4, help="parallel scrapers") + parser.add_argument("--cache-dir", default=".cache/upstream") + parser.add_argument("--platforms-dir", default="platforms") + parser.add_argument("--emulators-dir", default="emulators") + args = parser.parse_args() + + areas = tuple(a.strip() for a in args.only.split(",")) if args.only else AREAS + unknown = sorted(set(areas) - set(AREAS)) + if unknown: + parser.error(f"unknown area: {', '.join(unknown)}") + + platforms_dir = Path(args.platforms_dir) + profiles = load_emulator_profiles(args.emulators_dir, skip_aliases=False) + findings: list[Finding] = [] + with tempfile.TemporaryDirectory(prefix="freshness-", dir=REPO_ROOT / "tmp" if (REPO_ROOT / "tmp").is_dir() else None) as scratch: + workdir = Path(scratch) + if "platforms" in areas: + findings.extend( + [Finding("platforms", "scrapers", SKIPPED, detail="offline")] + if args.offline + else check_platforms(platforms_dir, workdir, args.jobs) + ) + if "targets" in areas: + findings.extend(check_targets(platforms_dir, workdir, profiles, args.jobs, args.offline)) + if "coreinfo" in areas: + findings.extend(check_coreinfo(profiles, args.cache_dir, args.offline)) + if "data" in areas: + findings.extend(check_data(args.offline)) + if "catalogs" in areas: + findings.extend(check_catalogs(args.cache_dir, args.offline)) + if "ci" in areas: + findings.extend(check_ci(args.cache_dir, args.offline)) + if "profiles" in areas: + findings.extend(check_profiles(args.profiles, args.offline)) + + if args.json: + print(json.dumps([asdict(f) for f in findings], indent=2)) + else: + print(render(findings)) + return 1 if any(f.status in (STALE, ERROR) for f in findings) else 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/common.py b/scripts/common.py index 6710b22e..53376652 100644 --- a/scripts/common.py +++ b/scripts/common.py @@ -276,7 +276,12 @@ def load_target_config( cores = set(str(c) for c in targets[canonical].get("cores", [])) + default_ovr = overrides.get("_default", {}) ovr = overrides.get(canonical, {}) + for c in default_ovr.get("add_cores", []): + cores.add(str(c)) + for c in default_ovr.get("remove_cores", []): + cores.discard(str(c)) for c in ovr.get("add_cores", []): cores.add(str(c)) for c in ovr.get("remove_cores", []): @@ -1096,6 +1101,29 @@ def group_identical_platforms( return result +def runs_standalone( + emu_name: str, profile: dict, standalone_cores: set[str] +) -> bool: + """Whether a platform lays this emulator's files out for its standalone build. + + A platform names the emulators it launches standalone in + ``standalone_cores``, by profile key or by any name in ``cores:``. The + name alone is not enough: Recalbox and Batocera call their standalone + ScummVM ``scummvm``, which is also the key of the libretro core's + profile, and that profile documents no standalone build. Laid out + "standalone" it would lose every ``path:`` and drop its files at the + root, so a profile whose ``type`` has no standalone build keeps its + libretro layout whatever the platform calls it. + """ + if not standalone_cores: + return False + if "standalone" not in str(profile.get("type", "")): + return False + return emu_name in standalone_cores or bool( + standalone_cores & {str(c) for c in profile.get("cores", [])} + ) + + def resolve_platform_cores( config: dict, profiles: dict[str, dict], @@ -1193,6 +1221,8 @@ MANUFACTURER_PREFIXES = ( "interton-", "texas-instruments-", "videoton-", + "wenquxing-", + "aquaplus-", ) diff --git a/scripts/refresh_stale.py b/scripts/refresh_stale.py new file mode 100755 index 00000000..8b096dac --- /dev/null +++ b/scripts/refresh_stale.py @@ -0,0 +1,430 @@ +#!/usr/bin/env python3 +"""Refresh everything check_freshness.py reports as STALE, in parallel. + +check_freshness.py already answers "what is out of date". This script turns +that answer into actions: it parses the JSON report, matches each stale +subject to the command that pulls its upstream, runs them concurrently, and +leaves every write in the working tree for human review. Nothing is ever +committed. + +Mapping (stale -> command): + platforms/ python -m scripts.scraper._scraper + -o platforms/.yml + targets/ python -m scripts.scraper.targets. + -o platforms/targets/.yml + data/ python scripts/refresh_data_dirs.py --key + --force + catalogs/mame recipes python -m scripts.scraper.romset_dat_importer + --source mame --fetch + catalogs/fbneo recipes python -m scripts.scraper.romset_dat_importer + --source fbneo --fetch + +Surfaced only (no refresh): coreinfo gaps, redump/no-intro/tosec packs, +CI pins (install.py SHA-256, PyPI, pinned actions), profile_sync. Each needs +a human read (new .info file, manual DAT download, PyPI version bump). + +Usage: + python scripts/refresh_stale.py --dry-run + python scripts/refresh_stale.py --only platforms,targets + python scripts/refresh_stale.py --jobs 6 +""" + +from __future__ import annotations + +import argparse +import json +import os +import re +import subprocess +import sys +import time +from concurrent.futures import ThreadPoolExecutor, as_completed +from dataclasses import dataclass +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[1] +LOG_DIR = REPO_ROOT / "tmp" / "refresh_stale" +CHECK_FRESHNESS = REPO_ROOT / "scripts" / "check_freshness.py" +PLATFORMS_REGISTRY = REPO_ROOT / "platforms" / "_registry.yml" + +AUTO_AREAS = ("platforms", "targets", "data", "catalogs") +MANUAL_AREAS = ("coreinfo", "ci", "profiles") + +JOB_TIMEOUT = 1800 # 30 minutes per refresher; scrapers rarely exceed 10 + + +@dataclass(frozen=True) +class Job: + """One refresh action derived from a stale finding.""" + + area: str + subject: str + command: list[str] + log_path: Path + + +@dataclass(frozen=True) +class JobResult: + job: Job + returncode: int + duration: float + tail: str # last non-empty line of stderr or stdout + + +def _load_platform_registry() -> dict[str, dict]: + """Return the platforms map from _registry.yml. + + pyyaml is already a project dependency; the registry is small so we read + it once at startup rather than parsing each dispatch. + """ + import yaml + + with PLATFORMS_REGISTRY.open(encoding="utf-8") as fh: + return (yaml.safe_load(fh) or {}).get("platforms") or {} + + +def _run_check_freshness(areas: tuple[str, ...], extra: list[str]) -> list[dict]: + """Run check_freshness.py --json and return its findings.""" + cmd = [sys.executable, str(CHECK_FRESHNESS), "--json"] + if areas and set(areas) != set(AUTO_AREAS + MANUAL_AREAS): + cmd += ["--only", ",".join(areas)] + cmd += extra + proc = subprocess.run( + cmd, cwd=REPO_ROOT, capture_output=True, text=True, check=False, timeout=1800 + ) + if proc.returncode not in (0, 1): + tail = (proc.stderr or proc.stdout).strip().splitlines()[-3:] + raise RuntimeError( + "check_freshness.py failed:\n" + "\n".join(tail or ["(no output)"]) + ) + try: + return json.loads(proc.stdout) + except json.JSONDecodeError as exc: + raise RuntimeError(f"check_freshness.py returned unreadable JSON: {exc}") from exc + + +def _slug(subject: str) -> str: + """A safe filename stem for a finding's subject.""" + return re.sub(r"[^A-Za-z0-9_.-]+", "_", subject).strip("_") or "job" + + +def plan_jobs( + findings: list[dict], registry: dict[str, dict] +) -> tuple[list[Job], list[dict], list[dict]]: + """Split findings into (refreshable jobs, surfaced-only, ignored). + + Surfaced-only entries carry stale state that only a human can resolve + (new .info file, manual DAT download, PyPI bump). Ignored entries are + anything not STALE (OK, SKIPPED, UNKNOWN, ERROR); ERROR is also surfaced + so the final report mentions it. + """ + jobs: list[Job] = [] + seen: set[tuple[str, str]] = set() + surfaced: list[dict] = [] + ignored: list[dict] = [] + + for finding in findings: + status = finding.get("status") + area = finding.get("area") + subject = str(finding.get("subject") or "") + + if status == "ERROR": + surfaced.append(finding) + continue + if status != "STALE": + ignored.append(finding) + continue + + command = _command_for(area, subject, registry) + if command is None: + surfaced.append(finding) + continue + + key = (area, " ".join(command)) + if key in seen: + # Two findings funneling to the same command (e.g. mame recipes + # listed twice) collapse to one job. + continue + seen.add(key) + + log = LOG_DIR / f"{area}__{_slug(subject)}.log" + jobs.append(Job(area=area, subject=subject, command=command, log_path=log)) + + return jobs, surfaced, ignored + + +def _command_for( + area: str, subject: str, registry: dict[str, dict] +) -> list[str] | None: + """Map a stale finding to its refresh command, or None if manual.""" + if area == "platforms": + entry = registry.get(subject) or {} + scraper = entry.get("scraper") + if not scraper: + return None + return [ + sys.executable, + "-m", + f"scripts.scraper.{scraper}_scraper", + "-o", + f"platforms/{subject}.yml", + ] + + if area == "targets": + # " cores" rows list buildbot names without a profile; no + # scraper fixes that, a human writes `cores:` or _overrides.yml. + if subject.endswith(" cores"): + return None + entry = registry.get(subject) or {} + module = entry.get("target_scraper") + if not module: + return None + return [ + sys.executable, + "-m", + f"scripts.scraper.targets.{module}_scraper", + "-o", + f"platforms/targets/{subject}.yml", + ] + + if area == "data": + if subject == "_data_dirs.yml": + # The registry file itself cannot be refreshed; its entries can. + return None + return [ + sys.executable, + "scripts/refresh_data_dirs.py", + "--key", + subject, + "--force", + ] + + if area == "catalogs": + if subject == "mame recipes": + return [ + sys.executable, + "-m", + "scripts.scraper.romset_dat_importer", + "--source", + "mame", + "--fetch", + ] + if subject == "fbneo recipes": + return [ + sys.executable, + "-m", + "scripts.scraper.romset_dat_importer", + "--source", + "fbneo", + "--fetch", + ] + # redump / no-intro / tosec need manual DAT download. + return None + + # coreinfo, ci, profiles: never auto-refresh. + return None + + +def _github_token_env() -> dict[str, str]: + """Hand `gh auth token` to the subprocess environment. + + emudeck and retropie target scrapers hit the GitHub API; without a token + the unauthenticated rate limit (60/h) is burnt within a few platforms. + """ + env = os.environ.copy() + if env.get("GITHUB_TOKEN"): + return env + try: + proc = subprocess.run( + ["gh", "auth", "token"], capture_output=True, text=True, timeout=10, check=False + ) + except (FileNotFoundError, subprocess.TimeoutExpired): + return env + token = (proc.stdout or "").strip() + if proc.returncode == 0 and token: + env["GITHUB_TOKEN"] = token + return env + + +def run_job(job: Job, env: dict[str, str]) -> JobResult: + """Execute one refresh command and collect its output.""" + start = time.monotonic() + job.log_path.parent.mkdir(parents=True, exist_ok=True) + with job.log_path.open("w", encoding="utf-8") as log: + log.write(f"$ {' '.join(job.command)}\n") + log.flush() + try: + proc = subprocess.run( + job.command, + cwd=REPO_ROOT, + stdout=log, + stderr=subprocess.STDOUT, + env=env, + timeout=JOB_TIMEOUT, + check=False, + ) + returncode = proc.returncode + except subprocess.TimeoutExpired: + log.write(f"\nTIMEOUT after {JOB_TIMEOUT}s\n") + returncode = 124 + duration = time.monotonic() - start + tail = _log_tail(job.log_path) + return JobResult(job=job, returncode=returncode, duration=duration, tail=tail) + + +def _log_tail(path: Path) -> str: + """Last non-empty line of a log, truncated for the summary table.""" + try: + text = path.read_text(encoding="utf-8", errors="replace") + except OSError: + return "" + for line in reversed(text.splitlines()): + stripped = line.strip() + if stripped and not stripped.startswith("$ "): + return stripped[:120] + return "" + + +def render( + results: list[JobResult], + surfaced: list[dict], + ignored: list[dict], +) -> str: + """Human-readable table of what ran and what still needs a human.""" + lines: list[str] = [] + subject_w = max((len(r.job.subject) for r in results), default=10) + subject_w = min(max(subject_w, 10), 44) + + if results: + lines.append("\n[refreshed]") + lines.append( + f" {'STATUS':8} {'AREA':10} {'SUBJECT':{subject_w}} " + f"{'DUR':>7} LOG" + ) + for r in sorted(results, key=lambda x: (x.job.area, x.job.subject)): + status = "OK" if r.returncode == 0 else f"FAIL ({r.returncode})" + rel_log = r.job.log_path.relative_to(REPO_ROOT) + lines.append( + f" {status:8} {r.job.area:10} {r.job.subject[:subject_w]:{subject_w}} " + f"{r.duration:6.1f}s {rel_log}" + ) + if r.returncode != 0 and r.tail: + lines.append(f" -> {r.tail}") + + if surfaced: + lines.append("\n[manual review needed]") + for f in surfaced: + detail = str(f.get("detail") or "").strip() + lines.append( + f" {f.get('status', ''):7} {f.get('area', ''):10} " + f"{f.get('subject', ''):28} {detail[:100]}" + ) + + if ignored: + counts: dict[str, int] = {} + for f in ignored: + counts[str(f.get("status") or "?")] = counts.get(str(f.get("status") or "?"), 0) + 1 + parts = ", ".join(f"{n} {status.lower()}" for status, n in sorted(counts.items())) + lines.append(f"\n[ignored] {parts}") + + ok = sum(1 for r in results if r.returncode == 0) + fail = len(results) - ok + lines.append( + f"\nREFRESH: {ok} ok, {fail} failed, {len(surfaced)} to review manually" + ) + return "\n".join(lines) + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) + parser.add_argument( + "--only", + help="comma-separated areas to check: " + ",".join(AUTO_AREAS), + ) + parser.add_argument( + "--dry-run", + action="store_true", + help="list what would be refreshed; do not run any command", + ) + parser.add_argument( + "--jobs", + type=int, + default=4, + help="parallel refreshers (default: 4)", + ) + parser.add_argument( + "--json", + action="store_true", + help="machine-readable summary (jobs, surfaced, ignored)", + ) + parser.add_argument( + "--freshness-arg", + action="append", + default=[], + help="extra flag to pass through to check_freshness.py (repeatable)", + ) + args = parser.parse_args() + + all_areas = AUTO_AREAS + MANUAL_AREAS + if args.only: + areas = tuple(a.strip() for a in args.only.split(",") if a.strip()) + unknown = sorted(set(areas) - set(all_areas)) + if unknown: + parser.error(f"unknown area: {', '.join(unknown)}") + else: + areas = all_areas + + try: + findings = _run_check_freshness(areas, args.freshness_arg) + except RuntimeError as exc: + print(exc, file=sys.stderr) + return 2 + + registry = _load_platform_registry() + jobs, surfaced, ignored = plan_jobs(findings, registry) + + if args.dry_run: + print(f"\n[plan] {len(jobs)} refreshable, {len(surfaced)} manual, " + f"{len(ignored)} already clean/skipped") + for job in sorted(jobs, key=lambda j: (j.area, j.subject)): + print(f" {job.area:10} {job.subject:28} {' '.join(job.command)}") + if surfaced: + print("\n[manual review needed]") + for f in surfaced: + print(f" {f.get('status', ''):7} {f.get('area', ''):10} " + f"{f.get('subject', ''):28} {str(f.get('detail') or '')[:100]}") + return 0 + + LOG_DIR.mkdir(parents=True, exist_ok=True) + env = _github_token_env() + results: list[JobResult] = [] + with ThreadPoolExecutor(max_workers=max(1, args.jobs)) as pool: + futures = [pool.submit(run_job, job, env) for job in jobs] + for future in as_completed(futures): + results.append(future.result()) + + if args.json: + payload = { + "refreshed": [ + { + "area": r.job.area, + "subject": r.job.subject, + "returncode": r.returncode, + "duration_sec": round(r.duration, 2), + "log": str(r.job.log_path.relative_to(REPO_ROOT)), + "tail": r.tail, + } + for r in results + ], + "surfaced": surfaced, + "ignored_count": len(ignored), + } + print(json.dumps(payload, indent=2)) + else: + print(render(results, surfaced, ignored)) + + return 0 if all(r.returncode == 0 for r in results) else 1 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/tests/test_check_freshness.py b/tests/test_check_freshness.py new file mode 100644 index 00000000..d17b15cf --- /dev/null +++ b/tests/test_check_freshness.py @@ -0,0 +1,218 @@ +"""check_freshness: the pure parts, no network. + +The script answers "is the local copy the one upstream serves" for every +transcribed layer. What can be locked without a network is how a diff is +read, how a core name is resolved, how a pin is parsed, and how a profile +report is folded, since each of those decides whether a row says STALE. +""" + +from __future__ import annotations + +import sys +import unittest +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(REPO_ROOT / "scripts")) + +import check_freshness as cf # noqa: E402 +from common import yaml_load # noqa: E402 + + +class PlatformDiffTests(unittest.TestCase): + def _platform(self, version="1", files=None, cores=None): + return { + "version": version, + "cores": cores or ["a", "b"], + "systems": { + "sys": {"files": files if files is not None else [ + {"name": "x.bin", "destination": "x.bin", "md5": "1"}, + ]}, + }, + } + + def test_identical_files_are_not_a_change(self): + diff = cf.diff_platform(self._platform(), self._platform()) + self.assertFalse(diff.changed) + self.assertEqual(diff.summary(), "identical") + + def test_version_alone_is_a_change_but_not_content(self): + diff = cf.diff_platform(self._platform("1"), self._platform("2")) + self.assertTrue(diff.changed) + self.assertFalse(diff.content_changed) + self.assertEqual(diff.version, ("1", "2")) + + def test_hash_change_on_a_destination_is_a_change_not_a_swap(self): + old = self._platform(files=[{"name": "x.bin", "destination": "x.bin", "md5": "1"}]) + new = self._platform(files=[{"name": "x.bin", "destination": "x.bin", "md5": "2"}]) + diff = cf.diff_platform(old, new) + self.assertEqual(diff.files_changed, ["sys/x.bin"]) + self.assertEqual(diff.files_added, []) + self.assertEqual(diff.files_removed, []) + + def test_systems_files_and_cores_are_reported_separately(self): + old = self._platform() + new = self._platform(cores=["a", "c"]) + new["systems"]["other"] = {"files": []} + new["systems"]["sys"]["files"].append({"name": "y.bin", "destination": "y.bin"}) + new["standalone_cores"] = ["dolphin"] + diff = cf.diff_platform(old, new) + self.assertEqual(diff.systems_added, ["other"]) + self.assertEqual(diff.files_added, ["sys/y.bin"]) + self.assertEqual(diff.cores_added, ["c", "standalone:dolphin"]) + self.assertEqual(diff.cores_removed, ["b"]) + self.assertIn("+cores 2", diff.summary()) + + def test_target_stamp_is_not_a_change(self): + old = {"scraped_at": "2026-01-01", "targets": {"t": {"cores": ["a"]}}} + new = {"scraped_at": "2026-02-01", "targets": {"t": {"cores": ["a"]}}} + self.assertFalse(cf.diff_targets(old, new).content_changed) + new["targets"]["t"]["cores"].append("b") + diff = cf.diff_targets(old, new) + self.assertEqual(diff.cores_added, ["t/b"]) + + +class CoreResolutionTests(unittest.TestCase): + PROFILES = { + "beetle_psx": {"cores": ["mednafen_psx"], "type": "libretro"}, + "eka2l1": {"cores": ["eka2l1"], "type": "standalone"}, + "FreeIntv": {"cores": ["freeintvtsoverlay"], "type": "alias"}, + } + + def test_index_maps_key_and_every_core_alias(self): + index = cf.profile_name_index(self.PROFILES) + self.assertEqual(index["mednafen_psx"], "beetle_psx") + self.assertEqual(index["beetle_psx"], "beetle_psx") + + def test_unresolved_honours_remove_cores_per_target(self): + index = cf.profile_name_index(self.PROFILES) + targets = { + "targets": { + "android": {"cores": ["mednafen_psx", "na", "ghost"]}, + "linux": {"cores": ["ghost"]}, + } + } + removed = {"android": {"na"}} + self.assertEqual( + cf.unresolved_target_cores(targets, index, removed), + {"ghost": ["android", "linux"]}, + ) + + def test_coreinfo_gaps_fold_case_and_flag_standalone_profiles(self): + names = ["FreeIntvTSOverlay", "eka2l1", "wqxemu", "mednafen_psx"] + unprofiled, standalone = cf.coreinfo_gaps(names, self.PROFILES) + self.assertEqual(unprofiled, ["wqxemu"]) + self.assertEqual(standalone, [("eka2l1", "eka2l1")]) + + +class PinParsingTests(unittest.TestCase): + WORKFLOW = """ + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7 + - run: pip install pyyaml jsonschema==4.23.0 "mkdocs-material>=9.7.5,<10" "pymdown-extensions>=10.14" + - uses: actions/deploy-pages@cd2ce8fcbc39b97be8ca5fce6e763baed58fa128 + """ + + def test_pip_pins_keep_the_specifier(self): + pins = cf.parse_pip_pins(self.WORKFLOW) + self.assertEqual(pins["pyyaml"], "") + self.assertEqual(pins["jsonschema"], "==4.23.0") + self.assertEqual(pins["mkdocs-material"], ">=9.7.5,<10") + self.assertEqual(pins["pymdown-extensions"], ">=10.14") + + def test_action_pins_carry_sha_and_optional_tag(self): + pins = cf.parse_action_pins(self.WORKFLOW) + self.assertEqual(pins["actions/checkout"], ("3d3c42e5aac5ba805825da76410c181273ba90b1", "v7")) + self.assertEqual(pins["actions/deploy-pages"][1], "") + + def test_specifier_admits_or_refuses_the_latest(self): + self.assertTrue(cf.specifier_allows("", "9.9")) + self.assertTrue(cf.specifier_allows(">=9.7.5,<10", "9.7.7")) + self.assertFalse(cf.specifier_allows(">=9.7.5,<10", "10.0.0")) + self.assertFalse(cf.specifier_allows("==4.23.0", "4.26.0")) + self.assertTrue(cf.specifier_allows("==4.23.0", "4.23.0")) + self.assertTrue(cf.specifier_allows(">=10.14", "12.0.1")) + + +class CatalogParsingTests(unittest.TestCase): + def test_tosec_newest_release_from_category_links(self): + html = ( + 'x' + 'y' + 'z' + ) + self.assertEqual(cf.tosec_latest_pack(html), "2025-03-13") + self.assertIsNone(cf.tosec_latest_pack("")) + + def test_fbneo_drift_compares_blob_shas_and_reports_untracked(self): + listing = [ + {"type": "file", "name": "A.dat", "sha": "1"}, + {"type": "file", "name": "B.dat", "sha": "2"}, + {"type": "dir", "name": "old"}, + ] + snapshot = {"upstream": {"blobs": {"A.dat": "1", "B.dat": "9"}}} + self.assertEqual(cf.fbneo_blob_drift(snapshot, listing), (["B.dat"], True)) + self.assertEqual(cf.fbneo_blob_drift({"upstream": {}}, listing), ([], False)) + snapshot = {"upstream": {"blobs": {"A.dat": "1", "B.dat": "2", "C.dat": "3"}}} + self.assertEqual(cf.fbneo_blob_drift(snapshot, listing), (["C.dat"], True)) + + def test_mame_versions_read_from_dat_labels(self): + recipes = {"dats": {"MAME mame0250": "0.250", "MAME mame0289": "0.289", "FinalBurn Neo": "x"}} + self.assertEqual(cf.mame_versions(recipes), ["mame0250", "mame0289"]) + + +class ProfileSummaryTests(unittest.TestCase): + def test_summary_separates_review_moved_and_unreachable(self): + report = [ + {"name": "a", "pin": "1" * 40, "head": "1" * 40, "needs_review": 0}, + {"name": "b", "pin": "1" * 40, "head": "2" * 40, "needs_review": 0}, + {"name": "c", "pin": "1" * 40, "head": "2" * 40, "needs_review": 3}, + {"name": "d", "skipped": "host does not resolve"}, + ] + findings = cf.summarize_profile_sync(report) + by_subject = {f.subject: f for f in findings} + self.assertEqual(by_subject["profile_sync"].status, cf.STALE) + self.assertEqual(by_subject["profile_sync"].local, "1 at tip, 1 anchored past the tip") + self.assertEqual(by_subject["c"].status, cf.STALE) + self.assertEqual(by_subject["d"].status, cf.UNKNOWN) + self.assertNotIn("a", by_subject) + self.assertNotIn("b", by_subject) + + def test_a_non_list_report_is_an_error_not_a_crash(self): + findings = cf.summarize_profile_sync({"oops": 1}) + self.assertEqual(findings[0].status, cf.ERROR) + + +class RepositoryWiringTests(unittest.TestCase): + """The script derives its work from the registry: a platform whose scraper + module does not exist would be skipped in silence.""" + + def test_every_registered_scraper_module_exists(self): + rows = cf._scrapable_platforms(REPO_ROOT / "platforms") + self.assertGreaterEqual(len(rows), 10) + for _name, module, _path in rows: + self.assertTrue( + (REPO_ROOT / (module.replace(".", "/") + ".py")).is_file(), module + ) + + def test_inheriting_platforms_without_a_source_are_not_scraped(self): + names = {name for name, _m, _p in cf._scrapable_platforms(REPO_ROOT / "platforms")} + with (REPO_ROOT / "platforms" / "lakka.yml").open(encoding="utf-8") as fh: + lakka = yaml_load(fh) + self.assertTrue(lakka.get("inherits")) + self.assertNotIn("lakka", names) + self.assertIn("retropie", names) + + def test_render_counts_every_status_once(self): + findings = [ + cf.Finding("ci", "a", cf.OK), + cf.Finding("ci", "b", cf.STALE, detail="x"), + cf.Finding("data", "c", cf.UNKNOWN), + ] + text = cf.render(findings) + self.assertIn("FRESHNESS: 1 stale, 1 unknown, 1 ok", text) + self.assertIn("[ci]", text) + self.assertIn("[data]", text) + + +if __name__ == "__main__": + unittest.main()