Files
libretro/scripts/check_freshness.py
T

1092 lines
41 KiB
Python

#!/usr/bin/env python3
"""Confront every transcribed layer with what its upstream serves today.
Each layer the repository transcribes moves on its own schedule: platform
lists, hardware targets, core-info, buildbot system assets, dump catalogs,
romset DATs, the CI toolchain. This script asks each of them the same
question, is the local copy the one upstream serves, and answers with one
row per subject.
Platform and target files are re-scraped through the scrapers' own write path
into a throwaway copy, then diffed against the committed file. The diff
decides, not a version string: a scraper that pins a new tag but produces
the same entries is reported as a version change and nothing else.
The native export patches each platform's own file, kept in a local cache.
That cache is checked against the platform file it was transcribed with: an
original cached before the last rescrape, or under another pin, is no longer
the file our data describes.
Emulator profiles are covered by profile_sync, which is slow (one API pass
per profile). ``--profiles`` runs it here and folds its verdict in.
Usage:
python scripts/check_freshness.py
python scripts/check_freshness.py --only platforms,targets
python scripts/check_freshness.py --json
python scripts/check_freshness.py --profiles
python scripts/check_freshness.py --offline # local age checks only
"""
from __future__ import annotations
import argparse
import contextlib
import hashlib
import io
import json
import re
import shutil
import subprocess
import sys
import tempfile
from concurrent.futures import ThreadPoolExecutor
from dataclasses import asdict, dataclass, field
from datetime import datetime, timezone
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
import export_native # noqa: E402
import refresh_data_dirs # noqa: E402
import upstream # noqa: E402
from common import ( # noqa: E402
list_registered_platforms,
load_emulator_profiles,
yaml_load,
)
from exporter import discover_exporters # noqa: E402
from exporter.baseline import build_native_model # noqa: E402
from scripts.scraper.redump_dat_scraper import fetch_snapshot # noqa: E402
from scripts.scraper.targets import discover_target_scrapers # noqa: E402
REPO_ROOT = Path(__file__).resolve().parents[1]
CORE_INFO_REPO = "https://github.com/libretro/libretro-core-info"
NO_INTRO_MIRROR = "https://github.com/hugo19941994/auto-datfile-generator"
NO_INTRO_RELEASE = "Daily_Rebuild"
NO_INTRO_ASSET = "no-intro.zip"
TOSEC_DOWNLOADS = "https://www.tosecdev.org/downloads"
MAME_REPO = "https://github.com/mamedev/mame"
FBNEO_REPO = "https://github.com/libretro/FBNeo"
PYPI_URL = "https://pypi.org/pypi/{name}/json"
INSTALLERS = ("install.sh", "install.ps1")
SHOWN_ITEMS = 4 # items named in a diff summary before ", ..."
SHOWN_NAMES = 12 # core names named in an unresolved row before ", ..."
AREAS = ("platforms", "native", "targets", "coreinfo", "data", "catalogs", "ci", "profiles")
OK, STALE, UNKNOWN, ERROR, SKIPPED = "OK", "STALE", "UNKNOWN", "ERROR", "SKIPPED"
# Every status a finding may carry, in the order the summary counts them.
STATUSES = (STALE, UNKNOWN, ERROR, SKIPPED, OK)
@dataclass(frozen=True)
class Finding:
area: str
subject: str
status: str
local: str = ""
remote: str = ""
detail: str = ""
@dataclass
class PlatformDiff:
"""What a fresh scrape changes in a platform or target file."""
version: tuple[str, str] | None = None
systems_added: list[str] = field(default_factory=list)
systems_removed: list[str] = field(default_factory=list)
files_added: list[str] = field(default_factory=list)
files_removed: list[str] = field(default_factory=list)
files_changed: list[str] = field(default_factory=list)
cores_added: list[str] = field(default_factory=list)
cores_removed: list[str] = field(default_factory=list)
@property
def content_changed(self) -> bool:
return any(
(
self.systems_added,
self.systems_removed,
self.files_added,
self.files_removed,
self.files_changed,
self.cores_added,
self.cores_removed,
)
)
@property
def changed(self) -> bool:
return self.content_changed or self.version is not None
def summary(self) -> str:
parts: list[str] = []
if self.version:
parts.append(f"version {self.version[0]} -> {self.version[1]}")
for label, items in (
("+systems", self.systems_added),
("-systems", self.systems_removed),
("+files", self.files_added),
("-files", self.files_removed),
("~files", self.files_changed),
("+cores", self.cores_added),
("-cores", self.cores_removed),
):
if items:
shown = ", ".join(items[:SHOWN_ITEMS]) + (", ..." if len(items) > SHOWN_ITEMS else "")
parts.append(f"{label} {len(items)} ({shown})")
return "; ".join(parts) if parts else "identical"
# --- platforms --------------------------------------------------------------
def _file_key(entry: dict) -> str:
return str(entry.get("destination") or entry.get("name") or "")
def _core_names(config: dict) -> set[str]:
cores = config.get("cores")
names = set(map(str, cores)) if isinstance(cores, list) else set()
standalone = config.get("standalone_cores")
if isinstance(standalone, list):
names.update(f"standalone:{c}" for c in standalone)
return names
def diff_platform(old: dict, new: dict) -> PlatformDiff:
"""Compare two platform files the way a reviewer reads the diff.
Files are keyed by system and destination, so a hash or flag change on
an existing destination is a change, not a removal plus an addition.
"""
diff = PlatformDiff()
old_version, new_version = str(old.get("version", "")), str(new.get("version", ""))
if old_version != new_version:
diff.version = (old_version, new_version)
old_systems = old.get("systems") or {}
new_systems = new.get("systems") or {}
diff.systems_added = sorted(set(new_systems) - set(old_systems))
diff.systems_removed = sorted(set(old_systems) - set(new_systems))
for system in sorted(set(old_systems) & set(new_systems)):
old_files = {
_file_key(f): f for f in (old_systems[system] or {}).get("files") or []
}
new_files = {
_file_key(f): f for f in (new_systems[system] or {}).get("files") or []
}
diff.files_added.extend(
f"{system}/{k}" for k in sorted(set(new_files) - set(old_files))
)
diff.files_removed.extend(
f"{system}/{k}" for k in sorted(set(old_files) - set(new_files))
)
diff.files_changed.extend(
f"{system}/{k}"
for k in sorted(set(old_files) & set(new_files))
if old_files[k] != new_files[k]
)
old_cores, new_cores = _core_names(old), _core_names(new)
diff.cores_added = sorted(new_cores - old_cores)
diff.cores_removed = sorted(old_cores - new_cores)
return diff
def diff_targets(old: dict, new: dict) -> PlatformDiff:
"""Compare two target files. The scrape stamp is not a change."""
diff = PlatformDiff()
old_targets = old.get("targets") or {}
new_targets = new.get("targets") or {}
diff.systems_added = sorted(set(new_targets) - set(old_targets))
diff.systems_removed = sorted(set(old_targets) - set(new_targets))
for name in sorted(set(old_targets) & set(new_targets)):
before = set(map(str, (old_targets[name] or {}).get("cores") or []))
after = set(map(str, (new_targets[name] or {}).get("cores") or []))
diff.cores_added.extend(f"{name}/{c}" for c in sorted(after - before))
diff.cores_removed.extend(f"{name}/{c}" for c in sorted(before - after))
return diff
def _scrape_into_copy(
module: str, source: Path, workdir: Path, timeout: int = 900
) -> tuple[Path, str | None]:
"""Run a scraper's own write path on a throwaway copy of *source*.
The CLI merges into the file it writes, so the copy carries every field
the scraper preserves and the diff shows only what the scrape changes.
"""
target = workdir / source.name
shutil.copyfile(source, target)
proc = subprocess.run(
[sys.executable, "-m", module, "-o", str(target)],
cwd=REPO_ROOT,
capture_output=True,
text=True,
timeout=timeout,
check=False,
)
if proc.returncode != 0:
tail = (proc.stderr or proc.stdout).strip().splitlines()[-1:]
return target, tail[0] if tail else f"exit {proc.returncode}"
return target, None
def _scrapable_platforms(platforms_dir: Path) -> list[tuple[str, str, Path]]:
"""(platform, scraper module, file) for every registered platform.
A platform that only inherits another and declares no source of its own
(Lakka) has nothing to scrape: its scraper writes the parent's file.
"""
with (platforms_dir / "_registry.yml").open(encoding="utf-8") as fh:
registry = (yaml_load(fh) or {}).get("platforms") or {}
rows = []
for name in list_registered_platforms(str(platforms_dir), include_archived=True):
path = platforms_dir / f"{name}.yml"
if not path.is_file():
continue
with path.open(encoding="utf-8") as fh:
config = yaml_load(fh) or {}
scraper = (registry.get(name) or {}).get("scraper")
if not scraper or (config.get("inherits") and not config.get("source")):
continue
rows.append((name, f"scripts.scraper.{scraper}_scraper", path))
return rows
def check_platforms(platforms_dir: Path, workdir: Path, jobs: int) -> list[Finding]:
rows = _scrapable_platforms(platforms_dir)
findings: list[Finding] = []
def run(row: tuple[str, str, Path]) -> Finding:
name, module, path = row
try:
fresh, error = _scrape_into_copy(module, path, workdir)
except subprocess.TimeoutExpired:
return Finding("platforms", name, ERROR, detail="scraper timed out")
if error:
return Finding("platforms", name, ERROR, detail=error)
with path.open(encoding="utf-8") as fh:
old = yaml_load(fh) or {}
with fresh.open(encoding="utf-8") as fh:
new = yaml_load(fh) or {}
diff = diff_platform(old, new)
status = STALE if diff.changed else OK
return Finding(
"platforms",
name,
status,
local=str(old.get("version", "")),
remote=str(new.get("version", "")),
detail=diff.summary(),
)
with ThreadPoolExecutor(max_workers=jobs) as pool:
findings.extend(pool.map(run, rows))
return sorted(findings, key=lambda f: f.subject)
# --- native originals --------------------------------------------------------
def native_cache_state(
wanted: dict[str, str],
root: Path,
recorded: dict[str, str],
index: Path,
transcribed_at: float,
) -> str:
"""Why a platform's cached originals cannot be patched, '' when they can.
The export corrects the file our data was transcribed from. A cached
original is that file only if it came from the URL the platform file
names today and was fetched no earlier than the platform file was last
rewritten: a branch URL keeps its name while its content moves.
"""
for relative, url in sorted(wanted.items()):
path = root / relative
if not path.is_file():
return f"{relative} not cached"
known = recorded.get(export_native.source_key(path, index))
if known is None:
return f"{relative} cached from an unrecorded URL"
if known != url:
return f"{relative} cached from another revision"
if path.stat().st_mtime < transcribed_at:
return f"{relative} cached before the platform file was rewritten"
return ""
def _transcribed_at(platforms_dir: Path, platform: str) -> float:
"""Last rewrite of the platform file or of any file it inherits."""
newest = 0.0
seen: set[str] = set()
name: str | None = platform
while name and name not in seen:
seen.add(name)
path = platforms_dir / f"{name}.yml"
if not path.is_file():
break
newest = max(newest, path.stat().st_mtime)
with path.open(encoding="utf-8") as fh:
name = (yaml_load(fh) or {}).get("inherits")
return newest
def check_native(platforms_dir: Path, cache_dir: Path, truth_dir: Path) -> list[Finding]:
exporters = discover_exporters()
index = cache_dir / export_native.SOURCES_INDEX
recorded = export_native.load_sources(cache_dir)
findings: list[Finding] = []
for name in list_registered_platforms(str(platforms_dir), include_archived=True):
exporter_class = exporters.get(name)
if not exporter_class:
continue
truth, scraped = export_native.load_inputs(name, truth_dir, str(platforms_dir))
systems, _ = build_native_model(truth or {}, scraped)
wanted = export_native.wanted_sources(exporter_class(), systems, scraped)
reason = native_cache_state(
wanted, cache_dir / name, recorded, index, _transcribed_at(platforms_dir, name)
)
findings.append(
Finding(
"native",
name,
STALE if reason else OK,
local=f"{len(wanted)} file(s)",
detail=f"{reason}, export_native.py --refresh-cache fetches it" if reason else "",
)
)
return sorted(findings, key=lambda f: f.subject)
# --- targets ----------------------------------------------------------------
def profile_name_index(profiles: dict[str, dict]) -> dict[str, str]:
"""Upstream core name -> profile key, the index target filtering uses."""
index: dict[str, str] = {}
for key, profile in profiles.items():
index.setdefault(key, key)
for alias in profile.get("cores") or []:
index.setdefault(str(alias), key)
return index
def _removed_cores(overrides: dict, platform: str) -> dict[str, set[str]]:
targets = ((overrides.get(platform) or {}).get("targets")) or {}
default = set(map(str, (targets.get("_default") or {}).get("remove_cores") or []))
return {
name: default | set(map(str, (entry or {}).get("remove_cores") or []))
for name, entry in targets.items()
if name != "_default"
} | {"_default": default}
def unresolved_target_cores(
targets: dict, index: dict[str, str], removed: dict[str, set[str]]
) -> dict[str, list[str]]:
"""Core names a target lists that no profile claims and no override drops."""
unresolved: dict[str, list[str]] = {}
default_drop = removed.get("_default", set())
for name, entry in (targets.get("targets") or {}).items():
dropped = removed.get(name, default_drop) | default_drop
for core in map(str, (entry or {}).get("cores") or []):
if core not in index and core not in dropped:
unresolved.setdefault(core, []).append(name)
return {core: sorted(names) for core, names in sorted(unresolved.items())}
def _target_scrapers() -> dict[str, str]:
"""platform -> scraper module for every target scraper on disk."""
return {
platform: cls.__module__
for platform, cls in discover_target_scrapers().items()
}
def check_targets(
platforms_dir: Path, workdir: Path, profiles: dict[str, dict], jobs: int,
offline: bool,
) -> list[Finding]:
targets_dir = platforms_dir / "targets"
overrides_path = targets_dir / "_overrides.yml"
overrides = {}
if overrides_path.is_file():
with overrides_path.open(encoding="utf-8") as fh:
overrides = yaml_load(fh) or {}
index = profile_name_index(profiles)
scrapers = {} if offline else _target_scrapers()
findings: list[Finding] = []
today = datetime.now(timezone.utc)
def run(platform: str) -> list[Finding]:
path = targets_dir / f"{platform}.yml"
if not path.is_file():
return [Finding("targets", platform, ERROR, detail="no target file")]
with path.open(encoding="utf-8") as fh:
old = yaml_load(fh) or {}
module = scrapers.get(platform)
if module is None:
# No scraper: a hand-written file has no upstream to diff against,
# so only its age and its core names can be reported.
stamp = str(old.get("scraped_at", ""))[:10]
age = _age_days(stamp, today)
out = [
Finding(
"targets",
platform,
OK if age is not None else UNKNOWN,
local=f"static {stamp}",
detail=f"{age} days old" if age is not None else "no scrape stamp",
)
]
missing = unresolved_target_cores(old, index, _removed_cores(overrides, platform))
out.extend(_unresolved_findings(platform, missing))
return out
try:
fresh, error = _scrape_into_copy(module, path, workdir)
except subprocess.TimeoutExpired:
return [Finding("targets", platform, ERROR, detail="scraper timed out")]
if error:
return [Finding("targets", platform, ERROR, detail=error)]
with fresh.open(encoding="utf-8") as fh:
new = yaml_load(fh) or {}
diff = diff_targets(old, new)
out = [
Finding(
"targets",
platform,
STALE if diff.content_changed else OK,
local=str(old.get("scraped_at", ""))[:10],
remote=str(new.get("scraped_at", ""))[:10],
detail=diff.summary(),
)
]
missing = unresolved_target_cores(new, index, _removed_cores(overrides, platform))
out.extend(_unresolved_findings(platform, missing))
return out
names = sorted(p.stem for p in targets_dir.glob("*.yml") if not p.name.startswith("_"))
with ThreadPoolExecutor(max_workers=jobs) as pool:
for rows in pool.map(run, names):
findings.extend(rows)
return findings
def _unresolved_findings(platform: str, missing: dict[str, list[str]]) -> list[Finding]:
"""One row per platform: the names its targets list that no profile claims.
A buildbot name without a profile is a core the target filter cannot
see; a platform's own emulator id without one is an emulator nobody has
profiled yet. Both are actionable, so both are listed, but as one row so
the backlog does not bury the day's changes.
"""
if not missing:
return []
names = sorted(missing)
shown = ", ".join(names[:SHOWN_NAMES]) + (", ..." if len(names) > SHOWN_NAMES else "")
return [
Finding(
"targets",
f"{platform} cores",
STALE,
local=f"{len(names)} without a profile",
detail=f"cores: alias or _overrides remove_cores needed for {shown}",
)
]
def _age_days(stamp: str, now: datetime) -> int | None:
try:
then = datetime.strptime(stamp[:10], "%Y-%m-%d").replace(tzinfo=timezone.utc)
except ValueError:
return None
return (now - then).days
# --- core-info --------------------------------------------------------------
def core_info_names(cache_dir: str, offline: bool) -> list[str] | None:
"""Every core the libretro core-info repository describes."""
repo = upstream.parse_repo(CORE_INFO_REPO)
if repo is None:
return None
payload = upstream._api( # noqa: SLF001
f"{repo.api_base}/repos/{repo.slug}/contents?per_page=100",
cache_dir,
offline,
upstream.MOVING_TTL,
)
if not isinstance(payload, list):
return None
names = []
for entry in payload:
name = entry.get("name", "") if isinstance(entry, dict) else ""
if name.endswith("_libretro.info"):
names.append(name[: -len("_libretro.info")])
return sorted(names)
def coreinfo_gaps(
names: list[str], profiles: dict[str, dict]
) -> tuple[list[str], list[tuple[str, str]]]:
"""Cores core-info knows that the profiles do not, and profiles that
call standalone a core libretro now builds.
Matching folds case: the buildbot serves lower-case names while a few
.info files keep the project's own casing.
"""
index = {k.casefold(): v for k, v in profile_name_index(profiles).items()}
unprofiled = [n for n in names if n.casefold() not in index]
standalone = [
(n, index[n.casefold()])
for n in names
if n.casefold() in index
and str(profiles[index[n.casefold()]].get("type", "")).strip() == "standalone"
]
return unprofiled, standalone
def check_coreinfo(profiles: dict[str, dict], cache_dir: str, offline: bool) -> list[Finding]:
names = core_info_names(cache_dir, offline)
if names is None:
return [Finding("coreinfo", "libretro-core-info", UNKNOWN, detail="listing unavailable")]
unprofiled, standalone = coreinfo_gaps(names, profiles)
findings = [
Finding(
"coreinfo",
"libretro-core-info",
STALE if unprofiled or standalone else OK,
local=f"{len(names) - len(unprofiled)} profiled",
remote=f"{len(names)} .info files",
)
]
findings.extend(
Finding("coreinfo", name, STALE, detail="no profile claims this core name")
for name in unprofiled
)
findings.extend(
Finding(
"coreinfo",
name,
STALE,
local=f"{key}: type standalone",
detail="core-info describes a libretro build of it",
)
for name, key in standalone
)
return findings
# --- data directories -------------------------------------------------------
def check_data(offline: bool) -> list[Finding]:
if offline:
return [Finding("data", "_data_dirs.yml", SKIPPED, detail="offline")]
registry = refresh_data_dirs.load_registry(str(REPO_ROOT / "platforms" / "_data_dirs.yml"))
sink = io.StringIO()
with contextlib.redirect_stdout(sink):
results = refresh_data_dirs.refresh_all(registry, dry_run=True)
findings = []
for key, result in sorted(results.items()):
if result is None:
findings.append(Finding("data", key, UNKNOWN, detail="remote unreachable"))
elif result:
findings.append(Finding("data", key, STALE, detail="upstream changed, refresh_data_dirs.py fetches it"))
else:
findings.append(Finding("data", key, OK))
return findings
# --- catalogs and recipes ---------------------------------------------------
def _snapshot(path: Path) -> dict:
if not path.is_file():
return {}
with path.open(encoding="utf-8") as fh:
return json.load(fh)
def check_catalogs(cache_dir: str, offline: bool) -> list[Finding]:
findings = []
findings.append(_check_redump(offline))
findings.append(_check_no_intro(cache_dir, offline))
findings.append(_check_tosec(offline))
findings.append(_check_mame(cache_dir, offline))
findings.append(_check_fbneo(cache_dir, offline))
return findings
def _check_redump(offline: bool) -> Finding:
local = _snapshot(REPO_ROOT / "provenance" / "redump.json")
dats = local.get("dats") or {}
stamp = ", ".join(sorted(set(dats.values()))) or "none"
if offline:
return Finding("catalogs", "redump", SKIPPED, local=stamp, detail="offline")
try:
remote_dats, _ = fetch_snapshot()
except (ConnectionError, ValueError) as exc:
return Finding("catalogs", "redump", UNKNOWN, local=stamp, detail=str(exc))
remote_stamp = ", ".join(sorted(set(remote_dats.values())))
changed = sorted(
name for name in set(dats) | set(remote_dats) if dats.get(name) != remote_dats.get(name)
)
return Finding(
"catalogs",
"redump",
STALE if changed else OK,
local=stamp,
remote=remote_stamp,
detail=("changed: " + ", ".join(changed)) if changed else f"{len(dats)} DATs",
)
def _check_no_intro(cache_dir: str, offline: bool) -> Finding:
local = _snapshot(REPO_ROOT / "provenance" / "no-intro.json")
imported = str(local.get("imported_at", ""))
if offline:
return Finding("catalogs", "no-intro", SKIPPED, local=imported, detail="offline")
repo = upstream.parse_repo(NO_INTRO_MIRROR)
payload = upstream._api( # noqa: SLF001
f"{repo.api_base}/repos/{repo.slug}/releases/tags/{NO_INTRO_RELEASE}",
cache_dir,
offline,
upstream.MOVING_TTL,
)
assets = payload.get("assets") if isinstance(payload, dict) else None
updated = ""
for asset in assets or []:
if isinstance(asset, dict) and asset.get("name") == NO_INTRO_ASSET:
updated = str(asset.get("updated_at", ""))[:10]
if not updated:
return Finding("catalogs", "no-intro", UNKNOWN, local=imported, detail="asset not listed")
return Finding(
"catalogs",
"no-intro",
STALE if updated > imported else OK,
local=imported,
remote=updated,
detail=f"{NO_INTRO_ASSET} rebuilt {updated}",
)
def tosec_latest_pack(html: str) -> str | None:
"""Date of the newest TOSEC release the downloads page lists."""
dates = re.findall(r'/downloads/category/\d+-(\d{4}-\d{2}-\d{2})"', html)
return max(dates) if dates else None
def _check_tosec(offline: bool) -> Finding:
local = _snapshot(REPO_ROOT / "provenance" / "tosec.json")
imported = str(local.get("imported_at", ""))
if offline:
return Finding("catalogs", "tosec", SKIPPED, local=imported, detail="offline")
try:
html = upstream._http_text(TOSEC_DOWNLOADS) # noqa: SLF001
except upstream.UpstreamError as exc:
return Finding("catalogs", "tosec", UNKNOWN, local=imported, detail=str(exc))
latest = tosec_latest_pack(html or "")
if latest is None:
return Finding("catalogs", "tosec", UNKNOWN, local=imported, detail="no release listed")
return Finding(
"catalogs",
"tosec",
STALE if latest > imported else OK,
local=imported,
remote=latest,
detail=f"newest pack TOSEC-v{latest}",
)
def mame_versions(recipes: dict) -> list[str]:
"""MAME version tags a recipe snapshot has absorbed, oldest first."""
tags = set()
for label in (recipes.get("dats") or {}):
match = re.search(r"\b(mame\d{4})\b", str(label))
if match:
tags.add(match.group(1))
return sorted(tags)
def _check_mame(cache_dir: str, offline: bool) -> Finding:
local = _snapshot(REPO_ROOT / "recipes" / "mame.json")
versions = mame_versions(local)
newest = versions[-1] if versions else "none"
if offline:
return Finding("catalogs", "mame recipes", SKIPPED, local=newest, detail="offline")
release = upstream.latest_release(upstream.parse_repo(MAME_REPO), cache_dir, offline)
if release is None:
return Finding("catalogs", "mame recipes", UNKNOWN, local=newest, detail="no release seen")
return Finding(
"catalogs",
"mame recipes",
OK if release.tag in versions else STALE,
local=newest,
remote=f"{release.tag} ({release.date})",
detail=f"{len(versions)} versions imported",
)
def fbneo_blob_drift(snapshot: dict, listing: object) -> tuple[list[str], bool]:
"""DAT files whose git blob differs from the one the snapshot was read
from. The second value says whether the snapshot records blobs at all:
one written before that field existed can only be compared by date.
"""
recorded = (snapshot.get("upstream") or {}).get("blobs") or {}
if not recorded:
return [], False
remote = {
item["name"]: item.get("sha", "")
for item in (listing if isinstance(listing, list) else [])
if isinstance(item, dict)
and item.get("type") == "file"
and str(item.get("name", "")).lower().endswith(".dat")
}
drifted = sorted(
name for name in set(recorded) | set(remote) if recorded.get(name) != remote.get(name)
)
return drifted, True
def _check_fbneo(cache_dir: str, offline: bool) -> Finding:
local = _snapshot(REPO_ROOT / "recipes" / "fbneo.json")
imported = str(local.get("imported_at", ""))
if offline:
return Finding("catalogs", "fbneo recipes", SKIPPED, local=imported, detail="offline")
repo = upstream.parse_repo(FBNEO_REPO)
listing = upstream._api( # noqa: SLF001
f"{repo.api_base}/repos/{repo.slug}/contents/dats",
cache_dir,
offline,
upstream.MOVING_TTL,
)
if not isinstance(listing, list):
return Finding("catalogs", "fbneo recipes", UNKNOWN, local=imported, detail="dats/ listing unavailable")
drifted, tracked = fbneo_blob_drift(local, listing)
if not tracked:
return Finding(
"catalogs",
"fbneo recipes",
STALE,
local=imported,
detail="snapshot records no upstream blobs: re-import with --fetch",
)
shown = ", ".join(n.removeprefix("FinalBurn Neo (ClrMame Pro XML, ").removesuffix(" only).dat") for n in drifted[:4])
return Finding(
"catalogs",
"fbneo recipes",
STALE if drifted else OK,
local=imported,
remote=f"{len(listing)} DATs listed",
detail=(f"{len(drifted)} DAT(s) changed upstream: {shown}" if drifted else "every DAT blob matches"),
)
# --- CI toolchain -----------------------------------------------------------
_PIN_RE = re.compile(r'"?([A-Za-z0-9_.\-]+)((?:[<>=!~]=?[^,"\s]+)(?:,[<>=!~]=?[^,"\s]+)*)?"?')
_USES_RE = re.compile(r"uses:\s*([\w.\-]+/[\w.\-]+)@([0-9a-f]{40})\s*(?:#\s*(\S+))?")
def parse_pip_pins(text: str) -> dict[str, str]:
"""Package -> specifier for every ``pip install`` line of a workflow."""
pins: dict[str, str] = {}
for line in text.splitlines():
stripped = line.strip()
if "pip install" not in stripped:
continue
args = stripped.split("pip install", 1)[1]
for raw in re.findall(r'"[^"]+"|\S+', args):
token = raw.strip('"')
if token.startswith("-") or not token:
continue
match = _PIN_RE.fullmatch(token)
if match:
pins.setdefault(match.group(1).lower(), match.group(2) or "")
return pins
def parse_action_pins(text: str) -> dict[str, tuple[str, str]]:
"""Action -> (pinned sha, commented tag) for every ``uses:`` line."""
return {
repo: (sha, tag or "") for repo, sha, tag in _USES_RE.findall(text)
}
def _version_tuple(version: str) -> tuple[int, ...]:
return tuple(int(p) for p in re.findall(r"\d+", version))
_SPEC_CLAUSE = re.compile(r"([<>=!~]=?)\s*(.+)")
_SPEC_TESTS = {
"==": lambda target, bound: target == bound,
"!=": lambda target, bound: target != bound,
">=": lambda target, bound: target >= bound,
">": lambda target, bound: target > bound,
"<=": lambda target, bound: target <= bound,
"<": lambda target, bound: target < bound,
}
def specifier_allows(spec: str, version: str) -> bool:
"""Whether a pip specifier admits a version. Pre-releases are compared
on their numeric parts, which is enough for the operators pip uses here."""
target = _version_tuple(version)
for clause in filter(None, (c.strip() for c in spec.split(","))):
match = _SPEC_CLAUSE.fullmatch(clause)
test = _SPEC_TESTS.get(match.group(1)) if match else None
if test is None or not test(target, _version_tuple(match.group(2))):
return False
return True
def pypi_latest(name: str) -> str | None:
payload = upstream._http_json(PYPI_URL.format(name=name)) # noqa: SLF001
if not isinstance(payload, dict):
return None
info = payload.get("info") or {}
return str(info.get("version") or "") or None
def check_ci(cache_dir: str, offline: bool) -> list[Finding]:
findings = [_check_installer_pins()]
workflows = sorted((REPO_ROOT / ".github" / "workflows").glob("*.yml"))
pins: dict[str, str] = {}
actions: dict[str, tuple[str, str]] = {}
for path in workflows:
text = path.read_text(encoding="utf-8")
for name, spec in parse_pip_pins(text).items():
pins.setdefault(name, spec)
actions.update(parse_action_pins(text))
if offline:
findings.append(Finding("ci", "workflow pins", SKIPPED, detail="offline"))
return findings
for name, spec in sorted(pins.items()):
try:
latest = pypi_latest(name)
except upstream.UpstreamError as exc:
findings.append(Finding("ci", f"pypi:{name}", UNKNOWN, local=spec, detail=str(exc)))
continue
if latest is None:
findings.append(Finding("ci", f"pypi:{name}", UNKNOWN, local=spec, detail="not on PyPI"))
continue
allowed = specifier_allows(spec, latest)
findings.append(
Finding(
"ci",
f"pypi:{name}",
OK if allowed else STALE,
local=spec or "unpinned",
remote=latest,
detail="latest admitted" if allowed else "latest release outside the pin",
)
)
for action, (sha, tag) in sorted(actions.items()):
repo = upstream.parse_repo(f"https://github.com/{action}")
release = upstream.latest_release(repo, cache_dir, offline)
if release is None:
findings.append(Finding("ci", f"action:{action}", UNKNOWN, local=f"{sha[:8]} {tag}", detail="no release seen"))
continue
latest_sha = upstream.tag_commit(repo, release.tag, cache_dir, offline) or ""
findings.append(
Finding(
"ci",
f"action:{action}",
OK if latest_sha == sha else STALE,
local=f"{sha[:8]} {tag}".strip(),
remote=f"{latest_sha[:8]} {release.tag}",
detail="" if latest_sha == sha else f"{release.tag} released {release.date}",
)
)
return findings
def _check_installer_pins() -> Finding:
"""The bootstraps pin install.py by SHA-256; a stale pin refuses every install."""
digest = hashlib.sha256((REPO_ROOT / "install.py").read_bytes()).hexdigest()
stale = []
for name in INSTALLERS:
text = (REPO_ROOT / name).read_text(encoding="utf-8", errors="replace")
if digest not in text:
stale.append(name)
return Finding(
"ci",
"install.py pin",
STALE if stale else OK,
local=digest[:12],
detail=("pin outdated in " + ", ".join(stale)) if stale else "install.sh and install.ps1 match",
)
# --- profiles ---------------------------------------------------------------
def check_profiles(run: bool, offline: bool) -> list[Finding]:
if not run:
return [
Finding(
"profiles",
"profile_sync",
SKIPPED,
detail="pass --profiles, or run profile_sync.py --all --triage",
)
]
cmd = [
sys.executable,
str(REPO_ROOT / "scripts" / "profile_sync.py"),
"--all",
"--json",
"--check-version",
"--detect-new-files",
"--watch-hashes",
]
if offline:
cmd.append("--offline")
proc = subprocess.run(cmd, cwd=REPO_ROOT, capture_output=True, text=True, check=False)
if proc.returncode not in (0, 1):
tail = proc.stderr.strip().splitlines()[-1:]
return [Finding("profiles", "profile_sync", ERROR, detail=tail[0] if tail else f"exit {proc.returncode}")]
try:
report = json.loads(proc.stdout)
except json.JSONDecodeError as exc:
return [Finding("profiles", "profile_sync", ERROR, detail=f"unreadable report: {exc}")]
return summarize_profile_sync(report)
def summarize_profile_sync(report: object) -> list[Finding]:
"""One row per profile with something to look at, from profile_sync's JSON.
The JSON carries the ref verdicts and the pin against the tip; a profile
whose refs need review is stale, one the forge could not serve is unknown,
and one whose upstream moved while every ref still anchors is reported as
such without being counted as stale.
"""
if not isinstance(report, list):
return [Finding("profiles", "profile_sync", ERROR, detail="report is not a list")]
findings = []
clean = moved = 0
for row in report:
if not isinstance(row, dict):
continue
name = str(row.get("name") or "?")
if row.get("skipped"):
findings.append(Finding("profiles", name, UNKNOWN, detail=str(row["skipped"])))
continue
review = row.get("needs_review") or 0
pin, head = str(row.get("pin") or ""), str(row.get("head") or "")
if review:
findings.append(
Finding(
"profiles",
name,
STALE,
local=pin[:8],
remote=head[:8],
detail=f"{review} refs to review",
)
)
elif pin and head and pin != head:
moved += 1
else:
clean += 1
stale = sum(1 for f in findings if f.status == STALE)
findings.insert(
0,
Finding(
"profiles",
"profile_sync",
STALE if stale else OK,
local=f"{clean} at tip, {moved} anchored past the tip",
remote=f"{stale} to review",
),
)
return findings
# --- reporting --------------------------------------------------------------
def render(findings: list[Finding]) -> str:
lines = []
width = max((len(f.subject) for f in findings), default=10)
width = min(width, 44)
for area in AREAS:
rows = [f for f in findings if f.area == area]
if not rows:
continue
lines.append(f"\n[{area}]")
for f in rows:
cols = f" {f.status:7} {f.subject[:width]:{width}}"
if f.local or f.remote:
cols += f" {f.local[:32]:32} -> {f.remote[:32]}"
if f.detail:
cols += f" {f.detail}"
lines.append(cols.rstrip())
counts = {status: sum(1 for f in findings if f.status == status) for status in STATUSES}
summary = ", ".join(f"{n} {status.lower()}" for status, n in counts.items() if n)
lines.append(f"\nFRESHNESS: {summary}")
return "\n".join(lines)
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__.split("\n\n")[0])
parser.add_argument("--only", help="comma-separated areas: " + ",".join(AREAS))
parser.add_argument("--json", action="store_true", help="machine-readable output")
parser.add_argument("--profiles", action="store_true", help="also run profile_sync (slow)")
parser.add_argument("--offline", action="store_true", help="no network, local ages only")
parser.add_argument("--jobs", type=int, default=4, help="parallel scrapers")
parser.add_argument("--cache-dir", default=".cache/upstream")
parser.add_argument("--native-cache-dir", default=export_native.DEFAULT_CACHE)
parser.add_argument("--truth-dir", default="dist/truth")
parser.add_argument("--platforms-dir", default="platforms")
parser.add_argument("--emulators-dir", default="emulators")
args = parser.parse_args()
areas = tuple(a.strip() for a in args.only.split(",")) if args.only else AREAS
unknown = sorted(set(areas) - set(AREAS))
if unknown:
parser.error(f"unknown area: {', '.join(unknown)}")
platforms_dir = Path(args.platforms_dir)
profiles = load_emulator_profiles(args.emulators_dir, skip_aliases=False)
findings: list[Finding] = []
with tempfile.TemporaryDirectory(prefix="freshness-", dir=REPO_ROOT / "tmp" if (REPO_ROOT / "tmp").is_dir() else None) as scratch:
workdir = Path(scratch)
if "platforms" in areas:
findings.extend(
[Finding("platforms", "scrapers", SKIPPED, detail="offline")]
if args.offline
else check_platforms(platforms_dir, workdir, args.jobs)
)
if "native" in areas:
findings.extend(
check_native(platforms_dir, Path(args.native_cache_dir), Path(args.truth_dir))
)
if "targets" in areas:
findings.extend(check_targets(platforms_dir, workdir, profiles, args.jobs, args.offline))
if "coreinfo" in areas:
findings.extend(check_coreinfo(profiles, args.cache_dir, args.offline))
if "data" in areas:
findings.extend(check_data(args.offline))
if "catalogs" in areas:
findings.extend(check_catalogs(args.cache_dir, args.offline))
if "ci" in areas:
findings.extend(check_ci(args.cache_dir, args.offline))
if "profiles" in areas:
findings.extend(check_profiles(args.profiles, args.offline))
if args.json:
print(json.dumps([asdict(f) for f in findings], indent=2))
else:
print(render(findings))
return 1 if any(f.status in (STALE, ERROR) for f in findings) else 0
if __name__ == "__main__":
sys.exit(main())