mirror of
https://github.com/Abdess/retroarch_system.git
synced 2026-10-10 13:33:24 -05:00
feat: keep cached platform originals in step
This commit is contained in:
1 parent
d50af62cde
commit
b81199ff26
8 files changed
+654
-68
No files matched your search
@@ -12,6 +12,11 @@ into a throwaway copy, then diffed against the committed file. The diff
|
|||||||
decides, not a version string: a scraper that pins a new tag but produces
|
decides, not a version string: a scraper that pins a new tag but produces
|
||||||
the same entries is reported as a version change and nothing else.
|
the same entries is reported as a version change and nothing else.
|
||||||
|
|
||||||
|
The native export patches each platform's own file, kept in a local cache.
|
||||||
|
That cache is checked against the platform file it was transcribed with: an
|
||||||
|
original cached before the last rescrape, or under another pin, is no longer
|
||||||
|
the file our data describes.
|
||||||
|
|
||||||
Emulator profiles are covered by profile_sync, which is slow (one API pass
|
Emulator profiles are covered by profile_sync, which is slow (one API pass
|
||||||
per profile). ``--profiles`` runs it here and folds its verdict in.
|
per profile). ``--profiles`` runs it here and folds its verdict in.
|
||||||
|
|
||||||
@@ -43,6 +48,7 @@ from pathlib import Path
|
|||||||
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
||||||
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
|
||||||
|
|
||||||
|
import export_native # noqa: E402
|
||||||
import refresh_data_dirs # noqa: E402
|
import refresh_data_dirs # noqa: E402
|
||||||
import upstream # noqa: E402
|
import upstream # noqa: E402
|
||||||
from common import ( # noqa: E402
|
from common import ( # noqa: E402
|
||||||
@@ -50,6 +56,8 @@ from common import ( # noqa: E402
|
|||||||
load_emulator_profiles,
|
load_emulator_profiles,
|
||||||
yaml_load,
|
yaml_load,
|
||||||
)
|
)
|
||||||
|
from exporter import discover_exporters # noqa: E402
|
||||||
|
from exporter.baseline import build_native_model # noqa: E402
|
||||||
from scripts.scraper.redump_dat_scraper import fetch_snapshot # noqa: E402
|
from scripts.scraper.redump_dat_scraper import fetch_snapshot # noqa: E402
|
||||||
from scripts.scraper.targets import discover_target_scrapers # noqa: E402
|
from scripts.scraper.targets import discover_target_scrapers # noqa: E402
|
||||||
|
|
||||||
@@ -66,7 +74,7 @@ INSTALLERS = ("install.sh", "install.ps1")
|
|||||||
|
|
||||||
SHOWN_ITEMS = 4 # items named in a diff summary before ", ..."
|
SHOWN_ITEMS = 4 # items named in a diff summary before ", ..."
|
||||||
SHOWN_NAMES = 12 # core names named in an unresolved row before ", ..."
|
SHOWN_NAMES = 12 # core names named in an unresolved row before ", ..."
|
||||||
AREAS = ("platforms", "targets", "coreinfo", "data", "catalogs", "ci", "profiles")
|
AREAS = ("platforms", "native", "targets", "coreinfo", "data", "catalogs", "ci", "profiles")
|
||||||
OK, STALE, UNKNOWN, ERROR, SKIPPED = "OK", "STALE", "UNKNOWN", "ERROR", "SKIPPED"
|
OK, STALE, UNKNOWN, ERROR, SKIPPED = "OK", "STALE", "UNKNOWN", "ERROR", "SKIPPED"
|
||||||
# Every status a finding may carry, in the order the summary counts them.
|
# Every status a finding may carry, in the order the summary counts them.
|
||||||
STATUSES = (STALE, UNKNOWN, ERROR, SKIPPED, OK)
|
STATUSES = (STALE, UNKNOWN, ERROR, SKIPPED, OK)
|
||||||
@@ -279,6 +287,80 @@ def check_platforms(platforms_dir: Path, workdir: Path, jobs: int) -> list[Findi
|
|||||||
return sorted(findings, key=lambda f: f.subject)
|
return sorted(findings, key=lambda f: f.subject)
|
||||||
|
|
||||||
|
|
||||||
|
# --- native originals --------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
def native_cache_state(
|
||||||
|
wanted: dict[str, str],
|
||||||
|
root: Path,
|
||||||
|
recorded: dict[str, str],
|
||||||
|
index: Path,
|
||||||
|
transcribed_at: float,
|
||||||
|
) -> str:
|
||||||
|
"""Why a platform's cached originals cannot be patched, '' when they can.
|
||||||
|
|
||||||
|
The export corrects the file our data was transcribed from. A cached
|
||||||
|
original is that file only if it came from the URL the platform file
|
||||||
|
names today and was fetched no earlier than the platform file was last
|
||||||
|
rewritten: a branch URL keeps its name while its content moves.
|
||||||
|
"""
|
||||||
|
for relative, url in sorted(wanted.items()):
|
||||||
|
path = root / relative
|
||||||
|
if not path.is_file():
|
||||||
|
return f"{relative} not cached"
|
||||||
|
known = recorded.get(export_native.source_key(path, index))
|
||||||
|
if known is None:
|
||||||
|
return f"{relative} cached from an unrecorded URL"
|
||||||
|
if known != url:
|
||||||
|
return f"{relative} cached from another revision"
|
||||||
|
if path.stat().st_mtime < transcribed_at:
|
||||||
|
return f"{relative} cached before the platform file was rewritten"
|
||||||
|
return ""
|
||||||
|
|
||||||
|
|
||||||
|
def _transcribed_at(platforms_dir: Path, platform: str) -> float:
|
||||||
|
"""Last rewrite of the platform file or of any file it inherits."""
|
||||||
|
newest = 0.0
|
||||||
|
seen: set[str] = set()
|
||||||
|
name: str | None = platform
|
||||||
|
while name and name not in seen:
|
||||||
|
seen.add(name)
|
||||||
|
path = platforms_dir / f"{name}.yml"
|
||||||
|
if not path.is_file():
|
||||||
|
break
|
||||||
|
newest = max(newest, path.stat().st_mtime)
|
||||||
|
with path.open(encoding="utf-8") as fh:
|
||||||
|
name = (yaml_load(fh) or {}).get("inherits")
|
||||||
|
return newest
|
||||||
|
|
||||||
|
|
||||||
|
def check_native(platforms_dir: Path, cache_dir: Path, truth_dir: Path) -> list[Finding]:
|
||||||
|
exporters = discover_exporters()
|
||||||
|
index = cache_dir / export_native.SOURCES_INDEX
|
||||||
|
recorded = export_native.load_sources(cache_dir)
|
||||||
|
findings: list[Finding] = []
|
||||||
|
for name in list_registered_platforms(str(platforms_dir), include_archived=True):
|
||||||
|
exporter_class = exporters.get(name)
|
||||||
|
if not exporter_class:
|
||||||
|
continue
|
||||||
|
truth, scraped = export_native.load_inputs(name, truth_dir, str(platforms_dir))
|
||||||
|
systems, _ = build_native_model(truth or {}, scraped)
|
||||||
|
wanted = export_native.wanted_sources(exporter_class(), systems, scraped)
|
||||||
|
reason = native_cache_state(
|
||||||
|
wanted, cache_dir / name, recorded, index, _transcribed_at(platforms_dir, name)
|
||||||
|
)
|
||||||
|
findings.append(
|
||||||
|
Finding(
|
||||||
|
"native",
|
||||||
|
name,
|
||||||
|
STALE if reason else OK,
|
||||||
|
local=f"{len(wanted)} file(s)",
|
||||||
|
detail=f"{reason}, export_native.py --refresh-cache fetches it" if reason else "",
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return sorted(findings, key=lambda f: f.subject)
|
||||||
|
|
||||||
|
|
||||||
# --- targets ----------------------------------------------------------------
|
# --- targets ----------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
@@ -959,6 +1041,8 @@ def main() -> int:
|
|||||||
parser.add_argument("--offline", action="store_true", help="no network, local ages only")
|
parser.add_argument("--offline", action="store_true", help="no network, local ages only")
|
||||||
parser.add_argument("--jobs", type=int, default=4, help="parallel scrapers")
|
parser.add_argument("--jobs", type=int, default=4, help="parallel scrapers")
|
||||||
parser.add_argument("--cache-dir", default=".cache/upstream")
|
parser.add_argument("--cache-dir", default=".cache/upstream")
|
||||||
|
parser.add_argument("--native-cache-dir", default=export_native.DEFAULT_CACHE)
|
||||||
|
parser.add_argument("--truth-dir", default="dist/truth")
|
||||||
parser.add_argument("--platforms-dir", default="platforms")
|
parser.add_argument("--platforms-dir", default="platforms")
|
||||||
parser.add_argument("--emulators-dir", default="emulators")
|
parser.add_argument("--emulators-dir", default="emulators")
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
@@ -979,6 +1063,10 @@ def main() -> int:
|
|||||||
if args.offline
|
if args.offline
|
||||||
else check_platforms(platforms_dir, workdir, args.jobs)
|
else check_platforms(platforms_dir, workdir, args.jobs)
|
||||||
)
|
)
|
||||||
|
if "native" in areas:
|
||||||
|
findings.extend(
|
||||||
|
check_native(platforms_dir, Path(args.native_cache_dir), Path(args.truth_dir))
|
||||||
|
)
|
||||||
if "targets" in areas:
|
if "targets" in areas:
|
||||||
findings.extend(check_targets(platforms_dir, workdir, profiles, args.jobs, args.offline))
|
findings.extend(check_targets(platforms_dir, workdir, profiles, args.jobs, args.offline))
|
||||||
if "coreinfo" in areas:
|
if "coreinfo" in areas:
|
||||||
|
|||||||
+124
-36
@@ -10,6 +10,7 @@ a platform ships keeps working.
|
|||||||
Usage:
|
Usage:
|
||||||
python scripts/export_native.py --all --fetch
|
python scripts/export_native.py --all --fetch
|
||||||
python scripts/export_native.py --platform recalbox --upstream-dir up/
|
python scripts/export_native.py --platform recalbox --upstream-dir up/
|
||||||
|
python scripts/export_native.py --platform retrobat --refresh-cache
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
@@ -28,6 +29,7 @@ from exporter import discover_exporters
|
|||||||
from exporter.baseline import build_native_model
|
from exporter.baseline import build_native_model
|
||||||
|
|
||||||
DEFAULT_CACHE = ".cache/upstream-native"
|
DEFAULT_CACHE = ".cache/upstream-native"
|
||||||
|
SOURCES_INDEX = ".sources.json"
|
||||||
_USER_AGENT = "retrobios-exporter/1.0"
|
_USER_AGENT = "retrobios-exporter/1.0"
|
||||||
_MAX_BYTES = 64 * 1024 * 1024
|
_MAX_BYTES = 64 * 1024 * 1024
|
||||||
|
|
||||||
@@ -41,7 +43,29 @@ def _load_sources(index: Path | None) -> dict[str, str]:
|
|||||||
return {}
|
return {}
|
||||||
|
|
||||||
|
|
||||||
def fetch(url: str, destination: Path, index: Path | None = None) -> bytes:
|
def source_key(destination: Path, index: Path | None) -> str:
|
||||||
|
"""How the index names a cached file: its path below the cache root.
|
||||||
|
|
||||||
|
A key that repeated the cache directory as it was typed changed with the
|
||||||
|
working directory, and a file recorded from one place looked unrecorded
|
||||||
|
from another.
|
||||||
|
"""
|
||||||
|
if index is not None:
|
||||||
|
try:
|
||||||
|
return destination.resolve().relative_to(index.parent.resolve()).as_posix()
|
||||||
|
except ValueError:
|
||||||
|
pass
|
||||||
|
return str(destination)
|
||||||
|
|
||||||
|
|
||||||
|
def load_sources(upstream_dir: Path) -> dict[str, str]:
|
||||||
|
"""Cached file -> URL it was fetched from, for a whole cache directory."""
|
||||||
|
return _load_sources(upstream_dir / SOURCES_INDEX)
|
||||||
|
|
||||||
|
|
||||||
|
def fetch(
|
||||||
|
url: str, destination: Path, index: Path | None = None, refresh: bool = False
|
||||||
|
) -> bytes:
|
||||||
"""Download an original once, then read it from the cache.
|
"""Download an original once, then read it from the cache.
|
||||||
|
|
||||||
The cache path carries the file's own name and nothing of the revision it
|
The cache path carries the file's own name and nothing of the revision it
|
||||||
@@ -50,10 +74,15 @@ def fetch(url: str, destination: Path, index: Path | None = None) -> bytes:
|
|||||||
URL that produced each cached file is recorded beside the cache, and a
|
URL that produced each cached file is recorded beside the cache, and a
|
||||||
different URL refetches. A cache written before this index existed keeps
|
different URL refetches. A cache written before this index existed keeps
|
||||||
being served: nothing recorded means nothing contradicted.
|
being served: nothing recorded means nothing contradicted.
|
||||||
|
|
||||||
|
A branch URL names no revision, so the same URL serves new bytes after
|
||||||
|
the platform moves. refresh downloads again whatever the cache holds:
|
||||||
|
it is what a rescrape calls, so the original and its transcription
|
||||||
|
describe the same moment.
|
||||||
"""
|
"""
|
||||||
recorded = _load_sources(index)
|
recorded = _load_sources(index)
|
||||||
key = str(destination)
|
key = source_key(destination, index)
|
||||||
if destination.exists() and recorded.get(key, url) == url:
|
if not refresh and destination.exists() and recorded.get(key, url) == url:
|
||||||
return destination.read_bytes()
|
return destination.read_bytes()
|
||||||
request = urllib.request.Request(url, headers={"User-Agent": _USER_AGENT})
|
request = urllib.request.Request(url, headers={"User-Agent": _USER_AGENT})
|
||||||
with urllib.request.urlopen(request, timeout=60) as response:
|
with urllib.request.urlopen(request, timeout=60) as response:
|
||||||
@@ -94,14 +123,10 @@ def _raw_url(url: str) -> str:
|
|||||||
return url
|
return url
|
||||||
|
|
||||||
|
|
||||||
def collect_originals(
|
def wanted_sources(
|
||||||
exporter: object,
|
exporter: object, systems: dict, scraped: dict | None = None
|
||||||
systems: dict,
|
) -> dict[str, str]:
|
||||||
upstream_dir: Path,
|
"""Cache-relative name -> URL of every file the exporter patches."""
|
||||||
allow_fetch: bool,
|
|
||||||
scraped: dict | None = None,
|
|
||||||
) -> tuple[dict[str, str], list[str]]:
|
|
||||||
"""Gather the platform's own files, from disk or from upstream."""
|
|
||||||
wanted = dict(exporter.native_sources())
|
wanted = dict(exporter.native_sources())
|
||||||
components = getattr(exporter, "components", None)
|
components = getattr(exporter, "components", None)
|
||||||
if callable(components):
|
if callable(components):
|
||||||
@@ -113,21 +138,39 @@ def collect_originals(
|
|||||||
base = pinned_base(wanted, scraped)
|
base = pinned_base(wanted, scraped)
|
||||||
if base:
|
if base:
|
||||||
wanted = {relative: base + relative for relative in wanted}
|
wanted = {relative: base + relative for relative in wanted}
|
||||||
|
return wanted
|
||||||
|
|
||||||
|
|
||||||
|
def collect_originals(
|
||||||
|
exporter: object,
|
||||||
|
systems: dict,
|
||||||
|
upstream_dir: Path,
|
||||||
|
allow_fetch: bool,
|
||||||
|
scraped: dict | None = None,
|
||||||
|
refresh: bool = False,
|
||||||
|
) -> tuple[dict[str, str], list[str]]:
|
||||||
|
"""Gather the platform's own files, from disk or from upstream.
|
||||||
|
|
||||||
|
With fetching on, an existing file still goes through fetch(): reading
|
||||||
|
it straight from disk skipped the recorded URL, so a file cached under
|
||||||
|
one pin kept being patched after the platform YAML moved to the next.
|
||||||
|
"""
|
||||||
|
wanted = wanted_sources(exporter, systems, scraped)
|
||||||
root = upstream_dir / exporter.platform_name()
|
root = upstream_dir / exporter.platform_name()
|
||||||
|
index = upstream_dir / SOURCES_INDEX
|
||||||
originals: dict[str, str] = {}
|
originals: dict[str, str] = {}
|
||||||
missing: list[str] = []
|
missing: list[str] = []
|
||||||
for relative, url in wanted.items():
|
for relative, url in wanted.items():
|
||||||
path = root / relative
|
path = root / relative
|
||||||
payload: bytes | None = None
|
payload: bytes | None = None
|
||||||
if path.exists():
|
if allow_fetch:
|
||||||
payload = path.read_bytes()
|
|
||||||
elif allow_fetch:
|
|
||||||
try:
|
try:
|
||||||
payload = fetch(url, path, upstream_dir / ".sources.json")
|
payload = fetch(url, path, index, refresh)
|
||||||
except (urllib.error.URLError, urllib.error.HTTPError, OSError) as exc:
|
except (urllib.error.URLError, urllib.error.HTTPError, OSError) as exc:
|
||||||
missing.append(f"{relative}: {exc}")
|
missing.append(f"{relative}: {exc}")
|
||||||
continue
|
continue
|
||||||
|
elif path.exists():
|
||||||
|
payload = path.read_bytes()
|
||||||
else:
|
else:
|
||||||
missing.append(f"{relative}: absent and fetching is off")
|
missing.append(f"{relative}: absent and fetching is off")
|
||||||
continue
|
continue
|
||||||
@@ -141,6 +184,42 @@ def collect_originals(
|
|||||||
return originals, missing
|
return originals, missing
|
||||||
|
|
||||||
|
|
||||||
|
def load_inputs(
|
||||||
|
platform: str, truth_dir: Path, platforms_dir: str
|
||||||
|
) -> tuple[dict | None, dict | None]:
|
||||||
|
"""The truth and the scraped config of a platform, None where absent."""
|
||||||
|
truth_file = truth_dir / f"{platform}.yml"
|
||||||
|
truth: dict | None = None
|
||||||
|
if truth_file.exists():
|
||||||
|
with open(truth_file) as handle:
|
||||||
|
truth = yaml_load(handle) or {}
|
||||||
|
try:
|
||||||
|
scraped = load_platform_config(platform, platforms_dir)
|
||||||
|
except (FileNotFoundError, OSError):
|
||||||
|
scraped = None
|
||||||
|
return truth, scraped
|
||||||
|
|
||||||
|
|
||||||
|
def refresh_cache(
|
||||||
|
platform: str,
|
||||||
|
exporter_class: type,
|
||||||
|
truth_dir: Path,
|
||||||
|
platforms_dir: str,
|
||||||
|
upstream_dir: Path,
|
||||||
|
) -> tuple[bool, list[str]]:
|
||||||
|
"""Download a platform's own files again. Returns (ok, messages)."""
|
||||||
|
truth, scraped = load_inputs(platform, truth_dir, platforms_dir)
|
||||||
|
systems, _ = build_native_model(truth or {}, scraped)
|
||||||
|
exporter = exporter_class()
|
||||||
|
wanted = wanted_sources(exporter, systems, scraped)
|
||||||
|
_, missing = collect_originals(
|
||||||
|
exporter, systems, upstream_dir, True, scraped, refresh=True
|
||||||
|
)
|
||||||
|
if missing:
|
||||||
|
return False, missing
|
||||||
|
return True, [f"{len(wanted)} file(s) fetched"]
|
||||||
|
|
||||||
|
|
||||||
def export_platform(
|
def export_platform(
|
||||||
platform: str,
|
platform: str,
|
||||||
exporter_class: type,
|
exporter_class: type,
|
||||||
@@ -153,19 +232,11 @@ def export_platform(
|
|||||||
"""Write one platform's corrected file. Returns (ok, messages)."""
|
"""Write one platform's corrected file. Returns (ok, messages)."""
|
||||||
messages: list[str] = []
|
messages: list[str] = []
|
||||||
|
|
||||||
truth_file = truth_dir / f"{platform}.yml"
|
truth, scraped = load_inputs(platform, truth_dir, platforms_dir)
|
||||||
truth: dict = {}
|
if truth is None:
|
||||||
if truth_file.exists():
|
truth = {}
|
||||||
with open(truth_file) as handle:
|
|
||||||
truth = yaml_load(handle) or {}
|
|
||||||
else:
|
|
||||||
messages.append(f"no truth for {platform}, only the platform's own data")
|
messages.append(f"no truth for {platform}, only the platform's own data")
|
||||||
|
|
||||||
try:
|
|
||||||
scraped = load_platform_config(platform, platforms_dir)
|
|
||||||
except (FileNotFoundError, OSError):
|
|
||||||
scraped = None
|
|
||||||
|
|
||||||
systems, report = build_native_model(truth, scraped)
|
systems, report = build_native_model(truth, scraped)
|
||||||
if not systems:
|
if not systems:
|
||||||
return False, ["nothing to write: neither the platform nor the truth has data"]
|
return False, ["nothing to write: neither the platform nor the truth has data"]
|
||||||
@@ -253,6 +324,7 @@ def run(
|
|||||||
platforms_dir: str,
|
platforms_dir: str,
|
||||||
upstream_dir: str,
|
upstream_dir: str,
|
||||||
allow_fetch: bool,
|
allow_fetch: bool,
|
||||||
|
refresh_only: bool = False,
|
||||||
) -> int:
|
) -> int:
|
||||||
exporters = discover_exporters()
|
exporters = discover_exporters()
|
||||||
failures = 0
|
failures = 0
|
||||||
@@ -265,15 +337,24 @@ def run(
|
|||||||
print(f" SKIP {platform}: no exporter")
|
print(f" SKIP {platform}: no exporter")
|
||||||
continue
|
continue
|
||||||
|
|
||||||
ok, messages = export_platform(
|
if refresh_only:
|
||||||
platform,
|
ok, messages = refresh_cache(
|
||||||
exporter_class,
|
platform,
|
||||||
Path(truth_dir),
|
exporter_class,
|
||||||
Path(output_dir),
|
Path(truth_dir),
|
||||||
platforms_dir,
|
platforms_dir,
|
||||||
Path(upstream_dir),
|
Path(upstream_dir),
|
||||||
allow_fetch,
|
)
|
||||||
)
|
else:
|
||||||
|
ok, messages = export_platform(
|
||||||
|
platform,
|
||||||
|
exporter_class,
|
||||||
|
Path(truth_dir),
|
||||||
|
Path(output_dir),
|
||||||
|
platforms_dir,
|
||||||
|
Path(upstream_dir),
|
||||||
|
allow_fetch,
|
||||||
|
)
|
||||||
label = "OK " if ok else "FAIL"
|
label = "OK " if ok else "FAIL"
|
||||||
print(f" {label} {platform}")
|
print(f" {label} {platform}")
|
||||||
for message in messages:
|
for message in messages:
|
||||||
@@ -306,7 +387,13 @@ def main() -> None:
|
|||||||
parser.add_argument(
|
parser.add_argument(
|
||||||
"--fetch",
|
"--fetch",
|
||||||
action="store_true",
|
action="store_true",
|
||||||
help="download a platform's file when it is not in the cache",
|
help="download a platform's file when it is not in the cache, "
|
||||||
|
"or when the cache came from another URL",
|
||||||
|
)
|
||||||
|
parser.add_argument(
|
||||||
|
"--refresh-cache",
|
||||||
|
action="store_true",
|
||||||
|
help="download every platform file again and write no export",
|
||||||
)
|
)
|
||||||
args = parser.parse_args()
|
args = parser.parse_args()
|
||||||
|
|
||||||
@@ -326,6 +413,7 @@ def main() -> None:
|
|||||||
args.platforms_dir,
|
args.platforms_dir,
|
||||||
args.upstream_dir,
|
args.upstream_dir,
|
||||||
args.fetch,
|
args.fetch,
|
||||||
|
args.refresh_cache,
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|||||||
+61
-18
@@ -10,6 +10,10 @@ committed.
|
|||||||
Mapping (stale -> command):
|
Mapping (stale -> command):
|
||||||
platforms/<name> python -m scripts.scraper.<module>_scraper
|
platforms/<name> python -m scripts.scraper.<module>_scraper
|
||||||
-o platforms/<name>.yml
|
-o platforms/<name>.yml
|
||||||
|
then the native refresh below: the original
|
||||||
|
the export patches must be the one just scraped
|
||||||
|
native/<name> python scripts/export_native.py --platform <name>
|
||||||
|
--refresh-cache
|
||||||
targets/<name> python -m scripts.scraper.targets.<module>
|
targets/<name> python -m scripts.scraper.targets.<module>
|
||||||
-o platforms/targets/<name>.yml
|
-o platforms/targets/<name>.yml
|
||||||
data/<key> python scripts/refresh_data_dirs.py --key <key>
|
data/<key> python scripts/refresh_data_dirs.py --key <key>
|
||||||
@@ -47,7 +51,7 @@ LOG_DIR = REPO_ROOT / "tmp" / "refresh_stale"
|
|||||||
CHECK_FRESHNESS = REPO_ROOT / "scripts" / "check_freshness.py"
|
CHECK_FRESHNESS = REPO_ROOT / "scripts" / "check_freshness.py"
|
||||||
PLATFORMS_REGISTRY = REPO_ROOT / "platforms" / "_registry.yml"
|
PLATFORMS_REGISTRY = REPO_ROOT / "platforms" / "_registry.yml"
|
||||||
|
|
||||||
AUTO_AREAS = ("platforms", "targets", "data", "catalogs")
|
AUTO_AREAS = ("platforms", "native", "targets", "data", "catalogs")
|
||||||
MANUAL_AREAS = ("coreinfo", "ci", "profiles")
|
MANUAL_AREAS = ("coreinfo", "ci", "profiles")
|
||||||
|
|
||||||
JOB_TIMEOUT = 1800 # 30 minutes per refresher; scrapers rarely exceed 10
|
JOB_TIMEOUT = 1800 # 30 minutes per refresher; scrapers rarely exceed 10
|
||||||
@@ -61,6 +65,8 @@ class Job:
|
|||||||
subject: str
|
subject: str
|
||||||
command: list[str]
|
command: list[str]
|
||||||
log_path: Path
|
log_path: Path
|
||||||
|
# Commands run after `command`, in order, only while each one succeeds.
|
||||||
|
then: tuple[tuple[str, ...], ...] = ()
|
||||||
|
|
||||||
|
|
||||||
@dataclass(frozen=True)
|
@dataclass(frozen=True)
|
||||||
@@ -122,6 +128,15 @@ def plan_jobs(
|
|||||||
seen: set[tuple[str, str]] = set()
|
seen: set[tuple[str, str]] = set()
|
||||||
surfaced: list[dict] = []
|
surfaced: list[dict] = []
|
||||||
ignored: list[dict] = []
|
ignored: list[dict] = []
|
||||||
|
# A platform about to be rescraped refreshes its cached original itself,
|
||||||
|
# after the scrape: a separate native job would race the scraper.
|
||||||
|
rescraped = {
|
||||||
|
str(f.get("subject") or "")
|
||||||
|
for f in findings
|
||||||
|
if f.get("status") == "STALE"
|
||||||
|
and f.get("area") == "platforms"
|
||||||
|
and _command_for("platforms", str(f.get("subject") or ""), registry)
|
||||||
|
}
|
||||||
|
|
||||||
for finding in findings:
|
for finding in findings:
|
||||||
status = finding.get("status")
|
status = finding.get("status")
|
||||||
@@ -135,6 +150,10 @@ def plan_jobs(
|
|||||||
ignored.append(finding)
|
ignored.append(finding)
|
||||||
continue
|
continue
|
||||||
|
|
||||||
|
if area == "native" and subject in rescraped:
|
||||||
|
ignored.append(finding)
|
||||||
|
continue
|
||||||
|
|
||||||
command = _command_for(area, subject, registry)
|
command = _command_for(area, subject, registry)
|
||||||
if command is None:
|
if command is None:
|
||||||
surfaced.append(finding)
|
surfaced.append(finding)
|
||||||
@@ -148,15 +167,33 @@ def plan_jobs(
|
|||||||
seen.add(key)
|
seen.add(key)
|
||||||
|
|
||||||
log = LOG_DIR / f"{area}__{_slug(subject)}.log"
|
log = LOG_DIR / f"{area}__{_slug(subject)}.log"
|
||||||
jobs.append(Job(area=area, subject=subject, command=command, log_path=log))
|
then: tuple[tuple[str, ...], ...] = ()
|
||||||
|
if area == "platforms":
|
||||||
|
then = (tuple(_native_refresh(subject)),)
|
||||||
|
jobs.append(
|
||||||
|
Job(area=area, subject=subject, command=command, log_path=log, then=then)
|
||||||
|
)
|
||||||
|
|
||||||
return jobs, surfaced, ignored
|
return jobs, surfaced, ignored
|
||||||
|
|
||||||
|
|
||||||
|
def _native_refresh(platform: str) -> list[str]:
|
||||||
|
return [
|
||||||
|
sys.executable,
|
||||||
|
"scripts/export_native.py",
|
||||||
|
"--platform",
|
||||||
|
platform,
|
||||||
|
"--refresh-cache",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
def _command_for(
|
def _command_for(
|
||||||
area: str, subject: str, registry: dict[str, dict]
|
area: str, subject: str, registry: dict[str, dict]
|
||||||
) -> list[str] | None:
|
) -> list[str] | None:
|
||||||
"""Map a stale finding to its refresh command, or None if manual."""
|
"""Map a stale finding to its refresh command, or None if manual."""
|
||||||
|
if area == "native":
|
||||||
|
return _native_refresh(subject)
|
||||||
|
|
||||||
if area == "platforms":
|
if area == "platforms":
|
||||||
entry = registry.get(subject) or {}
|
entry = registry.get(subject) or {}
|
||||||
scraper = entry.get("scraper")
|
scraper = entry.get("scraper")
|
||||||
@@ -250,23 +287,27 @@ def run_job(job: Job, env: dict[str, str]) -> JobResult:
|
|||||||
"""Execute one refresh command and collect its output."""
|
"""Execute one refresh command and collect its output."""
|
||||||
start = time.monotonic()
|
start = time.monotonic()
|
||||||
job.log_path.parent.mkdir(parents=True, exist_ok=True)
|
job.log_path.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
returncode = 0
|
||||||
with job.log_path.open("w", encoding="utf-8") as log:
|
with job.log_path.open("w", encoding="utf-8") as log:
|
||||||
log.write(f"$ {' '.join(job.command)}\n")
|
for command in (job.command, *job.then):
|
||||||
log.flush()
|
log.write(f"$ {' '.join(command)}\n")
|
||||||
try:
|
log.flush()
|
||||||
proc = subprocess.run(
|
try:
|
||||||
job.command,
|
proc = subprocess.run(
|
||||||
cwd=REPO_ROOT,
|
list(command),
|
||||||
stdout=log,
|
cwd=REPO_ROOT,
|
||||||
stderr=subprocess.STDOUT,
|
stdout=log,
|
||||||
env=env,
|
stderr=subprocess.STDOUT,
|
||||||
timeout=JOB_TIMEOUT,
|
env=env,
|
||||||
check=False,
|
timeout=JOB_TIMEOUT,
|
||||||
)
|
check=False,
|
||||||
returncode = proc.returncode
|
)
|
||||||
except subprocess.TimeoutExpired:
|
returncode = proc.returncode
|
||||||
log.write(f"\nTIMEOUT after {JOB_TIMEOUT}s\n")
|
except subprocess.TimeoutExpired:
|
||||||
returncode = 124
|
log.write(f"\nTIMEOUT after {JOB_TIMEOUT}s\n")
|
||||||
|
returncode = 124
|
||||||
|
if returncode != 0:
|
||||||
|
break
|
||||||
duration = time.monotonic() - start
|
duration = time.monotonic() - start
|
||||||
tail = _log_tail(job.log_path)
|
tail = _log_tail(job.log_path)
|
||||||
return JobResult(job=job, returncode=returncode, duration=duration, tail=tail)
|
return JobResult(job=job, returncode=returncode, duration=duration, tail=tail)
|
||||||
@@ -388,6 +429,8 @@ def main() -> int:
|
|||||||
f"{len(ignored)} already clean/skipped")
|
f"{len(ignored)} already clean/skipped")
|
||||||
for job in sorted(jobs, key=lambda j: (j.area, j.subject)):
|
for job in sorted(jobs, key=lambda j: (j.area, j.subject)):
|
||||||
print(f" {job.area:10} {job.subject:28} {' '.join(job.command)}")
|
print(f" {job.area:10} {job.subject:28} {' '.join(job.command)}")
|
||||||
|
for command in job.then:
|
||||||
|
print(f" {'':10} {'':28} then {' '.join(command)}")
|
||||||
if surfaced:
|
if surfaced:
|
||||||
print("\n[manual review needed]")
|
print("\n[manual review needed]")
|
||||||
for f in surfaced:
|
for f in surfaced:
|
||||||
|
|||||||
@@ -8,7 +8,9 @@ report is folded, since each of those decides whether a row says STALE.
|
|||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import os
|
||||||
import sys
|
import sys
|
||||||
|
import tempfile
|
||||||
import unittest
|
import unittest
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
|
|
||||||
@@ -19,6 +21,72 @@ import check_freshness as cf # noqa: E402
|
|||||||
from common import yaml_load # noqa: E402
|
from common import yaml_load # noqa: E402
|
||||||
|
|
||||||
|
|
||||||
|
class NativeCacheTests(unittest.TestCase):
|
||||||
|
"""A cached original is patched only if it is the file we transcribed."""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
self._tmp = tempfile.TemporaryDirectory()
|
||||||
|
self.addCleanup(self._tmp.cleanup)
|
||||||
|
self.cache = Path(self._tmp.name)
|
||||||
|
self.index = self.cache / cf.export_native.SOURCES_INDEX
|
||||||
|
self.root = self.cache / "plat"
|
||||||
|
self.root.mkdir()
|
||||||
|
self.wanted = {"list.txt": "https://example.invalid/tag-2/list.txt"}
|
||||||
|
|
||||||
|
def _cached(self, mtime: float) -> None:
|
||||||
|
path = self.root / "list.txt"
|
||||||
|
path.write_text("x", encoding="utf-8")
|
||||||
|
os.utime(path, (mtime, mtime))
|
||||||
|
|
||||||
|
def _state(self, recorded: dict, transcribed_at: float) -> str:
|
||||||
|
return cf.native_cache_state(
|
||||||
|
self.wanted, self.root, recorded, self.index, transcribed_at
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_a_file_fetched_after_the_scrape_from_the_named_url_is_fresh(self):
|
||||||
|
self._cached(200.0)
|
||||||
|
recorded = {"plat/list.txt": self.wanted["list.txt"]}
|
||||||
|
self.assertEqual(self._state(recorded, 100.0), "")
|
||||||
|
|
||||||
|
def test_an_absent_file_is_named(self):
|
||||||
|
self.assertEqual(self._state({}, 0.0), "list.txt not cached")
|
||||||
|
|
||||||
|
def test_a_file_with_no_recorded_url_proves_nothing(self):
|
||||||
|
self._cached(200.0)
|
||||||
|
self.assertEqual(self._state({}, 100.0), "list.txt cached from an unrecorded URL")
|
||||||
|
|
||||||
|
def test_a_file_from_the_previous_pin_is_stale(self):
|
||||||
|
self._cached(200.0)
|
||||||
|
recorded = {"plat/list.txt": "https://example.invalid/tag-1/list.txt"}
|
||||||
|
self.assertEqual(self._state(recorded, 100.0), "list.txt cached from another revision")
|
||||||
|
|
||||||
|
def test_a_file_older_than_the_rescrape_is_stale(self):
|
||||||
|
"""A branch URL does not change when its content does."""
|
||||||
|
self._cached(100.0)
|
||||||
|
recorded = {"plat/list.txt": self.wanted["list.txt"]}
|
||||||
|
self.assertEqual(
|
||||||
|
self._state(recorded, 200.0),
|
||||||
|
"list.txt cached before the platform file was rewritten",
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_the_transcription_date_follows_inheritance(self):
|
||||||
|
platforms = self.cache / "platforms"
|
||||||
|
platforms.mkdir()
|
||||||
|
(platforms / "parent.yml").write_text("platform: P\n", encoding="utf-8")
|
||||||
|
(platforms / "child.yml").write_text("inherits: parent\n", encoding="utf-8")
|
||||||
|
os.utime(platforms / "parent.yml", (300.0, 300.0))
|
||||||
|
os.utime(platforms / "child.yml", (100.0, 100.0))
|
||||||
|
self.assertEqual(cf._transcribed_at(platforms, "child"), 300.0)
|
||||||
|
self.assertEqual(cf._transcribed_at(platforms, "parent"), 300.0)
|
||||||
|
|
||||||
|
def test_every_registered_platform_with_an_exporter_gets_a_row(self):
|
||||||
|
rows = cf.check_native(REPO_ROOT / "platforms", self.cache, self.cache / "truth")
|
||||||
|
self.assertTrue(rows)
|
||||||
|
self.assertEqual({r.area for r in rows}, {"native"})
|
||||||
|
self.assertEqual({r.status for r in rows}, {cf.STALE})
|
||||||
|
self.assertIn("batocera", {r.subject for r in rows})
|
||||||
|
|
||||||
|
|
||||||
class PlatformDiffTests(unittest.TestCase):
|
class PlatformDiffTests(unittest.TestCase):
|
||||||
def _platform(self, version="1", files=None, cores=None):
|
def _platform(self, version="1", files=None, cores=None):
|
||||||
return {
|
return {
|
||||||
|
|||||||
+82
-1
@@ -15,6 +15,7 @@ and skip when it has not been populated (python scripts/export_native.py
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import json
|
import json
|
||||||
|
import os
|
||||||
import re
|
import re
|
||||||
import subprocess
|
import subprocess
|
||||||
import sys
|
import sys
|
||||||
@@ -33,7 +34,13 @@ from common import ( # noqa: E402
|
|||||||
)
|
)
|
||||||
from exporter import discover_exporters # noqa: E402
|
from exporter import discover_exporters # noqa: E402
|
||||||
from exporter.base_exporter import BaseExporter # noqa: E402
|
from exporter.base_exporter import BaseExporter # noqa: E402
|
||||||
from export_native import pinned_base # noqa: E402
|
from export_native import ( # noqa: E402
|
||||||
|
SOURCES_INDEX,
|
||||||
|
collect_originals,
|
||||||
|
fetch,
|
||||||
|
pinned_base,
|
||||||
|
source_key,
|
||||||
|
)
|
||||||
from exporter.baseline import ( # noqa: E402
|
from exporter.baseline import ( # noqa: E402
|
||||||
NativeFile,
|
NativeFile,
|
||||||
build_native_model,
|
build_native_model,
|
||||||
@@ -437,6 +444,80 @@ class PinnedRevision(unittest.TestCase):
|
|||||||
self.assertTrue(base.startswith("https://"), base)
|
self.assertTrue(base.startswith("https://"), base)
|
||||||
|
|
||||||
|
|
||||||
|
class _OneFileExporter:
|
||||||
|
"""An exporter reduced to what collect_originals reads."""
|
||||||
|
|
||||||
|
def __init__(self, url: str):
|
||||||
|
self._url = url
|
||||||
|
|
||||||
|
def platform_name(self) -> str:
|
||||||
|
return "plat"
|
||||||
|
|
||||||
|
def native_sources(self) -> dict[str, str]:
|
||||||
|
return {"list.txt": self._url}
|
||||||
|
|
||||||
|
|
||||||
|
class CachedOriginal(unittest.TestCase):
|
||||||
|
"""The cached original is the file the platform data was read from."""
|
||||||
|
|
||||||
|
def setUp(self):
|
||||||
|
self._tmp = tempfile.TemporaryDirectory()
|
||||||
|
self.addCleanup(self._tmp.cleanup)
|
||||||
|
self.root = Path(self._tmp.name)
|
||||||
|
self.cache = self.root / "cache"
|
||||||
|
self.index = self.cache / SOURCES_INDEX
|
||||||
|
|
||||||
|
def _upstream(self, name: str, text: str) -> str:
|
||||||
|
path = self.root / name
|
||||||
|
path.write_text(text, encoding="utf-8")
|
||||||
|
return path.as_uri()
|
||||||
|
|
||||||
|
def test_the_same_url_is_served_from_the_cache(self):
|
||||||
|
url = self._upstream("v1.txt", "one")
|
||||||
|
target = self.cache / "plat" / "list.txt"
|
||||||
|
self.assertEqual(fetch(url, target, self.index), b"one")
|
||||||
|
self._upstream("v1.txt", "moved")
|
||||||
|
self.assertEqual(fetch(url, target, self.index), b"one")
|
||||||
|
|
||||||
|
def test_refresh_reads_a_branch_url_again(self):
|
||||||
|
"""A branch URL keeps its name while the platform moves."""
|
||||||
|
url = self._upstream("branch.txt", "one")
|
||||||
|
target = self.cache / "plat" / "list.txt"
|
||||||
|
fetch(url, target, self.index)
|
||||||
|
self._upstream("branch.txt", "two")
|
||||||
|
self.assertEqual(fetch(url, target, self.index, refresh=True), b"two")
|
||||||
|
self.assertEqual(target.read_bytes(), b"two")
|
||||||
|
|
||||||
|
def test_a_new_pin_refetches_an_existing_file(self):
|
||||||
|
"""Reading the file straight from disk served the old pin forever."""
|
||||||
|
old = self._upstream("tag-1.txt", "one")
|
||||||
|
new = self._upstream("tag-2.txt", "two")
|
||||||
|
originals, missing = collect_originals(
|
||||||
|
_OneFileExporter(old), {}, self.cache, True
|
||||||
|
)
|
||||||
|
self.assertEqual((originals, missing), ({"list.txt": "one"}, []))
|
||||||
|
originals, missing = collect_originals(
|
||||||
|
_OneFileExporter(new), {}, self.cache, True
|
||||||
|
)
|
||||||
|
self.assertEqual((originals, missing), ({"list.txt": "two"}, []))
|
||||||
|
|
||||||
|
def test_offline_serves_what_is_cached_and_names_what_is_not(self):
|
||||||
|
url = self._upstream("v1.txt", "one")
|
||||||
|
_, missing = collect_originals(_OneFileExporter(url), {}, self.cache, False)
|
||||||
|
self.assertEqual(missing, ["list.txt: absent and fetching is off"])
|
||||||
|
collect_originals(_OneFileExporter(url), {}, self.cache, True)
|
||||||
|
originals, missing = collect_originals(
|
||||||
|
_OneFileExporter("file:///nowhere/else.txt"), {}, self.cache, False
|
||||||
|
)
|
||||||
|
self.assertEqual((originals, missing), ({"list.txt": "one"}, []))
|
||||||
|
|
||||||
|
def test_the_index_names_a_file_the_same_from_any_directory(self):
|
||||||
|
target = self.cache / "plat" / "list.txt"
|
||||||
|
self.assertEqual(source_key(target, self.index), "plat/list.txt")
|
||||||
|
relative = Path(os.path.relpath(target, Path.cwd()))
|
||||||
|
self.assertEqual(source_key(relative, self.index), "plat/list.txt")
|
||||||
|
|
||||||
|
|
||||||
class RecalboxExport(unittest.TestCase):
|
class RecalboxExport(unittest.TestCase):
|
||||||
def _render(self):
|
def _render(self):
|
||||||
systems, report = model()
|
systems, report = model()
|
||||||
|
|||||||
@@ -0,0 +1,145 @@
|
|||||||
|
"""refresh_stale: which command answers which stale row, and in what order.
|
||||||
|
|
||||||
|
The network side is check_freshness and the scrapers themselves. What is
|
||||||
|
locked here is the planning: a stale row maps to one command, rows a human
|
||||||
|
must read are surfaced instead of run, and a platform rescrape is followed
|
||||||
|
by the refresh of the original its export patches.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import sys
|
||||||
|
import tempfile
|
||||||
|
import unittest
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||||
|
sys.path.insert(0, str(REPO_ROOT / "scripts"))
|
||||||
|
|
||||||
|
import refresh_stale as rs # noqa: E402
|
||||||
|
|
||||||
|
REGISTRY = {
|
||||||
|
"retrobat": {"scraper": "retrobat", "target_scraper": None},
|
||||||
|
"batocera": {"scraper": "batocera", "target_scraper": "batocera_targets"},
|
||||||
|
"lakka": {},
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def finding(area: str, subject: str, status: str = "STALE") -> dict:
|
||||||
|
return {"area": area, "subject": subject, "status": status, "detail": ""}
|
||||||
|
|
||||||
|
|
||||||
|
class Planning(unittest.TestCase):
|
||||||
|
def _plan(self, *findings: dict):
|
||||||
|
return rs.plan_jobs(list(findings), REGISTRY)
|
||||||
|
|
||||||
|
def test_a_stale_platform_is_rescraped_then_its_original_is_refreshed(self):
|
||||||
|
jobs, surfaced, _ = self._plan(finding("platforms", "retrobat"))
|
||||||
|
self.assertEqual(surfaced, [])
|
||||||
|
self.assertEqual(len(jobs), 1)
|
||||||
|
self.assertEqual(
|
||||||
|
jobs[0].command[1:],
|
||||||
|
["-m", "scripts.scraper.retrobat_scraper", "-o", "platforms/retrobat.yml"],
|
||||||
|
)
|
||||||
|
self.assertEqual(
|
||||||
|
[list(c[1:]) for c in jobs[0].then],
|
||||||
|
[["scripts/export_native.py", "--platform", "retrobat", "--refresh-cache"]],
|
||||||
|
)
|
||||||
|
|
||||||
|
def test_a_native_row_of_a_rescraped_platform_is_not_a_second_job(self):
|
||||||
|
"""Two jobs would race: the refresh must follow the scrape."""
|
||||||
|
jobs, _, ignored = self._plan(
|
||||||
|
finding("native", "retrobat"), finding("platforms", "retrobat")
|
||||||
|
)
|
||||||
|
self.assertEqual([j.area for j in jobs], ["platforms"])
|
||||||
|
self.assertEqual([f["area"] for f in ignored], ["native"])
|
||||||
|
|
||||||
|
def test_a_stale_original_alone_is_refreshed(self):
|
||||||
|
jobs, _, _ = self._plan(finding("native", "lakka"))
|
||||||
|
self.assertEqual(
|
||||||
|
jobs[0].command[1:],
|
||||||
|
["scripts/export_native.py", "--platform", "lakka", "--refresh-cache"],
|
||||||
|
)
|
||||||
|
self.assertEqual(jobs[0].then, ())
|
||||||
|
|
||||||
|
def test_a_platform_without_a_scraper_is_surfaced(self):
|
||||||
|
jobs, surfaced, _ = self._plan(finding("platforms", "lakka"))
|
||||||
|
self.assertEqual(jobs, [])
|
||||||
|
self.assertEqual([f["subject"] for f in surfaced], ["lakka"])
|
||||||
|
|
||||||
|
def test_unprofiled_cores_need_a_human(self):
|
||||||
|
jobs, surfaced, _ = self._plan(finding("targets", "batocera cores"))
|
||||||
|
self.assertEqual(jobs, [])
|
||||||
|
self.assertEqual(len(surfaced), 1)
|
||||||
|
|
||||||
|
def test_targets_data_and_recipes_map_to_their_refreshers(self):
|
||||||
|
jobs, surfaced, _ = self._plan(
|
||||||
|
finding("targets", "batocera"),
|
||||||
|
finding("data", "dolphin-sys"),
|
||||||
|
finding("catalogs", "fbneo recipes"),
|
||||||
|
finding("catalogs", "redump"),
|
||||||
|
)
|
||||||
|
commands = {j.subject: " ".join(j.command[1:]) for j in jobs}
|
||||||
|
self.assertEqual(
|
||||||
|
commands,
|
||||||
|
{
|
||||||
|
"batocera": "-m scripts.scraper.targets.batocera_targets_scraper"
|
||||||
|
" -o platforms/targets/batocera.yml",
|
||||||
|
"dolphin-sys": "scripts/refresh_data_dirs.py --key dolphin-sys --force",
|
||||||
|
"fbneo recipes": "-m scripts.scraper.romset_dat_importer"
|
||||||
|
" --source fbneo --fetch",
|
||||||
|
},
|
||||||
|
)
|
||||||
|
self.assertEqual([f["subject"] for f in surfaced], ["redump"])
|
||||||
|
|
||||||
|
def test_only_stale_rows_run_and_errors_are_shown(self):
|
||||||
|
jobs, surfaced, ignored = self._plan(
|
||||||
|
finding("data", "a", "OK"),
|
||||||
|
finding("data", "b", "SKIPPED"),
|
||||||
|
finding("platforms", "retrobat", "ERROR"),
|
||||||
|
)
|
||||||
|
self.assertEqual(jobs, [])
|
||||||
|
self.assertEqual([f["status"] for f in surfaced], ["ERROR"])
|
||||||
|
self.assertEqual(len(ignored), 2)
|
||||||
|
|
||||||
|
def test_pins_and_profiles_are_never_run(self):
|
||||||
|
jobs, surfaced, _ = self._plan(
|
||||||
|
finding("ci", "pypi:jsonschema"),
|
||||||
|
finding("coreinfo", "libretro-core-info"),
|
||||||
|
finding("profiles", "profile_sync"),
|
||||||
|
)
|
||||||
|
self.assertEqual(jobs, [])
|
||||||
|
self.assertEqual(len(surfaced), 3)
|
||||||
|
|
||||||
|
|
||||||
|
class Running(unittest.TestCase):
|
||||||
|
def _job(self, directory: str, command: list[str], then=()) -> rs.Job:
|
||||||
|
return rs.Job("platforms", "x", command, Path(directory) / "x.log", then)
|
||||||
|
|
||||||
|
def test_a_follow_up_runs_after_a_success(self):
|
||||||
|
with tempfile.TemporaryDirectory() as directory:
|
||||||
|
marker = Path(directory) / "ran"
|
||||||
|
job = self._job(
|
||||||
|
directory,
|
||||||
|
[sys.executable, "-c", "pass"],
|
||||||
|
((sys.executable, "-c", f"open({str(marker)!r}, 'w').close()"),),
|
||||||
|
)
|
||||||
|
result = rs.run_job(job, {})
|
||||||
|
self.assertEqual(result.returncode, 0)
|
||||||
|
self.assertTrue(marker.exists())
|
||||||
|
|
||||||
|
def test_a_failed_scrape_leaves_the_original_alone(self):
|
||||||
|
with tempfile.TemporaryDirectory() as directory:
|
||||||
|
marker = Path(directory) / "ran"
|
||||||
|
job = self._job(
|
||||||
|
directory,
|
||||||
|
[sys.executable, "-c", "raise SystemExit(3)"],
|
||||||
|
((sys.executable, "-c", f"open({str(marker)!r}, 'w').close()"),),
|
||||||
|
)
|
||||||
|
result = rs.run_job(job, {})
|
||||||
|
self.assertEqual(result.returncode, 3)
|
||||||
|
self.assertFalse(marker.exists())
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -227,7 +227,7 @@ python scripts/generate_pack.py --all --verify-packs --output-dir dist/
|
|||||||
```
|
```
|
||||||
|
|
||||||
Integrated as pipeline step 6/8 (runs after consistency check, before
|
Integrated as pipeline step 6/8 (runs after consistency check, before
|
||||||
README generation). Requires packs in `dist/` — skip with `--skip-packs`.
|
README generation). Requires packs in `dist/`; skip with `--skip-packs`.
|
||||||
|
|
||||||
## Verification discipline
|
## Verification discipline
|
||||||
|
|
||||||
|
|||||||
+84
-11
@@ -236,14 +236,26 @@ python scripts/export_native.py --platform batocera --fetch
|
|||||||
python scripts/export_native.py --all --fetch --output-dir dist/upstream/
|
python scripts/export_native.py --all --fetch --output-dir dist/upstream/
|
||||||
```
|
```
|
||||||
|
|
||||||
`--fetch` downloads the platform's own file once into
|
`--fetch` downloads the platform's own file into `.cache/upstream-native/`,
|
||||||
`.cache/upstream-native/`, at the revision the platform YAML's `source:`
|
at the revision the platform YAML's `source:` names. Formats that carry
|
||||||
names. Formats that carry code are patched from it rather than
|
code are patched from it rather than regenerated, and the export fails
|
||||||
regenerated, and the export fails without it.
|
without it. The URL each cached file came from is recorded beside the
|
||||||
|
cache, so a platform file that moves to a new tag is fetched again instead
|
||||||
|
of patching the previous one.
|
||||||
|
|
||||||
|
A branch URL keeps its name while its content moves, and no URL comparison
|
||||||
|
can see that. `--refresh-cache` downloads every file again and writes no
|
||||||
|
export; it is what follows a rescrape, so the original and the platform
|
||||||
|
YAML describe the same moment.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python scripts/export_native.py --platform retrobat --refresh-cache
|
||||||
|
python scripts/export_native.py --all --refresh-cache
|
||||||
|
```
|
||||||
|
|
||||||
A list is written in the order the code looks: `priority:` first (lowest
|
A list is written in the order the code looks: `priority:` first (lowest
|
||||||
wins), then the profile's own declaration order. What a format cannot
|
wins), then the profile's own declaration order. What a format cannot
|
||||||
state is reported rather than written — EmuDeck's arrays are shared
|
state is reported rather than written: EmuDeck's arrays are shared
|
||||||
between emulators its file does not name, so a hash is corrected in place
|
between emulators its file does not name, so a hash is corrected in place
|
||||||
and never added.
|
and never added.
|
||||||
|
|
||||||
@@ -326,9 +338,11 @@ Per part of a ref: `ANCHORED` (same content, same lines), `SHIFTED` (same
|
|||||||
content, moved), `RENAMED` (the source file moved), `CHANGED` (content
|
content, moved), `RENAMED` (the source file moved), `CHANGED` (content
|
||||||
edited), `AMBIGUOUS` (several equally good candidates), `GONE` (nothing
|
edited), `AMBIGUOUS` (several equally good candidates), `GONE` (nothing
|
||||||
left to anchor to), `EXTERNAL` (the ref names a project the profile does
|
left to anchor to), `EXTERNAL` (the ref names a project the profile does
|
||||||
not declare, so no revision can confirm it). An entry carries the worst
|
not declare, so no revision can confirm it), `UNCHECKED` (a pin equal to
|
||||||
|
HEAD is judged on self-consistency, and an entry with neither a hash nor
|
||||||
|
a name gives that check nothing to look for). An entry carries the worst
|
||||||
status of its parts. Only `CHANGED`, `GONE` and `AMBIGUOUS` count as
|
status of its parts. Only `CHANGED`, `GONE` and `AMBIGUOUS` count as
|
||||||
needing a re-read.
|
needing a re-read; a zero next to a row of `unchecked` is not a proof.
|
||||||
|
|
||||||
A single cited line is often not distinctive, so the anchor widens by
|
A single cited line is often not distinctive, so the anchor widens by
|
||||||
steps of ±3, ±6, ±12, ±25 and ±50 lines until it is unique. Three further
|
steps of ±3, ±6, ±12, ±25 and ±50 lines until it is unique. Three further
|
||||||
@@ -424,9 +438,10 @@ directories the refs point at.
|
|||||||
|
|
||||||
Writes are explicit and mechanical only. `--backfill-commits` fills a
|
Writes are explicit and mechanical only. `--backfill-commits` fills a
|
||||||
missing `source_commit`, `--rebase-refs` recales `SHIFTED` and `RENAMED`
|
missing `source_commit`, `--rebase-refs` recales `SHIFTED` and `RENAMED`
|
||||||
line ranges, `--bump-commit` advances `source_commit` to HEAD only when
|
line ranges and moves `source_commit` to HEAD with them, all or nothing,
|
||||||
nothing needs a re-read. All three refuse to run on a dirty `emulators/`
|
`--bump-commit` advances `source_commit` to HEAD only when nothing needs a
|
||||||
without `--force`, and every write is verified by reparsing the document.
|
re-read. All three refuse to run on a dirty `emulators/` without `--force`,
|
||||||
|
and every write is verified by reparsing the document.
|
||||||
|
|
||||||
Uses `GITHUB_TOKEN` when set, which `--all` requires. Responses are cached
|
Uses `GITHUB_TOKEN` when set, which `--all` requires. Responses are cached
|
||||||
under `.cache/upstream/`, addressed by commit sha, so `--offline` replays a
|
under `.cache/upstream/`, addressed by commit sha, so `--offline` replays a
|
||||||
@@ -457,6 +472,62 @@ python scripts/refresh_data_dirs.py --platform batocera # single platform onl
|
|||||||
python scripts/refresh_data_dirs.py --registry path/to/_data_dirs.yml
|
python scripts/refresh_data_dirs.py --registry path/to/_data_dirs.yml
|
||||||
```
|
```
|
||||||
|
|
||||||
|
### check_freshness.py
|
||||||
|
|
||||||
|
Ask every transcribed layer whether the local copy is the one upstream serves
|
||||||
|
today, and answer with one row per subject. Platform and target files are
|
||||||
|
re-scraped through the scrapers' own write path into a throwaway copy and
|
||||||
|
diffed against the committed file, so the diff decides rather than a version
|
||||||
|
string. Core-info is listed and compared with the profiles (a core with no
|
||||||
|
profile, a profile that calls standalone a core libretro now builds). Data
|
||||||
|
directories are checked against their upstream, dump catalogs and romset
|
||||||
|
snapshots against the catalog they were read from (Redump DAT versions, the
|
||||||
|
No-Intro daily rebuild, the newest TOSEC pack, the newest MAME release, the
|
||||||
|
git blobs of the FBNeo DATs), and the CI toolchain against PyPI and the
|
||||||
|
actions' latest releases.
|
||||||
|
|
||||||
|
The `native` area needs no network. It reads the cache `export_native.py`
|
||||||
|
patches from and reports an original that is absent, that came from another
|
||||||
|
URL than the one the platform YAML names, or that was fetched before the
|
||||||
|
platform YAML was last rewritten.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python scripts/check_freshness.py
|
||||||
|
python scripts/check_freshness.py --only platforms,targets
|
||||||
|
python scripts/check_freshness.py --json
|
||||||
|
python scripts/check_freshness.py --profiles # also runs profile_sync (slow)
|
||||||
|
python scripts/check_freshness.py --offline # local ages only
|
||||||
|
```
|
||||||
|
|
||||||
|
The exit code is non-zero when anything is stale. Emulator profiles are the
|
||||||
|
one layer left to `profile_sync.py`, which the `--profiles` flag folds in.
|
||||||
|
|
||||||
|
### refresh_stale.py
|
||||||
|
|
||||||
|
Run the refresher behind every stale row of `check_freshness.py`, several at
|
||||||
|
a time, and leave the result in the working tree for review. Nothing is
|
||||||
|
committed.
|
||||||
|
|
||||||
|
| Stale row | Command |
|
||||||
|
|-----------|---------|
|
||||||
|
| `platforms/<name>` | the platform scraper, then the `native` refresh for the same platform |
|
||||||
|
| `native/<name>` | `export_native.py --platform <name> --refresh-cache` |
|
||||||
|
| `targets/<name>` | the target scraper |
|
||||||
|
| `data/<key>` | `refresh_data_dirs.py --key <key> --force` |
|
||||||
|
| `catalogs/mame recipes`, `catalogs/fbneo recipes` | `scraper/romset_dat_importer.py --source <source> --fetch` |
|
||||||
|
|
||||||
|
Rows that need a decision are listed and never run: a buildbot core with no
|
||||||
|
profile, a new `.info` file, the Redump, No-Intro and TOSEC packs, the CI
|
||||||
|
pins, and `profile_sync`. A failed scraper leaves its cached original
|
||||||
|
untouched. Each job writes its output to `tmp/refresh_stale/`, and a GitHub
|
||||||
|
token is taken from `GITHUB_TOKEN` or from `gh auth token`.
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python scripts/refresh_stale.py --dry-run
|
||||||
|
python scripts/refresh_stale.py
|
||||||
|
python scripts/refresh_stale.py --only platforms,targets --jobs 6
|
||||||
|
```
|
||||||
|
|
||||||
### Other tools
|
### Other tools
|
||||||
|
|
||||||
| Script | Purpose |
|
| Script | Purpose |
|
||||||
@@ -474,12 +545,14 @@ python scripts/refresh_data_dirs.py --registry path/to/_data_dirs.yml
|
|||||||
| `generate_site.py` | Generate all MkDocs site pages (this documentation) |
|
| `generate_site.py` | Generate all MkDocs site pages (this documentation) |
|
||||||
| `validate_site.py` | Validate rendered metadata, headings, image alternatives, JSON-LD, local resources, links and fragments |
|
| `validate_site.py` | Validate rendered metadata, headings, image alternatives, JSON-LD, local resources, links and fragments |
|
||||||
| `romset_recipes.py` | Identify which emulator version an arcade archive matches, and rebuild a pinned archive from ROMs already held |
|
| `romset_recipes.py` | Identify which emulator version an arcade archive matches, and rebuild a pinned archive from ROMs already held |
|
||||||
| `scraper/romset_dat_importer.py` | Fetch and import per-set recipes from MAME `-listxml` or FBNeo DATs into `recipes/` |
|
| `scraper/romset_dat_importer.py` | Fetch and import per-set recipes from MAME `-listxml` or FBNeo DATs into `recipes/`; `--fetch` alone takes the newest MAME release, and FBNeo DATs are refreshed by git blob sha |
|
||||||
| `deterministic_zip.py` | Rebuild MAME BIOS ZIPs deterministically (same ROMs = same hash) |
|
| `deterministic_zip.py` | Rebuild MAME BIOS ZIPs deterministically (same ROMs = same hash) |
|
||||||
| `torrentzip.py` | Build TorrentZip archives for MAME/FBNeo ROM sets (archive bytes depend only on contents) |
|
| `torrentzip.py` | Build TorrentZip archives for MAME/FBNeo ROM sets (archive bytes depend only on contents) |
|
||||||
| `crypto_verify.py` | 3DS RSA signature and AES crypto verification |
|
| `crypto_verify.py` | 3DS RSA signature and AES crypto verification |
|
||||||
| `sect233r1.py` | Pure Python ECDSA verification on sect233r1 curve (3DS OTP cert) |
|
| `sect233r1.py` | Pure Python ECDSA verification on sect233r1 curve (3DS OTP cert) |
|
||||||
| `check_buildbot_system.py` | Detect stale data directories by comparing with buildbot |
|
| `check_buildbot_system.py` | Detect stale data directories by comparing with buildbot |
|
||||||
|
| `check_freshness.py` | One report on every upstream the repository transcribes (see above) |
|
||||||
|
| `refresh_stale.py` | Run the refresher behind every stale row of that report (see above) |
|
||||||
| `migrate.py` | Migrate flat bios structure to Manufacturer/Console/ hierarchy |
|
| `migrate.py` | Migrate flat bios structure to Manufacturer/Console/ hierarchy |
|
||||||
|
|
||||||
## Installation tools
|
## Installation tools
|
||||||
|
|||||||
Reference in new issue
Block a user