diff --git a/platforms/_registry.yml b/platforms/_registry.yml index fd4b3f3e..69891881 100644 --- a/platforms/_registry.yml +++ b/platforms/_registry.yml @@ -969,7 +969,7 @@ platforms: bizhawk: config: bizhawk.yml status: active - logo: https://raw.githubusercontent.com/TASEmulators/BizHawk/master/Assets/bizhawk.ico + logo: https://avatars.githubusercontent.com/u/11743303 scraper: bizhawk source_url: https://raw.githubusercontent.com/TASEmulators/BizHawk/master/src/BizHawk.Emulation.Common/Database/FirmwareDatabase.cs source_format: csharp_firmware_database diff --git a/scripts/check_profile_refs.py b/scripts/check_profile_refs.py deleted file mode 100644 index f0294bb1..00000000 --- a/scripts/check_profile_refs.py +++ /dev/null @@ -1,277 +0,0 @@ -#!/usr/bin/env python3 -"""Check emulator profile source_refs against the profiled upstream. - -Resolves the upstream commit in effect at each profile's profiled_date -(or uses source_commit when the profile carries one), fetches every file -a source_ref points to at that commit and at HEAD, and reports whether -the values the profile declares still sit where the ref says. - -A ref is checked by looking for the entry's declared hashes (or filename -when no hash exists) in a window around the cited lines. Three outcomes -per ref at each revision: anchored (found in the window), moved (found -elsewhere in the file), gone (absent from the file). - -Usage: - python scripts/check_profile_refs.py --emulator vice - python scripts/check_profile_refs.py --all - python scripts/check_profile_refs.py --emulator clk --json -""" - -from __future__ import annotations - -import argparse -import json -import os -import re -import sys -import urllib.error -import urllib.parse -import urllib.request - -sys.path.insert(0, os.path.dirname(__file__)) -from common import load_emulator_profiles - -WINDOW = 40 -RAW_URL = "https://raw.githubusercontent.com/{owner}/{repo}/{ref}/{path}" -API_COMMITS = ( - "https://api.github.com/repos/{owner}/{repo}/commits" - "?until={date}T23:59:59Z&per_page=1" -) - -_GITHUB_RE = re.compile(r"https?://github\.com/([^/]+)/([^/]+?)(?:\.git)?/?$") -_REF_RE = re.compile(r"^(?P[^:]+?)(?::(?P\d+)(?:-(?P\d+))?)?$") - -_file_cache: dict[tuple[str, str, str, str], list[str] | None] = {} - - -def parse_source_ref(ref: str) -> tuple[str, int | None, int | None]: - """Split 'path/file.cpp:123-129' into (path, start, end).""" - m = _REF_RE.match(ref.strip()) - if not m: - return ref.strip(), None, None - start = int(m.group("start")) if m.group("start") else None - end = int(m.group("end")) if m.group("end") else start - return m.group("path"), start, end - - -def collect_tokens(entry: dict) -> list[str]: - """Declared values worth searching for near the ref: hashes, then name.""" - tokens: list[str] = [] - for field in ("sha1", "md5", "crc32", "sha256", "known_hash_adler32"): - val = entry.get(field) - vals = val if isinstance(val, list) else [val] if val else [] - for v in vals: - v = str(v).lower().removeprefix("0x") - if v: - tokens.append(v) - if not tokens: - name = entry.get("name", "") - if name: - tokens.append(os.path.basename(name).lower()) - return tokens - - -def match_ref( - lines: list[str] | None, - start: int | None, - end: int | None, - tokens: list[str], -) -> str: - """Return anchored / moved / gone for one ref against one revision.""" - if lines is None: - return "gone" - text = [ln.lower() for ln in lines] - if start is not None: - lo = max(0, start - 1 - WINDOW) - hi = min(len(text), (end or start) + WINDOW) - window = text[lo:hi] - if any(tok in ln for tok in tokens for ln in window): - return "anchored" - if any(tok in ln for tok in tokens for ln in text): - return "moved" if start is not None else "anchored" - return "gone" - - -def _github_repo(url: str) -> tuple[str, str] | None: - m = _GITHUB_RE.match(url.strip()) - return (m.group(1), m.group(2)) if m else None - - -def _http_json(url: str) -> object: - headers = {"User-Agent": "retrobios-refcheck/1.0"} - token = os.environ.get("GITHUB_TOKEN", "") - if token: - headers["Authorization"] = f"token {token}" - req = urllib.request.Request(url, headers=headers) - with urllib.request.urlopen(req, timeout=30) as resp: - return json.loads(resp.read().decode()) - - -def resolve_commit_at(owner: str, repo: str, date: str) -> str | None: - """Last default-branch commit on or before the given date.""" - url = API_COMMITS.format(owner=owner, repo=repo, date=date) - try: - data = _http_json(url) - except (urllib.error.URLError, urllib.error.HTTPError) as exc: - print(f" {owner}/{repo}: commit lookup failed: {exc}", file=sys.stderr) - return None - if isinstance(data, list) and data: - return data[0].get("sha") - return None - - -def fetch_lines(owner: str, repo: str, ref: str, path: str) -> list[str] | None: - key = (owner, repo, ref, path) - if key in _file_cache: - return _file_cache[key] - url = RAW_URL.format( - owner=owner, repo=repo, ref=ref, path=urllib.parse.quote(path) - ) - headers = {"User-Agent": "retrobios-refcheck/1.0"} - try: - req = urllib.request.Request(url, headers=headers) - with urllib.request.urlopen(req, timeout=30) as resp: - lines = resp.read().decode("utf-8", errors="replace").splitlines() - except urllib.error.HTTPError as exc: - if exc.code != 404: - print(f" fetch {url}: HTTP {exc.code}", file=sys.stderr) - lines = None - except urllib.error.URLError as exc: - print(f" fetch {url}: {exc}", file=sys.stderr) - lines = None - _file_cache[key] = lines - return lines - - -def check_profile(name: str, profile: dict) -> dict: - """Check every source_ref of one profile. Returns a report dict.""" - report: dict = { - "profile": name, - "repos": [], - "pinned_commit": None, - "commit_source": None, - "refs": [], - "skipped": None, - } - - repos = [] - for field in ("upstream", "source"): - url = profile.get(field, "") - gh = _github_repo(url) if url else None - if gh and gh not in repos: - repos.append(gh) - if not repos: - report["skipped"] = "no github repository in upstream/source" - return report - report["repos"] = [f"{o}/{r}" for o, r in repos] - - pinned = profile.get("source_commit") - if pinned: - report["commit_source"] = "source_commit" - else: - date = str(profile.get("profiled_date", "")) - if not date: - report["skipped"] = "no source_commit and no profiled_date" - return report - for owner, repo in repos: - pinned = resolve_commit_at(owner, repo, date) - if pinned: - repos = [(owner, repo)] + [r for r in repos if r != (owner, repo)] - break - report["commit_source"] = f"resolved from profiled_date {date}" - if not pinned: - report["skipped"] = "commit resolution failed" - return report - report["pinned_commit"] = pinned - - rank = {"gone": 0, "moved": 1, "anchored": 2} - for entry in profile.get("files", []): - ref = entry.get("source_ref", "") - if not ref: - continue - tokens = collect_tokens(entry) - - # A source_ref may carry several comma-separated references; - # the entry holds as long as one of them does. - pin_status = "gone" - head_status = "gone" - for part in (p.strip() for p in ref.split(",") if p.strip()): - path, start, end = parse_source_ref(part) - for owner, repo in repos: - pin_lines = fetch_lines(owner, repo, pinned, path) - if pin_lines is None and (owner, repo) != repos[-1]: - continue - head_lines = fetch_lines(owner, repo, "HEAD", path) - pin_part = match_ref(pin_lines, start, end, tokens) - head_part = match_ref(head_lines, start, end, tokens) - if rank[pin_part] > rank[pin_status]: - pin_status = pin_part - if rank[head_part] > rank[head_status]: - head_status = head_part - break - - report["refs"].append({ - "name": entry.get("name", ""), - "source_ref": ref, - "at_pin": pin_status, - "at_head": head_status, - }) - return report - - -def _print_report(report: dict) -> None: - name = report["profile"] - if report["skipped"]: - print(f"{name}: skipped ({report['skipped']})") - return - refs = report["refs"] - pin = report["pinned_commit"][:12] - print(f"{name}: {len(refs)} refs against {pin} ({report['commit_source']})") - for state, where in (("at_pin", "pin"), ("at_head", "HEAD")): - counts: dict[str, int] = {} - for r in refs: - counts[r[state]] = counts.get(r[state], 0) + 1 - summary = ", ".join(f"{v} {k}" for k, v in sorted(counts.items())) - print(f" at {where}: {summary}") - for r in refs: - if r["at_pin"] != "anchored" or r["at_head"] != "anchored": - print( - f" {r['source_ref']} ({r['name']}): " - f"pin={r['at_pin']} HEAD={r['at_head']}" - ) - - -def main() -> None: - parser = argparse.ArgumentParser( - description="Check profile source_refs against upstream revisions" - ) - group = parser.add_mutually_exclusive_group(required=True) - group.add_argument("--emulator", help="check a single profile") - group.add_argument("--all", action="store_true", help="check every profile") - parser.add_argument("--emulators-dir", default="emulators") - parser.add_argument("--json", action="store_true", dest="as_json") - args = parser.parse_args() - - profiles = load_emulator_profiles(args.emulators_dir, skip_aliases=False) - if args.emulator: - if args.emulator not in profiles: - print(f"unknown profile: {args.emulator}", file=sys.stderr) - sys.exit(1) - selected = {args.emulator: profiles[args.emulator]} - else: - selected = { - k: v - for k, v in sorted(profiles.items()) - if v.get("type") not in ("alias", "test") - } - - reports = [check_profile(name, prof) for name, prof in selected.items()] - if args.as_json: - print(json.dumps(reports, indent=2)) - else: - for report in reports: - _print_report(report) - - -if __name__ == "__main__": - main() diff --git a/scripts/generate_site.py b/scripts/generate_site.py index 28ce704c..22ecf663 100644 --- a/scripts/generate_site.py +++ b/scripts/generate_site.py @@ -16,6 +16,10 @@ import json import os import shutil import sys +import urllib.error +import urllib.parse +import urllib.request +from concurrent.futures import ThreadPoolExecutor from datetime import datetime, timezone from pathlib import Path @@ -42,6 +46,75 @@ RELEASE_URL = f"{REPO_URL}/releases/latest" GENERATED_DIRS = ["platforms", "systems", "emulators"] WIKI_SRC_DIR = "wiki" # manually maintained wiki sources SYSTEM_ICON_BASE = "https://raw.githubusercontent.com/libretro/retroarch-assets/master/xmb/systematic/png" +ICON_CACHE_PATH = Path(".cache") / "system_icons.json" + +# Icon names confirmed to exist upstream. A name absent from this map has not +# been checked yet; a name mapped to False has no icon and gets none rendered, +# because a heading with a broken image reads worse than a heading without one. +_icon_available: dict[str, bool] = {} + + +def _icon_name(manufacturer: str, console_name: str) -> str: + return f"{manufacturer} - {console_name}".replace("/", " ") + + +def _icon_url(icon_name: str) -> str: + return f"{SYSTEM_ICON_BASE}/{urllib.parse.quote(icon_name)}.png" + + +def prime_system_icons(names: set[str]) -> None: + """Record which system icons upstream actually serves. + + Results persist in ``.cache`` so later runs skip the network. Names that + cannot be checked stay unavailable: the site never links an image it has + not seen answer. + """ + cached: dict[str, bool] = {} + if ICON_CACHE_PATH.exists(): + try: + with open(ICON_CACHE_PATH) as f: + cached = json.load(f) + except (json.JSONDecodeError, OSError): + cached = {} + + unknown = sorted(n for n in names if n not in cached) + if unknown: + def check(name: str) -> tuple[str, bool | None]: + """True when served, False when upstream says it is gone. + + None on a transient failure, so a flaky run never records an + icon as absent for every later build. + """ + req = urllib.request.Request(_icon_url(name), method="HEAD") + try: + with urllib.request.urlopen(req, timeout=15) as resp: + return name, resp.status == 200 + except urllib.error.HTTPError as exc: + return name, False if exc.code == 404 else None + except (urllib.error.URLError, OSError): + return name, None + + print(f"Checking {len(unknown)} system icons...") + with ThreadPoolExecutor(max_workers=8) as pool: + for name, ok in pool.map(check, unknown): + if ok is not None: + cached[name] = ok + ICON_CACHE_PATH.parent.mkdir(parents=True, exist_ok=True) + with open(ICON_CACHE_PATH, "w") as f: + json.dump(cached, f, indent=2, sort_keys=True) + + _icon_available.update(cached) + missing = sum(1 for n in names if not cached.get(n)) + if missing: + print(f" {missing} systems have no upstream icon") + + +def system_icon_markdown(manufacturer: str, console_name: str) -> str: + """Icon image for a system heading, empty when upstream serves none.""" + name = _icon_name(manufacturer, console_name) + if not _icon_available.get(name): + return "" + return f"![{console_name}]({_icon_url(name)}){{ width=24 }} " CLS_LABELS = { "official_port": "Official ports", @@ -955,9 +1028,8 @@ def generate_system_page( for console_name in sorted(consoles.keys()): files = consoles[console_name] - icon_name = f"{manufacturer} - {console_name}".replace("/", " ") - icon_url = f"{SYSTEM_ICON_BASE}/{icon_name.replace(' ', '%20')}.png" - lines.append(f"## ![{console_name}]({icon_url}){{ width=24 }} {console_name}") + icon_md = system_icon_markdown(manufacturer, console_name) + lines.append(f"## {icon_md}{console_name}") lines.append("") # Separate main files from variants main_files = [f for f in files if "/.variants/" not in f["path"]] @@ -2889,6 +2961,12 @@ def main(): # Generate system pages print("Generating system pages...") + prime_system_icons({ + _icon_name(mfr, console) + for mfr, consoles in manufacturers.items() + for console in consoles + }) + write_if_changed( str(docs / "systems" / "index.md"), generate_systems_index(manufacturers) ) diff --git a/tests/test_profile_refs.py b/tests/test_profile_refs.py deleted file mode 100644 index fbe7cf6a..00000000 --- a/tests/test_profile_refs.py +++ /dev/null @@ -1,83 +0,0 @@ -"""Tests for check_profile_refs pure functions (no network).""" - -from __future__ import annotations - -import os -import sys -import unittest - -sys.path.insert(0, os.path.join(os.path.dirname(__file__), "..", "scripts")) - -from check_profile_refs import collect_tokens, match_ref, parse_source_ref - - -class TestParseSourceRef(unittest.TestCase): - def test_path_with_single_line(self): - self.assertEqual( - parse_source_ref("src/core/bios.cpp:258"), - ("src/core/bios.cpp", 258, 258), - ) - - def test_path_with_line_range(self): - self.assertEqual( - parse_source_ref("Machines/Utility/ROMCatalogue.cpp:123-129"), - ("Machines/Utility/ROMCatalogue.cpp", 123, 129), - ) - - def test_path_without_line(self): - self.assertEqual( - parse_source_ref("FirmwareDatabase.cs"), - ("FirmwareDatabase.cs", None, None), - ) - - -class TestCollectTokens(unittest.TestCase): - def test_hashes_preferred_over_name(self): - entry = { - "name": "kernal.bin", - "sha1": "6C4FA9465F6091B174DF27DFE679499DF447503C", - "crc32": "789c8cc5", - } - tokens = collect_tokens(entry) - self.assertIn("6c4fa9465f6091b174df27dfe679499df447503c", tokens) - self.assertIn("789c8cc5", tokens) - self.assertNotIn("kernal.bin", tokens) - - def test_name_fallback_without_hashes(self): - tokens = collect_tokens({"name": "GC/USA/IPL.bin"}) - self.assertEqual(tokens, ["ipl.bin"]) - - def test_adler_prefix_stripped(self): - tokens = collect_tokens({"known_hash_adler32": "0x4f1f6f5c"}) - self.assertEqual(tokens, ["4f1f6f5c"]) - - def test_hash_lists(self): - tokens = collect_tokens({"md5": ["AABB", "ccdd"]}) - self.assertEqual(tokens, ["aabb", "ccdd"]) - - -class TestMatchRef(unittest.TestCase): - LINES = ["x = 0"] * 100 + ['hash = "789c8cc5"'] + ["y = 1"] * 100 - - def test_anchored_inside_window(self): - self.assertEqual( - match_ref(self.LINES, 101, 101, ["789c8cc5"]), "anchored" - ) - - def test_moved_outside_window(self): - self.assertEqual(match_ref(self.LINES, 5, 5, ["789c8cc5"]), "moved") - - def test_gone_when_absent(self): - self.assertEqual(match_ref(self.LINES, 101, 101, ["deadbeef"]), "gone") - - def test_gone_when_file_missing(self): - self.assertEqual(match_ref(None, 1, 1, ["789c8cc5"]), "gone") - - def test_anchored_no_line_searches_whole_file(self): - self.assertEqual( - match_ref(self.LINES, None, None, ["789c8cc5"]), "anchored" - ) - - -if __name__ == "__main__": - unittest.main()