diff --git a/scripts/scraper/base_scraper.py b/scripts/scraper/base_scraper.py index c9c38632..44d1c37f 100644 --- a/scripts/scraper/base_scraper.py +++ b/scripts/scraper/base_scraper.py @@ -3,6 +3,7 @@ from __future__ import annotations import json +import os import sys import urllib.error import urllib.request @@ -225,22 +226,38 @@ class BaseScraper(ABC): ... -def fetch_github_latest_version(repo: str) -> str | None: - """Fetch the latest release version tag from a GitHub repo.""" +def github_headers() -> dict[str, str]: + """Headers for api.github.com, authenticated when a token is set. + + Anonymous calls share a quota of 60 per hour, so a scrape run after a + few others gets 403 for every request. + """ + headers = { + "User-Agent": "retrobios-scraper/1.0", + "Accept": "application/vnd.github.v3+json", + } + token = os.environ.get("GITHUB_TOKEN") + if token: + headers["Authorization"] = f"Bearer {token}" + return headers + + +def fetch_github_latest_version(repo: str) -> str: + """Return the tag of the latest release of a GitHub repo. + + Raises instead of returning a fallback: a platform YAML written without + the release it was read from cannot be patched back by the exporter. + """ url = f"https://api.github.com/repos/{repo}/releases/latest" try: - req = urllib.request.Request( - url, - headers={ - "User-Agent": "retrobios-scraper/1.0", - "Accept": "application/vnd.github.v3+json", - }, - ) + req = urllib.request.Request(url, headers=github_headers()) with urllib.request.urlopen(req, timeout=15) as resp: - data = json.loads(resp.read()) - return data.get("tag_name", "") - except (urllib.error.URLError, urllib.error.HTTPError, json.JSONDecodeError): - return None + tag = json.loads(resp.read()).get("tag_name", "") + except (urllib.error.URLError, json.JSONDecodeError) as e: + raise RuntimeError(f"cannot read the latest release of {repo}: {e}") from e + if not tag: + raise RuntimeError(f"the latest release of {repo} names no tag") + return tag class _PlatformDumper(yaml.SafeDumper): diff --git a/scripts/scraper/batocera_scraper.py b/scripts/scraper/batocera_scraper.py index 74e33a80..b878c090 100644 --- a/scripts/scraper/batocera_scraper.py +++ b/scripts/scraper/batocera_scraper.py @@ -19,7 +19,12 @@ from pathlib import Path from common import yaml_load -from .base_scraper import BaseScraper, BiosRequirement, requirement_entry +from .base_scraper import ( + BaseScraper, + BiosRequirement, + github_headers, + requirement_entry, +) PLATFORM_NAME = "batocera" @@ -38,39 +43,32 @@ def pick_stable_tag(names: list[str]) -> str | None: return max(stable)[1] if stable else None -def fetch_stable_tag() -> str | None: - """Return the newest stable batocera-N(.M) tag name.""" - import json - import os - import urllib.error - import urllib.request +def fetch_stable_tag() -> str: + """Return the newest stable batocera-N(.M) tag name. + Raises rather than falling back to master: the YAML records the tag + it was read from, and the exporter patches that revision. + """ url = "https://api.github.com/repos/batocera-linux/batocera.linux/tags?per_page=100" - headers = { - "User-Agent": "retrobios-scraper/1.0", - "Accept": "application/vnd.github.v3+json", - } - token = os.environ.get("GITHUB_TOKEN") - if token: - headers["Authorization"] = f"Bearer {token}" try: - req = urllib.request.Request(url, headers=headers) + req = urllib.request.Request(url, headers=github_headers()) with urllib.request.urlopen(req, timeout=15) as resp: tags = json.loads(resp.read()) - except (urllib.error.URLError, urllib.error.HTTPError, json.JSONDecodeError): - return None - return pick_stable_tag([tag["name"] for tag in tags]) + except (urllib.error.URLError, json.JSONDecodeError) as e: + raise RuntimeError(f"cannot list batocera tags: {e}") from e + tag = pick_stable_tag([tag["name"] for tag in tags]) + if not tag: + raise RuntimeError("no stable batocera tag among the latest 100") + return tag -_STABLE_TAG = fetch_stable_tag() or "master" - SOURCE_URL = ( - f"{_RAW_BASE}/{_STABLE_TAG}" + _RAW_BASE + "/{tag}" "/package/batocera/core/batocera-scripts/scripts/batocera-systems" ) CONFIGGEN_DEFAULTS_URL = ( - f"{_RAW_BASE}/{_STABLE_TAG}" + _RAW_BASE + "/{tag}" "/package/batocera/core/batocera-configgen/configs/" "configgen-defaults.yml" ) @@ -186,8 +184,9 @@ def _resolve_truncated_md5(md5: str, md5_index: dict[str, str]) -> str: class Scraper(BaseScraper): """Scraper for batocera-systems Python dict.""" - def __init__(self, url: str = SOURCE_URL): - super().__init__(url=url) + def __init__(self): + self.tag = fetch_stable_tag() + super().__init__(url=SOURCE_URL.format(tag=self.tag)) def _fetch_cores(self) -> tuple[list[str], list[str]]: """Extract core names and standalone cores from configgen-defaults.yml. @@ -195,16 +194,17 @@ class Scraper(BaseScraper): Returns (all_cores, standalone_cores) where standalone_cores are those with emulator != "libretro". """ + url = CONFIGGEN_DEFAULTS_URL.format(tag=self.tag) try: req = urllib.request.Request( - CONFIGGEN_DEFAULTS_URL, + url, headers={"User-Agent": "retrobios-scraper/1.0"}, ) with urllib.request.urlopen(req, timeout=30) as resp: raw = resp.read().decode("utf-8") except urllib.error.URLError as e: raise ConnectionError( - f"Failed to fetch {CONFIGGEN_DEFAULTS_URL}: {e}" + f"Failed to fetch {url}: {e}" ) from e data = yaml_load(raw) cores: set[str] = set() @@ -372,25 +372,12 @@ class Scraper(BaseScraper): systems[req.system]["files"].append(requirement_entry(req)) - batocera_version = "" - if _STABLE_TAG != "master": - batocera_version = _STABLE_TAG.removeprefix("batocera-") - if not batocera_version: - # Preserve existing version when fetch fails (offline mode) - existing = ( - Path(__file__).resolve().parents[2] / "platforms" / "batocera.yml" - ) - if existing.exists(): - with open(existing) as f: - old = yaml_load(f) or {} - batocera_version = str(old.get("version", "")) - cores, standalone = self._fetch_cores() result = { "platform": "Batocera", - "version": batocera_version or "", + "version": self.tag.removeprefix("batocera-"), "homepage": "https://batocera.org", - "source": SOURCE_URL, + "source": self.url, "base_destination": "bios", "hash_type": "md5", "verification_mode": "md5", diff --git a/scripts/scraper/bizhawk_scraper.py b/scripts/scraper/bizhawk_scraper.py index 946b10df..e511f22a 100644 --- a/scripts/scraper/bizhawk_scraper.py +++ b/scripts/scraper/bizhawk_scraper.py @@ -41,10 +41,8 @@ PLATFORM_NAME = "bizhawk" GITHUB_REPO = "TASEmulators/BizHawk" -_STABLE_TAG = fetch_github_latest_version(GITHUB_REPO) or "master" - SOURCE_URL = ( - f"https://raw.githubusercontent.com/TASEmulators/BizHawk/{_STABLE_TAG}" + "https://raw.githubusercontent.com/TASEmulators/BizHawk/{tag}" "/src/BizHawk.Emulation.Common/Database/FirmwareDatabase.cs" ) @@ -332,7 +330,8 @@ class Scraper(BaseScraper): """BizHawk firmware database scraper.""" def __init__(self): - super().__init__(url=SOURCE_URL) + self.tag = fetch_github_latest_version(GITHUB_REPO) + super().__init__(url=SOURCE_URL.format(tag=self.tag)) def validate_format(self, raw_data: str) -> bool: return "FirmwareDatabase" in raw_data and "FirmwareAndOption" in raw_data @@ -372,13 +371,11 @@ class Scraper(BaseScraper): systems[req.system]["files"].append(requirement_entry(req)) - version = _STABLE_TAG if _STABLE_TAG != "master" else "" - return { "platform": "BizHawk", - "version": version, + "version": self.tag, "homepage": "https://tasvideos.org/BizHawk", - "source": SOURCE_URL, + "source": self.url, "base_destination": "Firmware", "hash_type": "sha1", "verification_mode": "sha1", diff --git a/scripts/scraper/coreinfo_scraper.py b/scripts/scraper/coreinfo_scraper.py index 3d471b0a..322a1b05 100644 --- a/scripts/scraper/coreinfo_scraper.py +++ b/scripts/scraper/coreinfo_scraper.py @@ -295,7 +295,7 @@ class Scraper(BaseScraper): def fetch_metadata(self) -> dict: """Fetch version info from GitHub.""" version = fetch_github_latest_version("libretro/libretro-core-info") - return {"version": version or ""} + return {"version": version} def main(): diff --git a/scripts/scraper/emudeck_scraper.py b/scripts/scraper/emudeck_scraper.py index f8a64138..45484723 100644 --- a/scripts/scraper/emudeck_scraper.py +++ b/scripts/scraper/emudeck_scraper.py @@ -426,13 +426,7 @@ class Scraper(BaseScraper): systems[req.system]["files"].append(requirement_entry(req)) - version = "" - try: - v = fetch_github_latest_version("dragoonDorise/EmuDeck") - if v: - version = v - except (ConnectionError, ValueError, OSError): - pass + version = fetch_github_latest_version("dragoonDorise/EmuDeck") cores = self._fetch_installed_emulators() diff --git a/scripts/scraper/recalbox_scraper.py b/scripts/scraper/recalbox_scraper.py index ecd60e88..22bfc1fb 100644 --- a/scripts/scraper/recalbox_scraper.py +++ b/scripts/scraper/recalbox_scraper.py @@ -21,7 +21,7 @@ from .base_scraper import BaseScraper, BiosRequirement, requirement_entry PLATFORM_NAME = "recalbox" -def _fetch_gitlab_stable_tag() -> str | None: +def _fetch_gitlab_stable_tag() -> str: """Fetch the latest stable x.y.z tag from the Recalbox GitLab.""" import json import re @@ -33,16 +33,16 @@ def _fetch_gitlab_stable_tag() -> str | None: req = urllib.request.Request(url, headers={"User-Agent": "retrobios-scraper/1.0"}) with urllib.request.urlopen(req, timeout=15) as resp: tags = json.loads(resp.read()) - except (urllib.error.URLError, urllib.error.HTTPError, json.JSONDecodeError): - return None + except (urllib.error.URLError, json.JSONDecodeError) as e: + raise RuntimeError(f"cannot list Recalbox tags: {e}") from e stable = [t["name"] for t in tags if re.fullmatch(r"[0-9]+\.[0-9]+(\.[0-9]+)?", t["name"])] - return stable[0] if stable else None + if not stable: + raise RuntimeError("no stable Recalbox tag among the latest 50") + return stable[0] -_STABLE_TAG = _fetch_gitlab_stable_tag() or "master" - SOURCE_URL = ( - f"https://gitlab.com/recalbox/recalbox/-/raw/{_STABLE_TAG}/" + "https://gitlab.com/recalbox/recalbox/-/raw/{tag}/" "board/recalbox/fsoverlay/recalbox/share_init/system/" ".emulationstation/es_bios.xml" ) @@ -109,8 +109,9 @@ def split_cores(names) -> tuple[list[str], list[str]]: class Scraper(BaseScraper): """Scraper for Recalbox es_bios.xml.""" - def __init__(self, url: str = SOURCE_URL): - super().__init__(url=url) + def __init__(self): + self.tag = _fetch_gitlab_stable_tag() + super().__init__(url=SOURCE_URL.format(tag=self.tag)) def _fetch_cores(self) -> tuple[list[str], list[str]]: """Core names from es_bios.xml, and those Recalbox runs standalone. @@ -217,16 +218,12 @@ class Scraper(BaseScraper): systems[req.system]["files"].append(requirement_entry(req)) - version = _STABLE_TAG if _STABLE_TAG != "master" else "" - if not version: - version = "10.0" - cores, standalone = self._fetch_cores() return { "platform": "Recalbox", - "version": version, + "version": self.tag, "homepage": "https://www.recalbox.com", - "source": SOURCE_URL, + "source": self.url, "base_destination": "bios", "hash_type": "md5", "verification_mode": "md5", diff --git a/scripts/scraper/retrobat_scraper.py b/scripts/scraper/retrobat_scraper.py index 1ca0f164..0e6e5b52 100644 --- a/scripts/scraper/retrobat_scraper.py +++ b/scripts/scraper/retrobat_scraper.py @@ -160,10 +160,7 @@ class Scraper(BaseScraper): systems[req.system]["files"].append(requirement_entry(req)) - version = "" - tag = fetch_github_latest_version(GITHUB_REPO) - if tag: - version = tag + version = fetch_github_latest_version(GITHUB_REPO) return { "platform": "RetroBat", diff --git a/scripts/scraper/retrodeck_scraper.py b/scripts/scraper/retrodeck_scraper.py index 874e9fb6..e5a580a2 100644 --- a/scripts/scraper/retrodeck_scraper.py +++ b/scripts/scraper/retrodeck_scraper.py @@ -431,7 +431,7 @@ class Scraper(BaseScraper): from .base_scraper import fetch_github_latest_version except ImportError: from scraper.base_scraper import fetch_github_latest_version - version = fetch_github_latest_version("RetroDECK/RetroDECK") or "" + version = fetch_github_latest_version("RetroDECK/RetroDECK") return { "platform": "RetroDECK", diff --git a/scripts/scraper/rocknix_scraper.py b/scripts/scraper/rocknix_scraper.py index bf05cc66..2bb1053d 100644 --- a/scripts/scraper/rocknix_scraper.py +++ b/scripts/scraper/rocknix_scraper.py @@ -181,7 +181,7 @@ class Scraper(BaseScraper): return { "platform": "ROCKNIX", - "version": fetch_github_latest_version(GITHUB_REPO) or "", + "version": fetch_github_latest_version(GITHUB_REPO), "homepage": "https://rocknix.org", "source": SOURCE_URL, "base_destination": "bios", diff --git a/scripts/scraper/romm_scraper.py b/scripts/scraper/romm_scraper.py index fd4e25b0..c386d328 100644 --- a/scripts/scraper/romm_scraper.py +++ b/scripts/scraper/romm_scraper.py @@ -44,10 +44,8 @@ PLATFORM_NAME = "romm" GITHUB_REPO = "rommapp/romm" -_STABLE_TAG = fetch_github_latest_version(GITHUB_REPO) or "master" - SOURCE_URL = ( - f"https://raw.githubusercontent.com/rommapp/romm/{_STABLE_TAG}" + "https://raw.githubusercontent.com/rommapp/romm/{tag}" "/backend/models/fixtures/known_bios_files.json" ) @@ -124,8 +122,9 @@ FIRMWARE_MIRRORS: dict[str, tuple[str, ...]] = { class Scraper(BaseScraper): """Scraper for RomM known_bios_files.json.""" - def __init__(self, url: str = SOURCE_URL): - super().__init__(url=url) + def __init__(self): + self.tag = fetch_github_latest_version(GITHUB_REPO) + super().__init__(url=SOURCE_URL.format(tag=self.tag)) self._parsed: dict | None = None def _parse_json(self) -> dict: @@ -212,14 +211,12 @@ class Scraper(BaseScraper): systems[req.system]["files"].append(requirement_entry(req)) - version = _STABLE_TAG if _STABLE_TAG != "master" else "" - return { "inherits": "emulatorjs", "platform": "RomM", - "version": version, + "version": self.tag, "homepage": "https://romm.app", - "source": SOURCE_URL, + "source": self.url, "base_destination": "bios", "hash_type": "sha1", "verification_mode": "md5", diff --git a/tests/test_scraper_contract.py b/tests/test_scraper_contract.py index 36a45514..c66f4251 100644 --- a/tests/test_scraper_contract.py +++ b/tests/test_scraper_contract.py @@ -231,3 +231,50 @@ class RecalboxBuildMode(unittest.TestCase): if __name__ == "__main__": unittest.main() + + +class UnreadableReleaseStopsTheScrape(unittest.TestCase): + """A failed tag lookup fell back to master, or to an empty or invented + version. BizHawk, RomM, Batocera and Recalbox then wrote a YAML read from + master under the previous release's version (Recalbox under "10.0"), and + the exporter patched a revision the data never described.""" + + @staticmethod + def _refuse(*_args, **_kwargs): + import urllib.error + + raise urllib.error.HTTPError("https://api.github.com/x", 403, "rate limited", {}, None) + + def test_every_pinned_scraper_raises(self): + import importlib + from unittest import mock + + for name in ( + "scraper.bizhawk_scraper", + "scraper.romm_scraper", + "scraper.batocera_scraper", + "scraper.recalbox_scraper", + ): + module = importlib.import_module(name) + with self.subTest(scraper=name), mock.patch( + "urllib.request.urlopen", self._refuse + ), self.assertRaises(RuntimeError): + module.Scraper() + + def test_a_release_lookup_raises(self): + from unittest import mock + + from scraper.base_scraper import fetch_github_latest_version + + with mock.patch("urllib.request.urlopen", self._refuse), self.assertRaises( + RuntimeError + ): + fetch_github_latest_version("libretro/RetroArch") + + def test_the_token_is_sent(self): + from unittest import mock + + from scraper.base_scraper import github_headers + + with mock.patch.dict("os.environ", {"GITHUB_TOKEN": "t0k"}): + self.assertEqual(github_headers()["Authorization"], "Bearer t0k")