diff --git a/scripts/scraper/targets/emudeck_targets_scraper.py b/scripts/scraper/targets/emudeck_targets_scraper.py index a1d2b6df..c1ea7b87 100644 --- a/scripts/scraper/targets/emudeck_targets_scraper.py +++ b/scripts/scraper/targets/emudeck_targets_scraper.py @@ -49,7 +49,7 @@ _NAME_OVERRIDES: dict[str, str] = { _SKIP = {"retroarch_maincfg", "retroarch"} -def _fetch(url: str) -> str | None: +def _fetch(url: str) -> str: headers = {"User-Agent": "retrobios-scraper/1.0"} token = os.environ.get("GITHUB_TOKEN") if token and "api.github.com" in url: @@ -59,21 +59,18 @@ def _fetch(url: str) -> str | None: with urllib.request.urlopen(req, timeout=30) as resp: return resp.read().decode("utf-8") except urllib.error.URLError as e: - print(f" skip {url}: {e}", file=sys.stderr) - return None + # A target written from a failed request loses its cores in silence. + raise RuntimeError(f"cannot fetch {url}: {e}") from e def _list_emuscripts(api_url: str) -> list[str]: """List emulator script filenames from GitHub API.""" - raw = _fetch(api_url) - if not raw: - return [] - entries = json.loads(raw) - names = [] - for e in entries: - name = e.get("name", "") - if name.endswith(".sh") or name.endswith(".ps1"): - names.append(name) + entries = json.loads(_fetch(api_url)) + names = [ + e["name"] for e in entries if e.get("name", "").endswith((".sh", ".ps1")) + ] + if not names: + raise RuntimeError(f"no emulator scripts listed at {api_url}") return names @@ -136,15 +133,13 @@ class Scraper(BaseTargetScraper): import os target_path = os.path.join("platforms", "targets", "retroarch.yml") - if not os.path.exists(target_path): - return [] with open(target_path) as f: data = yaml_load(f) or {} # Find a target matching the architecture - for tname, tinfo in data.get("targets", {}).items(): - if tinfo.get("architecture") == arch: - return tinfo.get("cores", []) - return [] + for tinfo in data.get("targets", {}).values(): + if tinfo.get("architecture") == arch and tinfo.get("cores"): + return tinfo["cores"] + raise RuntimeError(f"no {arch} core list in {target_path}") def fetch_targets(self) -> dict: steamos_cores = self._fetch_cores_for_target(STEAMOS_API, "SteamOS") diff --git a/scripts/scraper/targets/retroarch_targets_scraper.py b/scripts/scraper/targets/retroarch_targets_scraper.py index 033f5cc9..2ba8d62a 100644 --- a/scripts/scraper/targets/retroarch_targets_scraper.py +++ b/scripts/scraper/targets/retroarch_targets_scraper.py @@ -87,7 +87,7 @@ class Scraper(BaseTargetScraper): def __init__(self, url: str = BUILDBOT_URL): super().__init__(url=url) - def _fetch_url(self, url: str) -> str | None: + def _fetch_url(self, url: str) -> str: try: req = urllib.request.Request( url, headers={"User-Agent": "retrobios-scraper/1.0"} @@ -95,14 +95,12 @@ class Scraper(BaseTargetScraper): with urllib.request.urlopen(req, timeout=30) as resp: return resp.read().decode("utf-8") except urllib.error.URLError as e: - print(f" skip {url}: {e}", file=sys.stderr) - return None + # A target written from a failed request loses its cores in silence. + raise RuntimeError(f"cannot fetch {url}: {e}") from e def _fetch_cores_for_target(self, path: str) -> list[str]: url = f"{self.url}{path}/" html = self._fetch_url(url) - if html is None: - return [] cores: list[str] = [] seen: set[str] = set() for match in _HREF_RE.finditer(html): @@ -134,10 +132,7 @@ class Scraper(BaseTargetScraper): def _fetch_cores_for_recipe(self, recipe_path: str) -> list[str]: url = f"{RECIPE_BASE_URL}{recipe_path}" - text = self._fetch_url(url) - if text is None: - return [] - return self._parse_recipe_cores(text) + return self._parse_recipe_cores(self._fetch_url(url)) def fetch_targets(self) -> dict: targets: dict[str, dict] = {} @@ -145,8 +140,7 @@ class Scraper(BaseTargetScraper): print(f" fetching {target_name}...", file=sys.stderr) cores = self._fetch_cores_for_target(path) if not cores: - print(f" warning: no cores found for {target_name}", file=sys.stderr) - continue + raise RuntimeError(f"no cores listed for {target_name}") targets[target_name] = { "architecture": arch, "cores": cores, @@ -156,8 +150,7 @@ class Scraper(BaseTargetScraper): print(f" fetching {target_name} (recipe)...", file=sys.stderr) cores = self._fetch_cores_for_recipe(recipe_path) if not cores: - print(f" warning: no cores found for {target_name}", file=sys.stderr) - continue + raise RuntimeError(f"no cores listed for {target_name}") targets[target_name] = { "architecture": arch, "cores": cores, diff --git a/scripts/scraper/targets/retropie_targets_scraper.py b/scripts/scraper/targets/retropie_targets_scraper.py index c266d69a..66009351 100644 --- a/scripts/scraper/targets/retropie_targets_scraper.py +++ b/scripts/scraper/targets/retropie_targets_scraper.py @@ -59,7 +59,7 @@ _MODULE_ID_RE = re.compile(r'rp_module_id\s*=\s*["\']([^"\']+)["\']') _MODULE_FLAGS_RE = re.compile(r'rp_module_flags\s*=\s*["\']([^"\']*)["\']') -def _fetch(url: str, accept: str = "text/plain") -> str | None: +def _fetch(url: str, accept: str = "text/plain") -> str: headers = {"User-Agent": "retrobios-scraper/1.0", "Accept": accept} token = os.environ.get("GITHUB_TOKEN") if token and "api.github.com" in url: @@ -69,8 +69,8 @@ def _fetch(url: str, accept: str = "text/plain") -> str | None: with urllib.request.urlopen(req, timeout=30) as resp: return resp.read().decode("utf-8") except urllib.error.URLError as e: - print(f" skip {url}: {e}", file=sys.stderr) - return None + # A target written from a failed request loses its cores in silence. + raise RuntimeError(f"cannot fetch {url}: {e}") from e def _is_available(flags_str: str, platform: str) -> bool: @@ -111,32 +111,24 @@ class Scraper(BaseTargetScraper): def _list_scriptmodules(self) -> list[str]: """Return list of .sh filenames from the libretrocores directory.""" - raw = _fetch(self.url, accept="application/vnd.github+json") - if raw is None: - return [] - try: - entries = json.loads(raw) - except json.JSONDecodeError as e: - print(f" JSON parse error: {e}", file=sys.stderr) - return [] - return [e["name"] for e in entries if e.get("name", "").endswith(".sh")] + entries = json.loads(_fetch(self.url, accept="application/vnd.github+json")) + names = [e["name"] for e in entries if e.get("name", "").endswith(".sh")] + if not names: + raise RuntimeError(f"no scriptmodules listed at {self.url}") + return names - def _fetch_module(self, filename: str) -> str | None: + def _fetch_module(self, filename: str) -> str: return _fetch(f"{RAW_BASE_URL}{filename}") def fetch_targets(self) -> dict: print(" listing RetroPie scriptmodules...", file=sys.stderr) filenames = self._list_scriptmodules() - if not filenames: - print(" warning: no scriptmodules found", file=sys.stderr) # {platform: [core_id, ...]} platform_cores: dict[str, list[str]] = {p: [] for p in PLATFORM_FLAGS} for filename in filenames: content = self._fetch_module(filename) - if content is None: - continue module_id, flags = _parse_module(content) if not module_id: print(f" warning: no rp_module_id in {filename}", file=sys.stderr) diff --git a/tests/test_target_scrapers.py b/tests/test_target_scrapers.py new file mode 100644 index 00000000..0fb8d61d --- /dev/null +++ b/tests/test_target_scrapers.py @@ -0,0 +1,64 @@ +"""A target scraper that cannot read its source writes nothing. + +The EmuDeck, RetroPie and RetroArch scrapers turned a failed request (an +anonymous API quota, a moved buildbot directory) into an empty core list +and wrote the file anyway: a target vanished, or kept no cores, and the +packs built for it shrank without an error. +""" + +from __future__ import annotations + +import sys +import unittest +import urllib.error +from pathlib import Path +from unittest import mock + +REPO_ROOT = Path(__file__).resolve().parent.parent +sys.path.insert(0, str(REPO_ROOT)) + +from scripts.scraper.targets import ( # noqa: E402 + emudeck_targets_scraper, + retroarch_targets_scraper, + retropie_targets_scraper, +) + + +def _refuse(*_args, **_kwargs): + raise urllib.error.HTTPError("https://api.github.com/x", 403, "rate limited", {}, None) + + +class FailedRequestsStopTheScrape(unittest.TestCase): + def test_every_scraper_raises(self): + for module in ( + emudeck_targets_scraper, + retropie_targets_scraper, + retroarch_targets_scraper, + ): + with self.subTest(scraper=module.__name__): + with mock.patch.object(module.urllib.request, "urlopen", _refuse): + with self.assertRaises(RuntimeError): + module.Scraper().fetch_targets() + + def test_an_empty_listing_is_not_a_target(self): + class _Empty: + def __enter__(self): + return self + + def __exit__(self, *exc): + return False + + def read(self): + return b"[]" + + for module in (emudeck_targets_scraper, retropie_targets_scraper): + with self.subTest(scraper=module.__name__): + with mock.patch.object( + module.urllib.request, "urlopen", lambda *a, **k: _Empty() + ): + with self.assertRaises(RuntimeError): + module.Scraper().fetch_targets() + + +if __name__ == "__main__": + unittest.main()