fix: stop target scrapes on a failed request

This commit is contained in:
Abdessamad Derraz committed 2026-10-06 05:19:05 +02:00
1 parent f0bab9766c
commit c6ba139493
4 files changed
+92 -48

No files matched your search

@@ -49,7 +49,7 @@ _NAME_OVERRIDES: dict[str, str] = {
_SKIP = {"retroarch_maincfg", "retroarch"}
def _fetch(url: str) -> str | None:
def _fetch(url: str) -> str:
headers = {"User-Agent": "retrobios-scraper/1.0"}
token = os.environ.get("GITHUB_TOKEN")
if token and "api.github.com" in url:
@@ -59,21 +59,18 @@ def _fetch(url: str) -> str | None:
with urllib.request.urlopen(req, timeout=30) as resp:
return resp.read().decode("utf-8")
except urllib.error.URLError as e:
print(f" skip {url}: {e}", file=sys.stderr)
return None
# A target written from a failed request loses its cores in silence.
raise RuntimeError(f"cannot fetch {url}: {e}") from e
def _list_emuscripts(api_url: str) -> list[str]:
"""List emulator script filenames from GitHub API."""
raw = _fetch(api_url)
if not raw:
return []
entries = json.loads(raw)
names = []
for e in entries:
name = e.get("name", "")
if name.endswith(".sh") or name.endswith(".ps1"):
names.append(name)
entries = json.loads(_fetch(api_url))
names = [
e["name"] for e in entries if e.get("name", "").endswith((".sh", ".ps1"))
]
if not names:
raise RuntimeError(f"no emulator scripts listed at {api_url}")
return names
@@ -136,15 +133,13 @@ class Scraper(BaseTargetScraper):
import os
target_path = os.path.join("platforms", "targets", "retroarch.yml")
if not os.path.exists(target_path):
return []
with open(target_path) as f:
data = yaml_load(f) or {}
# Find a target matching the architecture
for tname, tinfo in data.get("targets", {}).items():
if tinfo.get("architecture") == arch:
return tinfo.get("cores", [])
return []
for tinfo in data.get("targets", {}).values():
if tinfo.get("architecture") == arch and tinfo.get("cores"):
return tinfo["cores"]
raise RuntimeError(f"no {arch} core list in {target_path}")
def fetch_targets(self) -> dict:
steamos_cores = self._fetch_cores_for_target(STEAMOS_API, "SteamOS")
@@ -87,7 +87,7 @@ class Scraper(BaseTargetScraper):
def __init__(self, url: str = BUILDBOT_URL):
super().__init__(url=url)
def _fetch_url(self, url: str) -> str | None:
def _fetch_url(self, url: str) -> str:
try:
req = urllib.request.Request(
url, headers={"User-Agent": "retrobios-scraper/1.0"}
@@ -95,14 +95,12 @@ class Scraper(BaseTargetScraper):
with urllib.request.urlopen(req, timeout=30) as resp:
return resp.read().decode("utf-8")
except urllib.error.URLError as e:
print(f" skip {url}: {e}", file=sys.stderr)
return None
# A target written from a failed request loses its cores in silence.
raise RuntimeError(f"cannot fetch {url}: {e}") from e
def _fetch_cores_for_target(self, path: str) -> list[str]:
url = f"{self.url}{path}/"
html = self._fetch_url(url)
if html is None:
return []
cores: list[str] = []
seen: set[str] = set()
for match in _HREF_RE.finditer(html):
@@ -134,10 +132,7 @@ class Scraper(BaseTargetScraper):
def _fetch_cores_for_recipe(self, recipe_path: str) -> list[str]:
url = f"{RECIPE_BASE_URL}{recipe_path}"
text = self._fetch_url(url)
if text is None:
return []
return self._parse_recipe_cores(text)
return self._parse_recipe_cores(self._fetch_url(url))
def fetch_targets(self) -> dict:
targets: dict[str, dict] = {}
@@ -145,8 +140,7 @@ class Scraper(BaseTargetScraper):
print(f" fetching {target_name}...", file=sys.stderr)
cores = self._fetch_cores_for_target(path)
if not cores:
print(f" warning: no cores found for {target_name}", file=sys.stderr)
continue
raise RuntimeError(f"no cores listed for {target_name}")
targets[target_name] = {
"architecture": arch,
"cores": cores,
@@ -156,8 +150,7 @@ class Scraper(BaseTargetScraper):
print(f" fetching {target_name} (recipe)...", file=sys.stderr)
cores = self._fetch_cores_for_recipe(recipe_path)
if not cores:
print(f" warning: no cores found for {target_name}", file=sys.stderr)
continue
raise RuntimeError(f"no cores listed for {target_name}")
targets[target_name] = {
"architecture": arch,
"cores": cores,
@@ -59,7 +59,7 @@ _MODULE_ID_RE = re.compile(r'rp_module_id\s*=\s*["\']([^"\']+)["\']')
_MODULE_FLAGS_RE = re.compile(r'rp_module_flags\s*=\s*["\']([^"\']*)["\']')
def _fetch(url: str, accept: str = "text/plain") -> str | None:
def _fetch(url: str, accept: str = "text/plain") -> str:
headers = {"User-Agent": "retrobios-scraper/1.0", "Accept": accept}
token = os.environ.get("GITHUB_TOKEN")
if token and "api.github.com" in url:
@@ -69,8 +69,8 @@ def _fetch(url: str, accept: str = "text/plain") -> str | None:
with urllib.request.urlopen(req, timeout=30) as resp:
return resp.read().decode("utf-8")
except urllib.error.URLError as e:
print(f" skip {url}: {e}", file=sys.stderr)
return None
# A target written from a failed request loses its cores in silence.
raise RuntimeError(f"cannot fetch {url}: {e}") from e
def _is_available(flags_str: str, platform: str) -> bool:
@@ -111,32 +111,24 @@ class Scraper(BaseTargetScraper):
def _list_scriptmodules(self) -> list[str]:
"""Return list of .sh filenames from the libretrocores directory."""
raw = _fetch(self.url, accept="application/vnd.github+json")
if raw is None:
return []
try:
entries = json.loads(raw)
except json.JSONDecodeError as e:
print(f" JSON parse error: {e}", file=sys.stderr)
return []
return [e["name"] for e in entries if e.get("name", "").endswith(".sh")]
entries = json.loads(_fetch(self.url, accept="application/vnd.github+json"))
names = [e["name"] for e in entries if e.get("name", "").endswith(".sh")]
if not names:
raise RuntimeError(f"no scriptmodules listed at {self.url}")
return names
def _fetch_module(self, filename: str) -> str | None:
def _fetch_module(self, filename: str) -> str:
return _fetch(f"{RAW_BASE_URL}{filename}")
def fetch_targets(self) -> dict:
print(" listing RetroPie scriptmodules...", file=sys.stderr)
filenames = self._list_scriptmodules()
if not filenames:
print(" warning: no scriptmodules found", file=sys.stderr)
# {platform: [core_id, ...]}
platform_cores: dict[str, list[str]] = {p: [] for p in PLATFORM_FLAGS}
for filename in filenames:
content = self._fetch_module(filename)
if content is None:
continue
module_id, flags = _parse_module(content)
if not module_id:
print(f" warning: no rp_module_id in {filename}", file=sys.stderr)