fix: stop a scrape whose release tag is unread

This commit is contained in:
Abdessamad Derraz committed 2026-10-06 12:22:51 +02:00
1 parent e35aa9a323
commit 139bb99088
11 files changed
+133 -100

No files matched your search

+30 -13
View File
@@ -3,6 +3,7 @@
from __future__ import annotations
import json
import os
import sys
import urllib.error
import urllib.request
@@ -225,22 +226,38 @@ class BaseScraper(ABC):
...
def fetch_github_latest_version(repo: str) -> str | None:
"""Fetch the latest release version tag from a GitHub repo."""
def github_headers() -> dict[str, str]:
"""Headers for api.github.com, authenticated when a token is set.
Anonymous calls share a quota of 60 per hour, so a scrape run after a
few others gets 403 for every request.
"""
headers = {
"User-Agent": "retrobios-scraper/1.0",
"Accept": "application/vnd.github.v3+json",
}
token = os.environ.get("GITHUB_TOKEN")
if token:
headers["Authorization"] = f"Bearer {token}"
return headers
def fetch_github_latest_version(repo: str) -> str:
"""Return the tag of the latest release of a GitHub repo.
Raises instead of returning a fallback: a platform YAML written without
the release it was read from cannot be patched back by the exporter.
"""
url = f"https://api.github.com/repos/{repo}/releases/latest"
try:
req = urllib.request.Request(
url,
headers={
"User-Agent": "retrobios-scraper/1.0",
"Accept": "application/vnd.github.v3+json",
},
)
req = urllib.request.Request(url, headers=github_headers())
with urllib.request.urlopen(req, timeout=15) as resp:
data = json.loads(resp.read())
return data.get("tag_name", "")
except (urllib.error.URLError, urllib.error.HTTPError, json.JSONDecodeError):
return None
tag = json.loads(resp.read()).get("tag_name", "")
except (urllib.error.URLError, json.JSONDecodeError) as e:
raise RuntimeError(f"cannot read the latest release of {repo}: {e}") from e
if not tag:
raise RuntimeError(f"the latest release of {repo} names no tag")
return tag
class _PlatformDumper(yaml.SafeDumper):
+28 -41
View File
@@ -19,7 +19,12 @@ from pathlib import Path
from common import yaml_load
from .base_scraper import BaseScraper, BiosRequirement, requirement_entry
from .base_scraper import (
BaseScraper,
BiosRequirement,
github_headers,
requirement_entry,
)
PLATFORM_NAME = "batocera"
@@ -38,39 +43,32 @@ def pick_stable_tag(names: list[str]) -> str | None:
return max(stable)[1] if stable else None
def fetch_stable_tag() -> str | None:
"""Return the newest stable batocera-N(.M) tag name."""
import json
import os
import urllib.error
import urllib.request
def fetch_stable_tag() -> str:
"""Return the newest stable batocera-N(.M) tag name.
Raises rather than falling back to master: the YAML records the tag
it was read from, and the exporter patches that revision.
"""
url = "https://api.github.com/repos/batocera-linux/batocera.linux/tags?per_page=100"
headers = {
"User-Agent": "retrobios-scraper/1.0",
"Accept": "application/vnd.github.v3+json",
}
token = os.environ.get("GITHUB_TOKEN")
if token:
headers["Authorization"] = f"Bearer {token}"
try:
req = urllib.request.Request(url, headers=headers)
req = urllib.request.Request(url, headers=github_headers())
with urllib.request.urlopen(req, timeout=15) as resp:
tags = json.loads(resp.read())
except (urllib.error.URLError, urllib.error.HTTPError, json.JSONDecodeError):
return None
return pick_stable_tag([tag["name"] for tag in tags])
except (urllib.error.URLError, json.JSONDecodeError) as e:
raise RuntimeError(f"cannot list batocera tags: {e}") from e
tag = pick_stable_tag([tag["name"] for tag in tags])
if not tag:
raise RuntimeError("no stable batocera tag among the latest 100")
return tag
_STABLE_TAG = fetch_stable_tag() or "master"
SOURCE_URL = (
f"{_RAW_BASE}/{_STABLE_TAG}"
_RAW_BASE + "/{tag}"
"/package/batocera/core/batocera-scripts/scripts/batocera-systems"
)
CONFIGGEN_DEFAULTS_URL = (
f"{_RAW_BASE}/{_STABLE_TAG}"
_RAW_BASE + "/{tag}"
"/package/batocera/core/batocera-configgen/configs/"
"configgen-defaults.yml"
)
@@ -186,8 +184,9 @@ def _resolve_truncated_md5(md5: str, md5_index: dict[str, str]) -> str:
class Scraper(BaseScraper):
"""Scraper for batocera-systems Python dict."""
def __init__(self, url: str = SOURCE_URL):
super().__init__(url=url)
def __init__(self):
self.tag = fetch_stable_tag()
super().__init__(url=SOURCE_URL.format(tag=self.tag))
def _fetch_cores(self) -> tuple[list[str], list[str]]:
"""Extract core names and standalone cores from configgen-defaults.yml.
@@ -195,16 +194,17 @@ class Scraper(BaseScraper):
Returns (all_cores, standalone_cores) where standalone_cores are
those with emulator != "libretro".
"""
url = CONFIGGEN_DEFAULTS_URL.format(tag=self.tag)
try:
req = urllib.request.Request(
CONFIGGEN_DEFAULTS_URL,
url,
headers={"User-Agent": "retrobios-scraper/1.0"},
)
with urllib.request.urlopen(req, timeout=30) as resp:
raw = resp.read().decode("utf-8")
except urllib.error.URLError as e:
raise ConnectionError(
f"Failed to fetch {CONFIGGEN_DEFAULTS_URL}: {e}"
f"Failed to fetch {url}: {e}"
) from e
data = yaml_load(raw)
cores: set[str] = set()
@@ -372,25 +372,12 @@ class Scraper(BaseScraper):
systems[req.system]["files"].append(requirement_entry(req))
batocera_version = ""
if _STABLE_TAG != "master":
batocera_version = _STABLE_TAG.removeprefix("batocera-")
if not batocera_version:
# Preserve existing version when fetch fails (offline mode)
existing = (
Path(__file__).resolve().parents[2] / "platforms" / "batocera.yml"
)
if existing.exists():
with open(existing) as f:
old = yaml_load(f) or {}
batocera_version = str(old.get("version", ""))
cores, standalone = self._fetch_cores()
result = {
"platform": "Batocera",
"version": batocera_version or "",
"version": self.tag.removeprefix("batocera-"),
"homepage": "https://batocera.org",
"source": SOURCE_URL,
"source": self.url,
"base_destination": "bios",
"hash_type": "md5",
"verification_mode": "md5",
+5 -8
View File
@@ -41,10 +41,8 @@ PLATFORM_NAME = "bizhawk"
GITHUB_REPO = "TASEmulators/BizHawk"
_STABLE_TAG = fetch_github_latest_version(GITHUB_REPO) or "master"
SOURCE_URL = (
f"https://raw.githubusercontent.com/TASEmulators/BizHawk/{_STABLE_TAG}"
"https://raw.githubusercontent.com/TASEmulators/BizHawk/{tag}"
"/src/BizHawk.Emulation.Common/Database/FirmwareDatabase.cs"
)
@@ -332,7 +330,8 @@ class Scraper(BaseScraper):
"""BizHawk firmware database scraper."""
def __init__(self):
super().__init__(url=SOURCE_URL)
self.tag = fetch_github_latest_version(GITHUB_REPO)
super().__init__(url=SOURCE_URL.format(tag=self.tag))
def validate_format(self, raw_data: str) -> bool:
return "FirmwareDatabase" in raw_data and "FirmwareAndOption" in raw_data
@@ -372,13 +371,11 @@ class Scraper(BaseScraper):
systems[req.system]["files"].append(requirement_entry(req))
version = _STABLE_TAG if _STABLE_TAG != "master" else ""
return {
"platform": "BizHawk",
"version": version,
"version": self.tag,
"homepage": "https://tasvideos.org/BizHawk",
"source": SOURCE_URL,
"source": self.url,
"base_destination": "Firmware",
"hash_type": "sha1",
"verification_mode": "sha1",
+1 -1
View File
@@ -295,7 +295,7 @@ class Scraper(BaseScraper):
def fetch_metadata(self) -> dict:
"""Fetch version info from GitHub."""
version = fetch_github_latest_version("libretro/libretro-core-info")
return {"version": version or ""}
return {"version": version}
def main():
+1 -7
View File
@@ -426,13 +426,7 @@ class Scraper(BaseScraper):
systems[req.system]["files"].append(requirement_entry(req))
version = ""
try:
v = fetch_github_latest_version("dragoonDorise/EmuDeck")
if v:
version = v
except (ConnectionError, ValueError, OSError):
pass
version = fetch_github_latest_version("dragoonDorise/EmuDeck")
cores = self._fetch_installed_emulators()
+12 -15
View File
@@ -21,7 +21,7 @@ from .base_scraper import BaseScraper, BiosRequirement, requirement_entry
PLATFORM_NAME = "recalbox"
def _fetch_gitlab_stable_tag() -> str | None:
def _fetch_gitlab_stable_tag() -> str:
"""Fetch the latest stable x.y.z tag from the Recalbox GitLab."""
import json
import re
@@ -33,16 +33,16 @@ def _fetch_gitlab_stable_tag() -> str | None:
req = urllib.request.Request(url, headers={"User-Agent": "retrobios-scraper/1.0"})
with urllib.request.urlopen(req, timeout=15) as resp:
tags = json.loads(resp.read())
except (urllib.error.URLError, urllib.error.HTTPError, json.JSONDecodeError):
return None
except (urllib.error.URLError, json.JSONDecodeError) as e:
raise RuntimeError(f"cannot list Recalbox tags: {e}") from e
stable = [t["name"] for t in tags if re.fullmatch(r"[0-9]+\.[0-9]+(\.[0-9]+)?", t["name"])]
return stable[0] if stable else None
if not stable:
raise RuntimeError("no stable Recalbox tag among the latest 50")
return stable[0]
_STABLE_TAG = _fetch_gitlab_stable_tag() or "master"
SOURCE_URL = (
f"https://gitlab.com/recalbox/recalbox/-/raw/{_STABLE_TAG}/"
"https://gitlab.com/recalbox/recalbox/-/raw/{tag}/"
"board/recalbox/fsoverlay/recalbox/share_init/system/"
".emulationstation/es_bios.xml"
)
@@ -109,8 +109,9 @@ def split_cores(names) -> tuple[list[str], list[str]]:
class Scraper(BaseScraper):
"""Scraper for Recalbox es_bios.xml."""
def __init__(self, url: str = SOURCE_URL):
super().__init__(url=url)
def __init__(self):
self.tag = _fetch_gitlab_stable_tag()
super().__init__(url=SOURCE_URL.format(tag=self.tag))
def _fetch_cores(self) -> tuple[list[str], list[str]]:
"""Core names from es_bios.xml, and those Recalbox runs standalone.
@@ -217,16 +218,12 @@ class Scraper(BaseScraper):
systems[req.system]["files"].append(requirement_entry(req))
version = _STABLE_TAG if _STABLE_TAG != "master" else ""
if not version:
version = "10.0"
cores, standalone = self._fetch_cores()
return {
"platform": "Recalbox",
"version": version,
"version": self.tag,
"homepage": "https://www.recalbox.com",
"source": SOURCE_URL,
"source": self.url,
"base_destination": "bios",
"hash_type": "md5",
"verification_mode": "md5",
+1 -4
View File
@@ -160,10 +160,7 @@ class Scraper(BaseScraper):
systems[req.system]["files"].append(requirement_entry(req))
version = ""
tag = fetch_github_latest_version(GITHUB_REPO)
if tag:
version = tag
version = fetch_github_latest_version(GITHUB_REPO)
return {
"platform": "RetroBat",
+1 -1
View File
@@ -431,7 +431,7 @@ class Scraper(BaseScraper):
from .base_scraper import fetch_github_latest_version
except ImportError:
from scraper.base_scraper import fetch_github_latest_version
version = fetch_github_latest_version("RetroDECK/RetroDECK") or ""
version = fetch_github_latest_version("RetroDECK/RetroDECK")
return {
"platform": "RetroDECK",
+1 -1
View File
@@ -181,7 +181,7 @@ class Scraper(BaseScraper):
return {
"platform": "ROCKNIX",
"version": fetch_github_latest_version(GITHUB_REPO) or "",
"version": fetch_github_latest_version(GITHUB_REPO),
"homepage": "https://rocknix.org",
"source": SOURCE_URL,
"base_destination": "bios",
+6 -9
View File
@@ -44,10 +44,8 @@ PLATFORM_NAME = "romm"
GITHUB_REPO = "rommapp/romm"
_STABLE_TAG = fetch_github_latest_version(GITHUB_REPO) or "master"
SOURCE_URL = (
f"https://raw.githubusercontent.com/rommapp/romm/{_STABLE_TAG}"
"https://raw.githubusercontent.com/rommapp/romm/{tag}"
"/backend/models/fixtures/known_bios_files.json"
)
@@ -124,8 +122,9 @@ FIRMWARE_MIRRORS: dict[str, tuple[str, ...]] = {
class Scraper(BaseScraper):
"""Scraper for RomM known_bios_files.json."""
def __init__(self, url: str = SOURCE_URL):
super().__init__(url=url)
def __init__(self):
self.tag = fetch_github_latest_version(GITHUB_REPO)
super().__init__(url=SOURCE_URL.format(tag=self.tag))
self._parsed: dict | None = None
def _parse_json(self) -> dict:
@@ -212,14 +211,12 @@ class Scraper(BaseScraper):
systems[req.system]["files"].append(requirement_entry(req))
version = _STABLE_TAG if _STABLE_TAG != "master" else ""
return {
"inherits": "emulatorjs",
"platform": "RomM",
"version": version,
"version": self.tag,
"homepage": "https://romm.app",
"source": SOURCE_URL,
"source": self.url,
"base_destination": "bios",
"hash_type": "sha1",
"verification_mode": "md5",