fix: resolve large file assets renamed by github

This commit is contained in:
Abdessamad Derraz committed 2026-08-07 11:53:29 +02:00
1 parent d4a58c4937
commit 3a5cbc4c6c
4 files changed
+76 -914707

No files matched your search

File diff suppressed because it is too large. Load diff
Binary file not shown.
+29 -14
View File
@@ -1144,22 +1144,37 @@ def fetch_large_file(
else:
return cached
encoded_name = urllib.parse.quote(name)
url = f"https://github.com/{LARGE_FILES_REPO}/releases/download/{LARGE_FILES_RELEASE}/{encoded_name}"
os.makedirs(dest_dir, exist_ok=True)
tmp_path = cached + ".tmp"
try:
req = urllib.request.Request(url, headers={"User-Agent": "retrobios/1.0"})
with urllib.request.urlopen(req, timeout=300) as resp:
with open(tmp_path, "wb") as f:
while True:
chunk = resp.read(65536)
if not chunk:
break
f.write(chunk)
except (urllib.error.URLError, urllib.error.HTTPError):
if os.path.exists(tmp_path):
os.unlink(tmp_path)
# GitHub rewrites spaces to dots in release asset names, so a file whose
# name contains spaces is published under a dotted name.
candidates = [name]
if " " in name:
candidates.append(name.replace(" ", "."))
downloaded = False
for candidate in candidates:
encoded_name = urllib.parse.quote(candidate)
url = (
f"https://github.com/{LARGE_FILES_REPO}/releases/download/"
f"{LARGE_FILES_RELEASE}/{encoded_name}"
)
try:
req = urllib.request.Request(url, headers={"User-Agent": "retrobios/1.0"})
with urllib.request.urlopen(req, timeout=300) as resp:
with open(tmp_path, "wb") as f:
while True:
chunk = resp.read(65536)
if not chunk:
break
f.write(chunk)
downloaded = True
break
except (urllib.error.URLError, urllib.error.HTTPError):
if os.path.exists(tmp_path):
os.unlink(tmp_path)
if not downloaded:
return None
if expected_sha1 or expected_md5:
+47
View File
@@ -4885,6 +4885,53 @@ struct BurnDriver BurnDrvneogeo = {
self.assertIsNone(local)
self.assertEqual(status, "not_found")
def test_220_large_file_dotted_asset_fallback(self):
"""Files with spaces resolve to GitHub's dotted asset name."""
import tempfile
import urllib.error
import common
attempts = []
class FakeResponse:
def __init__(self, payload):
self._payload = payload
def read(self, _size=None):
data, self._payload = self._payload, b""
return data
def __enter__(self):
return self
def __exit__(self, *_):
return False
def fake_urlopen(req, timeout=None):
url = req.full_url if hasattr(req, "full_url") else str(req)
attempts.append(url)
if "MAME%200.174" in url:
raise urllib.error.HTTPError(url, 404, "Not Found", {}, None)
return FakeResponse(b"payload")
original = common.urllib.request.urlopen
common.urllib.request.urlopen = fake_urlopen
try:
with tempfile.TemporaryDirectory() as tmpdir:
path = common.fetch_large_file(
"MAME 0.174 Arcade XML.dat", dest_dir=tmpdir
)
self.assertIsNotNone(path)
with open(path, "rb") as fh:
self.assertEqual(fh.read(), b"payload")
finally:
common.urllib.request.urlopen = original
self.assertEqual(len(attempts), 2, attempts)
self.assertIn("MAME%200.174", attempts[0])
self.assertIn("MAME.0.174.Arcade.XML.dat", attempts[1])
def test_219_batocera_stable_tag_selection(self):
"""Stable tag picker takes the highest batocera-N(.M) tag."""
from scripts.scraper.batocera_scraper import pick_stable_tag