Files
libretro/scripts/largefiles.py
T
2026-10-05 02:51:13 +02:00

224 lines
7.5 KiB
Python

"""Files too large for the repository.
They live as release assets and are fetched at build time, verified
against the hash the caller declares."""
from __future__ import annotations
import contextlib
import os
import re
import tempfile
import urllib.error
import urllib.parse
import urllib.request
from collections import Counter
from collections.abc import Iterable
from pathlib import Path
from hashing import compute_hashes
LARGE_FILES_RELEASE = "large-files"
LARGE_FILES_REPO = "Abdess/retrobios"
LARGE_FILES_CACHE = ".cache/large"
GITIGNORE = Path(__file__).resolve().parent.parent / ".gitignore"
_UNSAFE_ASSET_CHARS = re.compile(r"[^A-Za-z0-9._-]+")
def registered_paths(gitignore_text: str) -> list[str]:
"""The bios/ paths .gitignore lists, which are the release assets."""
return [
line.strip()
for line in gitignore_text.splitlines()
if line.strip().startswith("bios/")
]
def load_registered_paths(gitignore: str | Path = GITIGNORE) -> list[str]:
try:
return registered_paths(Path(gitignore).read_text(encoding="utf-8"))
except FileNotFoundError:
return []
def asset_names(registered: Iterable[str]) -> dict[str, str]:
"""Map each registered path to the name of its release asset.
A path whose basename no other registered path shares is published
under that basename. A path whose basename is shared is published under
its location below bios/, segments joined by "--" and every character
outside [A-Za-z0-9._-] replaced by "_":
bios/Id Software/Wolfenstein Enemy Territory/etmain/pak0.pk3 becomes
Id_Software--Wolfenstein_Enemy_Territory--etmain--pak0.pk3. The name
depends on the path only, never on the bytes, so rebuilding a file
keeps its asset.
"""
paths = sorted(set(registered))
counts = Counter(os.path.basename(p) for p in paths)
names: dict[str, str] = {}
for registered_path in paths:
base = os.path.basename(registered_path)
if counts[base] == 1:
names[registered_path] = base
continue
rel = registered_path.removeprefix("bios/")
names[registered_path] = "--".join(
_UNSAFE_ASSET_CHARS.sub("_", part) for part in rel.split("/")
)
owners: dict[str, str] = {}
for registered_path, name in names.items():
# GitHub publishes a space as a dot, so both spellings are taken.
for spelling in {name, name.replace(" ", ".")}:
other = owners.setdefault(spelling, registered_path)
if other != registered_path:
raise ValueError(
f"release asset {spelling!r} named by both {other!r} "
f"and {registered_path!r}"
)
return names
def asset_name(path: str, registered: Iterable[str]) -> str:
"""The release asset name of *path* among the *registered* paths."""
return asset_names([*registered, path])[path]
def asset_candidates(name: str, registered: Iterable[str]) -> list[str]:
"""Assets that may hold *name*, a registered path or a bare file name.
A bare name shared by several registered paths has one asset per path;
the caller's hash tells them apart.
"""
registered = list(registered)
if "/" in name:
return [asset_name(name, registered)]
names = asset_names(registered)
matches = [
names[p] for p in sorted(names) if os.path.basename(p) == name
]
return matches or [name]
def fetch_large_file(
name: str,
dest_dir: str = LARGE_FILES_CACHE,
expected_sha1: str = "",
expected_md5: str = "",
*,
offline: bool = False,
registered: Iterable[str] | None = None,
) -> str | None:
"""Return a verified cached large file, downloading it only when allowed.
*name* is a registered bios/ path or a bare file name; asset_names()
turns it into the asset to fetch, and a bare name shared by several
registered paths tries each of their assets until one verifies.
"""
if registered is None:
registered = load_registered_paths()
for asset in asset_candidates(name, registered):
cached = _fetch_asset(
asset, dest_dir, expected_sha1, expected_md5, offline=offline
)
if cached:
return cached
return None
def _fetch_asset(
name: str,
dest_dir: str,
expected_sha1: str,
expected_md5: str,
*,
offline: bool,
) -> str | None:
cached = os.path.join(dest_dir, name)
# Between the existence test and the hash, a concurrent run can drop the
# same stale entry: the file is gone by the time this one reads it, and
# both of them try to unlink it.
def _drop(path: str) -> None:
with contextlib.suppress(FileNotFoundError):
os.unlink(path)
if os.path.exists(cached):
try:
hashes = compute_hashes(cached) if (expected_sha1 or expected_md5) else {}
except FileNotFoundError:
hashes = None
if hashes is None:
pass
elif expected_sha1 or expected_md5:
if expected_sha1 and hashes["sha1"].lower() != expected_sha1.lower():
_drop(cached)
elif expected_md5:
md5_list = [
m.strip().lower() for m in expected_md5.split(",") if m.strip()
]
if hashes["md5"].lower() not in md5_list:
_drop(cached)
else:
return cached
else:
return cached
else:
return cached
if offline:
return None
os.makedirs(dest_dir, exist_ok=True)
# A per-process scratch name: two runs fetching the same asset into one
# shared path interleave their writes into a full-size, corrupt file.
tmp_fd, tmp_path = tempfile.mkstemp(
dir=dest_dir, prefix=os.path.basename(cached) + ".", suffix=".tmp"
)
os.close(tmp_fd)
# GitHub rewrites spaces to dots in release asset names, so a file whose
# name contains spaces is published under a dotted name.
candidates = [name]
if " " in name:
candidates.append(name.replace(" ", "."))
downloaded = False
for candidate in candidates:
encoded_name = urllib.parse.quote(candidate)
url = (
f"https://github.com/{LARGE_FILES_REPO}/releases/download/"
f"{LARGE_FILES_RELEASE}/{encoded_name}"
)
try:
req = urllib.request.Request(url, headers={"User-Agent": "retrobios/1.0"})
with urllib.request.urlopen(req, timeout=300) as resp:
with open(tmp_path, "wb") as f:
while True:
chunk = resp.read(65536)
if not chunk:
break
f.write(chunk)
downloaded = True
break
except (urllib.error.URLError, urllib.error.HTTPError):
if os.path.exists(tmp_path):
os.unlink(tmp_path)
if not downloaded:
if os.path.exists(tmp_path):
os.unlink(tmp_path)
return None
if expected_sha1 or expected_md5:
hashes = compute_hashes(tmp_path)
if expected_sha1 and hashes["sha1"].lower() != expected_sha1.lower():
os.unlink(tmp_path)
return None
if expected_md5:
md5_list = [m.strip().lower() for m in expected_md5.split(",") if m.strip()]
if hashes["md5"].lower() not in md5_list:
os.unlink(tmp_path)
return None
os.replace(tmp_path, cached)
return cached