feat: download packs published as split volumes

This commit is contained in:
Abdessamad Derraz committed 2026-09-04 10:18:14 +02:00
1 parent d368b37165
commit 361b958f41
4 files changed
+772 -141

No files matched your search

+179 -118
View File
@@ -3,11 +3,13 @@
Cross-platform tool (Linux/macOS/Windows) using only Python stdlib.
A pack over 2 GB is published as numbered volumes (`.zip.001`, `.zip.002`),
so a platform is a group of assets rather than a single one.
Usage:
python scripts/download.py --list # List platforms
python scripts/download.py retroarch ~/path/ # Download pack
python scripts/download.py --verify retroarch ~/path # Verify local files
python scripts/download.py --info retroarch # Show coverage info
python scripts/download.py --info retroarch # Show pack info
"""
from __future__ import annotations
@@ -15,22 +17,63 @@ from __future__ import annotations
import argparse
import json
import os
import re
import shutil
import sys
import urllib.error
import urllib.parse
import urllib.request
import zipfile
from dataclasses import dataclass
from pathlib import Path
sys.path.insert(0, os.path.dirname(__file__))
from common import compute_hashes, safe_extract_zip
GITHUB_API = "https://api.github.com"
DEFAULT_API = "https://api.github.com"
REPO = "Abdess/retrobios"
PACK_SUFFIX = "_BIOS_Pack.zip"
CHECKSUMS_ASSET = "SHA256SUMS.txt"
STAGING_DIR = ".retrobios-download"
MAX_CHECKSUMS_BYTES = 1 << 20
CHUNK = 1 << 20
_VOLUME_RE = re.compile(rf"^(?P<base>.+{re.escape(PACK_SUFFIX)})\.(?P<index>\d+)$")
_LOOPBACK_HOSTS = {"127.0.0.1", "::1", "localhost"}
def _checked_url(value: str, label: str) -> str:
"""Refuse a URL that is neither HTTPS nor loopback.
This endpoint names the assets and, through them, the bytes that land in
the BIOS directory, so it is the trust anchor of the whole download.
Plain HTTP to loopback stays allowed: it is how this is tested end to end.
"""
parsed = urllib.parse.urlparse(value)
if parsed.scheme == "https":
return value.rstrip("/")
host = (parsed.hostname or "").lower()
if parsed.scheme == "http" and host in _LOOPBACK_HOSTS:
return value.rstrip("/")
print(f"Error: {label} must use HTTPS, got {value!r}", file=sys.stderr)
sys.exit(1)
API = _checked_url(os.environ.get("RETROBIOS_API", DEFAULT_API), "RETROBIOS_API")
@dataclass(frozen=True)
class Pack:
"""A platform pack: one asset, or the volumes it was split into."""
name: str
platform: str
parts: tuple[dict, ...]
size: int
def get_latest_release() -> dict:
"""Fetch latest release info from GitHub API."""
url = f"{GITHUB_API}/repos/{REPO}/releases/latest"
url = f"{API}/repos/{REPO}/releases/latest"
req = urllib.request.Request(
url,
headers={
@@ -49,31 +92,75 @@ def get_latest_release() -> dict:
raise
def list_platforms(release: dict) -> list[str]:
"""List available platform packs from release assets."""
platforms = []
def group_packs(release: dict) -> list[Pack]:
"""Group release assets into packs, volumes folded into their archive."""
groups: dict[str, list[tuple[int, dict]]] = {}
for asset in release.get("assets", []):
name = asset["name"]
if name.endswith("_BIOS_Pack.zip"):
platform = name.replace("_BIOS_Pack.zip", "").replace("_", " ")
platforms.append(platform)
return sorted(platforms)
volume = _VOLUME_RE.match(name)
if volume:
groups.setdefault(volume["base"], []).append((int(volume["index"]), asset))
elif name.endswith(PACK_SUFFIX):
groups.setdefault(name, []).append((0, asset))
packs = []
for base, volumes in sorted(groups.items()):
parts = tuple(asset for _, asset in sorted(volumes, key=lambda v: v[0]))
packs.append(
Pack(
name=base,
platform=base[: -len(PACK_SUFFIX)].replace("_", " "),
parts=parts,
size=sum(part.get("size", 0) for part in parts),
)
)
return packs
def find_asset(release: dict, platform: str) -> dict | None:
"""Find the release asset for a specific platform."""
normalized = platform.lower().replace(" ", "_").replace("-", "_")
def list_platforms(release: dict) -> list[str]:
"""List available platform packs from release assets."""
return sorted(pack.platform for pack in group_packs(release))
for asset in release.get("assets", []):
asset_name = asset["name"].lower().replace(" ", "_").replace("-", "_")
if normalized in asset_name and asset_name.endswith("_bios_pack.zip"):
return asset
def _match_key(value: str) -> str:
"""Letters and digits only: the platform is `misterfpga`, the asset MiSTer_FPGA."""
return re.sub(r"[^a-z0-9]", "", value.lower())
def find_pack(release: dict, platform: str) -> Pack | None:
"""Find the pack for a platform name."""
needle = _match_key(platform)
if not needle:
return None
for pack in group_packs(release):
if needle in _match_key(pack.name):
return pack
return None
def download_file(url: str, dest: str, expected_size: int = 0):
def fetch_checksums(release: dict) -> dict[str, str]:
"""Read the published SHA-256 of each pack, keyed by archive name."""
for asset in release.get("assets", []):
if asset["name"] != CHECKSUMS_ASSET:
continue
url = _checked_url(asset["browser_download_url"], "asset URL")
req = urllib.request.Request(
url, headers={"User-Agent": "retrobios-downloader/1.0"}
)
with urllib.request.urlopen(req, timeout=60) as resp:
body = resp.read(MAX_CHECKSUMS_BYTES).decode("utf-8", "replace")
sums = {}
for line in body.splitlines():
digest, _, name = line.strip().partition(" ")
if digest and name:
sums[name] = digest.lower()
return sums
return {}
def download_file(url: str, dest: Path, expected_size: int = 0, label: str = ""):
"""Download a file with progress indication."""
url = _checked_url(url, "asset URL")
req = urllib.request.Request(
url, headers={"User-Agent": "retrobios-downloader/1.0"}
)
@@ -94,95 +181,76 @@ def download_file(url: str, dest: str, expected_size: int = 0):
pct = downloaded * 100 // total
bar = "=" * (pct // 2) + " " * (50 - pct // 2)
print(
f"\r [{bar}] {pct}% ({downloaded:,}/{total:,})",
f"\r {label}[{bar}] {pct}% ({downloaded:,}/{total:,})",
end="",
flush=True,
)
print()
if expected_size and downloaded != expected_size:
raise ValueError(f"downloaded {downloaded} bytes; expected {expected_size}")
def extract_pack(zip_path: str, dest_dir: str):
"""Extract a BIOS pack ZIP to destination."""
with zipfile.ZipFile(zip_path, "r") as zf:
members = zf.namelist()
print(f" Extracting {len(members)} files to {dest_dir}/")
safe_extract_zip(zip_path, dest_dir)
def join_volumes(parts: list[Path], dest: Path) -> None:
"""Concatenate split volumes into one archive, streaming."""
with open(dest, "wb") as out:
for part in parts:
with open(part, "rb") as src:
shutil.copyfileobj(src, out, CHUNK)
def verify_files(platform: str, dest_dir: str, release: dict):
"""Verify local files against database.json from release."""
db_asset = None
for asset in release.get("assets", []):
if asset["name"] == "database.json":
db_asset = asset
break
def fetch_pack(pack: Pack, staging: Path, checksums: dict[str, str]) -> Path:
"""Download every volume of a pack and return the assembled archive."""
staging.mkdir(parents=True, exist_ok=True)
volumes = []
for index, part in enumerate(pack.parts, start=1):
label = f"part {index}/{len(pack.parts)} " if len(pack.parts) > 1 else ""
target = staging / part["name"]
print(f"Downloading {part['name']} ({part.get('size', 0):,} bytes)...")
download_file(part["browser_download_url"], target, part.get("size", 0), label)
volumes.append(target)
if not db_asset:
print("No database.json found in release assets. Cannot verify.")
return
archive = staging / pack.name
if volumes != [archive]:
if len(volumes) > 1:
print(f"Joining {len(volumes)} parts into {pack.name}...")
join_volumes(volumes, archive)
for volume in volumes:
volume.unlink()
else:
volumes[0].replace(archive)
import tempfile
tmp = tempfile.NamedTemporaryFile(suffix=".json", delete=False)
tmp.close()
try:
download_file(
db_asset["browser_download_url"], tmp.name, db_asset.get("size", 0)
)
with open(tmp.name) as f:
db = json.load(f)
finally:
os.unlink(tmp.name)
dest = Path(dest_dir)
verified = 0
missing = 0
mismatched = 0
for sha1, entry in db.get("files", {}).items():
name = entry["name"]
found = False
for local_file in dest.rglob(name):
if local_file.is_file():
local_sha1 = compute_hashes(local_file)["sha1"]
if local_sha1 == sha1:
verified += 1
found = True
break
else:
mismatched += 1
print(
f" MISMATCH: {name} (expected {sha1[:12]}..., got {local_sha1[:12]}...)"
)
found = True
break
if not found:
missing += 1
total = verified + missing + mismatched
print(f"\n Verified: {verified}/{total}")
if missing:
print(f" Missing: {missing}")
if mismatched:
print(f" Mismatched: {mismatched}")
expected = checksums.get(pack.name)
if expected:
print("Checking the archive...")
actual = compute_hashes(str(archive))["sha256"].lower()
if actual != expected:
archive.unlink()
print(
f"Error: checksum mismatch for {pack.name}\n"
f" expected {expected}\n got {actual}\n"
"Download the parts again; a truncated part gives this.",
file=sys.stderr,
)
sys.exit(1)
else:
print(f"No {CHECKSUMS_ASSET} in the release, skipping the checksum.")
return archive
def show_info(platform: str, release: dict):
"""Show coverage information for a platform."""
asset = find_asset(release, platform)
if not asset:
"""Show pack information for a platform."""
pack = find_pack(release, platform)
if not pack:
print(f"Platform '{platform}' not found in release")
return
print(f" Platform: {platform}")
print(f" File: {asset['name']}")
print(f" Size: {asset['size']:,} bytes ({asset['size'] / (1024 * 1024):.1f} MB)")
print(f" Downloads: {asset.get('download_count', 'N/A')}")
print(f" Updated: {asset.get('updated_at', 'N/A')}")
print(f" Platform: {pack.platform}")
print(f" File: {pack.name}")
print(f" Parts: {len(pack.parts)} part{'s' if len(pack.parts) > 1 else ''}")
print(f" Size: {pack.size:,} bytes ({pack.size / (1024 * 1024):.1f} MB)")
for part in pack.parts:
print(f" {part['name']} ({part.get('size', 0):,} bytes)")
def main():
@@ -193,14 +261,14 @@ def main():
Examples:
%(prog)s --list List available platforms
%(prog)s retroarch ~/RetroArch/system Download RetroArch pack
%(prog)s --verify retroarch ~/path Verify local files
%(prog)s --info retroarch Show pack info
To check files already in place, use: python install.py --check
""",
)
parser.add_argument("platform", nargs="?", help="Platform name")
parser.add_argument("dest", nargs="?", help="Destination directory")
parser.add_argument("--list", action="store_true", help="List available platforms")
parser.add_argument("--verify", action="store_true", help="Verify existing files")
parser.add_argument("--info", action="store_true", help="Show platform info")
args = parser.parse_args()
@@ -220,7 +288,8 @@ Examples:
OSError,
json.JSONDecodeError,
) as e:
print(f"Error: {e}")
print(f"Error: {e}", file=sys.stderr)
sys.exit(1)
return
if not args.platform:
@@ -229,43 +298,35 @@ Examples:
try:
release = get_latest_release()
except (urllib.error.URLError, urllib.error.HTTPError, OSError) as e:
print(f"Error fetching release info: {e}")
print(f"Error fetching release info: {e}", file=sys.stderr)
sys.exit(1)
if args.info:
show_info(args.platform, release)
return
if args.verify:
if not args.dest:
parser.error("Destination directory required for --verify")
verify_files(args.platform, args.dest, release)
return
if not args.dest:
parser.error("Destination directory required")
asset = find_asset(release, args.platform)
if not asset:
pack = find_pack(release, args.platform)
if not pack:
print(f"Platform '{args.platform}' not found in release.")
print("Available:", ", ".join(list_platforms(release)))
sys.exit(1)
import tempfile
dest = Path(os.path.expanduser(args.dest))
dest.mkdir(parents=True, exist_ok=True)
fd, zip_path = tempfile.mkstemp(suffix=".zip")
os.close(fd)
print(f"Downloading {asset['name']} ({asset['size']:,} bytes)...")
download_file(asset["browser_download_url"], zip_path, asset["size"])
dest = os.path.expanduser(args.dest)
os.makedirs(dest, exist_ok=True)
print(f"Extracting to {dest}/...")
extract_pack(zip_path, dest)
os.unlink(zip_path)
# Staged inside the destination: a pack is gigabytes, and that is the
# filesystem the user picked for them. The system temp directory is a RAM
# disk on the appliances these packs target.
staging = dest / STAGING_DIR
try:
archive = fetch_pack(pack, staging, fetch_checksums(release))
print(f"Extracting to {dest}/...")
safe_extract_zip(str(archive), str(dest))
finally:
shutil.rmtree(staging, ignore_errors=True)
print("Done!")
+145 -23
View File
@@ -1,16 +1,32 @@
#!/usr/bin/env bash
# Download BIOS pack from GitHub Releases (Linux/macOS one-liner compatible)
#
# A pack over 2 GB is published as numbered volumes (.zip.001, .zip.002),
# which are downloaded, joined and checked here.
#
# Usage:
# bash scripts/download.sh retroarch ~/RetroArch/system/
# bash scripts/download.sh --list
#
# Requires: curl, unzip, jq (optional, for --list)
# Requires: curl, unzip
set -euo pipefail
REPO="Abdess/retrobios"
API="https://api.github.com/repos/${REPO}/releases/latest"
API_BASE="${RETROBIOS_API:-https://api.github.com}"
# This endpoint names the assets and so the bytes that land in the BIOS
# directory. Plain HTTP stays allowed to loopback, which is how the script
# is exercised end to end.
case "$API_BASE" in
https://*) ;;
http://127.0.0.1[:/]*|http://localhost[:/]*|"http://[::1]"[:/]*) ;;
*)
echo "Error: RETROBIOS_API must use HTTPS, got '$API_BASE'" >&2
exit 1
;;
esac
API="${API_BASE%/}/repos/${REPO}/releases/latest"
usage() {
echo "Usage: $0 <platform> <destination>"
@@ -25,53 +41,159 @@ usage() {
exit 1
}
# One asset URL per line, in the order the release lists them.
asset_urls() {
printf '%s' "$1" | tr ',' '\n' |
sed -n 's/.*"browser_download_url"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p'
}
# The archive name behind each asset, volumes folded into their archive.
pack_names() {
asset_urls "$1" | sed 's#.*/##' |
sed -n 's/\(.*_BIOS_Pack\.zip\)\(\.[0-9][0-9]*\)\{0,1\}$/\1/p' |
LC_ALL=C sort -u
}
platform_names() {
pack_names "$1" | sed 's/_BIOS_Pack\.zip$//' | tr '_' ' '
}
# Letters and digits only: the platform is `misterfpga`, the asset MiSTer_FPGA.
normalize() {
printf '%s' "$1" | tr '[:upper:]' '[:lower:]' | tr -cd 'a-z0-9'
}
fetch_release() {
curl -fsSL "$API"
}
list_platforms() {
echo "Fetching available platforms..."
if command -v jq &>/dev/null; then
curl -sL "$API" | jq -r '.assets[].name' | grep '_BIOS_Pack.zip' | sed 's/_BIOS_Pack.zip//' | tr '_' ' '
else
curl -sL "$API" | grep -oP '"name":\s*"\K[^"]*_BIOS_Pack\.zip' | sed 's/_BIOS_Pack.zip//' | tr '_' ' '
fi
platform_names "$(fetch_release)"
}
download_pack() {
local platform="$1"
local dest="$2"
local normalized
normalized=$(echo "$platform" | tr ' ' '_' | tr '[:upper:]' '[:lower:]')
normalized=$(normalize "$platform")
echo "Fetching release info..."
local release_json
release_json=$(curl -sL "$API")
release_json=$(fetch_release)
# Find matching asset URL
local download_url
download_url=$(echo "$release_json" | grep -oP "\"browser_download_url\":\s*\"[^\"]*${normalized}[^\"]*_BIOS_Pack\.zip\"" | head -1 | grep -oP 'https://[^"]+')
local archive=""
local candidate
while read -r candidate; do
[ -n "$candidate" ] || continue
case "$(normalize "$candidate")" in
*"$normalized"*)
archive="$candidate"
break
;;
esac
done <<EOF
$(pack_names "$release_json")
EOF
if [[ -z "$download_url" ]]; then
if [ -z "$archive" ]; then
echo "Error: Platform '$platform' not found in latest release."
echo "Available platforms:"
list_platforms
platform_names "$release_json"
exit 1
fi
local filename
filename=$(basename "$download_url")
local volumes
volumes=$(asset_urls "$release_json" |
grep -E "/${archive//./\\.}(\.[0-9]+)?$" | LC_ALL=C sort)
if [ -z "$volumes" ]; then
echo "Error: no asset found for ${archive}." >&2
exit 1
fi
local tmpfile
tmpfile=$(mktemp "/tmp/${filename}.XXXXXX")
# Staged inside the destination: a pack is gigabytes, and /tmp is a RAM
# disk on the appliances these packs target.
mkdir -p "$dest"
local staging="${dest%/}/.retrobios-download"
rm -rf "$staging"
mkdir -p "$staging"
trap 'rm -rf "$staging"' EXIT
echo "Downloading ${filename}..."
curl -L --progress-bar -o "$tmpfile" "$download_url"
local count
count=$(printf '%s\n' "$volumes" | wc -l | tr -d ' ')
local index=0
local url name
while read -r url; do
[ -n "$url" ] || continue
index=$((index + 1))
name=$(basename "$url")
if [ "$count" -gt 1 ]; then
echo "Downloading ${name} (part ${index}/${count})..."
else
echo "Downloading ${name}..."
fi
curl -fL --progress-bar -o "${staging}/${name}" "$url"
done <<EOF
$volumes
EOF
if [ "$count" -gt 1 ]; then
echo "Joining ${count} parts into ${archive}..."
# split(1) writes plain byte ranges, so concatenation rebuilds the ZIP.
cat "${staging}/${archive}".[0-9][0-9][0-9] > "${staging}/${archive}"
rm -f "${staging}/${archive}".[0-9][0-9][0-9]
fi
verify_checksum "$release_json" "$staging" "$archive"
echo "Extracting to ${dest}/..."
mkdir -p "$dest"
unzip -o -q "$tmpfile" -d "$dest"
unzip -o -q "${staging}/${archive}" -d "$dest"
rm -f "$tmpfile"
rm -rf "$staging"
trap - EXIT
echo "Done! BIOS files extracted to ${dest}/"
}
verify_checksum() {
local release_json="$1" staging="$2" archive="$3"
local sums_url
sums_url=$(asset_urls "$release_json" | grep -E '/SHA256SUMS\.txt$' | head -1)
if [ -z "$sums_url" ]; then
echo "No SHA256SUMS.txt in the release, skipping the checksum."
return 0
fi
local hasher
if command -v sha256sum >/dev/null 2>&1; then
hasher="sha256sum"
elif command -v shasum >/dev/null 2>&1; then
hasher="shasum -a 256"
else
echo "No sha256sum available, skipping the checksum."
return 0
fi
local expected
expected=$(curl -fsSL "$sums_url" | awk -v name="$archive" '$2 == name {print $1}')
if [ -z "$expected" ]; then
echo "No checksum published for ${archive}, skipping."
return 0
fi
echo "Checking the archive..."
local actual
actual=$($hasher "${staging}/${archive}" | cut -d' ' -f1)
if [ "$actual" != "$expected" ]; then
echo "Error: checksum mismatch for ${archive}" >&2
echo " expected ${expected}" >&2
echo " got ${actual}" >&2
echo "Download the parts again; a truncated part gives this." >&2
exit 1
fi
}
# Main
case "${1:-}" in
--list|-l)