feat: publish a large pack as zips, not byte ranges

This commit is contained in:
Abdessamad Derraz committed 2026-10-04 18:09:24 +02:00
1 parent 9ea6083b8a
commit 52f682cdd6
12 files changed
+768 -112

No files matched your search

+70 -32
View File
@@ -3,8 +3,11 @@
Cross-platform tool (Linux/macOS/Windows) using only Python stdlib.
A pack over 2 GB is published as numbered volumes (`.zip.001`, `.zip.002`),
so a platform is a group of assets rather than a single one.
A pack over the release asset limit is published in several files, so a
platform is a group of assets rather than a single one. The parts are each
a ZIP (`.part1of2.zip`), checked and extracted one after the other. Releases
up to v2026.09.04 carry byte ranges of one archive instead (`.zip.001`,
`.zip.002`), which are joined first.
Usage:
python scripts/download.py --list # List platforms
@@ -39,6 +42,9 @@ MAX_CHECKSUMS_BYTES = 1 << 20
CHUNK = 1 << 20
_VOLUME_RE = re.compile(rf"^(?P<base>.+{re.escape(PACK_SUFFIX)})\.(?P<index>\d+)$")
_PART_RE = re.compile(
r"^(?P<stem>.+_BIOS_Pack)\.part(?P<index>\d+)of(?P<count>\d+)\.zip$"
)
_LOOPBACK_HOSTS = {"127.0.0.1", "::1", "localhost"}
@@ -64,12 +70,19 @@ API = _checked_url(os.environ.get("RETROBIOS_API", DEFAULT_API), "RETROBIOS_API"
@dataclass(frozen=True)
class Pack:
"""A platform pack: one asset, or the volumes it was split into."""
"""A platform pack: one asset, or the files it was published in.
`joined` says the files are byte ranges of one archive. `expected_parts`
is the count a part's own name declares, so a release that lists fewer
is recognised before anything is downloaded.
"""
name: str
platform: str
parts: tuple[dict, ...]
size: int
joined: bool = False
expected_parts: int = 1
def get_latest_release() -> dict:
@@ -94,13 +107,21 @@ def get_latest_release() -> dict:
def group_packs(release: dict) -> list[Pack]:
"""Group release assets into packs, volumes folded into their archive."""
"""Group release assets into packs, the files of one pack folded together."""
groups: dict[str, list[tuple[int, dict]]] = {}
joined: set[str] = set()
expected: dict[str, int] = {}
for asset in release.get("assets", []):
name = asset["name"]
volume = _VOLUME_RE.match(name)
part = _PART_RE.match(name)
if volume:
groups.setdefault(volume["base"], []).append((int(volume["index"]), asset))
joined.add(volume["base"])
elif part:
base = f"{part['stem']}.zip"
groups.setdefault(base, []).append((int(part["index"]), asset))
expected[base] = int(part["count"])
elif name.endswith(PACK_SUFFIX):
groups.setdefault(name, []).append((0, asset))
@@ -113,6 +134,8 @@ def group_packs(release: dict) -> list[Pack]:
platform=base[: -len(PACK_SUFFIX)].replace("_", " "),
parts=parts,
size=sum(part.get("size", 0) for part in parts),
joined=base in joined,
expected_parts=expected.get(base, len(parts)),
)
)
return packs
@@ -213,8 +236,34 @@ def make_staging(dest: Path) -> Path:
return Path(tempfile.mkdtemp(prefix=STAGING_PREFIX, dir=dest))
def fetch_pack(pack: Pack, staging: Path, checksums: dict[str, str]) -> Path:
"""Download every volume of a pack and return the assembled archive."""
def _check(archive: Path, checksums: dict[str, str]) -> None:
"""Compare a downloaded file with its published SHA-256, when there is one."""
expected = checksums.get(archive.name)
if not expected:
print(f"No checksum published for {archive.name}, skipping.")
return
print(f"Checking {archive.name}...")
actual = compute_hashes(str(archive))["sha256"].lower()
if actual != expected:
archive.unlink()
print(
f"Error: checksum mismatch for {archive.name}\n"
f" expected {expected}\n got {actual}\n"
"Download the parts again; a truncated part gives this.",
file=sys.stderr,
)
sys.exit(1)
def fetch_pack(pack: Pack, staging: Path, checksums: dict[str, str]) -> list[Path]:
"""Download a pack and return the archives to extract, each one checked."""
if len(pack.parts) != pack.expected_parts:
print(
f"Error: the release lists {len(pack.parts)} of "
f"{pack.expected_parts} parts of {pack.name}.",
file=sys.stderr,
)
sys.exit(1)
staging.mkdir(parents=True, exist_ok=True)
volumes = []
for index, part in enumerate(pack.parts, start=1):
@@ -224,32 +273,20 @@ def fetch_pack(pack: Pack, staging: Path, checksums: dict[str, str]) -> Path:
download_file(part["browser_download_url"], target, part.get("size", 0), label)
volumes.append(target)
archive = staging / pack.name
if volumes != [archive]:
if len(volumes) > 1:
print(f"Joining {len(volumes)} parts into {pack.name}...")
join_volumes(volumes, archive)
for volume in volumes:
volume.unlink()
else:
volumes[0].replace(archive)
if pack.joined:
archive = staging / pack.name
print(f"Joining {len(volumes)} parts into {pack.name}...")
join_volumes(volumes, archive)
for volume in volumes:
volume.unlink()
volumes = [archive]
expected = checksums.get(pack.name)
if expected:
print("Checking the archive...")
actual = compute_hashes(str(archive))["sha256"].lower()
if actual != expected:
archive.unlink()
print(
f"Error: checksum mismatch for {pack.name}\n"
f" expected {expected}\n got {actual}\n"
"Download the parts again; a truncated part gives this.",
file=sys.stderr,
)
sys.exit(1)
else:
if not checksums:
print(f"No {CHECKSUMS_ASSET} in the release, skipping the checksum.")
return archive
return volumes
for archive in volumes:
_check(archive, checksums)
return volumes
def show_info(platform: str, release: dict):
@@ -333,9 +370,10 @@ To check files already in place, use: python install.py --check
staging = make_staging(dest)
try:
archive = fetch_pack(pack, staging, fetch_checksums(release))
archives = fetch_pack(pack, staging, fetch_checksums(release))
print(f"Extracting to {dest}/...")
safe_extract_zip(str(archive), str(dest))
for archive in archives:
safe_extract_zip(str(archive), str(dest))
finally:
shutil.rmtree(staging, ignore_errors=True)
print("Done!")
+32 -12
View File
@@ -1,8 +1,9 @@
#!/usr/bin/env bash
# Download BIOS pack from GitHub Releases (Linux/macOS one-liner compatible)
#
# A pack over 2 GB is published as numbered volumes (.zip.001, .zip.002),
# which are downloaded, joined and checked here.
# A pack over 2 GB is published in several parts. Each is a ZIP
# (.part1of2.zip), checked and extracted in turn. Releases up to v2026.09.04
# carry byte ranges of one archive instead (.zip.001), which are joined first.
#
# Usage:
# bash scripts/download.sh retroarch ~/RetroArch/system/
@@ -47,9 +48,11 @@ asset_urls() {
sed -n 's/.*"browser_download_url"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p'
}
# The archive name behind each asset, volumes folded into their archive.
# The archive name behind each asset, the files of one pack folded together:
# Pack.part1of2.zip and Pack.zip.001 both belong to Pack.zip.
pack_names() {
asset_urls "$1" | sed 's#.*/##' |
sed 's/\(_BIOS_Pack\)\.part[0-9][0-9]*of[0-9][0-9]*\.zip$/\1.zip/' |
sed -n 's/\(.*_BIOS_Pack\.zip\)\(\.[0-9][0-9]*\)\{0,1\}$/\1/p' |
LC_ALL=C sort -u
}
@@ -103,14 +106,28 @@ EOF
exit 1
fi
local stem="${archive%.zip}"
local volumes
volumes=$(asset_urls "$release_json" |
grep -E "/${archive//./\\.}(\.[0-9]+)?$" | LC_ALL=C sort)
grep -E "/${stem//./\\.}(\.zip(\.[0-9]+)?|\.part[0-9]+of[0-9]+\.zip)$" |
LC_ALL=C sort)
if [ -z "$volumes" ]; then
echo "Error: no asset found for ${archive}." >&2
exit 1
fi
local count
count=$(printf '%s\n' "$volumes" | wc -l | tr -d ' ')
# Parts that are each a ZIP say in their name how many there are.
local declared
declared=$(printf '%s\n' "$volumes" |
sed -n 's/.*\.part[0-9][0-9]*of\([0-9][0-9]*\)\.zip$/\1/p' | head -1)
if [ -n "$declared" ] && [ "$declared" != "$count" ]; then
echo "Error: the release lists ${count} of ${declared} parts of ${archive}." >&2
exit 1
fi
# Staged inside the destination: a pack is gigabytes, and /tmp is a RAM
# disk on the appliances these packs target.
mkdir -p "$dest"
@@ -119,9 +136,6 @@ EOF
mkdir -p "$staging"
trap 'rm -rf "$staging"' EXIT
local count
count=$(printf '%s\n' "$volumes" | wc -l | tr -d ' ')
local index=0
local url name
while read -r url; do
@@ -138,17 +152,23 @@ EOF
$volumes
EOF
if [ "$count" -gt 1 ]; then
if [ -z "$declared" ] && [ "$count" -gt 1 ]; then
echo "Joining ${count} parts into ${archive}..."
# split(1) writes plain byte ranges, so concatenation rebuilds the ZIP.
# Releases up to v2026.09.04: split(1) wrote plain byte ranges, so
# concatenation rebuilds the ZIP.
cat "${staging}/${archive}".[0-9][0-9][0-9] > "${staging}/${archive}"
rm -f "${staging}/${archive}".[0-9][0-9][0-9]
fi
verify_checksum "$release_json" "$staging" "$archive"
local file
for file in "${staging}"/*.zip; do
verify_checksum "$release_json" "$staging" "$(basename "$file")"
done
echo "Extracting to ${dest}/..."
unzip -o -q "${staging}/${archive}" -d "$dest"
for file in "${staging}"/*.zip; do
unzip -o -q "$file" -d "$dest"
done
rm -rf "$staging"
trap - EXIT
@@ -182,7 +202,7 @@ verify_checksum() {
return 0
fi
echo "Checking the archive..."
echo "Checking ${archive}..."
local actual
actual=$($hasher "${staging}/${archive}" | cut -d' ' -f1)
if [ "$actual" != "$expected" ]; then
+17 -6
View File
@@ -63,6 +63,7 @@ import packresolve
import region as region_mod
import slot as slot_mod
import slots
import split_pack
from deterministic_zip import _FIXED_DATE_TIME, rebuild_zip_deterministic
from nativemode import (
digest_algorithm,
@@ -3274,6 +3275,20 @@ def _narrows_contents(pack_name: str) -> bool:
return any(f"{tag}_" in pack_name for tag in _CONTENT_NARROWING_TAGS)
def _pack_archives(output_dir: str) -> list[str]:
"""The ZIPs of a directory that are packs.
A part of a split pack holds a share of a platform by design and was
checked against the whole when it was cut: judged as a pack it would
fail conformance, and injecting a manifest would change published bytes.
"""
return sorted(
name
for name in os.listdir(output_dir)
if name.endswith(".zip") and not split_pack.is_part(name)
)
def verify_and_finalize_packs(
output_dir: str,
db: dict,
@@ -3295,18 +3310,14 @@ def verify_and_finalize_packs(
# Map ZIP names to platform names
pack_to_platform: dict[str, list[str]] = {}
for name in sorted(os.listdir(output_dir)):
if not name.endswith(".zip"):
continue
for name in _pack_archives(output_dir):
for pname in list_registered_platforms(platforms_dir):
cfg = load_platform_config(pname, platforms_dir)
display = cfg.get("platform", pname).replace(" ", "_")
if display in name or display.replace("_", "") in name.replace("_", ""):
pack_to_platform.setdefault(name, []).append(pname)
for name in sorted(os.listdir(output_dir)):
if not name.endswith(".zip"):
continue
for name in _pack_archives(output_dir):
zip_path = os.path.join(output_dir, name)
# Stage 1: database integrity
+12 -5
View File
@@ -340,11 +340,18 @@ def generate_readme(db: dict, platforms_dir: str) -> str:
" next because a pack carries only what that platform's emulators"
" load."
" The size is what the files occupy once extracted; the ZIP itself"
" downloads smaller, and anything over 2 GB arrives split into"
" `.zip.001`, `.zip.002` volumes. Open the `.001` with 7-Zip or"
" PeaZip, or join them first"
" (`cat Pack.zip.0* > Pack.zip`, or"
" `copy /b Pack.zip.001+Pack.zip.002 Pack.zip` on Windows).",
" downloads smaller.",
"",
"A pack over 2 GB comes in several parts, and every part is needed."
" How to open them depends on their name:",
"",
"- `Pack.part1of2.zip`, `Pack.part2of2.zip`: each part is an ordinary"
" ZIP. Extract them all into the same folder.",
"- `Pack.zip.001`, `Pack.zip.002` (releases up to v2026.09.04): slices"
" of one ZIP, none of which opens on its own. Put them in one folder"
" and open the `.001` with 7-Zip or PeaZip, or join them first with"
" `cat Pack.zip.0* > Pack.zip` on Linux and macOS,"
" `cmd /c copy /b Pack.zip.001+Pack.zip.002 Pack.zip` on Windows.",
"",
"Every release ships `SHA256SUMS.txt` and a detached signature of it,"
" checkable against `allowed_signers` in this repository:"
+9 -5
View File
@@ -3406,12 +3406,16 @@ download it, and extract the files into the BIOS folder listed below.
The metadata and website can be newer than that release: pack publication is
manual and only happens after the release gates pass.
Packs over 2 GB are split into numbered volumes (`.zip.001`, `.zip.002`).
Download every part, then open the `.001` file with 7-Zip or PeaZip, which
extract the whole archive directly. To join the parts manually instead:
A pack over 2 GB comes in several parts, and every part is needed. How to
open them depends on their name:
- Linux/macOS: `cat PackName.zip.0* > PackName.zip`
- Windows (cmd): `copy /b PackName.zip.001+PackName.zip.002 PackName.zip`
- `PackName.part1of2.zip`, `PackName.part2of2.zip`: each part is an ordinary
ZIP. Extract them all into the same folder.
- `PackName.zip.001`, `PackName.zip.002` (releases up to v2026.09.04): slices
of one ZIP, none of which opens on its own. Put them in one folder and open
the `.001` with 7-Zip or PeaZip, or join them first:
- Linux/macOS: `cat PackName.zip.0* > PackName.zip`
- Windows: `cmd /c copy /b PackName.zip.001+PackName.zip.002 PackName.zip`
### Steam Deck
+233
View File
@@ -0,0 +1,233 @@
#!/usr/bin/env python3
"""Publish a pack too large for one release asset as several ZIPs.
GitHub refuses a release file of 2 GiB or more. A pack over that is written
as parts, each a complete archive of whole files: any tool opens one, and
the parts extracted into the same folder are the pack. Members are copied
as stored, never recompressed, so the parts depend on the pack alone.
Usage:
python scripts/split_pack.py dist/
python scripts/split_pack.py dist/RetroArch_Lakka_v1.22.2_BIOS_Pack.zip
python scripts/split_pack.py dist/ --max-size 1900M
"""
from __future__ import annotations
import argparse
import re
import struct
import sys
import zipfile
from pathlib import Path
# "Each file included in a release must be under 2 GiB."
ASSET_LIMIT = 2 * 1024**3
PACK_SUFFIX = "_BIOS_Pack.zip"
_PART_NAME = re.compile(r"^(?P<stem>.+_BIOS_Pack)\.part(?P<index>\d+)of(?P<count>\d+)\.zip$")
_LOCAL = struct.Struct("<4s2B4HL2L2H")
_CENTRAL = struct.Struct("<4s4B4HL2L5H2L")
_END = struct.Struct("<4s4H2LH")
_LOCAL_SIG = b"PK\x03\x04"
_CENTRAL_SIG = b"PK\x01\x02"
_END_SIG = b"PK\x05\x06"
_DATA_DESCRIPTOR = 0x08
_UTF8_NAME = 0x800
_CHUNK = 1024 * 1024
_MAX_ENTRIES = 0xFFFF
_MAX_FIELD = 0xFFFFFFFF
def is_part(name: str) -> bool:
"""Whether a file name is one part of a split pack."""
return _PART_NAME.match(name) is not None
def part_name(pack_name: str, index: int, count: int) -> str:
return f"{pack_name[: -len('.zip')]}.part{index}of{count}.zip"
def _name_bytes(info: zipfile.ZipInfo) -> bytes:
encoding = "utf-8" if info.flag_bits & _UTF8_NAME else "cp437"
return info.orig_filename.encode(encoding)
def _weight(info: zipfile.ZipInfo) -> int:
"""Bytes a member adds to a part: both headers and its stored data."""
name = len(_name_bytes(info))
return _LOCAL.size + name + info.compress_size + _CENTRAL.size + name
def plan_parts(
members: list[zipfile.ZipInfo], limit: int
) -> list[list[zipfile.ZipInfo]]:
"""Fill each part in archive order, up to the limit."""
parts: list[list[zipfile.ZipInfo]] = [[]]
size = _END.size
for info in members:
weight = _weight(info)
if _END.size + weight >= limit:
raise ValueError(
f"{info.filename} needs {weight:,} bytes, more than a part of "
f"{limit:,} can hold"
)
if info.compress_size > _MAX_FIELD or info.file_size > _MAX_FIELD:
raise ValueError(f"{info.filename} needs ZIP64, which a part does not use")
if size + weight >= limit or len(parts[-1]) >= _MAX_ENTRIES:
parts.append([])
size = _END.size
parts[-1].append(info)
size += weight
return parts
def _dos_stamp(info: zipfile.ZipInfo) -> tuple[int, int]:
year, month, day, hour, minute, second = info.date_time
return (
(hour << 11) | (minute << 5) | (second // 2),
((year - 1980) << 9) | (month << 5) | day,
)
def _data_offset(source, info: zipfile.ZipInfo) -> int:
"""Where a member's stored bytes begin, read from its own local header."""
source.seek(info.header_offset)
header = _LOCAL.unpack(source.read(_LOCAL.size))
if header[0] != _LOCAL_SIG:
raise ValueError(f"{info.filename}: no local header at its offset")
return info.header_offset + _LOCAL.size + header[10] + header[11]
def _write_part(source, members: list[zipfile.ZipInfo], dest: Path) -> None:
directory = []
with open(dest, "wb") as out:
for info in members:
name = _name_bytes(info)
flags = info.flag_bits & ~_DATA_DESCRIPTOR
dos_time, dos_date = _dos_stamp(info)
offset = out.tell()
out.write(
_LOCAL.pack(
_LOCAL_SIG, info.extract_version, info.reserved, flags,
info.compress_type, dos_time, dos_date, info.CRC,
info.compress_size, info.file_size, len(name), 0,
)
)
out.write(name)
source.seek(_data_offset(source, info))
remaining = info.compress_size
while remaining:
chunk = source.read(min(_CHUNK, remaining))
if not chunk:
raise ValueError(f"{info.filename}: the pack ends inside it")
out.write(chunk)
remaining -= len(chunk)
directory.append(
_CENTRAL.pack(
_CENTRAL_SIG, info.create_version, info.create_system,
info.extract_version, info.reserved, flags,
info.compress_type, dos_time, dos_date, info.CRC,
info.compress_size, info.file_size, len(name), 0, 0, 0,
info.internal_attr, info.external_attr, offset,
)
+ name
)
start = out.tell()
for record in directory:
out.write(record)
out.write(
_END.pack(
_END_SIG, 0, 0, len(directory), len(directory),
out.tell() - start, start, 0,
)
)
def _identity(archive: zipfile.ZipFile) -> list[tuple[str, int, int]]:
return [(info.filename, info.CRC, info.file_size) for info in archive.infolist()]
def split_pack(zip_path: Path, limit: int = ASSET_LIMIT) -> list[Path]:
"""Replace a pack at or over the limit by parts that each fit under it.
Returns the files to publish. The pack is removed only once every part
has been read back in full and the parts together list exactly its
members, so a failure leaves the pack as it was and no part behind.
"""
zip_path = Path(zip_path)
if zip_path.stat().st_size < limit:
return [zip_path]
with zipfile.ZipFile(zip_path) as archive:
members = archive.infolist()
expected = _identity(archive)
plan = plan_parts(members, limit)
parts = [
zip_path.with_name(part_name(zip_path.name, index, len(plan)))
for index in range(1, len(plan) + 1)
]
try:
with open(zip_path, "rb") as source:
for dest, chosen in zip(parts, plan):
_write_part(source, chosen, dest)
found: list[tuple[str, int, int]] = []
for dest in parts:
with zipfile.ZipFile(dest) as archive:
damaged = archive.testzip()
if damaged:
raise ValueError(f"{dest.name}: {damaged} does not read back")
found.extend(_identity(archive))
if dest.stat().st_size >= limit:
raise ValueError(f"{dest.name} reached the limit of {limit:,} bytes")
if found != expected:
raise ValueError(f"the parts of {zip_path.name} do not add up to it")
except (OSError, ValueError, zipfile.BadZipFile):
for dest in parts:
dest.unlink(missing_ok=True)
raise
zip_path.unlink()
return parts
def split_directory(directory: Path, limit: int = ASSET_LIMIT) -> list[Path]:
"""Split every pack of a directory that needs it. Returns what to publish."""
published: list[Path] = []
for path in sorted(Path(directory).glob(f"*{PACK_SUFFIX}")):
published.extend(split_pack(path, limit))
return published
def _size(text: str) -> int:
match = re.fullmatch(r"(\d+)([KMG]?)", text.strip().upper())
if not match:
raise argparse.ArgumentTypeError(f"not a size: {text}")
return int(match[1]) * 1024 ** " KMG".index(match[2] or " ")
def main() -> int:
parser = argparse.ArgumentParser(
description="Split packs over the release asset limit into ZIP parts"
)
parser.add_argument("target", type=Path, help="a pack, or a directory of packs")
parser.add_argument(
"--max-size", type=_size, default=ASSET_LIMIT, metavar="SIZE",
help="a part stays under this many bytes (K, M, G suffixes; default 2G)",
)
args = parser.parse_args()
try:
if args.target.is_dir():
published = split_directory(args.target, args.max_size)
else:
published = split_pack(args.target, args.max_size)
except (OSError, ValueError, zipfile.BadZipFile) as exc:
print(f"Error: {exc}", file=sys.stderr)
return 1
for path in published:
print(f"{path.stat().st_size:>13,} {path.name}")
return 0
if __name__ == "__main__":
sys.exit(main())