feat: publish a large pack as zips, not byte ranges

This commit is contained in:
Abdessamad Derraz committed 2026-10-04 18:09:24 +02:00
1 parent 9ea6083b8a
commit 52f682cdd6
12 files changed
+768 -112

No files matched your search

+70 -32
View File
@@ -3,8 +3,11 @@
Cross-platform tool (Linux/macOS/Windows) using only Python stdlib.
A pack over 2 GB is published as numbered volumes (`.zip.001`, `.zip.002`),
so a platform is a group of assets rather than a single one.
A pack over the release asset limit is published in several files, so a
platform is a group of assets rather than a single one. The parts are each
a ZIP (`.part1of2.zip`), checked and extracted one after the other. Releases
up to v2026.09.04 carry byte ranges of one archive instead (`.zip.001`,
`.zip.002`), which are joined first.
Usage:
python scripts/download.py --list # List platforms
@@ -39,6 +42,9 @@ MAX_CHECKSUMS_BYTES = 1 << 20
CHUNK = 1 << 20
_VOLUME_RE = re.compile(rf"^(?P<base>.+{re.escape(PACK_SUFFIX)})\.(?P<index>\d+)$")
_PART_RE = re.compile(
r"^(?P<stem>.+_BIOS_Pack)\.part(?P<index>\d+)of(?P<count>\d+)\.zip$"
)
_LOOPBACK_HOSTS = {"127.0.0.1", "::1", "localhost"}
@@ -64,12 +70,19 @@ API = _checked_url(os.environ.get("RETROBIOS_API", DEFAULT_API), "RETROBIOS_API"
@dataclass(frozen=True)
class Pack:
"""A platform pack: one asset, or the volumes it was split into."""
"""A platform pack: one asset, or the files it was published in.
`joined` says the files are byte ranges of one archive. `expected_parts`
is the count a part's own name declares, so a release that lists fewer
is recognised before anything is downloaded.
"""
name: str
platform: str
parts: tuple[dict, ...]
size: int
joined: bool = False
expected_parts: int = 1
def get_latest_release() -> dict:
@@ -94,13 +107,21 @@ def get_latest_release() -> dict:
def group_packs(release: dict) -> list[Pack]:
"""Group release assets into packs, volumes folded into their archive."""
"""Group release assets into packs, the files of one pack folded together."""
groups: dict[str, list[tuple[int, dict]]] = {}
joined: set[str] = set()
expected: dict[str, int] = {}
for asset in release.get("assets", []):
name = asset["name"]
volume = _VOLUME_RE.match(name)
part = _PART_RE.match(name)
if volume:
groups.setdefault(volume["base"], []).append((int(volume["index"]), asset))
joined.add(volume["base"])
elif part:
base = f"{part['stem']}.zip"
groups.setdefault(base, []).append((int(part["index"]), asset))
expected[base] = int(part["count"])
elif name.endswith(PACK_SUFFIX):
groups.setdefault(name, []).append((0, asset))
@@ -113,6 +134,8 @@ def group_packs(release: dict) -> list[Pack]:
platform=base[: -len(PACK_SUFFIX)].replace("_", " "),
parts=parts,
size=sum(part.get("size", 0) for part in parts),
joined=base in joined,
expected_parts=expected.get(base, len(parts)),
)
)
return packs
@@ -213,8 +236,34 @@ def make_staging(dest: Path) -> Path:
return Path(tempfile.mkdtemp(prefix=STAGING_PREFIX, dir=dest))
def fetch_pack(pack: Pack, staging: Path, checksums: dict[str, str]) -> Path:
"""Download every volume of a pack and return the assembled archive."""
def _check(archive: Path, checksums: dict[str, str]) -> None:
"""Compare a downloaded file with its published SHA-256, when there is one."""
expected = checksums.get(archive.name)
if not expected:
print(f"No checksum published for {archive.name}, skipping.")
return
print(f"Checking {archive.name}...")
actual = compute_hashes(str(archive))["sha256"].lower()
if actual != expected:
archive.unlink()
print(
f"Error: checksum mismatch for {archive.name}\n"
f" expected {expected}\n got {actual}\n"
"Download the parts again; a truncated part gives this.",
file=sys.stderr,
)
sys.exit(1)
def fetch_pack(pack: Pack, staging: Path, checksums: dict[str, str]) -> list[Path]:
"""Download a pack and return the archives to extract, each one checked."""
if len(pack.parts) != pack.expected_parts:
print(
f"Error: the release lists {len(pack.parts)} of "
f"{pack.expected_parts} parts of {pack.name}.",
file=sys.stderr,
)
sys.exit(1)
staging.mkdir(parents=True, exist_ok=True)
volumes = []
for index, part in enumerate(pack.parts, start=1):
@@ -224,32 +273,20 @@ def fetch_pack(pack: Pack, staging: Path, checksums: dict[str, str]) -> Path:
download_file(part["browser_download_url"], target, part.get("size", 0), label)
volumes.append(target)
archive = staging / pack.name
if volumes != [archive]:
if len(volumes) > 1:
print(f"Joining {len(volumes)} parts into {pack.name}...")
join_volumes(volumes, archive)
for volume in volumes:
volume.unlink()
else:
volumes[0].replace(archive)
if pack.joined:
archive = staging / pack.name
print(f"Joining {len(volumes)} parts into {pack.name}...")
join_volumes(volumes, archive)
for volume in volumes:
volume.unlink()
volumes = [archive]
expected = checksums.get(pack.name)
if expected:
print("Checking the archive...")
actual = compute_hashes(str(archive))["sha256"].lower()
if actual != expected:
archive.unlink()
print(
f"Error: checksum mismatch for {pack.name}\n"
f" expected {expected}\n got {actual}\n"
"Download the parts again; a truncated part gives this.",
file=sys.stderr,
)
sys.exit(1)
else:
if not checksums:
print(f"No {CHECKSUMS_ASSET} in the release, skipping the checksum.")
return archive
return volumes
for archive in volumes:
_check(archive, checksums)
return volumes
def show_info(platform: str, release: dict):
@@ -333,9 +370,10 @@ To check files already in place, use: python install.py --check
staging = make_staging(dest)
try:
archive = fetch_pack(pack, staging, fetch_checksums(release))
archives = fetch_pack(pack, staging, fetch_checksums(release))
print(f"Extracting to {dest}/...")
safe_extract_zip(str(archive), str(dest))
for archive in archives:
safe_extract_zip(str(archive), str(dest))
finally:
shutil.rmtree(staging, ignore_errors=True)
print("Done!")
+32 -12
View File
@@ -1,8 +1,9 @@
#!/usr/bin/env bash
# Download BIOS pack from GitHub Releases (Linux/macOS one-liner compatible)
#
# A pack over 2 GB is published as numbered volumes (.zip.001, .zip.002),
# which are downloaded, joined and checked here.
# A pack over 2 GB is published in several parts. Each is a ZIP
# (.part1of2.zip), checked and extracted in turn. Releases up to v2026.09.04
# carry byte ranges of one archive instead (.zip.001), which are joined first.
#
# Usage:
# bash scripts/download.sh retroarch ~/RetroArch/system/
@@ -47,9 +48,11 @@ asset_urls() {
sed -n 's/.*"browser_download_url"[[:space:]]*:[[:space:]]*"\([^"]*\)".*/\1/p'
}
# The archive name behind each asset, volumes folded into their archive.
# The archive name behind each asset, the files of one pack folded together:
# Pack.part1of2.zip and Pack.zip.001 both belong to Pack.zip.
pack_names() {
asset_urls "$1" | sed 's#.*/##' |
sed 's/\(_BIOS_Pack\)\.part[0-9][0-9]*of[0-9][0-9]*\.zip$/\1.zip/' |
sed -n 's/\(.*_BIOS_Pack\.zip\)\(\.[0-9][0-9]*\)\{0,1\}$/\1/p' |
LC_ALL=C sort -u
}
@@ -103,14 +106,28 @@ EOF
exit 1
fi
local stem="${archive%.zip}"
local volumes
volumes=$(asset_urls "$release_json" |
grep -E "/${archive//./\\.}(\.[0-9]+)?$" | LC_ALL=C sort)
grep -E "/${stem//./\\.}(\.zip(\.[0-9]+)?|\.part[0-9]+of[0-9]+\.zip)$" |
LC_ALL=C sort)
if [ -z "$volumes" ]; then
echo "Error: no asset found for ${archive}." >&2
exit 1
fi
local count
count=$(printf '%s\n' "$volumes" | wc -l | tr -d ' ')
# Parts that are each a ZIP say in their name how many there are.
local declared
declared=$(printf '%s\n' "$volumes" |
sed -n 's/.*\.part[0-9][0-9]*of\([0-9][0-9]*\)\.zip$/\1/p' | head -1)
if [ -n "$declared" ] && [ "$declared" != "$count" ]; then
echo "Error: the release lists ${count} of ${declared} parts of ${archive}." >&2
exit 1
fi
# Staged inside the destination: a pack is gigabytes, and /tmp is a RAM
# disk on the appliances these packs target.
mkdir -p "$dest"
@@ -119,9 +136,6 @@ EOF
mkdir -p "$staging"
trap 'rm -rf "$staging"' EXIT
local count
count=$(printf '%s\n' "$volumes" | wc -l | tr -d ' ')
local index=0
local url name
while read -r url; do
@@ -138,17 +152,23 @@ EOF
$volumes
EOF
if [ "$count" -gt 1 ]; then
if [ -z "$declared" ] && [ "$count" -gt 1 ]; then
echo "Joining ${count} parts into ${archive}..."
# split(1) writes plain byte ranges, so concatenation rebuilds the ZIP.
# Releases up to v2026.09.04: split(1) wrote plain byte ranges, so
# concatenation rebuilds the ZIP.
cat "${staging}/${archive}".[0-9][0-9][0-9] > "${staging}/${archive}"
rm -f "${staging}/${archive}".[0-9][0-9][0-9]
fi
verify_checksum "$release_json" "$staging" "$archive"
local file
for file in "${staging}"/*.zip; do
verify_checksum "$release_json" "$staging" "$(basename "$file")"
done
echo "Extracting to ${dest}/..."
unzip -o -q "${staging}/${archive}" -d "$dest"
for file in "${staging}"/*.zip; do
unzip -o -q "$file" -d "$dest"
done
rm -rf "$staging"
trap - EXIT
@@ -182,7 +202,7 @@ verify_checksum() {
return 0
fi
echo "Checking the archive..."
echo "Checking ${archive}..."
local actual
actual=$($hasher "${staging}/${archive}" | cut -d' ' -f1)
if [ "$actual" != "$expected" ]; then
+17 -6
View File
@@ -63,6 +63,7 @@ import packresolve
import region as region_mod
import slot as slot_mod
import slots
import split_pack
from deterministic_zip import _FIXED_DATE_TIME, rebuild_zip_deterministic
from nativemode import (
digest_algorithm,
@@ -3274,6 +3275,20 @@ def _narrows_contents(pack_name: str) -> bool:
return any(f"{tag}_" in pack_name for tag in _CONTENT_NARROWING_TAGS)
def _pack_archives(output_dir: str) -> list[str]:
"""The ZIPs of a directory that are packs.
A part of a split pack holds a share of a platform by design and was
checked against the whole when it was cut: judged as a pack it would
fail conformance, and injecting a manifest would change published bytes.
"""
return sorted(
name
for name in os.listdir(output_dir)
if name.endswith(".zip") and not split_pack.is_part(name)
)
def verify_and_finalize_packs(
output_dir: str,
db: dict,
@@ -3295,18 +3310,14 @@ def verify_and_finalize_packs(
# Map ZIP names to platform names
pack_to_platform: dict[str, list[str]] = {}
for name in sorted(os.listdir(output_dir)):
if not name.endswith(".zip"):
continue
for name in _pack_archives(output_dir):
for pname in list_registered_platforms(platforms_dir):
cfg = load_platform_config(pname, platforms_dir)
display = cfg.get("platform", pname).replace(" ", "_")
if display in name or display.replace("_", "") in name.replace("_", ""):
pack_to_platform.setdefault(name, []).append(pname)
for name in sorted(os.listdir(output_dir)):
if not name.endswith(".zip"):
continue
for name in _pack_archives(output_dir):
zip_path = os.path.join(output_dir, name)
# Stage 1: database integrity
+12 -5
View File
@@ -340,11 +340,18 @@ def generate_readme(db: dict, platforms_dir: str) -> str:
" next because a pack carries only what that platform's emulators"
" load."
" The size is what the files occupy once extracted; the ZIP itself"
" downloads smaller, and anything over 2 GB arrives split into"
" `.zip.001`, `.zip.002` volumes. Open the `.001` with 7-Zip or"
" PeaZip, or join them first"
" (`cat Pack.zip.0* > Pack.zip`, or"
" `copy /b Pack.zip.001+Pack.zip.002 Pack.zip` on Windows).",
" downloads smaller.",
"",
"A pack over 2 GB comes in several parts, and every part is needed."
" How to open them depends on their name:",
"",
"- `Pack.part1of2.zip`, `Pack.part2of2.zip`: each part is an ordinary"
" ZIP. Extract them all into the same folder.",
"- `Pack.zip.001`, `Pack.zip.002` (releases up to v2026.09.04): slices"
" of one ZIP, none of which opens on its own. Put them in one folder"
" and open the `.001` with 7-Zip or PeaZip, or join them first with"
" `cat Pack.zip.0* > Pack.zip` on Linux and macOS,"
" `cmd /c copy /b Pack.zip.001+Pack.zip.002 Pack.zip` on Windows.",
"",
"Every release ships `SHA256SUMS.txt` and a detached signature of it,"
" checkable against `allowed_signers` in this repository:"
+9 -5
View File
@@ -3406,12 +3406,16 @@ download it, and extract the files into the BIOS folder listed below.
The metadata and website can be newer than that release: pack publication is
manual and only happens after the release gates pass.
Packs over 2 GB are split into numbered volumes (`.zip.001`, `.zip.002`).
Download every part, then open the `.001` file with 7-Zip or PeaZip, which
extract the whole archive directly. To join the parts manually instead:
A pack over 2 GB comes in several parts, and every part is needed. How to
open them depends on their name:
- Linux/macOS: `cat PackName.zip.0* > PackName.zip`
- Windows (cmd): `copy /b PackName.zip.001+PackName.zip.002 PackName.zip`
- `PackName.part1of2.zip`, `PackName.part2of2.zip`: each part is an ordinary
ZIP. Extract them all into the same folder.
- `PackName.zip.001`, `PackName.zip.002` (releases up to v2026.09.04): slices
of one ZIP, none of which opens on its own. Put them in one folder and open
the `.001` with 7-Zip or PeaZip, or join them first:
- Linux/macOS: `cat PackName.zip.0* > PackName.zip`
- Windows: `cmd /c copy /b PackName.zip.001+PackName.zip.002 PackName.zip`
### Steam Deck
+233
View File
@@ -0,0 +1,233 @@
#!/usr/bin/env python3
"""Publish a pack too large for one release asset as several ZIPs.
GitHub refuses a release file of 2 GiB or more. A pack over that is written
as parts, each a complete archive of whole files: any tool opens one, and
the parts extracted into the same folder are the pack. Members are copied
as stored, never recompressed, so the parts depend on the pack alone.
Usage:
python scripts/split_pack.py dist/
python scripts/split_pack.py dist/RetroArch_Lakka_v1.22.2_BIOS_Pack.zip
python scripts/split_pack.py dist/ --max-size 1900M
"""
from __future__ import annotations
import argparse
import re
import struct
import sys
import zipfile
from pathlib import Path
# "Each file included in a release must be under 2 GiB."
ASSET_LIMIT = 2 * 1024**3
PACK_SUFFIX = "_BIOS_Pack.zip"
_PART_NAME = re.compile(r"^(?P<stem>.+_BIOS_Pack)\.part(?P<index>\d+)of(?P<count>\d+)\.zip$")
_LOCAL = struct.Struct("<4s2B4HL2L2H")
_CENTRAL = struct.Struct("<4s4B4HL2L5H2L")
_END = struct.Struct("<4s4H2LH")
_LOCAL_SIG = b"PK\x03\x04"
_CENTRAL_SIG = b"PK\x01\x02"
_END_SIG = b"PK\x05\x06"
_DATA_DESCRIPTOR = 0x08
_UTF8_NAME = 0x800
_CHUNK = 1024 * 1024
_MAX_ENTRIES = 0xFFFF
_MAX_FIELD = 0xFFFFFFFF
def is_part(name: str) -> bool:
"""Whether a file name is one part of a split pack."""
return _PART_NAME.match(name) is not None
def part_name(pack_name: str, index: int, count: int) -> str:
return f"{pack_name[: -len('.zip')]}.part{index}of{count}.zip"
def _name_bytes(info: zipfile.ZipInfo) -> bytes:
encoding = "utf-8" if info.flag_bits & _UTF8_NAME else "cp437"
return info.orig_filename.encode(encoding)
def _weight(info: zipfile.ZipInfo) -> int:
"""Bytes a member adds to a part: both headers and its stored data."""
name = len(_name_bytes(info))
return _LOCAL.size + name + info.compress_size + _CENTRAL.size + name
def plan_parts(
members: list[zipfile.ZipInfo], limit: int
) -> list[list[zipfile.ZipInfo]]:
"""Fill each part in archive order, up to the limit."""
parts: list[list[zipfile.ZipInfo]] = [[]]
size = _END.size
for info in members:
weight = _weight(info)
if _END.size + weight >= limit:
raise ValueError(
f"{info.filename} needs {weight:,} bytes, more than a part of "
f"{limit:,} can hold"
)
if info.compress_size > _MAX_FIELD or info.file_size > _MAX_FIELD:
raise ValueError(f"{info.filename} needs ZIP64, which a part does not use")
if size + weight >= limit or len(parts[-1]) >= _MAX_ENTRIES:
parts.append([])
size = _END.size
parts[-1].append(info)
size += weight
return parts
def _dos_stamp(info: zipfile.ZipInfo) -> tuple[int, int]:
year, month, day, hour, minute, second = info.date_time
return (
(hour << 11) | (minute << 5) | (second // 2),
((year - 1980) << 9) | (month << 5) | day,
)
def _data_offset(source, info: zipfile.ZipInfo) -> int:
"""Where a member's stored bytes begin, read from its own local header."""
source.seek(info.header_offset)
header = _LOCAL.unpack(source.read(_LOCAL.size))
if header[0] != _LOCAL_SIG:
raise ValueError(f"{info.filename}: no local header at its offset")
return info.header_offset + _LOCAL.size + header[10] + header[11]
def _write_part(source, members: list[zipfile.ZipInfo], dest: Path) -> None:
directory = []
with open(dest, "wb") as out:
for info in members:
name = _name_bytes(info)
flags = info.flag_bits & ~_DATA_DESCRIPTOR
dos_time, dos_date = _dos_stamp(info)
offset = out.tell()
out.write(
_LOCAL.pack(
_LOCAL_SIG, info.extract_version, info.reserved, flags,
info.compress_type, dos_time, dos_date, info.CRC,
info.compress_size, info.file_size, len(name), 0,
)
)
out.write(name)
source.seek(_data_offset(source, info))
remaining = info.compress_size
while remaining:
chunk = source.read(min(_CHUNK, remaining))
if not chunk:
raise ValueError(f"{info.filename}: the pack ends inside it")
out.write(chunk)
remaining -= len(chunk)
directory.append(
_CENTRAL.pack(
_CENTRAL_SIG, info.create_version, info.create_system,
info.extract_version, info.reserved, flags,
info.compress_type, dos_time, dos_date, info.CRC,
info.compress_size, info.file_size, len(name), 0, 0, 0,
info.internal_attr, info.external_attr, offset,
)
+ name
)
start = out.tell()
for record in directory:
out.write(record)
out.write(
_END.pack(
_END_SIG, 0, 0, len(directory), len(directory),
out.tell() - start, start, 0,
)
)
def _identity(archive: zipfile.ZipFile) -> list[tuple[str, int, int]]:
return [(info.filename, info.CRC, info.file_size) for info in archive.infolist()]
def split_pack(zip_path: Path, limit: int = ASSET_LIMIT) -> list[Path]:
"""Replace a pack at or over the limit by parts that each fit under it.
Returns the files to publish. The pack is removed only once every part
has been read back in full and the parts together list exactly its
members, so a failure leaves the pack as it was and no part behind.
"""
zip_path = Path(zip_path)
if zip_path.stat().st_size < limit:
return [zip_path]
with zipfile.ZipFile(zip_path) as archive:
members = archive.infolist()
expected = _identity(archive)
plan = plan_parts(members, limit)
parts = [
zip_path.with_name(part_name(zip_path.name, index, len(plan)))
for index in range(1, len(plan) + 1)
]
try:
with open(zip_path, "rb") as source:
for dest, chosen in zip(parts, plan):
_write_part(source, chosen, dest)
found: list[tuple[str, int, int]] = []
for dest in parts:
with zipfile.ZipFile(dest) as archive:
damaged = archive.testzip()
if damaged:
raise ValueError(f"{dest.name}: {damaged} does not read back")
found.extend(_identity(archive))
if dest.stat().st_size >= limit:
raise ValueError(f"{dest.name} reached the limit of {limit:,} bytes")
if found != expected:
raise ValueError(f"the parts of {zip_path.name} do not add up to it")
except (OSError, ValueError, zipfile.BadZipFile):
for dest in parts:
dest.unlink(missing_ok=True)
raise
zip_path.unlink()
return parts
def split_directory(directory: Path, limit: int = ASSET_LIMIT) -> list[Path]:
"""Split every pack of a directory that needs it. Returns what to publish."""
published: list[Path] = []
for path in sorted(Path(directory).glob(f"*{PACK_SUFFIX}")):
published.extend(split_pack(path, limit))
return published
def _size(text: str) -> int:
match = re.fullmatch(r"(\d+)([KMG]?)", text.strip().upper())
if not match:
raise argparse.ArgumentTypeError(f"not a size: {text}")
return int(match[1]) * 1024 ** " KMG".index(match[2] or " ")
def main() -> int:
parser = argparse.ArgumentParser(
description="Split packs over the release asset limit into ZIP parts"
)
parser.add_argument("target", type=Path, help="a pack, or a directory of packs")
parser.add_argument(
"--max-size", type=_size, default=ASSET_LIMIT, metavar="SIZE",
help="a part stays under this many bytes (K, M, G suffixes; default 2G)",
)
args = parser.parse_args()
try:
if args.target.is_dir():
published = split_directory(args.target, args.max_size)
else:
published = split_pack(args.target, args.max_size)
except (OSError, ValueError, zipfile.BadZipFile) as exc:
print(f"Error: {exc}", file=sys.stderr)
return 1
for path in published:
print(f"{path.stat().st_size:>13,} {path.name}")
return 0
if __name__ == "__main__":
sys.exit(main())
+118 -12
View File
@@ -1,8 +1,11 @@
"""Tests for the release pack downloaders.
Packs over 2 GB are published as numbered volumes (`.zip.001`, `.zip.002`),
so a downloader that expects one asset per platform finds nothing for most
of them. These tests pin the grouping, the join and the staging location.
A pack over the release asset limit is published in several files, so a
downloader that expects one asset per platform finds nothing for most of
them. Two layouts exist: parts that are each a ZIP (`.part1of2.zip`), and
the byte ranges of one archive that releases up to v2026.09.04 carry
(`.zip.001`, `.zip.002`). These tests pin the grouping, the join, the
checksum of each layout and the staging location.
"""
from __future__ import annotations
@@ -34,6 +37,9 @@ _spec.loader.exec_module(download)
SHELL = REPO_ROOT / "scripts" / "download.sh"
sys.path.insert(0, str(REPO_ROOT / "scripts"))
import split_pack # noqa: E402
# A slice of a real release: two whole packs, three split, and the checksums.
RELEASE_ASSETS = [
"Batocera_43.1_BIOS_Pack.zip.001",
@@ -125,6 +131,46 @@ class TestPackGrouping(unittest.TestCase):
["Big_BIOS_Pack.zip.001", "Big_BIOS_Pack.zip.009", "Big_BIOS_Pack.zip.010"],
)
def test_zip_parts_group_into_one_pack(self):
packs = download.group_packs(
_release(
[
"Batocera_43.1_BIOS_Pack.part2of2.zip",
"Batocera_43.1_BIOS_Pack.part1of2.zip",
"BizHawk_2.11.1_BIOS_Pack.zip",
]
)
)
self.assertEqual(
[(pack.name, [part["name"] for part in pack.parts]) for pack in packs],
[
(
"Batocera_43.1_BIOS_Pack.zip",
[
"Batocera_43.1_BIOS_Pack.part1of2.zip",
"Batocera_43.1_BIOS_Pack.part2of2.zip",
],
),
("BizHawk_2.11.1_BIOS_Pack.zip", ["BizHawk_2.11.1_BIOS_Pack.zip"]),
],
)
self.assertFalse(packs[0].joined)
def test_zip_parts_order_numerically_not_lexically(self):
names = [f"Big_BIOS_Pack.part{n}of12.zip" for n in (10, 9, 1)]
pack = download.group_packs(_release(names))[0]
self.assertEqual(
[part["name"] for part in pack.parts],
[f"Big_BIOS_Pack.part{n}of12.zip" for n in (1, 9, 10)],
)
self.assertEqual(pack.expected_parts, 12)
def test_byte_range_volumes_are_joined(self):
pack = download.group_packs(
_release(["Old_BIOS_Pack.zip.001", "Old_BIOS_Pack.zip.002"])
)[0]
self.assertTrue(pack.joined)
def test_other_assets_are_not_packs(self):
packs = download.group_packs(_release(["SHA256SUMS.txt", "database.json"]))
self.assertEqual(packs, [])
@@ -214,19 +260,39 @@ class TestStagingIsolation(unittest.TestCase):
class ReleaseServer:
"""Serves a release index and its assets over loopback."""
def __init__(self, pack_name: str, payload: dict[str, bytes], volumes: int):
def __init__(
self,
pack_name: str,
payload: dict[str, bytes],
volumes: int,
zip_parts: bool = False,
):
self.root = Path(tempfile.mkdtemp())
(self.root / "assets").mkdir()
archive = self.root / "assets" / pack_name
with zipfile.ZipFile(archive, "w") as zf:
with zipfile.ZipFile(archive, "w", zipfile.ZIP_DEFLATED) as zf:
for name, data in payload.items():
zf.writestr(name, data)
raw = archive.read_bytes()
self.digest = hashlib.sha256(raw).hexdigest()
archive.unlink()
sums = f"{self.digest} {pack_name}\n"
self.names: list[str] = []
if volumes == 1:
if zip_parts:
# One byte under the whole archive: the last member no longer fits.
parts = split_pack.split_pack(archive, limit=len(raw) - 1)
self.names = [part.name for part in parts]
sums = "".join(
f"{hashlib.sha256(part.read_bytes()).hexdigest()} {part.name}\n"
for part in parts
)
raw = b""
else:
archive.unlink()
if zip_parts:
pass
elif volumes == 1:
(self.root / "assets" / pack_name).write_bytes(raw)
self.names.append(pack_name)
else:
@@ -236,9 +302,7 @@ class ReleaseServer:
(self.root / "assets" / name).write_bytes(raw[start : start + cut])
self.names.append(name)
(self.root / "assets" / "SHA256SUMS.txt").write_text(
f"{self.digest} {pack_name}\n"
)
(self.root / "assets" / "SHA256SUMS.txt").write_text(sums)
handler = functools.partial(_QuietHandler, directory=str(self.root))
self.httpd = http.server.ThreadingHTTPServer(("127.0.0.1", 0), handler)
@@ -254,9 +318,18 @@ class ReleaseServer:
}
for name in self.names + ["SHA256SUMS.txt"]
]
(index / "latest").write_text(json.dumps({"assets": assets}))
self.index = index / "latest"
self.index.write_text(json.dumps({"assets": assets}))
threading.Thread(target=self.httpd.serve_forever, daemon=True).start()
def unlist_last_part(self) -> None:
"""An upload that stopped early: the release lists one part fewer."""
release = json.loads(self.index.read_text())
release["assets"] = [
asset for asset in release["assets"] if asset["name"] != self.names[-1]
]
self.index.write_text(json.dumps(release))
def corrupt_last_volume(self) -> None:
target = self.root / "assets" / self.names[-1]
target.write_bytes(target.read_bytes()[:-16] + b"0" * 16)
@@ -270,6 +343,7 @@ class DownloaderCase(unittest.TestCase):
"""Common fixture: a two-volume pack served over loopback."""
volumes = 2
zip_parts = False
payload = {
"bios/scph5501.bin": b"\x10\x20" * 4096,
"bios/dc/dc_boot.bin": b"\x30\x40" * 4096,
@@ -280,7 +354,9 @@ class DownloaderCase(unittest.TestCase):
platform_label = "Batocera 43.1"
def setUp(self):
self.server = ReleaseServer(self.pack_name, self.payload, self.volumes)
self.server = ReleaseServer(
self.pack_name, self.payload, self.volumes, self.zip_parts
)
self.addCleanup(self.server.close)
self.dest = Path(tempfile.mkdtemp()) / "bios"
# A pack is gigabytes: staging it in the system temp directory fills
@@ -314,6 +390,7 @@ class DownloaderCase(unittest.TestCase):
]
self.assertIn(self.platform_label, names)
self.assertNotIn(".001", out)
self.assertNotIn("part1of", out)
def assert_refused(self, proc):
self.assertNotEqual(proc.returncode, 0)
@@ -450,5 +527,34 @@ class TestWholePackShell(WholePackCase, TestDownloadShell):
pass
class ZipPartsCase(DownloaderCase):
"""A pack published as parts that are each an archive."""
zip_parts = True
class ZipPartsTests:
def test_each_part_is_an_archive_on_its_own(self):
self.assertEqual(len(self.server.names), 2)
for name in self.server.names:
with zipfile.ZipFile(self.server.root / "assets" / name) as archive:
self.assertIsNone(archive.testzip())
def test_a_release_missing_a_part_is_refused(self):
self.server.unlist_last_part()
proc = self.run_cli(self.platform, str(self.dest), expect_success=False)
self.assertNotEqual(proc.returncode, 0)
self.assertIn("1 of 2", proc.stdout + proc.stderr)
self.assertFalse((self.dest / "bios/scph5501.bin").exists())
class TestZipPartsPython(ZipPartsTests, ZipPartsCase, TestDownloadPython):
pass
class TestZipPartsShell(ZipPartsTests, ZipPartsCase, TestDownloadShell):
pass
if __name__ == "__main__":
unittest.main()
+210
View File
@@ -0,0 +1,210 @@
"""A pack too large for one release asset is published as ZIPs, not slices.
GitHub refuses a release file of 2 GiB or more, and seven packs are larger.
They used to be cut with split(1) into `.zip.001`, `.zip.002`: byte ranges
of one archive, each under a name that still said zip and each behind its
own download link. A range opened alone is not an archive, and no tool says
that a part is missing: 7-Zip answers "Unavailable start of archive",
PowerShell "not a supported archive file format", Windows has no handler
for `.002`. Five reports in six months took a part for a broken download.
Each part is now an archive of whole files. Any tool opens it, and the
parts extracted into one folder are the pack.
"""
from __future__ import annotations
import os
import stat
import sys
import tempfile
import unittest
import zipfile
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(REPO_ROOT / "scripts"))
import generate_pack as builder # noqa: E402
import split_pack # noqa: E402
LIMIT = 4096
def _payload(seed: int, size: int) -> bytes:
"""Bytes deflate cannot shrink, so a member weighs what it says."""
return bytes((seed * 131 + n * 7919 + (n * n) % 251) % 256 for n in range(size))
class SplitFixture(unittest.TestCase):
def setUp(self):
self._tmp = tempfile.TemporaryDirectory()
self.dist = Path(self._tmp.name)
self.pack = self.dist / "Demo_1.0_BIOS_Pack.zip"
self.members = {
f"system/dir{n % 3}/file{n:02d}.bin": _payload(n, 900) for n in range(12)
}
self.members["README.txt"] = b"read me\n"
self._build(self.pack, self.members)
def tearDown(self):
self._tmp.cleanup()
@staticmethod
def _build(path: Path, members: dict[str, bytes]) -> None:
with zipfile.ZipFile(path, "w", zipfile.ZIP_DEFLATED) as archive:
for name, data in members.items():
info = zipfile.ZipInfo(name, date_time=(1980, 1, 1, 0, 0, 0))
info.compress_type = zipfile.ZIP_DEFLATED
info.external_attr = (0o100000 | 0o644) << 16
archive.writestr(info, data)
def _read(self, parts: list[Path]) -> dict[str, bytes]:
found: dict[str, bytes] = {}
for part in parts:
with zipfile.ZipFile(part) as archive:
self.assertIsNone(archive.testzip())
for name in archive.namelist():
self.assertNotIn(name, found, "a file sits in two parts")
found[name] = archive.read(name)
return found
class PartsAreArchives(SplitFixture):
def test_a_pack_under_the_limit_stays_one_file(self):
parts = split_pack.split_pack(self.pack, limit=1 << 20)
self.assertEqual(parts, [self.pack])
self.assertTrue(self.pack.is_file())
def test_the_parts_extracted_together_are_the_pack(self):
parts = split_pack.split_pack(self.pack, limit=LIMIT)
self.assertGreater(len(parts), 1)
self.assertEqual(self._read(parts), self.members)
def test_every_part_fits_the_asset_limit(self):
for part in split_pack.split_pack(self.pack, limit=LIMIT):
self.assertLess(part.stat().st_size, LIMIT)
def test_a_part_name_says_how_many_there_are(self):
parts = split_pack.split_pack(self.pack, limit=LIMIT)
count = len(parts)
self.assertEqual(
[part.name for part in parts],
[
f"Demo_1.0_BIOS_Pack.part{index}of{count}.zip"
for index in range(1, count + 1)
],
)
def test_the_whole_archive_leaves_once_its_parts_are_checked(self):
split_pack.split_pack(self.pack, limit=LIMIT)
self.assertFalse(self.pack.exists())
def test_the_same_pack_gives_the_same_parts(self):
first = [p.read_bytes() for p in split_pack.split_pack(self.pack, limit=LIMIT)]
other = self.dist / "again"
other.mkdir()
again = other / self.pack.name
self._build(again, self.members)
second = [p.read_bytes() for p in split_pack.split_pack(again, limit=LIMIT)]
self.assertEqual(first, second)
def test_a_file_no_part_can_hold_is_refused(self):
big = self.dist / "Big_BIOS_Pack.zip"
self._build(big, {"huge.bin": _payload(1, LIMIT * 2), "small.bin": b"x"})
with self.assertRaises(ValueError):
split_pack.split_pack(big, limit=LIMIT)
self.assertTrue(big.is_file(), "the pack is kept when it cannot be split")
self.assertEqual(sorted(p.name for p in self.dist.glob("Big*")), [big.name])
def test_the_executable_bit_travels(self):
tool = self.dist / "Tools_BIOS_Pack.zip"
with zipfile.ZipFile(tool, "w", zipfile.ZIP_DEFLATED) as archive:
info = zipfile.ZipInfo("bin/engine", date_time=(1980, 1, 1, 0, 0, 0))
info.compress_type = zipfile.ZIP_DEFLATED
info.external_attr = (0o100000 | 0o755) << 16
archive.writestr(info, _payload(3, 3000))
for n in range(4):
archive.writestr(f"data{n}.bin", _payload(n, 900))
parts = split_pack.split_pack(tool, limit=LIMIT)
modes = {}
for part in parts:
with zipfile.ZipFile(part) as archive:
for info in archive.infolist():
modes[info.filename] = info.external_attr >> 16
self.assertTrue(modes["bin/engine"] & stat.S_IXUSR)
class PartsAreNotTakenForPacks(SplitFixture):
def test_a_part_is_recognised_by_its_name(self):
self.assertTrue(split_pack.is_part("RetroArch_BIOS_Pack.part1of2.zip"))
self.assertFalse(split_pack.is_part("RetroArch_BIOS_Pack.zip"))
self.assertFalse(split_pack.is_part("RetroArch_BIOS_Pack.zip.001"))
def test_pack_verification_skips_the_parts(self):
"""A part holds half a platform by design: judged as a pack it fails."""
parts = split_pack.split_pack(self.pack, limit=LIMIT)
self.assertEqual(
builder._pack_archives(str(self.dist)), [],
[p.name for p in parts],
)
def test_the_directory_form_splits_only_what_is_too_large(self):
small = self.dist / "Small_BIOS_Pack.zip"
self._build(small, {"one.bin": b"1"})
written = split_pack.split_directory(self.dist, limit=LIMIT)
names = sorted(path.name for path in written)
self.assertIn("Small_BIOS_Pack.zip", names)
self.assertNotIn("Demo_1.0_BIOS_Pack.zip", names)
self.assertTrue(all(os.path.exists(path) for path in written))
class ReleaseProcessPublishesArchives(unittest.TestCase):
def setUp(self):
self.text = (REPO_ROOT / "wiki" / "release-process.md").read_text(
encoding="utf-8"
)
def test_it_no_longer_cuts_byte_ranges(self):
# assertIn would print the whole page on failure.
self.assertFalse("split --bytes" in self.text)
self.assertTrue("scripts/split_pack.py" in self.text)
def test_the_checksums_are_those_of_the_published_files(self):
self.assertLess(
self.text.index("scripts/split_pack.py"),
self.text.index("sha256sum *.zip"),
)
class PartsAreExplainedWhereTheyAreDownloaded(unittest.TestCase):
PAGES = (
"scripts/generate_readme.py",
"scripts/generate_site.py",
"wiki/getting-started.md",
"wiki/troubleshooting.md",
)
def _text(self, relative: str) -> str:
return (REPO_ROOT / relative).read_text(encoding="utf-8")
def test_every_page_names_both_layouts(self):
"""The older release stays downloadable until the next one replaces it."""
for page in self.PAGES:
with self.subTest(page=page):
text = self._text(page)
self.assertTrue(".part1of2.zip" in text)
self.assertTrue(".zip.001" in text)
def test_the_windows_join_command_runs_from_powershell(self):
"""`copy /b A+B C` is a cmd builtin. Typed into PowerShell, the shell
the install instructions open, it is Copy-Item and fails."""
for page in self.PAGES:
with self.subTest(page=page):
for line in self._text(page).splitlines():
if "copy /b" in line:
self.assertTrue("cmd /c copy /b" in line, line.strip())
if __name__ == "__main__":
unittest.main()
+14 -10
View File
@@ -77,9 +77,9 @@ bash scripts/download.sh retroarch ~/RetroArch/system/
bash scripts/download.sh --list # show available packs
```
A pack published in several volumes is downloaded part by part, joined, and
checked against the SHA-256 the release publishes before anything is
extracted. `python scripts/download.py` does the same on Windows.
A pack published in several parts is downloaded part by part and checked
against the SHA-256 the release publishes before anything is extracted.
`python scripts/download.py` does the same on Windows.
### Option 3: manual download
@@ -87,15 +87,19 @@ extracted. `python scripts/download.py` does the same on Windows.
2. Download the ZIP pack for your platform
3. Extract to the BIOS directory listed below
Packs over 2 GB are split into numbered volumes (`.zip.001`, `.zip.002`).
Download every part into the same folder, then open the `.001` with 7-Zip or
PeaZip, which read the whole set. To join them into one ZIP first:
A pack over 2 GB comes in several parts, and every part is needed. How to
open them depends on their name:
- Linux/macOS: `cat Pack.zip.0* > Pack.zip`
- Windows (cmd): `copy /b Pack.zip.001+Pack.zip.002 Pack.zip`
- `Pack.part1of2.zip`, `Pack.part2of2.zip`: each part is an ordinary ZIP.
Extract them all into the same folder.
- `Pack.zip.001`, `Pack.zip.002` (releases up to v2026.09.04): slices of one
ZIP, none of which opens on its own. Put them in one folder and open the
`.001` with 7-Zip or PeaZip, or join them first:
- Linux/macOS: `cat Pack.zip.0* > Pack.zip`
- Windows: `cmd /c copy /b Pack.zip.001+Pack.zip.002 Pack.zip`
A frontend's own extractor may refuse a volume: Batocera answers `Archive
type: '001' is not yet supported`. Join the parts from a shell there. See
A frontend's own extractor may refuse a slice: Batocera answers `Archive
type: '001' is not yet supported`. Join the slices from a shell there. See
[Download](../which-pack.md) for the per-setup instructions.
## BIOS directory by platform
+33 -20
View File
@@ -161,19 +161,19 @@ python scripts/pipeline.py
python scripts/generate_pack.py --platform retropie --output-dir dist/
python scripts/generate_pack.py --platform retropie --verify-packs --output-dir dist/
# 3. Checksums of the full ZIPs, before splitting, then sign the list. The
# checksums answer corruption; the signature answers a rewritten release,
# which is the one thing a checksum published beside its own artifacts
# cannot answer.
# 3. A release file must be under 2 GiB. A larger pack becomes parts that are
# each a ZIP of whole files (Pack.part1of2.zip): any tool opens one, and
# the parts extracted into one folder are the pack. Each part is read back
# and the set compared with the pack before the pack is removed.
python scripts/split_pack.py dist/
# 4. Checksums of the files as published, then sign the list. The checksums
# answer corruption; the signature answers a rewritten release, which is
# the one thing a checksum published beside its own artifacts cannot
# answer.
(cd dist && sha256sum *.zip > SHA256SUMS.txt)
ssh-keygen -Y sign -f ~/.ssh/retrobios_signing -n file dist/SHA256SUMS.txt
# 4. Split anything over 2 GB (GitHub asset cap); 7-Zip and PeaZip open .001 directly
for f in dist/*.zip; do
[ "$(stat -c%s "$f")" -gt 2000000000 ] || continue
split --bytes=1900M --numeric-suffixes=1 --suffix-length=3 "$f" "$f." && rm "$f"
done
# 5. The two sizes a pack has, and its file count. Extracted runs well above
# downloaded, so the notes table names the one it carries. The count is
# what a file manager shows once the pack is extracted.
@@ -183,8 +183,8 @@ sys.path.insert(0, "scripts")
from download import _match_key
parts = {}
for path in sorted(pathlib.Path("dist").glob("*_BIOS_Pack.zip*")):
parts.setdefault(path.name.split(".zip")[0] + ".zip", []).append(path)
for path in sorted(pathlib.Path("dist").glob("*_BIOS_Pack*.zip")):
parts.setdefault(path.name.split("_BIOS_Pack")[0], []).append(path)
def size(n):
return f"{n / 1024 ** 3:.1f} GB" if n >= 1024 ** 3 else f"{n / 1024 ** 2:.0f} MB"
@@ -204,7 +204,7 @@ PY
# during the whole upload.
DATE=$(date +%Y.%m.%d)
gh release create "v${DATE}" --draft --title "BIOS Pack v${DATE}" --notes-file notes.md
for f in dist/SHA256SUMS.txt dist/SHA256SUMS.txt.sig dist/*.zip dist/*.zip.0*; do
for f in dist/SHA256SUMS.txt dist/SHA256SUMS.txt.sig dist/*.zip; do
[ -f "$f" ] && gh release upload "v${DATE}" "$f#$(basename "$f")" --clobber
done
gh release view "v${DATE}" --json assets --jq '.assets | length' # expect every file
@@ -233,14 +233,17 @@ sha256sum --check --ignore-missing SHA256SUMS.txt
```
The first command must print `Good "file" signature for releases@retrobios`.
Order matters: verify the list before trusting the sums in it, and join split
volumes before checking, since the sums are of the full ZIPs.
Order matters: verify the list before trusting the sums in it. The list names
every file as published, so a single part checks on its own. Releases up to
v2026.09.04 listed the whole ZIPs instead, and their `.zip.001` volumes have
to be joined before checking.
The signature and the reproducible build answer different questions. The
signature says the list came from the holder of the release key. The build
says the bytes are derivable: packs are deterministic, so rebuilding one from
the same collection yields the same archive, and its checksum can be compared
against the signed list without trusting either.
says the bytes are derivable: packs are deterministic and a part copies its
members as stored, so rebuilding and splitting from the same collection
yields the same files, and their checksums can be compared against the
signed list without trusting either.
The private half lives on the maintainer's machine and is generated with
`ssh-keygen -t ed25519 -f ~/.ssh/retrobios_signing -C releases@retrobios`. Its
@@ -261,8 +264,18 @@ the closed issues. A pack has two sizes and they are far apart: Batocera
downloads as 2.4 GB and extracts to 4.0 GB. Step 5 prints both, and whichever
one the table carries, the header names it. Someone sizing a USB drive is
reading that column. The README table is the extracted size, from the install
manifests. `SHA256SUMS.txt` lists the checksums of the full ZIPs before
splitting.
manifests. `SHA256SUMS.txt` lists the checksums of the files as published,
parts included.
A pack in parts is said so above the table, where the links are: every part
is needed, each is an ordinary ZIP, and they extract into the same folder.
Until v2026.09.04 the parts were byte ranges cut by `split`, named
`.zip.001`, and the sentence explaining them sat under the table. A range
opened alone is not an archive and no tool says a part is missing, so five
reports in six months took one for a broken download (#45, #51, #66, #77,
#79). The README, the download page and the troubleshooting page describe
both layouts for as long as a release cut that way is still published; once
step 7 has deleted it, the `.zip.001` paragraph leaves those three pages.
The table carries a Files column, the `pack_files` step 5 prints, and the
notes never open on the size of the collection. That total covers every
+7 -4
View File
@@ -538,8 +538,9 @@ python scripts/refresh_stale.py --only platforms,targets --jobs 6
| `validate_schemas.py` | Validate the data contracts: schemas and semantic invariants. `--source-only` checks `emulators/` and `platforms/` alone, which is what PR validation runs |
| `auto_fetch.py` | Fetch missing BIOS files from known sources (4-step pipeline) |
| `list_platforms.py` | List active platforms (`--all` includes archived, used by CI) |
| `download.py` | Download a pack from GitHub releases, split volumes joined and checked (Python, stdlib only) |
| `download.py` | Download a pack from GitHub releases, every part checked (Python, stdlib only) |
| `download.sh` | Same, as a shell one-liner (`curl` + `unzip`) |
| `split_pack.py` | Cut a pack over the release asset limit into parts that are each a ZIP of whole files, read back before the pack is removed |
| `provenance_report.py` | Dump-catalog coverage and acquisition targets (see above) |
| `generate_readme.py` | Generate README.md and CONTRIBUTING.md from database |
| `generate_site.py` | Generate all MkDocs site pages (this documentation) |
@@ -581,9 +582,11 @@ same-named file.
`scripts/download.sh` remains available for downloading a prebuilt platform
ZIP when a manually reviewed pack release contains it; it is separate from the
per-file automatic installer above. A pack over 2 GB is published as numbered
volumes: both downloaders group them under one platform name, download each
one, join them and check the result against the release's `SHA256SUMS.txt`.
per-file automatic installer above. A pack over 2 GB is published in several
parts: both downloaders group them under one platform name, download each
one and check it against the release's `SHA256SUMS.txt`. Parts that are each
a ZIP are extracted one after the other; the `.zip.001` slices of releases up
to v2026.09.04 are joined first.
That list carries a detached signature, `SHA256SUMS.txt.sig`, verifiable
against `allowed_signers` at the repository root; the
[release process](release-process.md#verifying-a-release) gives the commands.
+13 -6
View File
@@ -271,18 +271,25 @@ Some platforms share packs (Lakka uses the RetroArch pack). The installer handle
this mapping automatically, but if you're downloading manually, check which pack
name corresponds to your platform.
**A split pack will not extract:**
**A pack in several parts will not extract:**
A pack over 2 GB is published as `.zip.001`, `.zip.002`. Put every part in one
folder; 7-Zip and PeaZip open the `.001` directly. A frontend's own extractor
may refuse it, Batocera among them:
Parts named `Pack.part1of2.zip` are ordinary ZIPs: extract each one into the
same folder, and check that all of them were downloaded, the name says how
many there are.
Parts named `Pack.zip.001`, `Pack.zip.002` (releases up to v2026.09.04) are
slices of one ZIP. None opens on its own, and the error never says a part is
missing: 7-Zip reports `Unavailable start of archive` on a `.002`, Windows
has no program for the extension. Put every slice in one folder; 7-Zip and
PeaZip open the `.001` directly. A frontend's own extractor may refuse it,
Batocera among them:
```
Archive type: '001' is not yet supported
```
The volumes are plain byte ranges, so joining them from a shell rebuilds the
ZIP:
The slices are plain byte ranges, so joining them from a shell rebuilds the
ZIP (`cmd /c copy /b Pack.zip.001+Pack.zip.002 Pack.zip` on Windows):
```bash
cat Pack.zip.0* > Pack.zip