feat: add torrentzip builder for romset reconstruction

This commit is contained in:
Abdessamad Derraz committed 2026-08-06 16:41:47 +02:00
1 parent c4bef1010a
commit 966c0e5b59
2 files changed
+250

No files matched your search

+146
View File
@@ -0,0 +1,146 @@
#!/usr/bin/env python3
"""TorrentZip archive builder for MAME/FBNeo ROM sets.
TorrentZip is the archive format MAME romset distributions use. It fixes
every non-content variable (member order, timestamps, compression level,
extra fields, attributes), so the archive bytes are a pure function of the
member names and their contents.
That property makes it possible to rebuild a romset ZIP that matches an
upstream-published hash exactly, and to identify which romset revision a
published hash refers to.
Usage:
from torrentzip import build_torrentzip, identify_romset
data = build_torrentzip([("sp-s2.sp1", rom_bytes), ...])
md5 = hashlib.md5(data).hexdigest()
CLI:
python scripts/torrentzip.py --check bios/Arcade/Arcade/naomi2.zip
python scripts/torrentzip.py --rebuild in.zip --output out.zip
"""
from __future__ import annotations
import argparse
import hashlib
import struct
import sys
import zipfile
import zlib
from pathlib import Path
# TorrentZip constants: 1996-12-24 23:32:00, deflate level 9
_DOS_DATE = ((1996 - 1980) << 9) | (12 << 5) | 24
_DOS_TIME = (23 << 11) | (32 << 5) | 0
_COMMENT_PREFIX = "TORRENTZIPPED-"
def build_torrentzip(members: list[tuple[str, bytes]]) -> bytes:
"""Build a TorrentZip archive from (name, data) members.
Members are sorted case-insensitively by name, as the format requires.
Returns the complete archive bytes.
"""
ordered = sorted(members, key=lambda m: m[0].lower())
body = bytearray()
central = bytearray()
for name, data in ordered:
raw = name.encode("ascii")
compressor = zlib.compressobj(9, zlib.DEFLATED, -15)
blob = compressor.compress(data) + compressor.flush()
crc = zlib.crc32(data) & 0xFFFFFFFF
offset = len(body)
body += struct.pack(
"<IHHHHHIIIHH", 0x04034B50, 20, 2, 8, _DOS_TIME, _DOS_DATE,
crc, len(blob), len(data), len(raw), 0,
)
body += raw + blob
central += struct.pack(
"<IHHHHHHIIIHHHHHII", 0x02014B50, 0, 20, 2, 8, _DOS_TIME, _DOS_DATE,
crc, len(blob), len(data), len(raw), 0, 0, 0, 0, 0, offset,
)
central += raw
cd_offset = len(body)
body += central
comment = (
f"{_COMMENT_PREFIX}{zlib.crc32(bytes(central)) & 0xFFFFFFFF:08X}"
).encode("ascii")
body += struct.pack(
"<IHHHHIIH", 0x06054B50, 0, 0, len(ordered), len(ordered),
len(central), cd_offset, len(comment),
)
body += comment
return bytes(body)
def rebuild_torrentzip(source: str | Path) -> bytes:
"""Read an archive and return its TorrentZip normalization."""
with zipfile.ZipFile(source) as zf:
members = [
(info.filename, zf.read(info.filename))
for info in zf.infolist()
if not info.is_dir()
]
return build_torrentzip(members)
def is_torrentzip(source: str | Path) -> bool:
"""Report whether an archive already carries the TorrentZip signature."""
with zipfile.ZipFile(source) as zf:
return zf.comment.decode("ascii", "replace").startswith(_COMMENT_PREFIX)
def identify_romset(
recipes: dict[str, list[tuple[str, str]]],
atoms: dict[str, bytes],
target_md5: str,
) -> str | None:
"""Find which recipe rebuilds to a target archive MD5.
recipes maps a label (e.g. a MAME version) to a list of
(member name, CRC32 hex) pairs; atoms maps CRC32 hex to ROM bytes.
Returns the matching label, or None when no recipe reproduces the hash.
"""
target = target_md5.lower()
for label, recipe in recipes.items():
if any(crc.lower() not in atoms for _, crc in recipe):
continue
members = [(name, atoms[crc.lower()]) for name, crc in recipe]
if hashlib.md5(build_torrentzip(members)).hexdigest() == target:
return label
return None
def main() -> None:
"""Entry point."""
parser = argparse.ArgumentParser(description="TorrentZip builder and checker")
parser.add_argument("--check", metavar="ZIP", help="report TorrentZip conformance")
parser.add_argument("--rebuild", metavar="ZIP", help="normalize an archive")
parser.add_argument("--output", "-o", help="output path for --rebuild")
args = parser.parse_args()
if args.check:
data = Path(args.check).read_bytes()
rebuilt = rebuild_torrentzip(args.check)
print(f"{args.check}")
print(f" signature: {is_torrentzip(args.check)}")
print(f" md5: {hashlib.md5(data).hexdigest()}")
print(f" rebuilt: {hashlib.md5(rebuilt).hexdigest()}")
print(f" conform: {data == rebuilt}")
return
if args.rebuild:
data = rebuild_torrentzip(args.rebuild)
dest = args.output or args.rebuild
Path(dest).write_bytes(data)
print(f"{dest}: {len(data)} bytes, md5 {hashlib.md5(data).hexdigest()}")
return
parser.error("nothing to do: pass --check or --rebuild")
if __name__ == "__main__":
sys.exit(main())
+104
View File
@@ -0,0 +1,104 @@
"""Tests for the TorrentZip builder against real MAME romsets."""
from __future__ import annotations
import hashlib
import os
import sys
import tempfile
import unittest
import zipfile
REPO_ROOT = os.path.join(os.path.dirname(__file__), "..")
sys.path.insert(0, os.path.join(REPO_ROOT, "scripts"))
from torrentzip import ( # noqa: E402
build_torrentzip,
identify_romset,
is_torrentzip,
rebuild_torrentzip,
)
class TorrentZipTest(unittest.TestCase):
"""The format is deterministic: same members always give the same bytes."""
def test_deterministic_output(self):
members = [("b.rom", b"second"), ("a.rom", b"first")]
self.assertEqual(build_torrentzip(members), build_torrentzip(members))
def test_member_order_is_irrelevant(self):
a = build_torrentzip([("a.rom", b"one"), ("b.rom", b"two")])
b = build_torrentzip([("b.rom", b"two"), ("a.rom", b"one")])
self.assertEqual(a, b)
def test_case_insensitive_sorting(self):
data = build_torrentzip([("B.rom", b"x"), ("a.rom", b"y")])
with zipfile.ZipFile(_as_file(self, data)) as zf:
self.assertEqual([i.filename for i in zf.infolist()], ["a.rom", "B.rom"])
def test_fixed_timestamp_and_signature(self):
data = build_torrentzip([("a.rom", b"payload")])
with zipfile.ZipFile(_as_file(self, data)) as zf:
self.assertEqual(zf.infolist()[0].date_time, (1996, 12, 24, 23, 32, 0))
self.assertTrue(zf.comment.decode().startswith("TORRENTZIPPED-"))
def test_roundtrip_readable_content(self):
data = build_torrentzip([("a.rom", b"hello"), ("b.rom", b"world")])
with zipfile.ZipFile(_as_file(self, data)) as zf:
self.assertEqual(zf.read("a.rom"), b"hello")
self.assertEqual(zf.read("b.rom"), b"world")
def test_rebuild_matches_real_mame_set(self):
"""Rebuilding a shipped MAME romset reproduces it byte for byte."""
candidates = [
os.path.join(REPO_ROOT, "bios", "Arcade", "Arcade", "naomi2.zip"),
os.path.join(REPO_ROOT, "bios", "Arcade", "MAME", "naomi2.zip"),
]
checked = 0
for path in candidates:
if not os.path.exists(path) or not is_torrentzip(path):
continue
with open(path, "rb") as fh:
original = fh.read()
self.assertEqual(rebuild_torrentzip(path), original, path)
checked += 1
if not checked:
self.skipTest("no TorrentZip romset available")
def test_identify_romset_selects_matching_recipe(self):
atoms = {"3610a686": b"alpha", "9d0d1b46": b"beta"}
import zlib
atoms = {
f"{zlib.crc32(b'alpha') & 0xffffffff:08x}": b"alpha",
f"{zlib.crc32(b'beta') & 0xffffffff:08x}": b"beta",
}
crc_a, crc_b = list(atoms)
wanted = hashlib.md5(
build_torrentzip([("a.rom", b"alpha"), ("b.rom", b"beta")])
).hexdigest()
recipes = {
"v1": [("a.rom", crc_a)],
"v2": [("a.rom", crc_a), ("b.rom", crc_b)],
}
self.assertEqual(identify_romset(recipes, atoms, wanted), "v2")
self.assertIsNone(identify_romset(recipes, atoms, "0" * 32))
def test_identify_romset_skips_recipes_with_absent_atoms(self):
atoms = {"aaaaaaaa": b"x"}
recipes = {"v1": [("a.rom", "ffffffff")]}
self.assertIsNone(identify_romset(recipes, atoms, "0" * 32))
def _as_file(test: unittest.TestCase, data: bytes) -> str:
"""Write bytes to a temp file cleaned up with the test."""
fd, path = tempfile.mkstemp(suffix=".zip")
os.close(fd)
with open(path, "wb") as fh:
fh.write(data)
test.addCleanup(os.unlink, path)
return path
if __name__ == "__main__":
unittest.main()