mirror of
https://github.com/Abdess/retroarch_system.git
synced 2026-10-10 13:33:24 -05:00
feat: add torrentzip builder for romset reconstruction
This commit is contained in:
1 parent
c4bef1010a
commit
966c0e5b59
2 files changed
+250
No files matched your search
@@ -0,0 +1,146 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""TorrentZip archive builder for MAME/FBNeo ROM sets.
|
||||||
|
|
||||||
|
TorrentZip is the archive format MAME romset distributions use. It fixes
|
||||||
|
every non-content variable (member order, timestamps, compression level,
|
||||||
|
extra fields, attributes), so the archive bytes are a pure function of the
|
||||||
|
member names and their contents.
|
||||||
|
|
||||||
|
That property makes it possible to rebuild a romset ZIP that matches an
|
||||||
|
upstream-published hash exactly, and to identify which romset revision a
|
||||||
|
published hash refers to.
|
||||||
|
|
||||||
|
Usage:
|
||||||
|
from torrentzip import build_torrentzip, identify_romset
|
||||||
|
|
||||||
|
data = build_torrentzip([("sp-s2.sp1", rom_bytes), ...])
|
||||||
|
md5 = hashlib.md5(data).hexdigest()
|
||||||
|
|
||||||
|
CLI:
|
||||||
|
python scripts/torrentzip.py --check bios/Arcade/Arcade/naomi2.zip
|
||||||
|
python scripts/torrentzip.py --rebuild in.zip --output out.zip
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import hashlib
|
||||||
|
import struct
|
||||||
|
import sys
|
||||||
|
import zipfile
|
||||||
|
import zlib
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
# TorrentZip constants: 1996-12-24 23:32:00, deflate level 9
|
||||||
|
_DOS_DATE = ((1996 - 1980) << 9) | (12 << 5) | 24
|
||||||
|
_DOS_TIME = (23 << 11) | (32 << 5) | 0
|
||||||
|
_COMMENT_PREFIX = "TORRENTZIPPED-"
|
||||||
|
|
||||||
|
|
||||||
|
def build_torrentzip(members: list[tuple[str, bytes]]) -> bytes:
|
||||||
|
"""Build a TorrentZip archive from (name, data) members.
|
||||||
|
|
||||||
|
Members are sorted case-insensitively by name, as the format requires.
|
||||||
|
Returns the complete archive bytes.
|
||||||
|
"""
|
||||||
|
ordered = sorted(members, key=lambda m: m[0].lower())
|
||||||
|
body = bytearray()
|
||||||
|
central = bytearray()
|
||||||
|
|
||||||
|
for name, data in ordered:
|
||||||
|
raw = name.encode("ascii")
|
||||||
|
compressor = zlib.compressobj(9, zlib.DEFLATED, -15)
|
||||||
|
blob = compressor.compress(data) + compressor.flush()
|
||||||
|
crc = zlib.crc32(data) & 0xFFFFFFFF
|
||||||
|
offset = len(body)
|
||||||
|
body += struct.pack(
|
||||||
|
"<IHHHHHIIIHH", 0x04034B50, 20, 2, 8, _DOS_TIME, _DOS_DATE,
|
||||||
|
crc, len(blob), len(data), len(raw), 0,
|
||||||
|
)
|
||||||
|
body += raw + blob
|
||||||
|
central += struct.pack(
|
||||||
|
"<IHHHHHHIIIHHHHHII", 0x02014B50, 0, 20, 2, 8, _DOS_TIME, _DOS_DATE,
|
||||||
|
crc, len(blob), len(data), len(raw), 0, 0, 0, 0, 0, offset,
|
||||||
|
)
|
||||||
|
central += raw
|
||||||
|
|
||||||
|
cd_offset = len(body)
|
||||||
|
body += central
|
||||||
|
comment = (
|
||||||
|
f"{_COMMENT_PREFIX}{zlib.crc32(bytes(central)) & 0xFFFFFFFF:08X}"
|
||||||
|
).encode("ascii")
|
||||||
|
body += struct.pack(
|
||||||
|
"<IHHHHIIH", 0x06054B50, 0, 0, len(ordered), len(ordered),
|
||||||
|
len(central), cd_offset, len(comment),
|
||||||
|
)
|
||||||
|
body += comment
|
||||||
|
return bytes(body)
|
||||||
|
|
||||||
|
|
||||||
|
def rebuild_torrentzip(source: str | Path) -> bytes:
|
||||||
|
"""Read an archive and return its TorrentZip normalization."""
|
||||||
|
with zipfile.ZipFile(source) as zf:
|
||||||
|
members = [
|
||||||
|
(info.filename, zf.read(info.filename))
|
||||||
|
for info in zf.infolist()
|
||||||
|
if not info.is_dir()
|
||||||
|
]
|
||||||
|
return build_torrentzip(members)
|
||||||
|
|
||||||
|
|
||||||
|
def is_torrentzip(source: str | Path) -> bool:
|
||||||
|
"""Report whether an archive already carries the TorrentZip signature."""
|
||||||
|
with zipfile.ZipFile(source) as zf:
|
||||||
|
return zf.comment.decode("ascii", "replace").startswith(_COMMENT_PREFIX)
|
||||||
|
|
||||||
|
|
||||||
|
def identify_romset(
|
||||||
|
recipes: dict[str, list[tuple[str, str]]],
|
||||||
|
atoms: dict[str, bytes],
|
||||||
|
target_md5: str,
|
||||||
|
) -> str | None:
|
||||||
|
"""Find which recipe rebuilds to a target archive MD5.
|
||||||
|
|
||||||
|
recipes maps a label (e.g. a MAME version) to a list of
|
||||||
|
(member name, CRC32 hex) pairs; atoms maps CRC32 hex to ROM bytes.
|
||||||
|
Returns the matching label, or None when no recipe reproduces the hash.
|
||||||
|
"""
|
||||||
|
target = target_md5.lower()
|
||||||
|
for label, recipe in recipes.items():
|
||||||
|
if any(crc.lower() not in atoms for _, crc in recipe):
|
||||||
|
continue
|
||||||
|
members = [(name, atoms[crc.lower()]) for name, crc in recipe]
|
||||||
|
if hashlib.md5(build_torrentzip(members)).hexdigest() == target:
|
||||||
|
return label
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
"""Entry point."""
|
||||||
|
parser = argparse.ArgumentParser(description="TorrentZip builder and checker")
|
||||||
|
parser.add_argument("--check", metavar="ZIP", help="report TorrentZip conformance")
|
||||||
|
parser.add_argument("--rebuild", metavar="ZIP", help="normalize an archive")
|
||||||
|
parser.add_argument("--output", "-o", help="output path for --rebuild")
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
if args.check:
|
||||||
|
data = Path(args.check).read_bytes()
|
||||||
|
rebuilt = rebuild_torrentzip(args.check)
|
||||||
|
print(f"{args.check}")
|
||||||
|
print(f" signature: {is_torrentzip(args.check)}")
|
||||||
|
print(f" md5: {hashlib.md5(data).hexdigest()}")
|
||||||
|
print(f" rebuilt: {hashlib.md5(rebuilt).hexdigest()}")
|
||||||
|
print(f" conform: {data == rebuilt}")
|
||||||
|
return
|
||||||
|
|
||||||
|
if args.rebuild:
|
||||||
|
data = rebuild_torrentzip(args.rebuild)
|
||||||
|
dest = args.output or args.rebuild
|
||||||
|
Path(dest).write_bytes(data)
|
||||||
|
print(f"{dest}: {len(data)} bytes, md5 {hashlib.md5(data).hexdigest()}")
|
||||||
|
return
|
||||||
|
|
||||||
|
parser.error("nothing to do: pass --check or --rebuild")
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
sys.exit(main())
|
||||||
@@ -0,0 +1,104 @@
|
|||||||
|
"""Tests for the TorrentZip builder against real MAME romsets."""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import hashlib
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import tempfile
|
||||||
|
import unittest
|
||||||
|
import zipfile
|
||||||
|
|
||||||
|
REPO_ROOT = os.path.join(os.path.dirname(__file__), "..")
|
||||||
|
sys.path.insert(0, os.path.join(REPO_ROOT, "scripts"))
|
||||||
|
|
||||||
|
from torrentzip import ( # noqa: E402
|
||||||
|
build_torrentzip,
|
||||||
|
identify_romset,
|
||||||
|
is_torrentzip,
|
||||||
|
rebuild_torrentzip,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class TorrentZipTest(unittest.TestCase):
|
||||||
|
"""The format is deterministic: same members always give the same bytes."""
|
||||||
|
|
||||||
|
def test_deterministic_output(self):
|
||||||
|
members = [("b.rom", b"second"), ("a.rom", b"first")]
|
||||||
|
self.assertEqual(build_torrentzip(members), build_torrentzip(members))
|
||||||
|
|
||||||
|
def test_member_order_is_irrelevant(self):
|
||||||
|
a = build_torrentzip([("a.rom", b"one"), ("b.rom", b"two")])
|
||||||
|
b = build_torrentzip([("b.rom", b"two"), ("a.rom", b"one")])
|
||||||
|
self.assertEqual(a, b)
|
||||||
|
|
||||||
|
def test_case_insensitive_sorting(self):
|
||||||
|
data = build_torrentzip([("B.rom", b"x"), ("a.rom", b"y")])
|
||||||
|
with zipfile.ZipFile(_as_file(self, data)) as zf:
|
||||||
|
self.assertEqual([i.filename for i in zf.infolist()], ["a.rom", "B.rom"])
|
||||||
|
|
||||||
|
def test_fixed_timestamp_and_signature(self):
|
||||||
|
data = build_torrentzip([("a.rom", b"payload")])
|
||||||
|
with zipfile.ZipFile(_as_file(self, data)) as zf:
|
||||||
|
self.assertEqual(zf.infolist()[0].date_time, (1996, 12, 24, 23, 32, 0))
|
||||||
|
self.assertTrue(zf.comment.decode().startswith("TORRENTZIPPED-"))
|
||||||
|
|
||||||
|
def test_roundtrip_readable_content(self):
|
||||||
|
data = build_torrentzip([("a.rom", b"hello"), ("b.rom", b"world")])
|
||||||
|
with zipfile.ZipFile(_as_file(self, data)) as zf:
|
||||||
|
self.assertEqual(zf.read("a.rom"), b"hello")
|
||||||
|
self.assertEqual(zf.read("b.rom"), b"world")
|
||||||
|
|
||||||
|
def test_rebuild_matches_real_mame_set(self):
|
||||||
|
"""Rebuilding a shipped MAME romset reproduces it byte for byte."""
|
||||||
|
candidates = [
|
||||||
|
os.path.join(REPO_ROOT, "bios", "Arcade", "Arcade", "naomi2.zip"),
|
||||||
|
os.path.join(REPO_ROOT, "bios", "Arcade", "MAME", "naomi2.zip"),
|
||||||
|
]
|
||||||
|
checked = 0
|
||||||
|
for path in candidates:
|
||||||
|
if not os.path.exists(path) or not is_torrentzip(path):
|
||||||
|
continue
|
||||||
|
with open(path, "rb") as fh:
|
||||||
|
original = fh.read()
|
||||||
|
self.assertEqual(rebuild_torrentzip(path), original, path)
|
||||||
|
checked += 1
|
||||||
|
if not checked:
|
||||||
|
self.skipTest("no TorrentZip romset available")
|
||||||
|
|
||||||
|
def test_identify_romset_selects_matching_recipe(self):
|
||||||
|
atoms = {"3610a686": b"alpha", "9d0d1b46": b"beta"}
|
||||||
|
import zlib
|
||||||
|
|
||||||
|
atoms = {
|
||||||
|
f"{zlib.crc32(b'alpha') & 0xffffffff:08x}": b"alpha",
|
||||||
|
f"{zlib.crc32(b'beta') & 0xffffffff:08x}": b"beta",
|
||||||
|
}
|
||||||
|
crc_a, crc_b = list(atoms)
|
||||||
|
wanted = hashlib.md5(
|
||||||
|
build_torrentzip([("a.rom", b"alpha"), ("b.rom", b"beta")])
|
||||||
|
).hexdigest()
|
||||||
|
recipes = {
|
||||||
|
"v1": [("a.rom", crc_a)],
|
||||||
|
"v2": [("a.rom", crc_a), ("b.rom", crc_b)],
|
||||||
|
}
|
||||||
|
self.assertEqual(identify_romset(recipes, atoms, wanted), "v2")
|
||||||
|
self.assertIsNone(identify_romset(recipes, atoms, "0" * 32))
|
||||||
|
|
||||||
|
def test_identify_romset_skips_recipes_with_absent_atoms(self):
|
||||||
|
atoms = {"aaaaaaaa": b"x"}
|
||||||
|
recipes = {"v1": [("a.rom", "ffffffff")]}
|
||||||
|
self.assertIsNone(identify_romset(recipes, atoms, "0" * 32))
|
||||||
|
|
||||||
|
|
||||||
|
def _as_file(test: unittest.TestCase, data: bytes) -> str:
|
||||||
|
"""Write bytes to a temp file cleaned up with the test."""
|
||||||
|
fd, path = tempfile.mkstemp(suffix=".zip")
|
||||||
|
os.close(fd)
|
||||||
|
with open(path, "wb") as fh:
|
||||||
|
fh.write(data)
|
||||||
|
test.addCleanup(os.unlink, path)
|
||||||
|
return path
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
Reference in new issue
Block a user