fix: keep install manifests and release assets true

This commit is contained in:
Abdessamad Derraz committed 2026-09-14 01:17:49 +02:00
1 parent 4f67e3cad3
commit 08bce47668
22 files changed
+878 -263

No files matched your search

+336
View File
@@ -0,0 +1,336 @@
#!/usr/bin/env python3
"""Compare the large-files release with the files the collection keeps out of git.
Every gitignored file under bios/ is served to the installer as an asset of
the large-files release, and install.py refuses a download whose
Content-Length differs from the manifest size. The manifest size is that of
the local copy, so an asset uploaded before the local file was rebuilt fails
every install of it until it is uploaded again.
Usage:
python scripts/check_release_assets.py
python scripts/check_release_assets.py --json
"""
from __future__ import annotations
import argparse
import hashlib
import json
import os
import re
import subprocess
import sys
sys.path.insert(0, os.path.dirname(__file__))
from common import load_database
from largefiles import LARGE_FILES_RELEASE, LARGE_FILES_REPO
Finding = tuple[str, str, int, int | None]
def expected_assets(db: dict, gitignore_text: str) -> dict[str, int]:
"""Map each gitignored database path to the size the manifests carry."""
ignored = {
line.strip()
for line in gitignore_text.splitlines()
if line.strip().startswith("bios/")
}
return {
entry["path"]: entry["size"]
for entry in db.get("files", {}).values()
if entry.get("path") in ignored
}
def _asset_names(path: str) -> list[str]:
"""Names the release may publish a file under.
GitHub rewrites spaces to dots in asset names, so a file whose name
carries spaces is published under the dotted form.
"""
name = os.path.basename(path)
return [name, name.replace(" ", ".")] if " " in name else [name]
def compare(
expected: dict[str, int], assets: dict[str, int], notes_current: bool = True
) -> list[Finding]:
"""Files the release lacks or serves at another size, sorted by kind.
A release description that no longer matches what render_notes() would
write is a finding of its own: the page is what a reader checks a
download against.
"""
findings: list[Finding] = []
if not notes_current:
findings.append(("notes", "release description", 0, None))
for path, size in sorted(expected.items()):
published = next(
(assets[name] for name in _asset_names(path) if name in assets), None
)
if published is None:
findings.append(("missing", path, size, None))
elif published != size:
findings.append(("size", path, size, published))
return sorted(findings)
ASSET_URL = f"https://github.com/{LARGE_FILES_REPO}/releases/download/{LARGE_FILES_RELEASE}/"
NOT_INDEXED = "Not indexed"
SECTIONS = (
"Console firmware",
"Arcade",
"Computer",
"Game engine data",
"Virtual machine firmware",
"Other",
NOT_INDEXED,
)
_ROW = re.compile(
r"^\| \[(?P<name>[^\]]+)\]\([^)]*\) \| (?P<desc>[^|]*?) \| [^|]* \|"
r"(?: `(?P<sha1>[0-9a-f]{40})` \|)?$",
re.M,
)
_HEADING = re.compile(r"^## (.+)$", re.M)
def parse_notes(body: str) -> dict[str, tuple[str, str, str]]:
"""Name -> (section, description, sha1) for every row of a release page."""
rows: dict[str, tuple[str, str, str]] = {}
section = ""
for line in body.splitlines():
heading = _HEADING.match(line)
if heading:
section = heading.group(1).strip()
continue
row = _ROW.match(line)
if row:
rows[row.group("name")] = (
section, row.group("desc").strip(), row.group("sha1") or ""
)
return rows
def _section_for(path: str) -> str:
parts = path.split("/")
top = parts[1] if len(parts) > 1 else ""
if top in ("Nintendo", "Sony", "Sega", "Microsoft", "NEC", "SNK"):
return "Console firmware"
if top == "Arcade" or "/samples/" in path:
return "Arcade"
if top == "QEMU":
return "Virtual machine firmware"
if top in ("RPG Maker", "ScummVM"):
return "Game engine data"
if top in ("Apple", "Commodore", "Atari", "Sinclair", "Amstrad"):
return "Computer"
return "Other"
def _mib(size: int) -> str:
return f"{round(size / 1048576)} MB"
def render_notes(
db: dict,
gitignore_text: str,
assets: dict[str, int],
previous: str,
descriptions: dict[str, str],
bundles: dict[str, str],
cache_sha1: dict[str, str],
) -> str:
"""The release page, from what the release actually serves.
Every asset gets one row. A file the database indexes carries its
database SHA1; a data-directory bundle carries the hash of its cached
copy when one exists; anything else is listed apart with its size only,
so the page never vouches for bytes nobody indexed. Descriptions and
sections already written by hand are kept by name.
"""
known = parse_notes(previous)
ignored = {
line.strip()
for line in gitignore_text.splitlines()
if line.strip().startswith("bios/")
}
indexed: dict[str, tuple[str, str]] = {}
for sha1, entry in db.get("files", {}).items():
path = entry.get("path", "")
if path in ignored:
name = os.path.basename(path)
for candidate in (name, name.replace(" ", ".")):
indexed[candidate] = (sha1, path)
rows: dict[str, list[tuple[int, str]]] = {section: [] for section in SECTIONS}
for name, size in assets.items():
section, description, previous_sha1 = known.get(name, ("", "", ""))
if name in indexed:
sha1, path = indexed[name]
section = section if section in SECTIONS[:-1] else _section_for(path)
stem = name.split(".")[0] + "." + name.split(".")[1] if name.count(".") > 1 else name
description = description or descriptions.get(name) or descriptions.get(stem) or path
elif name in bundles:
sha1 = cache_sha1.get(name, previous_sha1)
section = section if section in SECTIONS[:-1] else "Game engine data"
description = description or bundles[name]
else:
sha1 = ""
section = NOT_INDEXED
description = description or name
link = f"[{name}]({ASSET_URL}{name})"
if section == NOT_INDEXED:
line = f"| {link} | {description} | {_mib(size)} |"
else:
line = f"| {link} | {description} | {_mib(size)} | `{sha1}` |"
rows[section].append((-size, line))
out = [
"Files too large for the git repository (GitHub 100MB limit), plus game engine data packs.",
"Downloaded automatically by `generate_pack.py` when not present locally.",
"Manual: `gh release download large-files`",
]
for section in SECTIONS:
if not rows[section]:
continue
out += ["", f"## {section}", ""]
if section == NOT_INDEXED:
out += [
"Assets the collection does not index: no database entry names them, "
"so no hash is vouched for here.",
"",
"| File | Description | Size |",
"|------|-------------|-----:|",
]
else:
out += ["| File | Description | Size | SHA1 |", "|------|-------------|-----:|------|"]
out += [line for _, line in sorted(rows[section])]
total = sum(assets.values())
out += ["", f"{len(assets)} files, {total / 1073741824:.1f} GB total. Verify a download with `sha1sum <file>`.", ""]
return "\n".join(out)
def profile_descriptions(emulators_dir: str = "emulators") -> dict[str, str]:
"""Name -> description, from the first profile that declares the file."""
from common import load_emulator_profiles
found: dict[str, str] = {}
for profile in load_emulator_profiles(emulators_dir).values():
for entry in profile.get("files", []):
name = str(entry.get("name") or "")
if name and entry.get("description"):
found.setdefault(name, str(entry["description"]))
return found
def registry_bundles(registry_path: str = "platforms/_data_dirs.yml") -> dict[str, str]:
"""Asset name -> description for data directories served from the release."""
from common import yaml_load
with open(registry_path, encoding="utf-8") as handle:
registry = yaml_load(handle) or {}
bundles: dict[str, str] = {}
for entry in (registry.get("data_directories") or registry).values():
if not isinstance(entry, dict):
continue
url = str(entry.get("source_url") or "")
if url.startswith(ASSET_URL):
bundles[url[len(ASSET_URL):]] = str(entry.get("description") or "")
return bundles
def cache_hashes(names: set[str], cache_dir: str = ".cache/large") -> dict[str, str]:
"""SHA1 of the cached copy of each named asset that the cache holds."""
hashes: dict[str, str] = {}
for name in names:
path = os.path.join(cache_dir, name)
if os.path.isfile(path):
digest = hashlib.sha1()
with open(path, "rb") as handle:
for chunk in iter(lambda: handle.read(1 << 20), b""):
digest.update(chunk)
hashes[name] = digest.hexdigest()
return hashes
def fetch_release() -> tuple[dict[str, int], str]:
"""Name and size of every asset of the large-files release, and its page."""
result = subprocess.run(
[
"gh", "release", "view", LARGE_FILES_RELEASE,
"-R", LARGE_FILES_REPO, "--json", "assets,body",
],
capture_output=True, text=True, check=True,
)
data = json.loads(result.stdout)
return {a["name"]: a["size"] for a in data["assets"]}, data.get("body", "")
def fetch_assets() -> dict[str, int]:
"""Name and size of every asset of the large-files release."""
result = subprocess.run(
[
"gh", "release", "view", LARGE_FILES_RELEASE,
"-R", LARGE_FILES_REPO, "--json", "assets",
],
capture_output=True, text=True, check=True,
)
return {a["name"]: a["size"] for a in json.loads(result.stdout)["assets"]}
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
parser.add_argument("--db", default="database.json")
parser.add_argument("--gitignore", default=".gitignore")
parser.add_argument("--json", action="store_true")
parser.add_argument(
"--notes", metavar="FILE",
help="write the release page rendered from the collection to FILE",
)
args = parser.parse_args()
with open(args.gitignore, encoding="utf-8") as handle:
gitignore_text = handle.read()
db = load_database(args.db)
expected = expected_assets(db, gitignore_text)
try:
assets, body = fetch_release()
except (subprocess.CalledProcessError, FileNotFoundError, KeyError) as exc:
print(f"ERROR: cannot list release assets: {exc}", file=sys.stderr)
return 2
bundles = registry_bundles()
rendered = render_notes(
db, gitignore_text, assets, body, profile_descriptions(), bundles,
cache_hashes(set(bundles)),
)
if args.notes:
with open(args.notes, "w", encoding="utf-8") as handle:
handle.write(rendered)
findings = compare(expected, assets, notes_current=rendered.strip() == body.strip())
if args.json:
print(json.dumps(
[
{"kind": kind, "path": path, "expected": size, "published": published}
for kind, path, size, published in findings
],
indent=2,
))
else:
print(f"{len(expected)} gitignored files, {len(assets)} release assets")
for kind, path, size, published in findings:
if kind == "missing":
print(f" MISSING {path} ({size} bytes)")
elif kind == "size":
print(f" SIZE {path}: local {size} != release {published}")
else:
print(" NOTES release description differs from the rendered page"
" (--notes FILE, then gh release edit large-files --notes-file FILE)")
if not findings:
print(" release assets and description match the collection")
return 1 if findings else 0
if __name__ == "__main__":
sys.exit(main())
+6 -1
View File
@@ -230,7 +230,12 @@ def deduplicate(bios_dir: str, dry_run: bool = False) -> dict:
# Write MAME clone mapping
if mame_clones:
clone_path = "_mame_clones.json"
# Beside the scanned tree, where get_mame_clone_map() reads it for
# the real layout. A path relative to the working directory let a
# scan of a fixture tree rewrite the repository's own map.
clone_path = os.path.join(
os.path.dirname(os.path.abspath(bios_dir)), "_mame_clones.json"
)
# A group is only visible while both copies are on disk, and this run
# has just deleted the clone. Writing only what was seen this time
# therefore erases every mapping an earlier run recorded, and the
+11 -8
View File
@@ -2707,13 +2707,16 @@ def _load_gitignore_entries(repo_root: str) -> set[str]:
return entries
def _is_large_file(local_path: str, repo_root: str) -> bool:
"""Check if a file is a large file (>50MB or in .gitignore)."""
if local_path and os.path.exists(local_path):
if os.path.getsize(local_path) > 50_000_000:
return True
def _is_release_asset(local_path: str, repo_root: str) -> bool:
"""Whether the installer fetches this file from the large-files release.
The repository serves every committed file from its raw URL; only the
files kept out of git live as release assets, and .gitignore is the
ledger of those. Size is not the criterion: a committed file over 50 MB
is still served by the repository, and announcing it as a release asset
sends the installer to an asset nobody uploaded.
"""
gitignore = _load_gitignore_entries(repo_root)
# Check if the path relative to repo root is in .gitignore
try:
rel = os.path.relpath(local_path, repo_root)
except ValueError:
@@ -2826,7 +2829,7 @@ def _manifest_core_entries(
"cores": [source_emu] if source_emu else [],
}
if _is_large_file(local_path or "", repo_root):
if _is_release_asset(local_path or "", repo_root):
entry["storage"] = "release"
entry["release_asset"] = (
os.path.basename(local_path) if local_path else fe["name"]
@@ -3021,7 +3024,7 @@ def generate_manifest(
sha256 = hashes["sha256"]
repo_path = _get_repo_path(sha1, db) if sha1 else ""
is_release_asset = _is_large_file(local_path or "", repo_root)
is_release_asset = _is_release_asset(local_path or "", repo_root)
# An entry needs somewhere to be fetched from. Resolution can
# land on a file the database does not index -- a data
+15
View File
@@ -6,6 +6,7 @@ Steps:
1b. provenance_report.py (dump-catalog coverage from provenance/)
1c. romset_recipes.py (archive identification, reconstruction targets)
2. refresh_data_dirs.py (update Dolphin Sys, PPSSPP, etc.)
2b2. check_release_assets.py (large-files release vs gitignored files, online)
3. verify.py --all (check all platforms)
4. generate_pack.py --all (build ZIP packs)
4b. generate install manifests
@@ -346,6 +347,20 @@ def main():
elif args.check_buildbot:
print("\n--- 2b check buildbot system: SKIPPED (--offline) ---")
# Step 2b2: The release must serve the bytes the manifests describe. The
# installer compares Content-Length with the manifest size, so an asset
# uploaded before its local file was rebuilt fails every install of it.
if not args.offline:
ok, _ = run(
[sys.executable, "scripts/check_release_assets.py"],
"2b2 check release assets",
)
results["check_release_assets"] = ok
all_ok = all_ok and ok
else:
print("\n--- 2b2 check release assets: SKIPPED (--offline) ---")
results["check_release_assets"] = SKIPPED
# Step 2c: Generate truth YAMLs
# A targeted run writes its model in a subdirectory of its own, and the
# diff has to read the same one or it compares a narrowed model against