Files
libretro/scripts/check_release_assets.py
T

337 lines
12 KiB
Python

#!/usr/bin/env python3
"""Compare the large-files release with the files the collection keeps out of git.
Every gitignored file under bios/ is served to the installer as an asset of
the large-files release, and install.py refuses a download whose
Content-Length differs from the manifest size. The manifest size is that of
the local copy, so an asset uploaded before the local file was rebuilt fails
every install of it until it is uploaded again.
Usage:
python scripts/check_release_assets.py
python scripts/check_release_assets.py --json
"""
from __future__ import annotations
import argparse
import hashlib
import json
import os
import re
import subprocess
import sys
sys.path.insert(0, os.path.dirname(__file__))
from common import load_database
from largefiles import LARGE_FILES_RELEASE, LARGE_FILES_REPO
Finding = tuple[str, str, int, int | None]
def expected_assets(db: dict, gitignore_text: str) -> dict[str, int]:
"""Map each gitignored database path to the size the manifests carry."""
ignored = {
line.strip()
for line in gitignore_text.splitlines()
if line.strip().startswith("bios/")
}
return {
entry["path"]: entry["size"]
for entry in db.get("files", {}).values()
if entry.get("path") in ignored
}
def _asset_names(path: str) -> list[str]:
"""Names the release may publish a file under.
GitHub rewrites spaces to dots in asset names, so a file whose name
carries spaces is published under the dotted form.
"""
name = os.path.basename(path)
return [name, name.replace(" ", ".")] if " " in name else [name]
def compare(
expected: dict[str, int], assets: dict[str, int], notes_current: bool = True
) -> list[Finding]:
"""Files the release lacks or serves at another size, sorted by kind.
A release description that no longer matches what render_notes() would
write is a finding of its own: the page is what a reader checks a
download against.
"""
findings: list[Finding] = []
if not notes_current:
findings.append(("notes", "release description", 0, None))
for path, size in sorted(expected.items()):
published = next(
(assets[name] for name in _asset_names(path) if name in assets), None
)
if published is None:
findings.append(("missing", path, size, None))
elif published != size:
findings.append(("size", path, size, published))
return sorted(findings)
ASSET_URL = f"https://github.com/{LARGE_FILES_REPO}/releases/download/{LARGE_FILES_RELEASE}/"
NOT_INDEXED = "Not indexed"
SECTIONS = (
"Console firmware",
"Arcade",
"Computer",
"Game engine data",
"Virtual machine firmware",
"Other",
NOT_INDEXED,
)
_ROW = re.compile(
r"^\| \[(?P<name>[^\]]+)\]\([^)]*\) \| (?P<desc>[^|]*?) \| [^|]* \|"
r"(?: `(?P<sha1>[0-9a-f]{40})` \|)?$",
re.M,
)
_HEADING = re.compile(r"^## (.+)$", re.M)
def parse_notes(body: str) -> dict[str, tuple[str, str, str]]:
"""Name -> (section, description, sha1) for every row of a release page."""
rows: dict[str, tuple[str, str, str]] = {}
section = ""
for line in body.splitlines():
heading = _HEADING.match(line)
if heading:
section = heading.group(1).strip()
continue
row = _ROW.match(line)
if row:
rows[row.group("name")] = (
section, row.group("desc").strip(), row.group("sha1") or ""
)
return rows
def _section_for(path: str) -> str:
parts = path.split("/")
top = parts[1] if len(parts) > 1 else ""
if top in ("Nintendo", "Sony", "Sega", "Microsoft", "NEC", "SNK"):
return "Console firmware"
if top == "Arcade" or "/samples/" in path:
return "Arcade"
if top == "QEMU":
return "Virtual machine firmware"
if top in ("RPG Maker", "ScummVM"):
return "Game engine data"
if top in ("Apple", "Commodore", "Atari", "Sinclair", "Amstrad"):
return "Computer"
return "Other"
def _mib(size: int) -> str:
return f"{round(size / 1048576)} MB"
def render_notes(
db: dict,
gitignore_text: str,
assets: dict[str, int],
previous: str,
descriptions: dict[str, str],
bundles: dict[str, str],
cache_sha1: dict[str, str],
) -> str:
"""The release page, from what the release actually serves.
Every asset gets one row. A file the database indexes carries its
database SHA1; a data-directory bundle carries the hash of its cached
copy when one exists; anything else is listed apart with its size only,
so the page never vouches for bytes nobody indexed. Descriptions and
sections already written by hand are kept by name.
"""
known = parse_notes(previous)
ignored = {
line.strip()
for line in gitignore_text.splitlines()
if line.strip().startswith("bios/")
}
indexed: dict[str, tuple[str, str]] = {}
for sha1, entry in db.get("files", {}).items():
path = entry.get("path", "")
if path in ignored:
name = os.path.basename(path)
for candidate in (name, name.replace(" ", ".")):
indexed[candidate] = (sha1, path)
rows: dict[str, list[tuple[int, str]]] = {section: [] for section in SECTIONS}
for name, size in assets.items():
section, description, previous_sha1 = known.get(name, ("", "", ""))
if name in indexed:
sha1, path = indexed[name]
section = section if section in SECTIONS[:-1] else _section_for(path)
stem = name.split(".")[0] + "." + name.split(".")[1] if name.count(".") > 1 else name
description = description or descriptions.get(name) or descriptions.get(stem) or path
elif name in bundles:
sha1 = cache_sha1.get(name, previous_sha1)
section = section if section in SECTIONS[:-1] else "Game engine data"
description = description or bundles[name]
else:
sha1 = ""
section = NOT_INDEXED
description = description or name
link = f"[{name}]({ASSET_URL}{name})"
if section == NOT_INDEXED:
line = f"| {link} | {description} | {_mib(size)} |"
else:
line = f"| {link} | {description} | {_mib(size)} | `{sha1}` |"
rows[section].append((-size, line))
out = [
"Files too large for the git repository (GitHub 100MB limit), plus game engine data packs.",
"Downloaded automatically by `generate_pack.py` when not present locally.",
"Manual: `gh release download large-files`",
]
for section in SECTIONS:
if not rows[section]:
continue
out += ["", f"## {section}", ""]
if section == NOT_INDEXED:
out += [
"Assets the collection does not index: no database entry names them, "
"so no hash is vouched for here.",
"",
"| File | Description | Size |",
"|------|-------------|-----:|",
]
else:
out += ["| File | Description | Size | SHA1 |", "|------|-------------|-----:|------|"]
out += [line for _, line in sorted(rows[section])]
total = sum(assets.values())
out += ["", f"{len(assets)} files, {total / 1073741824:.1f} GB total. Verify a download with `sha1sum <file>`.", ""]
return "\n".join(out)
def profile_descriptions(emulators_dir: str = "emulators") -> dict[str, str]:
"""Name -> description, from the first profile that declares the file."""
from common import load_emulator_profiles
found: dict[str, str] = {}
for profile in load_emulator_profiles(emulators_dir).values():
for entry in profile.get("files", []):
name = str(entry.get("name") or "")
if name and entry.get("description"):
found.setdefault(name, str(entry["description"]))
return found
def registry_bundles(registry_path: str = "platforms/_data_dirs.yml") -> dict[str, str]:
"""Asset name -> description for data directories served from the release."""
from common import yaml_load
with open(registry_path, encoding="utf-8") as handle:
registry = yaml_load(handle) or {}
bundles: dict[str, str] = {}
for entry in (registry.get("data_directories") or registry).values():
if not isinstance(entry, dict):
continue
url = str(entry.get("source_url") or "")
if url.startswith(ASSET_URL):
bundles[url[len(ASSET_URL):]] = str(entry.get("description") or "")
return bundles
def cache_hashes(names: set[str], cache_dir: str = ".cache/large") -> dict[str, str]:
"""SHA1 of the cached copy of each named asset that the cache holds."""
hashes: dict[str, str] = {}
for name in names:
path = os.path.join(cache_dir, name)
if os.path.isfile(path):
digest = hashlib.sha1()
with open(path, "rb") as handle:
for chunk in iter(lambda: handle.read(1 << 20), b""):
digest.update(chunk)
hashes[name] = digest.hexdigest()
return hashes
def fetch_release() -> tuple[dict[str, int], str]:
"""Name and size of every asset of the large-files release, and its page."""
result = subprocess.run(
[
"gh", "release", "view", LARGE_FILES_RELEASE,
"-R", LARGE_FILES_REPO, "--json", "assets,body",
],
capture_output=True, text=True, check=True,
)
data = json.loads(result.stdout)
return {a["name"]: a["size"] for a in data["assets"]}, data.get("body", "")
def fetch_assets() -> dict[str, int]:
"""Name and size of every asset of the large-files release."""
result = subprocess.run(
[
"gh", "release", "view", LARGE_FILES_RELEASE,
"-R", LARGE_FILES_REPO, "--json", "assets",
],
capture_output=True, text=True, check=True,
)
return {a["name"]: a["size"] for a in json.loads(result.stdout)["assets"]}
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__.split("\n")[0])
parser.add_argument("--db", default="database.json")
parser.add_argument("--gitignore", default=".gitignore")
parser.add_argument("--json", action="store_true")
parser.add_argument(
"--notes", metavar="FILE",
help="write the release page rendered from the collection to FILE",
)
args = parser.parse_args()
with open(args.gitignore, encoding="utf-8") as handle:
gitignore_text = handle.read()
db = load_database(args.db)
expected = expected_assets(db, gitignore_text)
try:
assets, body = fetch_release()
except (subprocess.CalledProcessError, FileNotFoundError, KeyError) as exc:
print(f"ERROR: cannot list release assets: {exc}", file=sys.stderr)
return 2
bundles = registry_bundles()
rendered = render_notes(
db, gitignore_text, assets, body, profile_descriptions(), bundles,
cache_hashes(set(bundles)),
)
if args.notes:
with open(args.notes, "w", encoding="utf-8") as handle:
handle.write(rendered)
findings = compare(expected, assets, notes_current=rendered.strip() == body.strip())
if args.json:
print(json.dumps(
[
{"kind": kind, "path": path, "expected": size, "published": published}
for kind, path, size, published in findings
],
indent=2,
))
else:
print(f"{len(expected)} gitignored files, {len(assets)} release assets")
for kind, path, size, published in findings:
if kind == "missing":
print(f" MISSING {path} ({size} bytes)")
elif kind == "size":
print(f" SIZE {path}: local {size} != release {published}")
else:
print(" NOTES release description differs from the rendered page"
" (--notes FILE, then gh release edit large-files --notes-file FILE)")
if not findings:
print(" release assets and description match the collection")
return 1 if findings else 0
if __name__ == "__main__":
sys.exit(main())