fix: name homonymous release assets by path

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
Abdessamad DerrazandClaude Opus 5.5 committed 2026-10-05 02:51:13 +02:00
1 parent 5816562bf6
commit 0a9d36cae3
5 files changed
+297 -27

No files matched your search

+31 -23
View File
@@ -21,21 +21,23 @@ import os
import re
import subprocess
import sys
from collections.abc import Iterable
sys.path.insert(0, os.path.dirname(__file__))
from common import load_database
from largefiles import LARGE_FILES_RELEASE, LARGE_FILES_REPO
from largefiles import (
LARGE_FILES_RELEASE,
LARGE_FILES_REPO,
asset_names,
registered_paths,
)
Finding = tuple[str, str, int, int | None]
def expected_assets(db: dict, gitignore_text: str) -> dict[str, int]:
"""Map each gitignored database path to the size the manifests carry."""
ignored = {
line.strip()
for line in gitignore_text.splitlines()
if line.strip().startswith("bios/")
}
ignored = set(registered_paths(gitignore_text))
return {
entry["path"]: entry["size"]
for entry in db.get("files", {}).values()
@@ -43,31 +45,37 @@ def expected_assets(db: dict, gitignore_text: str) -> dict[str, int]:
}
def _asset_names(path: str) -> list[str]:
"""Names the release may publish a file under.
def _spellings(name: str) -> list[str]:
"""Names the release may publish an asset under.
GitHub rewrites spaces to dots in asset names, so a file whose name
carries spaces is published under the dotted form.
"""
name = os.path.basename(path)
return [name, name.replace(" ", ".")] if " " in name else [name]
def compare(
expected: dict[str, int], assets: dict[str, int], notes_current: bool = True
expected: dict[str, int],
assets: dict[str, int],
notes_current: bool = True,
registered: Iterable[str] | None = None,
) -> list[Finding]:
"""Files the release lacks or serves at another size, sorted by kind.
A release description that no longer matches what render_notes() would
write is a finding of its own: the page is what a reader checks a
download against.
*registered* is every path .gitignore lists; an asset name depends on
which other paths share its basename, so it defaults to *expected* only
when the caller has nothing wider. A release description that no longer
matches what render_notes() would write is a finding of its own: the
page is what a reader checks a download against.
"""
findings: list[Finding] = []
if not notes_current:
findings.append(("notes", "release description", 0, None))
names = asset_names([*(registered or ()), *expected])
for path, size in sorted(expected.items()):
published = next(
(assets[name] for name in _asset_names(path) if name in assets), None
(assets[name] for name in _spellings(names[path]) if name in assets),
None,
)
if published is None:
findings.append(("missing", path, size, None))
@@ -150,17 +158,12 @@ def render_notes(
sections already written by hand are kept by name.
"""
known = parse_notes(previous)
ignored = {
line.strip()
for line in gitignore_text.splitlines()
if line.strip().startswith("bios/")
}
names = asset_names(registered_paths(gitignore_text))
indexed: dict[str, tuple[str, str]] = {}
for sha1, entry in db.get("files", {}).items():
path = entry.get("path", "")
if path in ignored:
name = os.path.basename(path)
for candidate in (name, name.replace(" ", ".")):
if path in names:
for candidate in _spellings(names[path]):
indexed[candidate] = (sha1, path)
rows: dict[str, list[tuple[int, str]]] = {section: [] for section in SECTIONS}
@@ -307,7 +310,12 @@ def main() -> int:
if args.notes:
with open(args.notes, "w", encoding="utf-8") as handle:
handle.write(rendered)
findings = compare(expected, assets, notes_current=rendered.strip() == body.strip())
findings = compare(
expected,
assets,
notes_current=rendered.strip() == body.strip(),
registered=registered_paths(gitignore_text),
)
if args.json:
print(json.dumps(
+1 -1
View File
@@ -307,7 +307,7 @@ def _preserve_large_file_entries(files: dict, db_path: str) -> int:
if path not in large_files.values() and name not in large_files:
continue
cached = fetch_large_file(
name,
path if path in large_files.values() else name,
expected_sha1=entry.get("sha1", ""),
expected_md5=entry.get("md5", ""),
)
+108 -1
View File
@@ -7,9 +7,14 @@ from __future__ import annotations
import contextlib
import os
import re
import tempfile
import urllib.error
import urllib.parse
import urllib.request
from collections import Counter
from collections.abc import Iterable
from pathlib import Path
from hashing import compute_hashes
@@ -17,6 +22,83 @@ from hashing import compute_hashes
LARGE_FILES_RELEASE = "large-files"
LARGE_FILES_REPO = "Abdess/retrobios"
LARGE_FILES_CACHE = ".cache/large"
GITIGNORE = Path(__file__).resolve().parent.parent / ".gitignore"
_UNSAFE_ASSET_CHARS = re.compile(r"[^A-Za-z0-9._-]+")
def registered_paths(gitignore_text: str) -> list[str]:
"""The bios/ paths .gitignore lists, which are the release assets."""
return [
line.strip()
for line in gitignore_text.splitlines()
if line.strip().startswith("bios/")
]
def load_registered_paths(gitignore: str | Path = GITIGNORE) -> list[str]:
try:
return registered_paths(Path(gitignore).read_text(encoding="utf-8"))
except FileNotFoundError:
return []
def asset_names(registered: Iterable[str]) -> dict[str, str]:
"""Map each registered path to the name of its release asset.
A path whose basename no other registered path shares is published
under that basename. A path whose basename is shared is published under
its location below bios/, segments joined by "--" and every character
outside [A-Za-z0-9._-] replaced by "_":
bios/Id Software/Wolfenstein Enemy Territory/etmain/pak0.pk3 becomes
Id_Software--Wolfenstein_Enemy_Territory--etmain--pak0.pk3. The name
depends on the path only, never on the bytes, so rebuilding a file
keeps its asset.
"""
paths = sorted(set(registered))
counts = Counter(os.path.basename(p) for p in paths)
names: dict[str, str] = {}
for registered_path in paths:
base = os.path.basename(registered_path)
if counts[base] == 1:
names[registered_path] = base
continue
rel = registered_path.removeprefix("bios/")
names[registered_path] = "--".join(
_UNSAFE_ASSET_CHARS.sub("_", part) for part in rel.split("/")
)
owners: dict[str, str] = {}
for registered_path, name in names.items():
# GitHub publishes a space as a dot, so both spellings are taken.
for spelling in {name, name.replace(" ", ".")}:
other = owners.setdefault(spelling, registered_path)
if other != registered_path:
raise ValueError(
f"release asset {spelling!r} named by both {other!r} "
f"and {registered_path!r}"
)
return names
def asset_name(path: str, registered: Iterable[str]) -> str:
"""The release asset name of *path* among the *registered* paths."""
return asset_names([*registered, path])[path]
def asset_candidates(name: str, registered: Iterable[str]) -> list[str]:
"""Assets that may hold *name*, a registered path or a bare file name.
A bare name shared by several registered paths has one asset per path;
the caller's hash tells them apart.
"""
registered = list(registered)
if "/" in name:
return [asset_name(name, registered)]
names = asset_names(registered)
matches = [
names[p] for p in sorted(names) if os.path.basename(p) == name
]
return matches or [name]
def fetch_large_file(
@@ -26,8 +108,33 @@ def fetch_large_file(
expected_md5: str = "",
*,
offline: bool = False,
registered: Iterable[str] | None = None,
) -> str | None:
"""Return a verified cached large file, downloading it only when allowed.
*name* is a registered bios/ path or a bare file name; asset_names()
turns it into the asset to fetch, and a bare name shared by several
registered paths tries each of their assets until one verifies.
"""
if registered is None:
registered = load_registered_paths()
for asset in asset_candidates(name, registered):
cached = _fetch_asset(
asset, dest_dir, expected_sha1, expected_md5, offline=offline
)
if cached:
return cached
return None
def _fetch_asset(
name: str,
dest_dir: str,
expected_sha1: str,
expected_md5: str,
*,
offline: bool,
) -> str | None:
"""Return a verified cached large file, downloading it only when allowed."""
cached = os.path.join(dest_dir, name)
# Between the existence test and the hash, a concurrent run can drop the
# same stale entry: the file is gone by the time this one reads it, and