diff --git a/scripts/exporter/baseline.py b/scripts/exporter/baseline.py index 81d02c4e..cebb25d9 100644 --- a/scripts/exporter/baseline.py +++ b/scripts/exporter/baseline.py @@ -322,13 +322,18 @@ def build_native_model( if candidate.truth is None ] - def by_destination(candidate: NativeFile) -> bool: + def by_destination(candidate: NativeFile, t_dest: str = t_dest) -> bool: # Dolphin writes dolphin-emu/Sys/GC/USA/IPL.bin for the truth's # GC/USA/IPL.bin: a destination ending in the path is the file. theirs = _match_key(candidate.platform or {})[0] return bool(t_dest) and (theirs == t_dest or theirs.endswith("/" + t_dest)) - def by_name(candidate: NativeFile) -> bool: + def by_name( + candidate: NativeFile, + t_dest: str = t_dest, + t_name: str = t_name, + truth_entry: dict = truth_entry, + ) -> bool: theirs_dest, theirs = _match_key(candidate.platform or {}) if not t_name or theirs != t_name: return False @@ -346,7 +351,7 @@ def build_native_model( isinstance(t_size, int) and isinstance(p_size, int) and t_size != p_size ) - def by_hash(candidate: NativeFile) -> bool: + def by_hash(candidate: NativeFile, t_hashes: set = t_hashes) -> bool: if not t_hashes: return False theirs = { diff --git a/scripts/generate_pack.py b/scripts/generate_pack.py index 71c56852..912a7d5b 100644 --- a/scripts/generate_pack.py +++ b/scripts/generate_pack.py @@ -577,6 +577,24 @@ def _select_variants( return region_drops, region_fallbacks, slot_undecidable +def _platform_pack_name( + platform_name: str, + platforms_dir: str, + narrowings: list[tuple[str, str]], + system_filter: list[str] | None, + pack_name: str | None = None, +) -> str: + """File name of a platform pack: stem, narrowing tags, system tag. + + A caller naming the pack itself (a --split part) is taken at its word. + """ + if pack_name: + return pack_name + stem = _platform_pack_stem([platform_name], platform_name, platforms_dir) + tags = "".join(tag for tag, _label in narrowings) + return f"{stem}{tags}_BIOS_Pack{_system_tag(system_filter)}.zip" + + def generate_pack( platform_name: str, platforms_dir: str, @@ -623,10 +641,12 @@ def generate_pack( narrowings = _narrowings( source, regions, target_name, one_per_slot, required_only ) - narrow_tags = "".join(tag for tag, _label in narrowings) - stem = _platform_pack_stem([platform_name], platform_name, platforms_dir) - zip_name = pack_name or f"{stem}{narrow_tags}_BIOS_Pack{_system_tag(system_filter)}.zip" - zip_path = os.path.join(output_dir, zip_name) + zip_path = os.path.join( + output_dir, + _platform_pack_name( + platform_name, platforms_dir, narrowings, system_filter, pack_name + ), + ) os.makedirs(output_dir, exist_ok=True) # Case-insensitive dedup only for platforms targeting Windows/macOS. @@ -2460,6 +2480,13 @@ def _refuse_refresh_data(args, parser) -> None: parser.error(f"--refresh-data is incompatible with {flag}") +def _refuse_include_archived(args, parser) -> None: + """Only --all chooses among registered platforms; every other mode names + its own platform, emulator, system or hash.""" + if args.include_archived and not args.all: + parser.error("--include-archived requires --all") + + def _refuse_unapplied_flags(args, parser) -> None: """Refuse every flag the requested mode would not apply. @@ -2467,10 +2494,7 @@ def _refuse_unapplied_flags(args, parser) -> None: about an artifact the caller did not name. Run before any quick-exit mode, since --verify-packs and --manifest-targets return early. """ - # Only --all chooses among registered platforms; every other mode names - # its own platform, emulator, system or hash. - if args.include_archived and not args.all: - parser.error("--include-archived requires --all") + _refuse_include_archived(args, parser) # Parsed before the quick-exit modes: --verify-packs returns early and # still needs the region priority list to narrow its expectation. args.regions = [] diff --git a/scripts/generate_readme.py b/scripts/generate_readme.py index a77c6e64..2620a3ff 100644 --- a/scripts/generate_readme.py +++ b/scripts/generate_readme.py @@ -147,6 +147,15 @@ def extract_notes(platforms_dir: str) -> list[str]: return notes +def _extract_cells(platforms_dir: str) -> dict[str, str]: + """The download table's "Extract to" cell, by display name.""" + return {display: f"`{folder}`" for display, folder in extract_targets(platforms_dir)} + + +def _paragraphs(texts: list[str]) -> list[str]: + return [line for text in texts for line in ("", text)] + + def download_table( coverages: dict, archived: set[str], @@ -393,9 +402,7 @@ def generate_readme(db: dict, platforms_dir: str) -> str: # Where the pack itself is extracted, which is not always the BIOS folder: # a pack whose entries already carry their own root (RetroDECK) extracts # one level above it. - extract_paths = { - display: f"`{folder}`" for display, folder in extract_targets(platforms_dir) - } + extract_paths = _extract_cells(platforms_dir) archived = { name for name, entry in load_platform_registry(platforms_dir).items() @@ -408,8 +415,7 @@ def generate_readme(db: dict, platforms_dir: str) -> str: ) ) - for note in extract_notes(platforms_dir): - lines.extend(["", note]) + lines.extend(_paragraphs(extract_notes(platforms_dir))) if archived: lines.extend( [ diff --git a/scripts/generate_site.py b/scripts/generate_site.py index 6e495ac9..269053ce 100644 --- a/scripts/generate_site.py +++ b/scripts/generate_site.py @@ -207,6 +207,16 @@ def _timestamp() -> str: # Home page +def _extract_table(platforms_dir: str) -> list[str]: + """The home page's extraction table, indented for its admonition.""" + rows = [ + f" | {display} | `{folder}` |" + for display, folder in extract_targets(platforms_dir) + ] + notes = [f" {note}" for note in extract_notes(platforms_dir)] + return [" | Platform | Extract to |", " |----------|-----------|", *rows, "", *notes] + + def generate_home( db: dict, coverages: dict, @@ -328,14 +338,7 @@ def generate_home( "", '??? info "Where to extract"', "", - " | Platform | Extract to |", - " |----------|-----------|", - *( - f" | {display} | `{folder}` |" - for display, folder in extract_targets(platforms_dir) - ), - "", - *(f" {note}" for note in extract_notes(platforms_dir)), + *_extract_table(platforms_dir), " Every other pack extracts straight into the BIOS folder." " [Full instructions per setup](which-pack.md).", "", @@ -1870,17 +1873,17 @@ def _file_badges(f: dict, in_repo: bool) -> list[str]: badges.append( 'optional' ) - if not in_repo and f.get("unsourceable"): + if in_repo: + badges.append( + 'in repo' + ) + elif f.get("unsourceable"): badges.append( 'unsourceable' ) - elif not in_repo: - badges.append( - 'missing' - ) else: badges.append( - 'in repo' + 'missing' ) if hle: badges.append( @@ -2247,6 +2250,40 @@ def _availability_check(db: dict, data_names): return _file_available +def _emulator_file_summary(files: list[dict], available) -> list[str]: + """The counts above an emulator's file table, and its categories.""" + in_repo_count = sum(1 for f in files if available(f)) + unsourceable_count = sum(1 for f in files if f.get("unsourceable")) + missing_count = len(files) - in_repo_count - unsourceable_count + req_count = sum(1 for f in files if f.get("required")) + hle_count = sum(1 for f in files if f.get("hle_fallback")) + + held = f"{in_repo_count} in repo, {missing_count} missing" + if unsourceable_count: + held += f", {unsourceable_count} unsourceable" + parts = [ + f"**{len(files)} files**", + f"{req_count} required, {len(files) - req_count} optional", + held, + ] + if hle_count: + parts.append(f"{hle_count} with HLE fallback") + lines = [" | ".join(parts)] + + categories = [f.get("category", "bios") for f in files] + if "game_data" in categories or "bios_zip" in categories: + cats = [ + f"{categories.count(key)} {label}" + for key, label in ( + ("bios", "BIOS"), ("game_data", "game data"), ("bios_zip", "BIOS ZIPs") + ) + if categories.count(key) + ] + lines.append(f"Categories: {', '.join(cats)}") + lines.append("") + return lines + + def generate_emulator_page( name: str, profile: dict, @@ -2363,38 +2400,7 @@ def generate_emulator_page( else: _file_available = _availability_check(db, data_names) - # Stats by category - bios_files = [f for f in files if f.get("category", "bios") == "bios"] - game_data = [f for f in files if f.get("category") == "game_data"] - bios_zips = [f for f in files if f.get("category") == "bios_zip"] - - in_repo_count = sum(1 for f in files if _file_available(f)) - unsourceable_count = sum(1 for f in files if f.get("unsourceable")) - missing_count = len(files) - in_repo_count - unsourceable_count - req_count = sum(1 for f in files if f.get("required")) - opt_count = len(files) - req_count - hle_count = sum(1 for f in files if f.get("hle_fallback")) - - parts = [f"**{len(files)} files**"] - parts.append(f"{req_count} required, {opt_count} optional") - held = f"{in_repo_count} in repo, {missing_count} missing" - if unsourceable_count: - held += f", {unsourceable_count} unsourceable" - parts.append(held) - if hle_count: - parts.append(f"{hle_count} with HLE fallback") - lines.append(" | ".join(parts)) - - if game_data or bios_zips: - cats = [] - if bios_files: - cats.append(f"{len(bios_files)} BIOS") - if game_data: - cats.append(f"{len(game_data)} game data") - if bios_zips: - cats.append(f"{len(bios_zips)} BIOS ZIPs") - lines.append(f"Categories: {', '.join(cats)}") - lines.append("") + lines.extend(_emulator_file_summary(files, _file_available)) # File table for f in files: diff --git a/scripts/largefiles.py b/scripts/largefiles.py index 79b8e9ab..d0be38ec 100644 --- a/scripts/largefiles.py +++ b/scripts/largefiles.py @@ -163,7 +163,7 @@ def _fetch_asset( hashes, expected_sha1, expected_md5 ): return cached - elif offline or _served_size(name) in (None, os.path.getsize(cached)): + elif not _replaced_on_release(name, cached, offline): # A copy of this asset that answers another hash: the caller # wants a different revision under the same name, and the # release still serves this one. Keeping it is what lets the @@ -234,6 +234,18 @@ def _asset_urls(name: str) -> list[tuple[str, str]]: ] +def _replaced_on_release(name: str, cached: str, offline: bool) -> bool: + """Whether the release now serves other bytes than the cached copy. + + Sizes are compared, as check_release_assets does: a request per asset, + no download. Offline or unreadable, the copy is assumed current. + """ + if offline: + return False + served = _served_size(name) + return served is not None and served != os.path.getsize(cached) + + def _served_size(name: str) -> int | None: """Size of the asset the release serves now, or None when unreadable.""" for candidate, url in _asset_urls(name): diff --git a/scripts/profile_sync.py b/scripts/profile_sync.py index c50f79b7..b5a87274 100644 --- a/scripts/profile_sync.py +++ b/scripts/profile_sync.py @@ -71,6 +71,11 @@ class RefPart: end: int | None raw: str = "" + @property + def last(self) -> int | None: + """Last cited line: the end of a range, or the single line.""" + return self.end or self.start + @dataclass(frozen=True) class AnchorResult: @@ -910,7 +915,7 @@ def anchor_part( reason = "written against HEAD, pin names an older revision" return PartResult(part, "GONE", None, None, None, [], reason, slug, url) - end = part.end or part.start + end = part.last if end > len(pin_lines): # The file is there and the line is not yet: the pinned revision is # shorter than the one the ref was written against. nestopia cited @@ -1173,7 +1178,7 @@ def verify_at_pin(part: RefPart, pin_lines, tokens, hash_tokens=()) -> PartResul ) if part.start is None: return PartResult(part, "ANCHORED", None, None, None, []) - if (part.end or part.start) > len(pin_lines): + if part.last > len(pin_lines): return PartResult( part, "GONE", None, None, None, [], "beyond the end of the file" ) @@ -1190,7 +1195,7 @@ def verify_at_pin(part: RefPart, pin_lines, tokens, hash_tokens=()) -> PartResul # per line, so the window reaches forward as far as the entry has members. reach = SELF_CHECK_CONTEXT + len(tokens) lo = max(0, part.start - 1 - SELF_CHECK_CONTEXT) - hi = min(len(pin_lines), (part.end or part.start) + reach) + hi = min(len(pin_lines), part.last + reach) window = "\n".join(pin_lines[lo:hi]).lower() if any(token in window for token in tokens): return PartResult(part, "ANCHORED", None, None, None, []) @@ -2823,6 +2828,22 @@ def realign_prose( return messages +def _refuse_dirty_tree(emulators_dir: str) -> None: + """Writes need a clean profile tree to roll back to.""" + try: + dirty = emulators_dir_is_dirty(emulators_dir) + except RuntimeError as e: + print(f"{e}. Pass --force to write anyway.", file=sys.stderr) + raise SystemExit(1) from e + if dirty: + print( + f"{emulators_dir} carries uncommitted changes. " + "Commit them first or pass --force.", + file=sys.stderr, + ) + raise SystemExit(1) + + def emulators_dir_is_dirty(emulators_dir: str) -> bool: """True when the profile directory carries uncommitted changes. @@ -3233,18 +3254,7 @@ def main() -> None: or args.realign_prose ) if writes and not args.dry_run and not args.force: - try: - dirty = emulators_dir_is_dirty(args.emulators_dir) - except RuntimeError as e: - print(f"{e}. Pass --force to write anyway.", file=sys.stderr) - raise SystemExit(1) from e - if dirty: - print( - f"{args.emulators_dir} carries uncommitted changes. " - "Commit them first or pass --force.", - file=sys.stderr, - ) - raise SystemExit(1) + _refuse_dirty_tree(args.emulators_dir) if args.realign_prose: for name in selected: diff --git a/scripts/scraper/libretro_scraper.py b/scripts/scraper/libretro_scraper.py index 9d62f691..e3609677 100644 --- a/scripts/scraper/libretro_scraper.py +++ b/scripts/scraper/libretro_scraper.py @@ -101,6 +101,27 @@ SYSTEM_SLUG_MAP = { } +def _core_info_files() -> list[str]: + """Paths of every .info file in libretro-core-info, or RuntimeError.""" + import json + + url = "https://api.github.com/repos/libretro/libretro-core-info/git/trees/master?recursive=1" + try: + req = urllib.request.Request(url, headers=github_headers()) + with urllib.request.urlopen(req, timeout=30) as resp: + tree = json.loads(resp.read()) + except (urllib.error.URLError, json.JSONDecodeError) as e: + raise RuntimeError(f"cannot list libretro-core-info: {e}") from e + info_files = [ + item["path"] + for item in tree.get("tree", []) + if item["path"].endswith("_libretro.info") + ] + if not info_files: + raise RuntimeError(f"no .info file in {url}") + return info_files + + class Scraper(BaseScraper): """Scraper for libretro System.dat.""" @@ -167,28 +188,10 @@ class Scraper(BaseScraper): def _fetch_core_metadata(self) -> dict[str, dict]: """Fetch per-core metadata from libretro-core-info .info files.""" - import json - from .coreinfo_scraper import CORE_SYSTEM_MAP - url = "https://api.github.com/repos/libretro/libretro-core-info/git/trees/master?recursive=1" - try: - req = urllib.request.Request(url, headers=github_headers()) - with urllib.request.urlopen(req, timeout=30) as resp: - tree = json.loads(resp.read()) - except (urllib.error.URLError, json.JSONDecodeError) as e: - raise RuntimeError(f"cannot list libretro-core-info: {e}") from e - - info_files = [ - item["path"] - for item in tree.get("tree", []) - if item["path"].endswith("_libretro.info") - ] - if not info_files: - raise RuntimeError(f"no .info file in {url}") - metadata: dict[str, dict] = {} - for filename in info_files: + for filename in _core_info_files(): core_name = filename.replace("_libretro.info", "") system_slug = CORE_SYSTEM_MAP.get(core_name) if not system_slug or system_slug in metadata: diff --git a/scripts/truth.py b/scripts/truth.py index 52bf651a..76583741 100644 --- a/scripts/truth.py +++ b/scripts/truth.py @@ -178,11 +178,17 @@ def _same_file(files: list[dict], file_entry: dict, emu_name: str) -> dict | Non ) +def _declared(entry: dict, field: str) -> set[str]: + """The values an entry declares for a hash field, one or a list.""" + value = entry.get(field) + values = value if isinstance(value, list) else [value] + return {str(v).lower() for v in values if v} + + def _contents_disagree(a: dict, b: dict) -> bool: """A hash both declare without a value in common, or two declared sizes.""" for field in ("sha1", "md5", "sha256", "crc32"): - ours = {str(v).lower() for v in (a.get(field) if isinstance(a.get(field), list) else [a.get(field)]) if v} - theirs = {str(v).lower() for v in (b.get(field) if isinstance(b.get(field), list) else [b.get(field)]) if v} + ours, theirs = _declared(a, field), _declared(b, field) if ours and theirs and not ours & theirs: return True size_a, size_b = a.get("size"), b.get("size") @@ -294,6 +300,16 @@ def _has_exploitable_data(entry: dict) -> bool: ) +def _system_dir_files(profile: dict, standalone: bool) -> list[dict]: + """Entries of the build the platform runs, read from the system directory.""" + return list( + filter( + read_from_system_dir, + filter_files_by_mode(profile.get("files", []), standalone=standalone), + ) + ) + + def generate_platform_truth( platform_name: str, config: dict, @@ -372,14 +388,11 @@ def generate_platform_truth( continue cores_profiled.add(emu_name) - filtered = filter_files_by_mode( - profile.get("files", []), - standalone=runs_standalone(emu_name, profile, standalone_set), + filtered = _system_dir_files( + profile, runs_standalone(emu_name, profile, standalone_set) ) for fe in filtered: - if not read_from_system_dir(fe): - continue profile_sid = fe.get("system", "") if not profile_sid: sys_ids = profile.get("systems", [])