fix: state one catalog ratio, correct the faq

The home page and the stats export counted every file carrying a
provenance record, the provenance page and the README only the system
files. The site published 553 and 566 for the same quantity, one click
apart, and the export paired the wider count with composition.systems as
its denominator. common.count_catalog_matched is now the single source,
scoped to the systems bucket.

The FAQ had drifted from the profiles it describes: MAME pinned at 0.287
against 0.289 in mame.yml, Adler-32 attributed to Dolphin's IPL rather
than the DSP ROMs that carry known_hash_adler32, and the per-emulator
verbose report named as the only content check on an existence platform,
which skips the DISCREPANCY line the platform report raises itself.

Tests read both sides: no generator may count matches inline, and each
FAQ claim is checked against the profile or the script that owns it.
This commit is contained in:
Abdessamad Derraz committed 2026-09-04 11:41:17 +02:00
1 parent 022888e9ea
commit fe77535c3b
5 files changed
+184 -52

No files matched your search

+27 -8
View File
@@ -950,6 +950,17 @@ def unique_emulator_profiles(profiles: dict[str, dict]) -> dict[str, dict]:
GAME_DATA_TOPS = ("RPG Maker", "ScummVM")
def composition_tier(path: str) -> str:
"""Which composition bucket a repository path belongs to."""
parts = path.split("/")
top = parts[1] if len(parts) > 1 else ""
if top == "Arcade":
return "arcade"
if top in GAME_DATA_TOPS:
return "game_data"
return "systems"
def compute_composition(db: dict) -> dict:
"""File and byte counts by tree area.
@@ -963,19 +974,27 @@ def compute_composition(db: dict) -> dict:
"game_data": {"files": 0, "size_bytes": 0},
}
for entry in db.get("files", {}).values():
parts = entry.get("path", "").split("/")
top = parts[1] if len(parts) > 1 else ""
if top == "Arcade":
bucket = buckets["arcade"]
elif top in GAME_DATA_TOPS:
bucket = buckets["game_data"]
else:
bucket = buckets["systems"]
bucket = buckets[composition_tier(entry.get("path", ""))]
bucket["files"] += 1
bucket["size_bytes"] += entry.get("size", 0)
return buckets
def count_catalog_matched(db: dict) -> int:
"""System files byte-identical to a dump-preservation catalog entry.
Scoped to the systems bucket, the denominator every surface pairs it
with: No-Intro, Redump and TOSEC index console and computer dumps, so
arcade ROM sets and engine data sit on neither side of the ratio.
"""
return sum(
1
for entry in db.get("files", {}).values()
if entry.get("provenance")
and composition_tier(entry.get("path", "")) == "systems"
)
def group_identical_platforms(
platforms: list[str],
platforms_dir: str,
+6 -18
View File
@@ -18,8 +18,8 @@ from datetime import datetime, timezone
sys.path.insert(0, os.path.dirname(__file__))
from common import (
GAME_DATA_TOPS,
compute_composition,
count_catalog_matched,
list_registered_platforms,
load_database,
load_emulator_profiles,
@@ -184,25 +184,13 @@ _CATALOG_LABELS = {"redump": "Redump", "no-intro": "No-Intro", "tosec": "TOSEC"}
def _catalog_matched_line(db: dict) -> list[str]:
"""Bullet line for files matched to dump-preservation catalogs.
Counted against the system files alone: those catalogs index console and
computer dumps, so arcade ROM sets and engine data sit outside their
scope and would only dilute the ratio.
"""
sources: set[str] = set()
matched = 0
for entry in db.get("files", {}).values():
provenance = entry.get("provenance")
if not provenance:
continue
sources.update(provenance)
parts = entry.get("path", "").split("/")
top = parts[1] if len(parts) > 1 else ""
if top != "Arcade" and top not in GAME_DATA_TOPS:
matched += 1
"""Bullet line for files matched to dump-preservation catalogs."""
matched = count_catalog_matched(db)
if not matched:
return []
sources: set[str] = set()
for entry in db.get("files", {}).values():
sources.update(entry.get("provenance") or ())
system_files = compute_composition(db)["systems"]["files"]
labels = ", ".join(_CATALOG_LABELS.get(s, s) for s in sorted(sources))
return [
+9 -21
View File
@@ -32,7 +32,7 @@ from pathlib import Path
sys.path.insert(0, os.path.dirname(__file__))
from common import (
compute_composition,
GAME_DATA_TOPS,
count_catalog_matched,
list_registered_platforms,
load_database,
load_emulator_profiles,
@@ -296,16 +296,16 @@ def generate_home(
]
)
catalog_matched = sum(
1 for f in db.get("files", {}).values() if f.get("provenance")
)
catalog_matched = count_catalog_matched(db)
if catalog_matched:
system_files = compute_composition(db)["systems"]["files"]
lines.extend(
[
"",
f"**{catalog_matched:,}** files are byte-identical to a dump "
"catalogued by No-Intro, Redump, or TOSEC, and say so on their "
"system page. [What that means](provenance.md).",
f"**{catalog_matched:,}** of {system_files:,} system files are "
"byte-identical to a dump catalogued by No-Intro, Redump, or "
"TOSEC, and say so on their system page. "
"[What that means](provenance.md).",
]
)
@@ -404,9 +404,7 @@ def compute_stats(db: dict, coverages: dict, profiles: dict) -> dict:
"platforms": len(coverages),
"emulators": len(unique),
"systems": len(systems),
"catalog_matched": sum(
1 for f in db.get("files", {}).values() if f.get("provenance")
),
"catalog_matched": count_catalog_matched(db),
"source": REPO_URL,
"downloads": RELEASE_URL,
}
@@ -1504,17 +1502,7 @@ def _prov_title(data: dict) -> str:
def generate_provenance_page(db: dict, report: dict) -> str:
"""Page explaining the verified dump badges and listing catalog gaps."""
# Scoped to system files: these catalogs index console and computer dumps,
# so arcade ROM sets and engine data can never match and would only make
# the ratio look worse than the work behind it.
matched_files = 0
for entry in db.get("files", {}).values():
if not entry.get("provenance"):
continue
parts = entry.get("path", "").split("/")
top = parts[1] if len(parts) > 1 else ""
if top != "Arcade" and top not in GAME_DATA_TOPS:
matched_files += 1
matched_files = count_catalog_matched(db)
total_files = compute_composition(db)["systems"]["files"]
lines = [