refactor: give common.py's parts their own modules

common.py had grown to 1833 lines by accumulation. Six coherent pieces
move out - untrusted parsing, digests, archives, generated artefacts,
release assets, dump catalogues - and common.py re-exports them, so the
sixty existing import sites keep working and migrating them stays
optional.

The site build is now reproducible, which is what made the move
checkable. It deleted its generated directories first, so every page was
new and write_if_changed had no earlier version to compare against: a
deploy republished six hundred pages for the clock alone. Directories
are swept instead, a page is removed only once nothing produces it, and
the body pass compares against the body of the file on disk rather than
against the decorated page. Two consecutive builds on the same inputs
now produce identical bytes; before, 1034 files differed.
This commit is contained in:
Abdessamad Derraz committed 2026-08-12 11:49:54 +02:00
1 parent 5e168b86c8
commit b5fd643ccf
10 files changed
+783 -589

No files matched your search

+96
View File
@@ -0,0 +1,96 @@
"""Dump catalogues joined onto the collection.
Redump, No-Intro and TOSEC annotate what the repo holds. They are an
opinion on provenance, never an authority: the emulator source is."""
from __future__ import annotations
import json
import os
from pathlib import Path
from artifacts import write_if_changed
DEFAULT_PROVENANCE_DIR = "provenance"
def load_provenance_snapshots(provenance_dir: str = DEFAULT_PROVENANCE_DIR) -> dict:
"""Load dump-catalog snapshots from provenance/*.json.
Returns {source_name: snapshot} where snapshot holds the normalized
entries written by the redump scraper or the pack importer. Missing
directory means no snapshots: returns an empty dict.
"""
snapshots = {}
prov_path = Path(provenance_dir)
if not prov_path.is_dir():
return snapshots
for path in sorted(prov_path.glob("*.json")):
with open(path) as f:
snapshot = json.load(f)
source = snapshot.get("source")
if source and snapshot.get("entries"):
snapshots[source] = snapshot
return snapshots
def build_provenance_index(snapshots: dict) -> dict:
"""Index snapshot entries by sha1 and by (md5, size) per source.
First entry wins on hash collisions within a source; entries are
pre-sorted at snapshot write time so the outcome is deterministic.
"""
index = {}
for source, snapshot in snapshots.items():
by_sha1 = {}
by_md5_size = {}
for entry in snapshot["entries"]:
sha1 = entry.get("sha1", "")
md5 = entry.get("md5", "")
if sha1 and sha1 not in by_sha1:
by_sha1[sha1] = entry
if md5 and entry.get("size"):
key = (md5, entry["size"])
if key not in by_md5_size:
by_md5_size[key] = entry
index[source] = {"by_sha1": by_sha1, "by_md5_size": by_md5_size}
return index
def annotate_provenance(files: dict, snapshots: dict) -> dict[str, int]:
"""Attach a provenance field to database file entries.
Matches by SHA1 first, then MD5 + size. Returns per-source match
counts. Files without any catalog match keep no provenance field.
"""
index = build_provenance_index(snapshots)
counts = dict.fromkeys(index, 0)
for sha1, entry in files.items():
matches = {}
for source in sorted(index):
src_index = index[source]
hit = src_index["by_sha1"].get(sha1) or src_index["by_md5_size"].get(
(entry.get("md5", ""), entry.get("size", 0))
)
if hit:
matches[source] = {
"dat": hit.get("dat", ""),
"name": hit.get("name", ""),
"description": hit.get("description", ""),
}
counts[source] += 1
if matches:
entry["provenance"] = matches
else:
entry.pop("provenance", None)
return counts
def write_provenance_snapshot(
path: str, source: str, imported_at: str, dats: dict, entries: list[dict]
) -> bool:
"""Write a normalized provenance snapshot, sorted for determinism."""
snapshot = {
"source": source,
"imported_at": imported_at,
"dats": dict(sorted(dats.items())),
"entries": sorted(entries, key=lambda e: (e["dat"], e["name"])),
}
return write_if_changed(path, json.dumps(snapshot, indent=2) + "\n")