mirror of
https://github.com/Abdess/retroarch_system.git
synced 2026-10-11 14:03:23 -05:00
108 lines
3.7 KiB
Python
108 lines
3.7 KiB
Python
"""Dump catalogues joined onto the collection.
|
|
|
|
Redump, No-Intro and TOSEC annotate what the repo holds. They are an
|
|
opinion on provenance, never an authority: the emulator source is."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import os
|
|
from pathlib import Path
|
|
|
|
from artifacts import write_if_changed
|
|
|
|
|
|
DEFAULT_PROVENANCE_DIR = "provenance"
|
|
|
|
def load_provenance_snapshots(provenance_dir: str = DEFAULT_PROVENANCE_DIR) -> dict:
|
|
"""Load dump-catalog snapshots from provenance/*.json.
|
|
|
|
Returns {source_name: snapshot} where snapshot holds the normalized
|
|
entries written by the redump scraper or the pack importer. Missing
|
|
directory means no snapshots: returns an empty dict.
|
|
"""
|
|
snapshots = {}
|
|
prov_path = Path(provenance_dir)
|
|
if not prov_path.is_dir():
|
|
return snapshots
|
|
for path in sorted(prov_path.glob("*.json")):
|
|
with open(path) as f:
|
|
snapshot = json.load(f)
|
|
source = snapshot.get("source")
|
|
if source and snapshot.get("entries"):
|
|
snapshots[source] = snapshot
|
|
return snapshots
|
|
|
|
def build_provenance_index(snapshots: dict) -> dict:
|
|
"""Index snapshot entries by sha1 and by (md5, size) per source.
|
|
|
|
First entry wins on hash collisions within a source; entries are
|
|
pre-sorted at snapshot write time so the outcome is deterministic.
|
|
"""
|
|
index = {}
|
|
for source, snapshot in snapshots.items():
|
|
by_sha1 = {}
|
|
by_md5_size = {}
|
|
for entry in snapshot["entries"]:
|
|
sha1 = entry.get("sha1", "")
|
|
md5 = entry.get("md5", "")
|
|
if sha1 and sha1 not in by_sha1:
|
|
by_sha1[sha1] = entry
|
|
if md5 and entry.get("size"):
|
|
key = (md5, entry["size"])
|
|
if key not in by_md5_size:
|
|
by_md5_size[key] = entry
|
|
index[source] = {"by_sha1": by_sha1, "by_md5_size": by_md5_size}
|
|
return index
|
|
|
|
def annotate_provenance(files: dict, snapshots: dict) -> dict[str, int]:
|
|
"""Attach a provenance field to database file entries.
|
|
|
|
Matches by SHA1 first, then MD5 + size. Returns per-source match
|
|
counts. Files without any catalog match keep no provenance field.
|
|
"""
|
|
index = build_provenance_index(snapshots)
|
|
counts = dict.fromkeys(index, 0)
|
|
for sha1, entry in files.items():
|
|
matches = {}
|
|
for source in sorted(index):
|
|
src_index = index[source]
|
|
hit = src_index["by_sha1"].get(sha1) or src_index["by_md5_size"].get(
|
|
(entry.get("md5", ""), entry.get("size", 0))
|
|
)
|
|
if hit:
|
|
matches[source] = {
|
|
"dat": hit.get("dat", ""),
|
|
"name": hit.get("name", ""),
|
|
"description": hit.get("description", ""),
|
|
}
|
|
counts[source] += 1
|
|
if matches:
|
|
entry["provenance"] = matches
|
|
else:
|
|
entry.pop("provenance", None)
|
|
return counts
|
|
|
|
def build_snapshot(
|
|
source: str, imported_at: str, dats: dict, entries: list[dict]
|
|
) -> dict:
|
|
"""A provenance snapshot, sorted for determinism."""
|
|
return {
|
|
"source": source,
|
|
"imported_at": imported_at,
|
|
"dats": dict(sorted(dats.items())),
|
|
"entries": sorted(entries, key=lambda e: (e["dat"], e["name"])),
|
|
}
|
|
|
|
|
|
def write_snapshot(path: str, snapshot: dict) -> bool:
|
|
"""Write a snapshot; timestamps alone never count as a change."""
|
|
return write_if_changed(path, json.dumps(snapshot, indent=2) + "\n")
|
|
|
|
|
|
def write_provenance_snapshot(
|
|
path: str, source: str, imported_at: str, dats: dict, entries: list[dict]
|
|
) -> bool:
|
|
"""Write a normalized provenance snapshot."""
|
|
return write_snapshot(path, build_snapshot(source, imported_at, dats, entries))
|