mirror of
https://github.com/Abdess/retroarch_system.git
synced 2026-10-11 05:53:23 -05:00
179 lines
6.6 KiB
Python
179 lines
6.6 KiB
Python
#!/usr/bin/env python3
|
|
"""Report dump-catalog coverage of the collection.
|
|
|
|
Reads provenance/*.json snapshots and database.json, then reports per
|
|
source how many catalog entries the collection holds and which are
|
|
missing. Missing entries are acquisition targets: catalog-verified
|
|
dumps the collection does not have yet.
|
|
|
|
Usage:
|
|
python scripts/provenance_report.py
|
|
python scripts/provenance_report.py --missing
|
|
python scripts/provenance_report.py --json
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import hashlib
|
|
import json
|
|
import os
|
|
import sys
|
|
import zipfile
|
|
|
|
sys.path.insert(0, os.path.dirname(__file__))
|
|
from common import DEFAULT_PROVENANCE_DIR, load_database, load_provenance_snapshots
|
|
|
|
|
|
def archive_members(db: dict) -> dict[tuple[str, int], list[tuple[str, str]]]:
|
|
"""(crc32, size) of every member of the collection's ZIPs -> (zip, member).
|
|
|
|
Read from the central directories alone: a romset member is a dump the
|
|
collection already holds, and listing it as an acquisition target sent
|
|
searches after astrocdw.zip's bioswhit.bin and the Gamate BIOS.
|
|
"""
|
|
members: dict[tuple[str, int], list[tuple[str, str]]] = {}
|
|
for entry in db.get("files", {}).values():
|
|
path = entry.get("path", "")
|
|
if not path.endswith(".zip") or not os.path.exists(path):
|
|
continue
|
|
try:
|
|
with zipfile.ZipFile(path) as archive:
|
|
for info in archive.infolist():
|
|
if info.is_dir():
|
|
continue
|
|
key = (f"{info.CRC:08x}", info.file_size)
|
|
members.setdefault(key, []).append((path, info.filename))
|
|
except (zipfile.BadZipFile, OSError) as exc:
|
|
print(f" WARNING: {path}: {exc}", file=sys.stderr)
|
|
return members
|
|
|
|
|
|
def _held_in_archive(entry: dict, members: dict) -> bool:
|
|
"""Whether a catalog entry is a member of one of the collection's ZIPs.
|
|
|
|
crc32 and size only nominate candidates; a declared sha1 must match the
|
|
member's bytes.
|
|
"""
|
|
crc = str(entry.get("crc32") or "").lower()
|
|
size = entry.get("size")
|
|
if not crc or size is None:
|
|
return False
|
|
candidates = members.get((crc.zfill(8), int(size)), [])
|
|
sha1 = str(entry.get("sha1") or "").lower()
|
|
if not sha1:
|
|
return bool(candidates)
|
|
for path, name in candidates:
|
|
try:
|
|
with zipfile.ZipFile(path) as archive:
|
|
if hashlib.sha1(archive.read(name)).hexdigest() == sha1:
|
|
return True
|
|
except (zipfile.BadZipFile, OSError, KeyError) as exc:
|
|
print(f" WARNING: {path}:{name}: {exc}", file=sys.stderr)
|
|
return False
|
|
|
|
|
|
def build_report(db: dict, snapshots: dict, members: dict | None = None) -> dict:
|
|
"""Compare each snapshot against the collection.
|
|
|
|
A DAT counts as covered when the collection holds at least one of
|
|
its entries. Missing entries from covered DATs are acquisition
|
|
targets; missing entries from DATs the collection does not cover at
|
|
all are out of scope and only counted. No-Intro tags every non-game
|
|
dump "[BIOS]", including digital title distribution, so without this
|
|
split the target list is swamped by content the project never ships.
|
|
"""
|
|
by_sha1 = db.get("files", {})
|
|
members = members or {}
|
|
by_md5_size = {
|
|
(entry.get("md5", ""), entry.get("size", 0)) for entry in by_sha1.values()
|
|
}
|
|
# Redump lists some dumps by crc32 alone (ps2-0101jd-20030110, DTL-H10100):
|
|
# collected as a loose file it would stay MISSING for good without this.
|
|
by_crc32_size = {
|
|
(str(entry.get("crc32", "")).lower(), entry.get("size", 0))
|
|
for entry in by_sha1.values()
|
|
if entry.get("crc32")
|
|
}
|
|
|
|
report = {}
|
|
for source, snapshot in sorted(snapshots.items()):
|
|
matched = 0
|
|
covered_dats = set()
|
|
unmatched = []
|
|
for entry in snapshot["entries"]:
|
|
if (
|
|
entry.get("sha1") in by_sha1
|
|
or (entry.get("md5"), entry.get("size")) in by_md5_size
|
|
or (str(entry.get("crc32") or "").lower(), entry.get("size")) in by_crc32_size
|
|
or _held_in_archive(entry, members)
|
|
):
|
|
matched += 1
|
|
covered_dats.add(entry.get("dat", ""))
|
|
else:
|
|
unmatched.append(entry)
|
|
missing = [e for e in unmatched if e.get("dat", "") in covered_dats]
|
|
out_of_scope = len(unmatched) - len(missing)
|
|
report[source] = {
|
|
"imported_at": snapshot.get("imported_at", ""),
|
|
"dats": snapshot.get("dats", {}),
|
|
"total": len(snapshot["entries"]),
|
|
"matched": matched,
|
|
"covered_dats": sorted(covered_dats),
|
|
"missing": missing,
|
|
"out_of_scope": out_of_scope,
|
|
}
|
|
return report
|
|
|
|
|
|
def main() -> int:
|
|
parser = argparse.ArgumentParser(description="Dump-catalog coverage report")
|
|
parser.add_argument("--db", default="database.json", help="Database path")
|
|
parser.add_argument(
|
|
"--provenance-dir",
|
|
default=DEFAULT_PROVENANCE_DIR,
|
|
help="Directory with dump-catalog snapshots",
|
|
)
|
|
parser.add_argument("--missing", action="store_true", help="List missing entries")
|
|
parser.add_argument("--json", action="store_true", help="Full report as JSON")
|
|
args = parser.parse_args()
|
|
|
|
db = load_database(args.db)
|
|
snapshots = load_provenance_snapshots(args.provenance_dir)
|
|
if not snapshots:
|
|
print(f"No provenance snapshots in {args.provenance_dir}/")
|
|
return 0
|
|
|
|
report = build_report(db, snapshots, archive_members(db))
|
|
|
|
if args.json:
|
|
print(json.dumps(report, indent=2))
|
|
return 0
|
|
|
|
for source, data in report.items():
|
|
in_scope = data["matched"] + len(data["missing"])
|
|
pct = 100 * data["matched"] / in_scope if in_scope else 0
|
|
print(
|
|
f" {source} ({data['imported_at']}): "
|
|
f"{data['matched']}/{in_scope} in collection ({pct:.0f}%) "
|
|
f"across {len(data['covered_dats'])} covered DATs"
|
|
)
|
|
if data["out_of_scope"]:
|
|
print(
|
|
f" {data['out_of_scope']} entries in DATs the collection "
|
|
f"does not cover, not counted as targets"
|
|
)
|
|
if args.missing:
|
|
for entry in data["missing"]:
|
|
label = entry.get("description") or entry["name"]
|
|
print(f" MISSING {entry['name']} ({label}) sha1={entry.get('sha1')}")
|
|
|
|
total_missing = sum(len(d["missing"]) for d in report.values())
|
|
if total_missing:
|
|
print(f" {total_missing} catalog entries missing from collection")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
sys.exit(main())
|