fix: keep the database free of stale entries

Two entries claimed bios/Sony/PlayStation 3/PS3UPDAT.PUP with different
SHA1s. Preserving large-file entries matched on path and keyed on
SHA1, so replacing a firmware revision on disk left the old entry
pointing at a path that now serves other bytes. A preserved entry whose
path the scan has already claimed is dropped, and validate_schemas
refuses a database where one path carries two entries.

Separately, a run without --force rebuilt each cached entry from a
hand-written list of four digests and wrote it back without adler32,
so one such run stripped the digest from every file permanently. A
cache entry missing any digest is now a miss.
This commit is contained in:
Abdessamad Derraz committed 2026-08-11 00:55:32 +02:00
1 parent 3b8f2d75d5
commit dc14089932
3 files changed
+153 -11

No files matched your search

+23 -10
View File
@@ -19,12 +19,13 @@ from pathlib import Path
sys.path.insert(0, os.path.dirname(__file__))
from common import (
DEFAULT_PROVENANCE_DIR,
annotate_provenance,
compute_hashes,
DEFAULT_PROVENANCE_DIR,
list_registered_platforms,
load_provenance_snapshots,
write_if_changed,
yaml_load,
)
CACHE_DIR = ".cache"
@@ -34,6 +35,10 @@ DEFAULT_OUTPUT = "database.json"
SKIP_PATTERNS = {".git", ".github", "__pycache__", ".cache", ".DS_Store", "desktop.ini"}
# Every digest database.json publishes per file. The schema requires all five,
# so a cache entry missing any of them cannot serve a database entry.
CACHED_HASHES = frozenset({"sha1", "md5", "sha256", "crc32", "adler32"})
def should_skip(path: Path) -> bool:
"""Check if a path should be skipped. Allows .variants/ directories."""
@@ -80,13 +85,14 @@ def scan_bios_dir(bios_dir: Path, cache: dict, force: bool) -> tuple[dict, dict,
if not force and cache_key in cache:
cached = cache[cache_key]
if cached.get("mtime") == mtime and cached.get("size") == size:
hashes = {
"sha1": cached["sha1"],
"md5": cached["md5"],
"sha256": cached["sha256"],
"crc32": cached["crc32"],
}
fresh = cached.get("mtime") == mtime and cached.get("size") == size
# Rebuilding the hash dict by hand dropped whichever digest the
# list forgot, and the entry was then written back to the cache
# without it, so the loss survived every later run. Requiring the
# full set instead turns a partial cache entry into a miss, which
# heals it.
if fresh and CACHED_HASHES.issubset(cached):
hashes = {name: cached[name] for name in CACHED_HASHES}
sha1 = hashes["sha1"]
is_variant = "/.variants/" in rel_path or "\\.variants\\" in rel_path
if sha1 in files:
@@ -273,6 +279,11 @@ def _preserve_large_file_entries(files: dict, db_path: str) -> int:
except (FileNotFoundError, json.JSONDecodeError):
return 0
# A path the scan just claimed holds known bytes. An older entry naming
# the same path describes a revision that is no longer there, and keeping
# it would publish a hash that contradicts the file it points at.
scanned_paths = {entry.get("path", "") for entry in files.values()}
count = 0
for sha1, entry in existing_db.get("files", {}).items():
if sha1 in files:
@@ -289,6 +300,8 @@ def _preserve_large_file_entries(files: dict, db_path: str) -> int:
)
if cached:
entry = {**entry, "path": cached}
elif path in scanned_paths:
continue
files[sha1] = entry
count += 1
return count
@@ -402,7 +415,7 @@ def _collect_all_aliases(files: dict) -> dict:
config_file = platforms_dir / f"{platform_name}.yml"
try:
with open(config_file) as f:
config = yaml.safe_load(f) or {}
config = yaml_load(f) or {}
except (yaml.YAMLError, OSError) as e:
print(f"Warning: {config_file.name}: {e}", file=sys.stderr)
continue
@@ -451,7 +464,7 @@ def _collect_all_aliases(files: dict) -> dict:
continue
try:
with open(emu_file) as f:
emu_config = yaml.safe_load(f) or {}
emu_config = yaml_load(f) or {}
except (yaml.YAMLError, OSError):
continue
for file_entry in emu_config.get("files", []):
+14 -1
View File
@@ -10,6 +10,8 @@ import zipfile
from pathlib import Path, PurePosixPath
import yaml
from common import yaml_load
from jsonschema import Draft202012Validator, FormatChecker
ROOT = Path(__file__).resolve().parent.parent
@@ -48,7 +50,7 @@ def _validate_yaml_directory(
continue
try:
with path.open(encoding="utf-8") as handle:
data = yaml.safe_load(handle)
data = yaml_load(handle)
except (OSError, yaml.YAMLError) as exc:
out.append(f"{path.relative_to(ROOT)}: {exc}")
continue
@@ -122,9 +124,20 @@ def _semantic_database_checks(database: dict) -> list[str]:
out.append("database.json: total_files does not equal len(files)")
if database.get("total_size") != sum(entry.get("size", 0) for entry in files.values()):
out.append("database.json: total_size does not equal the file-size sum")
# One path holds one content. Two entries naming the same path means one
# of them declares a hash the file at that path does not have.
owners: dict[str, str] = {}
for sha1, entry in files.items():
if entry.get("sha1") != sha1:
out.append(f"database.json: files/{sha1}: key and sha1 differ")
path = entry.get("path", "")
if not path:
continue
first = owners.setdefault(path, sha1)
if first != sha1:
out.append(
f"database.json: {path} is claimed by {first} and {sha1}"
)
return out
+116
View File
@@ -10,6 +10,7 @@ from __future__ import annotations
import hashlib
import io
import json
import os
import sys
import tempfile
@@ -157,5 +158,120 @@ class LargeFileCacheTest(unittest.TestCase):
)
class HashCacheKeepsEveryDigest(unittest.TestCase):
"""A cache hit must serve the same five digests a fresh hash produces.
The cache-hit path rebuilt the hash dict from a hand-written list that
omitted adler32, then wrote the entry back without it, so one run without
--force stripped the digest from all 7,850 entries for good.
"""
def setUp(self):
import generate_db
self.generate_db = generate_db
self._tmp = tempfile.TemporaryDirectory()
self.bios = Path(self._tmp.name) / "bios"
(self.bios / "Sony" / "PS").mkdir(parents=True)
(self.bios / "Sony" / "PS" / "boot.bin").write_bytes(b"CACHED PAYLOAD")
def tearDown(self):
self._tmp.cleanup()
def test_cache_hit_serves_the_full_digest_set(self):
files, _, cache = self.generate_db.scan_bios_dir(self.bios, {}, force=False)
entry = next(iter(files.values()))
self.assertTrue(self.generate_db.CACHED_HASHES.issubset(entry))
# Second pass, this time served entirely from the cache.
again, _, cache2 = self.generate_db.scan_bios_dir(self.bios, cache, force=False)
entry2 = next(iter(again.values()))
self.assertTrue(
self.generate_db.CACHED_HASHES.issubset(entry2),
f"cache hit lost {self.generate_db.CACHED_HASHES - set(entry2)}",
)
self.assertEqual(
{k: entry[k] for k in self.generate_db.CACHED_HASHES},
{k: entry2[k] for k in self.generate_db.CACHED_HASHES},
)
self.assertTrue(
self.generate_db.CACHED_HASHES.issubset(next(iter(cache2.values())))
)
def test_a_partial_cache_entry_is_rehashed_instead_of_trusted(self):
_, _, cache = self.generate_db.scan_bios_dir(self.bios, {}, force=False)
key = next(iter(cache))
cache[key].pop("adler32")
files, _, healed = self.generate_db.scan_bios_dir(self.bios, cache, force=False)
entry = next(iter(files.values()))
self.assertTrue(self.generate_db.CACHED_HASHES.issubset(entry))
self.assertTrue(self.generate_db.CACHED_HASHES.issubset(healed[key]))
class PreservedLargeFileEntries(unittest.TestCase):
"""A preserved entry must never claim a path another entry already owns.
A large file replaced on disk by a newer firmware revision left its old
entry in the database forever, pointing at a path that now serves other
bytes.
"""
def setUp(self):
import generate_db
self.generate_db = generate_db
self._tmp = tempfile.TemporaryDirectory()
self.tmp = Path(self._tmp.name)
self._cwd = os.getcwd()
os.chdir(self.tmp)
(self.tmp / ".gitignore").write_text("bios/Sony/PS3/FW.PUP\n")
self._real_fetch = common.fetch_large_file
generate_db.__dict__.pop("fetch_large_file", None)
def tearDown(self):
os.chdir(self._cwd)
common.fetch_large_file = self._real_fetch
self._tmp.cleanup()
def _write_db(self, entries: dict) -> str:
path = str(self.tmp / "database.json")
Path(path).write_text(json.dumps({"files": entries}))
return path
def test_stale_entry_for_a_rescanned_path_is_dropped(self):
common.fetch_large_file = lambda *a, **k: None
db_path = self._write_db(
{
"a" * 40: {"name": "FW.PUP", "path": "bios/Sony/PS3/FW.PUP"},
"b" * 40: {"name": "FW.PUP", "path": "bios/Sony/PS3/FW.PUP"},
}
)
# The scan found the current revision at that path.
files = {"a" * 40: {"name": "FW.PUP", "path": "bios/Sony/PS3/FW.PUP"}}
count = self.generate_db._preserve_large_file_entries(files, db_path)
self.assertEqual(count, 0)
self.assertEqual(list(files), ["a" * 40])
def test_absent_large_file_is_still_preserved(self):
common.fetch_large_file = lambda *a, **k: None
db_path = self._write_db(
{"b" * 40: {"name": "FW.PUP", "path": "bios/Sony/PS3/FW.PUP"}}
)
files: dict = {}
count = self.generate_db._preserve_large_file_entries(files, db_path)
self.assertEqual(count, 1)
self.assertIn("b" * 40, files)
def test_verified_cache_hit_repoints_the_entry(self):
common.fetch_large_file = lambda *a, **k: "/cache/large/FW.PUP"
db_path = self._write_db(
{"b" * 40: {"name": "FW.PUP", "path": "bios/Sony/PS3/FW.PUP"}}
)
files = {"a" * 40: {"name": "FW.PUP", "path": "bios/Sony/PS3/FW.PUP"}}
count = self.generate_db._preserve_large_file_entries(files, db_path)
self.assertEqual(count, 1)
self.assertEqual(files["b" * 40]["path"], "/cache/large/FW.PUP")
if __name__ == "__main__":
unittest.main()