diff --git a/scripts/generate_db.py b/scripts/generate_db.py index 9ac38e15..82923fc8 100644 --- a/scripts/generate_db.py +++ b/scripts/generate_db.py @@ -19,12 +19,13 @@ from pathlib import Path sys.path.insert(0, os.path.dirname(__file__)) from common import ( - DEFAULT_PROVENANCE_DIR, annotate_provenance, compute_hashes, + DEFAULT_PROVENANCE_DIR, list_registered_platforms, load_provenance_snapshots, write_if_changed, + yaml_load, ) CACHE_DIR = ".cache" @@ -34,6 +35,10 @@ DEFAULT_OUTPUT = "database.json" SKIP_PATTERNS = {".git", ".github", "__pycache__", ".cache", ".DS_Store", "desktop.ini"} +# Every digest database.json publishes per file. The schema requires all five, +# so a cache entry missing any of them cannot serve a database entry. +CACHED_HASHES = frozenset({"sha1", "md5", "sha256", "crc32", "adler32"}) + def should_skip(path: Path) -> bool: """Check if a path should be skipped. Allows .variants/ directories.""" @@ -80,13 +85,14 @@ def scan_bios_dir(bios_dir: Path, cache: dict, force: bool) -> tuple[dict, dict, if not force and cache_key in cache: cached = cache[cache_key] - if cached.get("mtime") == mtime and cached.get("size") == size: - hashes = { - "sha1": cached["sha1"], - "md5": cached["md5"], - "sha256": cached["sha256"], - "crc32": cached["crc32"], - } + fresh = cached.get("mtime") == mtime and cached.get("size") == size + # Rebuilding the hash dict by hand dropped whichever digest the + # list forgot, and the entry was then written back to the cache + # without it, so the loss survived every later run. Requiring the + # full set instead turns a partial cache entry into a miss, which + # heals it. + if fresh and CACHED_HASHES.issubset(cached): + hashes = {name: cached[name] for name in CACHED_HASHES} sha1 = hashes["sha1"] is_variant = "/.variants/" in rel_path or "\\.variants\\" in rel_path if sha1 in files: @@ -273,6 +279,11 @@ def _preserve_large_file_entries(files: dict, db_path: str) -> int: except (FileNotFoundError, json.JSONDecodeError): return 0 + # A path the scan just claimed holds known bytes. An older entry naming + # the same path describes a revision that is no longer there, and keeping + # it would publish a hash that contradicts the file it points at. + scanned_paths = {entry.get("path", "") for entry in files.values()} + count = 0 for sha1, entry in existing_db.get("files", {}).items(): if sha1 in files: @@ -289,6 +300,8 @@ def _preserve_large_file_entries(files: dict, db_path: str) -> int: ) if cached: entry = {**entry, "path": cached} + elif path in scanned_paths: + continue files[sha1] = entry count += 1 return count @@ -402,7 +415,7 @@ def _collect_all_aliases(files: dict) -> dict: config_file = platforms_dir / f"{platform_name}.yml" try: with open(config_file) as f: - config = yaml.safe_load(f) or {} + config = yaml_load(f) or {} except (yaml.YAMLError, OSError) as e: print(f"Warning: {config_file.name}: {e}", file=sys.stderr) continue @@ -451,7 +464,7 @@ def _collect_all_aliases(files: dict) -> dict: continue try: with open(emu_file) as f: - emu_config = yaml.safe_load(f) or {} + emu_config = yaml_load(f) or {} except (yaml.YAMLError, OSError): continue for file_entry in emu_config.get("files", []): diff --git a/scripts/validate_schemas.py b/scripts/validate_schemas.py index 9f28b1d2..22b2bde6 100644 --- a/scripts/validate_schemas.py +++ b/scripts/validate_schemas.py @@ -10,6 +10,8 @@ import zipfile from pathlib import Path, PurePosixPath import yaml + +from common import yaml_load from jsonschema import Draft202012Validator, FormatChecker ROOT = Path(__file__).resolve().parent.parent @@ -48,7 +50,7 @@ def _validate_yaml_directory( continue try: with path.open(encoding="utf-8") as handle: - data = yaml.safe_load(handle) + data = yaml_load(handle) except (OSError, yaml.YAMLError) as exc: out.append(f"{path.relative_to(ROOT)}: {exc}") continue @@ -122,9 +124,20 @@ def _semantic_database_checks(database: dict) -> list[str]: out.append("database.json: total_files does not equal len(files)") if database.get("total_size") != sum(entry.get("size", 0) for entry in files.values()): out.append("database.json: total_size does not equal the file-size sum") + # One path holds one content. Two entries naming the same path means one + # of them declares a hash the file at that path does not have. + owners: dict[str, str] = {} for sha1, entry in files.items(): if entry.get("sha1") != sha1: out.append(f"database.json: files/{sha1}: key and sha1 differ") + path = entry.get("path", "") + if not path: + continue + first = owners.setdefault(path, sha1) + if first != sha1: + out.append( + f"database.json: {path} is claimed by {first} and {sha1}" + ) return out diff --git a/tests/test_large_file_cache.py b/tests/test_large_file_cache.py index cae051d8..5a765466 100644 --- a/tests/test_large_file_cache.py +++ b/tests/test_large_file_cache.py @@ -10,6 +10,7 @@ from __future__ import annotations import hashlib import io +import json import os import sys import tempfile @@ -157,5 +158,120 @@ class LargeFileCacheTest(unittest.TestCase): ) +class HashCacheKeepsEveryDigest(unittest.TestCase): + """A cache hit must serve the same five digests a fresh hash produces. + + The cache-hit path rebuilt the hash dict from a hand-written list that + omitted adler32, then wrote the entry back without it, so one run without + --force stripped the digest from all 7,850 entries for good. + """ + + def setUp(self): + import generate_db + + self.generate_db = generate_db + self._tmp = tempfile.TemporaryDirectory() + self.bios = Path(self._tmp.name) / "bios" + (self.bios / "Sony" / "PS").mkdir(parents=True) + (self.bios / "Sony" / "PS" / "boot.bin").write_bytes(b"CACHED PAYLOAD") + + def tearDown(self): + self._tmp.cleanup() + + def test_cache_hit_serves_the_full_digest_set(self): + files, _, cache = self.generate_db.scan_bios_dir(self.bios, {}, force=False) + entry = next(iter(files.values())) + self.assertTrue(self.generate_db.CACHED_HASHES.issubset(entry)) + + # Second pass, this time served entirely from the cache. + again, _, cache2 = self.generate_db.scan_bios_dir(self.bios, cache, force=False) + entry2 = next(iter(again.values())) + self.assertTrue( + self.generate_db.CACHED_HASHES.issubset(entry2), + f"cache hit lost {self.generate_db.CACHED_HASHES - set(entry2)}", + ) + self.assertEqual( + {k: entry[k] for k in self.generate_db.CACHED_HASHES}, + {k: entry2[k] for k in self.generate_db.CACHED_HASHES}, + ) + self.assertTrue( + self.generate_db.CACHED_HASHES.issubset(next(iter(cache2.values()))) + ) + + def test_a_partial_cache_entry_is_rehashed_instead_of_trusted(self): + _, _, cache = self.generate_db.scan_bios_dir(self.bios, {}, force=False) + key = next(iter(cache)) + cache[key].pop("adler32") + files, _, healed = self.generate_db.scan_bios_dir(self.bios, cache, force=False) + entry = next(iter(files.values())) + self.assertTrue(self.generate_db.CACHED_HASHES.issubset(entry)) + self.assertTrue(self.generate_db.CACHED_HASHES.issubset(healed[key])) + + +class PreservedLargeFileEntries(unittest.TestCase): + """A preserved entry must never claim a path another entry already owns. + + A large file replaced on disk by a newer firmware revision left its old + entry in the database forever, pointing at a path that now serves other + bytes. + """ + + def setUp(self): + import generate_db + + self.generate_db = generate_db + self._tmp = tempfile.TemporaryDirectory() + self.tmp = Path(self._tmp.name) + self._cwd = os.getcwd() + os.chdir(self.tmp) + (self.tmp / ".gitignore").write_text("bios/Sony/PS3/FW.PUP\n") + self._real_fetch = common.fetch_large_file + generate_db.__dict__.pop("fetch_large_file", None) + + def tearDown(self): + os.chdir(self._cwd) + common.fetch_large_file = self._real_fetch + self._tmp.cleanup() + + def _write_db(self, entries: dict) -> str: + path = str(self.tmp / "database.json") + Path(path).write_text(json.dumps({"files": entries})) + return path + + def test_stale_entry_for_a_rescanned_path_is_dropped(self): + common.fetch_large_file = lambda *a, **k: None + db_path = self._write_db( + { + "a" * 40: {"name": "FW.PUP", "path": "bios/Sony/PS3/FW.PUP"}, + "b" * 40: {"name": "FW.PUP", "path": "bios/Sony/PS3/FW.PUP"}, + } + ) + # The scan found the current revision at that path. + files = {"a" * 40: {"name": "FW.PUP", "path": "bios/Sony/PS3/FW.PUP"}} + count = self.generate_db._preserve_large_file_entries(files, db_path) + self.assertEqual(count, 0) + self.assertEqual(list(files), ["a" * 40]) + + def test_absent_large_file_is_still_preserved(self): + common.fetch_large_file = lambda *a, **k: None + db_path = self._write_db( + {"b" * 40: {"name": "FW.PUP", "path": "bios/Sony/PS3/FW.PUP"}} + ) + files: dict = {} + count = self.generate_db._preserve_large_file_entries(files, db_path) + self.assertEqual(count, 1) + self.assertIn("b" * 40, files) + + def test_verified_cache_hit_repoints_the_entry(self): + common.fetch_large_file = lambda *a, **k: "/cache/large/FW.PUP" + db_path = self._write_db( + {"b" * 40: {"name": "FW.PUP", "path": "bios/Sony/PS3/FW.PUP"}} + ) + files = {"a" * 40: {"name": "FW.PUP", "path": "bios/Sony/PS3/FW.PUP"}} + count = self.generate_db._preserve_large_file_entries(files, db_path) + self.assertEqual(count, 1) + self.assertEqual(files["b" * 40]["path"], "/cache/large/FW.PUP") + + if __name__ == "__main__": unittest.main()