chore: restore large files before ci coverage checks

A CI checkout omits every file over 50 MB, so verify and generate_pack
resolve those database entries against a disk that does not hold them and
report them missing. The generated README then stops matching the
committed one for a reason that has nothing to do with staleness.

restore_large_files.py writes them back from the release cache, matched by
SHA1 rather than by name, and only where the path is gitignored and
absent. The site workflow runs it, and refreshes the data directories, before
generating.
This commit is contained in:
Abdessamad Derraz committed 2026-08-10 14:34:27 +02:00
1 parent b76edd2034
commit d588ebbde4
4 files changed
+182 -26

No files matched your search

+1 -26
View File
@@ -58,32 +58,7 @@ jobs:
run: |
mkdir -p .cache/large
gh release download large-files -D .cache/large/ 2>/dev/null || true
python3 -c "
import hashlib, json, os, shutil
db = json.load(open('database.json'))
with open('.gitignore') as f:
ignored = {l.strip() for l in f if l.strip().startswith('bios/')}
cache = '.cache/large'
if not os.path.isdir(cache):
exit(0)
idx = {}
for fn in os.listdir(cache):
fp = os.path.join(cache, fn)
if os.path.isfile(fp):
h = hashlib.sha1(open(fp, 'rb').read()).hexdigest()
idx[h] = fp
restored = 0
for sha1, entry in db['files'].items():
path = entry['path']
if path in ignored and not os.path.exists(path):
src = idx.get(sha1)
if src:
os.makedirs(os.path.dirname(path), exist_ok=True)
shutil.copy2(src, path)
print(f'Restored: {path}')
restored += 1
print(f'Total: {restored} files restored')
"
python scripts/restore_large_files.py
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+14
View File
@@ -52,6 +52,20 @@ jobs:
- name: Validate data contracts
run: python scripts/validate_schemas.py
# Coverage is resolved against the disk, so a checkout without the
# release assets and the data caches reports files as missing and the
# generated README stops matching the committed one.
- name: Restore large files from release
run: |
mkdir -p .cache/large
gh release download large-files -D .cache/large/ 2>/dev/null || true
python scripts/restore_large_files.py
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
- name: Refresh data directories
run: python scripts/refresh_data_dirs.py
- name: Generate site
run: |
python scripts/generate_site.py
+88
View File
@@ -0,0 +1,88 @@
#!/usr/bin/env python3
"""Restore gitignored large files into the checkout from the release cache.
Files over 50 MB live as assets of the `large-files` release instead of in
git. A CI checkout is therefore incomplete: every consumer that resolves a
database entry against the disk (verify.py, generate_pack.py, and through
them generate_readme.py) reports those paths as missing.
Assets are matched by content, not by name: the SHA1 of each cached file is
looked up in the database, and the entry's path is written only when it is
gitignored and absent.
Usage:
python scripts/restore_large_files.py [--cache .cache/large] [--db database.json]
"""
from __future__ import annotations
import argparse
import hashlib
import os
import shutil
import sys
sys.path.insert(0, os.path.dirname(__file__))
from common import load_database
def gitignored_paths(gitignore: str) -> set[str]:
"""Paths the repository keeps out of git, as written in .gitignore."""
try:
with open(gitignore) as f:
return {
line.strip() for line in f if line.strip().startswith("bios/")
}
except FileNotFoundError:
return set()
def index_cache(cache_dir: str) -> dict[str, str]:
"""Map SHA1 to cached file path for every asset in the cache."""
index: dict[str, str] = {}
for name in sorted(os.listdir(cache_dir)):
path = os.path.join(cache_dir, name)
if not os.path.isfile(path):
continue
digest = hashlib.sha1()
with open(path, "rb") as f:
for chunk in iter(lambda: f.read(1 << 20), b""):
digest.update(chunk)
index[digest.hexdigest()] = path
return index
def restore(cache_dir: str, db_path: str, gitignore: str) -> int:
if not os.path.isdir(cache_dir):
print(f"No cache at {cache_dir}, nothing to restore")
return 0
ignored = gitignored_paths(gitignore)
index = index_cache(cache_dir)
db = load_database(db_path)
restored = 0
for sha1, entry in db.get("files", {}).items():
path = entry.get("path", "")
if path not in ignored or os.path.exists(path):
continue
source = index.get(sha1)
if not source:
continue
os.makedirs(os.path.dirname(path), exist_ok=True)
shutil.copy2(source, path)
print(f"Restored: {path}")
restored += 1
print(f"Total: {restored} files restored")
return restored
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--cache", default=".cache/large")
parser.add_argument("--db", default="database.json")
parser.add_argument("--gitignore", default=".gitignore")
args = parser.parse_args()
restore(args.cache, args.db, args.gitignore)
if __name__ == "__main__":
main()
+79
View File
@@ -700,5 +700,84 @@ class TargetManifestRegressions(unittest.TestCase):
generate_target_manifests(str(source), str(output))
class CheckoutCompletenessRegressions(unittest.TestCase):
"""Coverage is resolved against the disk, so the checkout must be whole.
Files over 50 MB and the data directory caches are gitignored. A job that
regenerates the README or the site without restoring them counts those
files as missing and publishes a coverage the collection does not have.
"""
def _steps(self, workflow: str, job: str) -> list[dict]:
data = yaml.safe_load((ROOT / ".github" / "workflows" / workflow).read_text())
return data["jobs"][job]["steps"]
def _index(self, steps: list[dict], needle: str) -> int:
for position, step in enumerate(steps):
if needle in step.get("name", "") or needle in str(step.get("run", "")):
return position
self.fail(f"no step matching {needle!r}")
def test_site_deploy_completes_the_checkout_before_generating(self):
steps = self._steps("deploy-site.yml", "build")
generate = self._index(steps, "Generate site")
self.assertLess(self._index(steps, "restore_large_files.py"), generate)
self.assertLess(self._index(steps, "refresh_data_dirs.py"), generate)
def test_pack_build_completes_the_checkout_before_building(self):
steps = self._steps("build.yml", "release")
build = self._index(steps, "Build packs")
self.assertLess(self._index(steps, "restore_large_files.py"), build)
self.assertLess(self._index(steps, "refresh_data_dirs.py"), build)
def test_restore_matches_assets_by_content_not_by_name(self):
from scripts.restore_large_files import restore
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
root = Path(directory)
cache = root / "cache"
cache.mkdir()
payload = b"firmware bytes"
(cache / "renamed-asset.bin").write_bytes(payload)
sha1 = hashlib.sha1(payload).hexdigest()
(root / ".gitignore").write_text("bios/Sony/big.pup\n", encoding="utf-8")
(root / "database.json").write_text(
json.dumps({"files": {sha1: {"path": "bios/Sony/big.pup"}}}),
encoding="utf-8",
)
cwd = os.getcwd()
os.chdir(root)
try:
self.assertEqual(restore(str(cache), "database.json", ".gitignore"), 1)
self.assertEqual((root / "bios/Sony/big.pup").read_bytes(), payload)
# A path already in the checkout is never overwritten.
self.assertEqual(restore(str(cache), "database.json", ".gitignore"), 0)
finally:
os.chdir(cwd)
def test_restore_leaves_tracked_paths_alone(self):
from scripts.restore_large_files import restore
with tempfile.TemporaryDirectory(dir=TMP_ROOT) as directory:
root = Path(directory)
cache = root / "cache"
cache.mkdir()
payload = b"tracked bytes"
(cache / "asset.bin").write_bytes(payload)
sha1 = hashlib.sha1(payload).hexdigest()
(root / ".gitignore").write_text("bios/other.bin\n", encoding="utf-8")
(root / "database.json").write_text(
json.dumps({"files": {sha1: {"path": "bios/tracked.bin"}}}),
encoding="utf-8",
)
cwd = os.getcwd()
os.chdir(root)
try:
self.assertEqual(restore(str(cache), "database.json", ".gitignore"), 0)
self.assertFalse((root / "bios/tracked.bin").exists())
finally:
os.chdir(cwd)
if __name__ == "__main__":
unittest.main()