refactor: give common.py's parts their own modules

common.py had grown to 1833 lines by accumulation. Six coherent pieces
move out - untrusted parsing, digests, archives, generated artefacts,
release assets, dump catalogues - and common.py re-exports them, so the
sixty existing import sites keep working and migrating them stays
optional.

The site build is now reproducible, which is what made the move
checkable. It deleted its generated directories first, so every page was
new and write_if_changed had no earlier version to compare against: a
deploy republished six hundred pages for the clock alone. Directories
are swept instead, a page is removed only once nothing produces it, and
the body pass compares against the body of the file on disk rather than
against the decorated page. Two consecutive builds on the same inputs
now produce identical bytes; before, 1034 files differed.
This commit is contained in:
Abdessamad Derraz committed 2026-08-12 11:49:54 +02:00
1 parent 5e168b86c8
commit b5fd643ccf
10 files changed
+783 -589

No files matched your search

+10 -9
View File
@@ -23,6 +23,7 @@ REPO_ROOT = Path(__file__).resolve().parent.parent
sys.path.insert(0, str(REPO_ROOT / "scripts"))
import common # noqa: E402
import largefiles
PAYLOAD_A = b"A" * (256 * 1024)
PAYLOAD_B = b"B" * (256 * 1024)
@@ -54,10 +55,10 @@ class _SlowResponse(io.BytesIO):
class LargeFileCacheTest(unittest.TestCase):
def setUp(self):
self.dir = tempfile.mkdtemp()
self._urlopen = common.urllib.request.urlopen
self._urlopen = largefiles.urllib.request.urlopen
def tearDown(self):
common.urllib.request.urlopen = self._urlopen
largefiles.urllib.request.urlopen = self._urlopen
def test_concurrent_fetches_do_not_mix(self):
barrier = threading.Barrier(2, timeout=10)
@@ -70,7 +71,7 @@ class LargeFileCacheTest(unittest.TestCase):
i = next(index)
return _SlowResponse(payloads[i], barrier)
common.urllib.request.urlopen = fake_urlopen
largefiles.urllib.request.urlopen = fake_urlopen
results: list[str | None] = [None, None]
def worker(slot: int):
@@ -90,7 +91,7 @@ class LargeFileCacheTest(unittest.TestCase):
self.assertIn(digest, accepted)
def test_no_scratch_file_survives_a_successful_fetch(self):
common.urllib.request.urlopen = lambda req, timeout=None: _SlowResponse(
largefiles.urllib.request.urlopen = lambda req, timeout=None: _SlowResponse(
PAYLOAD_A
)
common.fetch_large_file("asset.bin", dest_dir=self.dir)
@@ -101,12 +102,12 @@ class LargeFileCacheTest(unittest.TestCase):
def fail(req, timeout=None):
raise urllib.error.URLError("offline")
common.urllib.request.urlopen = fail
largefiles.urllib.request.urlopen = fail
self.assertIsNone(common.fetch_large_file("asset.bin", dest_dir=self.dir))
self.assertEqual(os.listdir(self.dir), [])
def test_hash_mismatch_leaves_no_scratch_file(self):
common.urllib.request.urlopen = lambda req, timeout=None: _SlowResponse(
largefiles.urllib.request.urlopen = lambda req, timeout=None: _SlowResponse(
PAYLOAD_A
)
result = common.fetch_large_file(
@@ -122,7 +123,7 @@ class LargeFileCacheTest(unittest.TestCase):
def fail(req, timeout=None):
raise AssertionError("must not download when the cache is valid")
common.urllib.request.urlopen = fail
largefiles.urllib.request.urlopen = fail
self.assertEqual(
common.fetch_large_file("asset.bin", dest_dir=self.dir), str(cached)
)
@@ -131,7 +132,7 @@ class LargeFileCacheTest(unittest.TestCase):
def fail(req, timeout=None):
raise AssertionError("offline mode must not open the network")
common.urllib.request.urlopen = fail
largefiles.urllib.request.urlopen = fail
self.assertIsNone(
common.fetch_large_file(
"asset.bin", dest_dir=self.dir, offline=True
@@ -146,7 +147,7 @@ class LargeFileCacheTest(unittest.TestCase):
def fail(req, timeout=None):
raise AssertionError("offline cache hit must not open the network")
common.urllib.request.urlopen = fail
largefiles.urllib.request.urlopen = fail
self.assertEqual(
common.fetch_large_file(
"asset.bin",