From 29ff87a99b463afbebe8f0a6d73b7f9c6390c0fe Mon Sep 17 00:00:00 2001 From: Abdessamad Derraz <3028866+Abdess@users.noreply.github.com> Date: Mon, 10 Aug 2026 13:37:01 +0200 Subject: [PATCH] feat: publish versioned data exports The metadata behind the verifier, the pack builder and the site is now served as static files: a versioned JSON API, CSV extracts, a SQLite snapshot, and a catalog carrying a SHA-256 for each artifact. The gaps dataset covers both layers behind a layer column. It previously held one row, the single platform-verification anomaly, while the page offering it as a download led with the emulator-level count. source_ref renders as a permalink pinned to the revision the profile cites. When a profile declares two repositories, a path that belongs to neither by name is left as plain code: a citation without a link still names the file and the lines, a link to the wrong repository does not. The table filter, focus outlines and tap targets are progressive enhancement; the pages work without them. --- docs_assets/extra.css | 112 +++- docs_assets/site.js | 110 ++++ mkdocs.yml | 12 +- scripts/generate_readme.py | 1 + scripts/generate_site.py | 1012 +++++++++++++++++++++++++++++++++--- tests/test_site_exports.py | 207 ++++++++ 6 files changed, 1389 insertions(+), 65 deletions(-) create mode 100644 docs_assets/site.js create mode 100644 tests/test_site_exports.py diff --git a/docs_assets/extra.css b/docs_assets/extra.css index 6f636853..7dd58fa0 100644 --- a/docs_assets/extra.css +++ b/docs_assets/extra.css @@ -5,7 +5,8 @@ --rb-primary: #4a4e8a; --rb-primary-light: #6366a0; --rb-primary-dark: #363870; - --rb-accent: #e8594f; + --rb-accent: #b42318; + --rb-link: #273c8f; --rb-success: #2e7d32; --rb-warning: #f57c00; --rb-danger: #c62828; @@ -20,6 +21,8 @@ --rb-surface: #1e1e2e; --rb-border: #313244; --rb-text-secondary: #a6adc8; + --rb-accent: #ff9a8f; + --rb-link: #aebcff; } /* ── Material theme overrides ── */ @@ -509,7 +512,8 @@ /* Pack button in tables: smaller */ .md-typeset table .md-button { font-size: 0.75rem; - padding: 0.3em 0.8em; + min-height: 44px; + padding: 0.55em 0.8em; } /* ── Hide permalink anchors in hero ── */ @@ -576,3 +580,107 @@ font-size: 1.3rem; } } + +/* ── Accessibility and dense-data navigation ── */ +.md-typeset a:not(.md-button):not(.headerlink):not(.rb-badge) { + color: var(--rb-link); + text-decoration: underline; + text-decoration-thickness: 0.08em; + text-underline-offset: 0.14em; +} + +.md-typeset a:not(.md-button):not(.headerlink):hover { + text-decoration-thickness: 0.14em; +} + +:where(a, button, input, select, textarea, [tabindex]):focus-visible { + outline: 3px solid var(--rb-accent); + outline-offset: 3px; +} + +.md-typeset .md-button, +.md-typeset button, +.md-typeset input, +.md-typeset select { + min-height: 44px; +} + +.rb-visually-hidden { + position: absolute !important; + width: 1px !important; + height: 1px !important; + padding: 0 !important; + margin: -1px !important; + overflow: hidden !important; + clip: rect(0, 0, 0, 0) !important; + white-space: nowrap !important; + border: 0 !important; +} + +.rb-table-filter { + display: grid; + grid-template-columns: minmax(14rem, 28rem) auto; + gap: 0.35rem 0.8rem; + align-items: end; + margin: 1rem 0 0.45rem; +} + +.rb-table-filter label { + grid-column: 1 / -1; + color: var(--rb-text-secondary); + font-size: 0.78rem; + font-weight: 700; +} + +.rb-table-filter input { + width: 100%; + border: 2px solid var(--rb-border); + border-radius: 6px; + background: var(--md-default-bg-color); + color: var(--md-default-fg-color); + padding: 0.55rem 0.75rem; + font: inherit; +} + +.rb-table-filter-count { + align-self: center; + color: var(--rb-text-secondary); + font-size: 0.8rem; + white-space: nowrap; +} + +.md-typeset__scrollwrap[role="region"]:focus-visible { + border-radius: 8px; + outline: 3px solid var(--rb-accent); + outline-offset: 2px; +} + +/* Material's compact footer defaults sit just below WCAG AA on the dark bar. */ +.md-footer-meta .md-copyright { + color: #b5b5b5; +} + +.md-footer-meta.md-typeset .md-copyright a { + color: #b8c6ff !important; +} + +@media (max-width: 600px) { + .rb-table-filter { + grid-template-columns: 1fr; + } + + .rb-table-filter label { + grid-column: auto; + } +} + +@media (prefers-reduced-motion: reduce) { + *, + *::before, + *::after { + scroll-behavior: auto !important; + transition-duration: 0.01ms !important; + animation-duration: 0.01ms !important; + animation-iteration-count: 1 !important; + } +} diff --git a/docs_assets/site.js b/docs_assets/site.js new file mode 100644 index 00000000..f57c9151 --- /dev/null +++ b/docs_assets/site.js @@ -0,0 +1,110 @@ +/* Progressive enhancement for generated RetroBIOS reference tables. */ +(() => { + "use strict"; + + const textFromHeading = (element) => { + let cursor = element; + while (cursor) { + cursor = cursor.previousElementSibling; + if (!cursor) break; + if (/^H[1-6]$/.test(cursor.tagName)) { + return cursor.textContent.trim(); + } + } + return document.querySelector("h1")?.textContent.trim() || "Data table"; + }; + + const enhanceTables = () => { + document.querySelectorAll(".md-typeset table").forEach((table, index) => { + if (table.dataset.rbEnhanced === "true") return; + table.dataset.rbEnhanced = "true"; + + table.querySelectorAll("thead th").forEach((cell) => { + if (!cell.hasAttribute("scope")) cell.setAttribute("scope", "col"); + }); + table.querySelectorAll("tbody tr").forEach((row) => { + const first = row.querySelector(":scope > th"); + if (first && !first.hasAttribute("scope")) first.setAttribute("scope", "row"); + }); + + const container = table.parentElement; + const heading = textFromHeading(container); + if (!table.querySelector("caption")) { + const caption = document.createElement("caption"); + caption.className = "rb-visually-hidden"; + caption.textContent = heading; + table.prepend(caption); + } + if (container && container.scrollWidth > container.clientWidth) { + container.tabIndex = 0; + container.setAttribute("role", "region"); + container.setAttribute("aria-label", `${heading}, scrollable table`); + } + + const rows = Array.from(table.querySelectorAll("tbody tr")); + if (rows.length < 20 || !container || container.dataset.rbFilter === "true") { + return; + } + container.dataset.rbFilter = "true"; + + const controls = document.createElement("div"); + controls.className = "rb-table-filter"; + const id = `rb-table-filter-${index}-${Math.random().toString(36).slice(2, 8)}`; + const label = document.createElement("label"); + label.htmlFor = id; + label.textContent = `Filter ${heading}`; + const input = document.createElement("input"); + input.id = id; + input.type = "search"; + input.autocomplete = "off"; + input.placeholder = "Name, platform, system, hash..."; + const count = document.createElement("span"); + count.className = "rb-table-filter-count"; + count.setAttribute("aria-live", "polite"); + count.textContent = `${rows.length} rows`; + controls.append(label, input, count); + container.before(controls); + + const searchable = rows.map((row) => ({ + row, + text: row.textContent.normalize("NFKD").toLocaleLowerCase(), + })); + input.addEventListener("input", () => { + const query = input.value.trim().normalize("NFKD").toLocaleLowerCase(); + let visible = 0; + searchable.forEach((entry) => { + const matches = !query || entry.text.includes(query); + entry.row.hidden = !matches; + if (matches) visible += 1; + }); + count.textContent = `${visible} of ${rows.length} rows`; + }); + }); + }; + + const nameMaterialWidgets = () => { + document.querySelectorAll('[role="dialog"]').forEach((dialog) => { + if (dialog.hasAttribute("aria-label") || dialog.hasAttribute("aria-labelledby")) return; + const heading = dialog.querySelector("h1, h2, h3"); + dialog.setAttribute("aria-label", heading?.textContent.trim() || "Site search"); + }); + document.querySelectorAll('[role="progressbar"]').forEach((progress) => { + if (!progress.hasAttribute("aria-label") && !progress.hasAttribute("aria-labelledby")) { + progress.setAttribute("aria-label", "Loading page"); + } + }); + }; + + const enhance = () => { + enhanceTables(); + nameMaterialWidgets(); + }; + + if (window.document$?.subscribe) { + window.document$.subscribe(enhance); + } else if (document.readyState === "loading") { + document.addEventListener("DOMContentLoaded", enhance, { once: true }); + } else { + enhance(); + } +})(); diff --git a/mkdocs.yml b/mkdocs.yml index a79ae8e5..7da5c326 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -8,6 +8,10 @@ repo_name: Abdess/retrobios # Almost every page is generated from platforms/, emulators/ and database.json, # so a per-page edit link would point at a file that does not exist. edit_uri: '' +# Local implementation plans are preserved in docs/ for development sessions, +# but are not part of the public reference site. +exclude_docs: | + superpowers/** copyright: MIT for the tooling. BIOS and firmware files are third-party system software, preserved for personal backup, archival and interoperability. theme: @@ -53,6 +57,8 @@ theme: - toc.follow extra_css: - stylesheets/extra.css +extra_javascript: +- javascripts/site.js extra: social: - icon: fontawesome/brands/github @@ -64,6 +70,7 @@ markdown_extensions: - attr_list - def_list - footnotes +- meta - md_in_html - tables - toc: @@ -111,6 +118,7 @@ nav: - 3DO Company: systems/3do-company.md - APF: systems/apf.md - Acorn: systems/acorn.md + - Adobe: systems/adobe.md - Amstrad: systems/amstrad.md - Apple: systems/apple.md - Arcade: systems/arcade.md @@ -510,7 +518,7 @@ nav: - Launchers (2): - Dolphin Launcher: emulators/dolphin_launcher.md - Parallel Launcher: emulators/parallel-launcher.md - - Other (37): + - Other (36): - ares: emulators/ares.md - Basilisk II: emulators/basiliskii.md - Beetle GBA (Mednafen): emulators/beetle_gba.md @@ -541,7 +549,6 @@ nav: - SUPER ZSNES: emulators/superzsnes.md - ti99sim: emulators/ti99sim.md - tsugaru: emulators/tsugaru.md - - VBA-M: emulators/vba_m.md - veesem: emulators/veesem.md - VICE: emulators/vice.md - Vita3K: emulators/vita3k.md @@ -551,6 +558,7 @@ nav: - Cross-reference: cross-reference.md - Gap Analysis: gaps.md - Dump provenance: provenance.md +- Data & API: data.md - Wiki: - Overview: wiki/index.md - Getting started: wiki/getting-started.md diff --git a/scripts/generate_readme.py b/scripts/generate_readme.py index b61d9809..99a8d548 100644 --- a/scripts/generate_readme.py +++ b/scripts/generate_readme.py @@ -476,6 +476,7 @@ def generate_readme(db: dict, platforms_dir: str) -> str: "- **Per-system pages** showing which emulators and platforms cover each console", "- **Gap analysis** identifying missing files and undeclared core requirements", f"- **Cross-reference** mapping files across {len(coverages)} platforms and {emulator_count} emulators", + "- **Versioned data access** through JSON, CSV and SQLite exports with published SHA-256 checksums", "", "## How it works", "", diff --git a/scripts/generate_site.py b/scripts/generate_site.py index 2a1b703e..826dd3ba 100644 --- a/scripts/generate_site.py +++ b/scripts/generate_site.py @@ -12,9 +12,14 @@ Usage: from __future__ import annotations import argparse +import csv +import hashlib +import io import json import os +import re import shutil +import sqlite3 import sys import urllib.error import urllib.parse @@ -39,13 +44,16 @@ from common import ( yaml = require_yaml() from generate_readme import compute_coverage, manifest_totals +from profile_sync import source_ref_values, split_source_ref from provenance_report import build_report +import upstream DOCS_DIR = "docs" SITE_NAME = "RetroBIOS" REPO_URL = "https://github.com/Abdess/retrobios" RELEASE_URL = f"{REPO_URL}/releases/latest" -GENERATED_DIRS = ["platforms", "systems", "emulators"] +SITE_URL = "https://abdess.github.io/retrobios/" +GENERATED_DIRS = ["platforms", "systems", "emulators", "wiki", "api", "downloads"] WIKI_SRC_DIR = "wiki" # manually maintained wiki sources SYSTEM_ICON_BASE = "https://raw.githubusercontent.com/libretro/retroarch-assets/master/xmb/systematic/png" ICON_CACHE_PATH = Path(".cache") / "system_icons.json" @@ -56,6 +64,120 @@ ICON_CACHE_PATH = Path(".cache") / "system_icons.json" _icon_available: dict[str, bool] = {} +def _forge_sources(profile: dict, label: str = "") -> list[tuple[upstream.Repo, str]]: + """Supported source repositories and immutable revisions for a profile.""" + candidates: list[tuple[upstream.Repo, str]] = [] + seen: set[tuple[str, str, str]] = set() + source_repos: set[tuple[str, str]] = set() + + for field in ("source", "upstream"): + raw = profile.get(field, "") + values: list[str] = [] + if isinstance(raw, dict): + if label and isinstance(raw.get(label), str): + values.append(raw[label]) + values.extend(str(value) for value in raw.values() if value) + elif raw: + values.append(str(raw)) + + for value in values: + repo = upstream.parse_repo(value) + if repo is None: + continue + repo_key = (repo.host, repo.slug) + pin = str(profile.get(f"{field}_commit") or "") + if field == "upstream" and not pin and repo_key in source_repos: + pin = str(profile.get("source_commit") or "") + if not pin: + continue + key = (repo.host, repo.slug, pin) + if key not in seen: + candidates.append((repo, pin)) + seen.add(key) + if field == "source": + source_repos.add(repo_key) + return candidates + + +def _source_permalink(repo: upstream.Repo, pin: str, path: str, + start: int | None, end: int | None) -> str: + """Forge-specific browser URL for one file at one immutable revision.""" + quoted_path = urllib.parse.quote(path, safe="/@:+-._~") + base = f"https://{repo.host}/{repo.owner}/{repo.name}" + if repo.family == "github": + url = f"{base}/blob/{pin}/{quoted_path}" + elif repo.family == "gitlab": + url = f"{base}/-/blob/{pin}/{quoted_path}" + else: + url = f"{base}/src/commit/{pin}/{quoted_path}" + if start is not None: + if repo.family == "gitlab": + url += f"#L{start}" + if end is not None and end != start: + url += f"-{end}" + else: + url += f"#L{start}" + if end is not None and end != start: + url += f"-L{end}" + return url + + +def _source_ref_markdown(profile: dict, value) -> str: + """Render source_ref values as pinned links when their forge is known. + + A profile can declare two repositories: the libretro port in ``source`` and + the original emulator in ``upstream``. Which of the two carries a given + path cannot be decided without reading their trees, and this generator runs + offline. Rather than guess, an unattributable path is rendered as plain + code: a citation with no link still names the file and the lines, while a + link to the wrong repository is a false citation. + """ + rendered_groups: list[str] = [] + path_re = re.compile(r"^[A-Za-z0-9_.@/+~-]+$") + for label, refs in source_ref_values(value): + candidates = _forge_sources(profile, label) + rendered: list[str] = [] + for part in split_source_ref(refs): + display = part.path + if part.start is not None: + display += f":{part.start}" + if part.end is not None and part.end != part.start: + display += f"-{part.end}" + + selected: tuple[upstream.Repo, str] | None = None + real_path = part.path + if path_re.fullmatch(part.path) and candidates: + for repo, pin in candidates: + prefix = f"{repo.name}/" + if part.path.startswith(prefix): + selected = (repo, pin) + real_path = part.path[len(prefix):] + break + if selected is None and len(candidates) == 1: + selected = candidates[0] + + if selected is None: + rendered.append(f"`{display}`") + else: + repo, pin = selected + url = _source_permalink( + repo, pin, real_path, part.start, part.end + ) + rendered.append(f"[`{display}`]({url})") + + text = ", ".join(rendered) if rendered else f"`{refs}`" + if label: + text = f"**{label}:** {text}" + rendered_groups.append(text) + return "; ".join(rendered_groups) + + +def _admonition_body(text: str) -> str: + """Indent prose without turning source tokens such as ``#if`` into H1s.""" + escaped = re.sub(r"(?m)^(\s*)#", r"\1\\#", text) + return escaped.replace("\n", "\n ") + + def _content_check_ceiling(profiles: dict) -> str: """How far content checking can reach across every profiled entry.""" hashed = sized = neither = 0 @@ -372,8 +494,8 @@ def generate_home( "", f"# {SITE_NAME}", "", - "BIOS and firmware packs checked the way each platform checks them, " - "with emulator source code as the deciding authority.", + "BIOS and firmware metadata checked the way each platform checks it, " + "cross-referenced against revision-pinned emulator source code.", "", "", "", @@ -410,7 +532,7 @@ def generate_home( [ "## Platforms", "", - "| | Platform | Files | Checked by | Download |", + "| Icon | Platform | Files | Checked by | Download |", "|---|----------|-------|-----------|----------|", ] ) @@ -527,6 +649,7 @@ def generate_home( "[Cross-reference](cross-reference.md){ .md-button } " "[Gap Analysis](gaps.md){ .md-button } " "[Dump provenance](provenance.md){ .md-button } " + "[Data & API](data.md){ .md-button } " "[Contributing](contributing.md){ .md-button .md-button--primary }", "", f'
', @@ -542,6 +665,7 @@ def compute_stats(db: dict, coverages: dict, profiles: dict) -> dict: for p in unique.values(): systems.update(p.get("systems", [])) return { + "schema_version": 1, "generated_at": _timestamp(), "files": db.get("total_files", 0), "size_bytes": db.get("total_size", 0), @@ -570,6 +694,716 @@ def generate_stats(stats: dict) -> str: return json.dumps(stats, indent=2) + "\n" +def _json_text(value) -> str: + """Stable scalar representation for CSV and SQLite exports.""" + if value is None: + return "" + if isinstance(value, (dict, list, tuple, bool)): + return json.dumps(value, ensure_ascii=False, sort_keys=True) + return str(value) + + +def _csv_document(fieldnames: list[str], rows: list[dict]) -> str: + stream = io.StringIO(newline="") + writer = csv.DictWriter( + stream, fieldnames=fieldnames, extrasaction="ignore", lineterminator="\n" + ) + writer.writeheader() + writer.writerows(rows) + return stream.getvalue() + + +def _sha256_path(path: Path) -> str: + digest = hashlib.sha256() + with path.open("rb") as handle: + for chunk in iter(lambda: handle.read(1024 * 1024), b""): + digest.update(chunk) + return digest.hexdigest() + + +def _platform_export_rows(coverages: dict) -> tuple[list[dict], list[dict]]: + """Normalized platform summaries and one row per declared file.""" + items: list[dict] = [] + file_rows: list[dict] = [] + for key, coverage in sorted(coverages.items()): + config = coverage["config"] + items.append({ + "id": key, + "name": coverage["platform"], + "coverage": { + field: coverage.get(field, 0) + for field in ( + "total", "present", "verified", "untested", "missing", + "core_present", "core_missing", "core_unsourceable", + ) + }, + "contract": config, + }) + for system_id, system in sorted(config.get("systems", {}).items()): + for entry in system.get("files", []) or []: + file_rows.append({ + "platform_id": key, + "platform": coverage["platform"], + "system": system_id, + "name": entry.get("name", ""), + "destination": entry.get( + "destination", entry.get("dest", entry.get("name", "")) + ), + "required": bool(entry.get("required", True)), + "region": _json_text(entry.get("region")), + "variant_group": _json_text(entry.get("variant_group")), + "size": entry.get("size"), + "sha1": _json_text(entry.get("sha1")), + "sha256": _json_text(entry.get("sha256")), + "md5": _json_text(entry.get("md5")), + "crc32": _json_text(entry.get("crc32")), + }) + return items, file_rows + + +def _emulator_export_items(profiles: dict) -> list[dict]: + return [ + {"id": key, "profile": profile} + for key, profile in sorted(profiles.items()) + ] + + +def build_emulator_gap_report( + profiles: dict, + coverages: dict, + db: dict, + data_names: set[str] | None = None, +) -> dict: + """Files a core loads that no platform declares, per emulator profile. + + The gap analysis page and the published gaps export must not compute this + twice and drift; both read this one report. + """ + from common import expand_platform_declared_names + from cross_reference import cross_reference as run_cross_reference + + all_declared: set[str] = set() + declared: dict[str, set[str]] = {} + for _name, cov in coverages.items(): + config = cov["config"] + # Enrich with alias resolution (MD5 -> SHA1 -> canonical name + aliases) + all_declared.update(expand_platform_declared_names(config, db)) + for sys_id, system in config.get("systems", {}).items(): + for fe in system.get("files", []): + fname = fe.get("name", "") + if fname: + declared.setdefault(sys_id, set()).add(fname) + + unique_profiles = { + k: v + for k, v in profiles.items() + if v.get("type") not in ("alias", "test") + } + return run_cross_reference( + unique_profiles, declared, db, + data_names=data_names, all_declared=all_declared, + ) + + +def _gap_export_rows(coverages: dict, gap_report: dict | None = None) -> list[dict]: + """Every gap the site reports, both layers, in one table. + + ``platform`` rows are anomalies against a platform's own BIOS list. + ``emulator`` rows are files a profiled core loads that no platform + declares, which is the larger number the gap analysis page leads with. + A `layer` column keeps the two apart instead of publishing only one. + """ + rows: list[dict] = [] + for key, coverage in sorted(coverages.items()): + for detail in coverage.get("details", []): + if detail.get("status") == "ok" and not detail.get("discrepancy"): + continue + rows.append({ + "layer": "platform", + "platform_id": key, + "platform": coverage["platform"], + "emulator": "", + "system": detail.get("system", ""), + "name": detail.get("name", ""), + "status": detail.get("status", ""), + "required": bool(detail.get("required", True)), + "in_repo": "", + "reason": detail.get("reason", ""), + "discrepancy": detail.get("discrepancy", ""), + }) + for emu_key, data in sorted((gap_report or {}).items()): + systems = ";".join(str(s) for s in data.get("systems", []) or []) + for gap in data.get("gap_details", []) or []: + rows.append({ + "layer": "emulator", + "platform_id": "", + "platform": "", + "emulator": data.get("emulator", emu_key), + "system": systems, + "name": gap.get("name", ""), + "status": gap.get("source", ""), + "required": bool(gap.get("required", False)), + "in_repo": bool(gap.get("in_repo", False)), + "reason": gap.get("note", ""), + "discrepancy": "", + }) + for entry in data.get("unsourceable", []) or []: + rows.append({ + "layer": "emulator", + "platform_id": "", + "platform": "", + "emulator": data.get("emulator", emu_key), + "system": systems, + "name": entry.get("name", ""), + "status": "unsourceable", + "required": bool(entry.get("required", False)), + "in_repo": False, + "reason": entry.get("reason", ""), + "discrepancy": "", + }) + return rows + + +def _cross_reference_export_rows(coverages: dict, profiles: dict) -> list[dict]: + from common import resolve_platform_cores + + unique = { + key: value + for key, value in profiles.items() + if value.get("type") not in ("alias", "test") + } + rows: list[dict] = [] + for platform_id, coverage in sorted(coverages.items()): + for profile_id in sorted(resolve_platform_cores(coverage["config"], unique)): + profile = unique[profile_id] + cores = profile.get("cores") or [profile_id] + systems = profile.get("systems") or [""] + for core in cores: + for system in systems: + rows.append({ + "platform_id": platform_id, + "platform": coverage["platform"], + "profile_id": profile_id, + "emulator": profile.get("emulator", profile_id), + "core": core, + "system": system, + "classification": profile.get("core_classification", ""), + "type": profile.get("type", ""), + "source": _json_text(profile.get("source")), + "upstream": _json_text(profile.get("upstream")), + "profiled_commit": profile.get("source_commit", ""), + "file_count": len(profile.get("files", []) or []), + }) + return rows + + +def _write_sqlite_export( + destination: Path, + db: dict, + platform_items: list[dict], + platform_files: list[dict], + emulator_items: list[dict], + gap_rows: list[dict], +) -> None: + """Build a queryable, deterministic snapshot without embedding binaries.""" + temp_dir = Path("tmp") / "site" + temp_dir.mkdir(parents=True, exist_ok=True) + temp_path = temp_dir / f"retrobios-{os.getpid()}.sqlite" + temp_path.unlink(missing_ok=True) + destination.parent.mkdir(parents=True, exist_ok=True) + + connection = sqlite3.connect(temp_path) + try: + connection.executescript(""" + PRAGMA journal_mode = OFF; + PRAGMA synchronous = OFF; + CREATE TABLE metadata (key TEXT PRIMARY KEY, value TEXT NOT NULL); + CREATE TABLE files ( + sha1 TEXT PRIMARY KEY, path TEXT NOT NULL, name TEXT NOT NULL, + size INTEGER NOT NULL, md5 TEXT NOT NULL, sha256 TEXT NOT NULL, + crc32 TEXT NOT NULL, adler32 TEXT NOT NULL + ); + CREATE TABLE file_provenance ( + sha1 TEXT NOT NULL, catalog TEXT NOT NULL, details_json TEXT NOT NULL, + PRIMARY KEY (sha1, catalog), + FOREIGN KEY (sha1) REFERENCES files(sha1) + ); + CREATE TABLE platforms ( + id TEXT PRIMARY KEY, name TEXT NOT NULL, total INTEGER NOT NULL, + present INTEGER NOT NULL, verified INTEGER NOT NULL, + missing INTEGER NOT NULL, contract_json TEXT NOT NULL + ); + CREATE TABLE platform_files ( + platform_id TEXT NOT NULL, system TEXT NOT NULL, name TEXT NOT NULL, + destination TEXT NOT NULL, required INTEGER NOT NULL, + region TEXT, variant_group TEXT, size INTEGER, + sha1 TEXT, sha256 TEXT, md5 TEXT, crc32 TEXT + ); + CREATE TABLE emulators ( + id TEXT PRIMARY KEY, name TEXT NOT NULL, type TEXT, + classification TEXT, source TEXT, upstream TEXT, + source_commit TEXT, profile_json TEXT NOT NULL + ); + CREATE TABLE emulator_systems ( + emulator_id TEXT NOT NULL, system TEXT NOT NULL, + PRIMARY KEY (emulator_id, system) + ); + CREATE TABLE emulator_files ( + emulator_id TEXT NOT NULL, system TEXT, name TEXT NOT NULL, + path TEXT, required INTEGER NOT NULL, mode TEXT, region TEXT, + size TEXT, sha1 TEXT, sha256 TEXT, md5 TEXT, crc32 TEXT, + source_ref TEXT + ); + CREATE TABLE gaps ( + layer TEXT NOT NULL, platform_id TEXT, platform TEXT, + emulator TEXT, system TEXT, name TEXT NOT NULL, + status TEXT NOT NULL, required INTEGER NOT NULL, + in_repo TEXT, reason TEXT, discrepancy TEXT + ); + CREATE INDEX files_name_idx ON files(name); + CREATE INDEX files_md5_idx ON files(md5); + CREATE INDEX files_sha256_idx ON files(sha256); + CREATE INDEX platform_files_name_idx ON platform_files(name); + CREATE INDEX emulator_files_name_idx ON emulator_files(name); + CREATE INDEX gaps_status_idx ON gaps(layer, status); + """) + metadata = { + "schema_version": "1", + "generated_at": str(db.get("generated_at") or _timestamp()), + "source": REPO_URL, + "scope": "metadata only; no BIOS or firmware payload bytes", + } + connection.executemany( + "INSERT INTO metadata(key, value) VALUES (?, ?)", + sorted(metadata.items()), + ) + for sha1, entry in sorted(db.get("files", {}).items()): + connection.execute( + "INSERT INTO files VALUES (?, ?, ?, ?, ?, ?, ?, ?)", + ( + sha1, entry.get("path", ""), entry.get("name", ""), + entry.get("size", 0), entry.get("md5", ""), + entry.get("sha256", ""), entry.get("crc32", ""), + entry.get("adler32", ""), + ), + ) + for catalog, details in sorted((entry.get("provenance") or {}).items()): + connection.execute( + "INSERT INTO file_provenance VALUES (?, ?, ?)", + (sha1, catalog, _json_text(details)), + ) + for item in platform_items: + coverage = item["coverage"] + connection.execute( + "INSERT INTO platforms VALUES (?, ?, ?, ?, ?, ?, ?)", + ( + item["id"], item["name"], coverage.get("total", 0), + coverage.get("present", 0), coverage.get("verified", 0), + coverage.get("missing", 0), _json_text(item["contract"]), + ), + ) + connection.executemany( + "INSERT INTO platform_files VALUES " + "(:platform_id, :system, :name, :destination, :required, :region, " + ":variant_group, :size, :sha1, :sha256, :md5, :crc32)", + platform_files, + ) + for item in emulator_items: + profile = item["profile"] + connection.execute( + "INSERT INTO emulators VALUES (?, ?, ?, ?, ?, ?, ?, ?)", + ( + item["id"], profile.get("emulator", item["id"]), + profile.get("type", ""), + profile.get("core_classification", ""), + _json_text(profile.get("source")), + _json_text(profile.get("upstream")), + profile.get("source_commit", ""), _json_text(profile), + ), + ) + for system in sorted(set(profile.get("systems", []) or [])): + connection.execute( + "INSERT INTO emulator_systems VALUES (?, ?)", + (item["id"], system), + ) + for entry in profile.get("files", []) or []: + connection.execute( + "INSERT INTO emulator_files VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)", + ( + item["id"], _json_text(entry.get("system")), + entry.get("name", ""), entry.get("path", ""), + int(bool(entry.get("required", False))), + entry.get("mode", ""), _json_text(entry.get("region")), + _json_text(entry.get("size")), _json_text(entry.get("sha1")), + _json_text(entry.get("sha256")), _json_text(entry.get("md5")), + _json_text(entry.get("crc32")), + _json_text(entry.get("source_ref")), + ), + ) + connection.executemany( + "INSERT INTO gaps VALUES " + "(:layer, :platform_id, :platform, :emulator, :system, :name, " + ":status, :required, :in_repo, :reason, :discrepancy)", + gap_rows, + ) + connection.commit() + connection.execute("VACUUM") + finally: + connection.close() + os.replace(temp_path, destination) + + +def generate_data_exports( + docs: Path, + db: dict, + coverages: dict, + profiles: dict, + stats: dict, + gap_report: dict | None = None, +) -> list[dict]: + """Create versioned static API, CSV and SQLite metadata snapshots.""" + api = docs / "api" / "v1" + downloads = docs / "downloads" + schemas_dest = api / "schemas" + for directory in (api, downloads, schemas_dest): + directory.mkdir(parents=True, exist_ok=True) + + generated_at = str(db.get("generated_at") or _timestamp()) + platform_items, platform_files = _platform_export_rows(coverages) + emulator_items = _emulator_export_items(profiles) + gap_rows = _gap_export_rows(coverages, gap_report) + cross_rows = _cross_reference_export_rows(coverages, profiles) + + envelopes = { + "platforms.json": ("platforms", platform_items), + "emulators.json": ("emulators", emulator_items), + "gaps.json": ("verification-and-coverage-gaps", gap_rows), + } + for filename, (kind, items) in envelopes.items(): + document = { + "schema_version": 1, + "generated_at": generated_at, + "kind": kind, + "count": len(items), + "items": items, + } + write_if_changed( + str(api / filename), + json.dumps(document, ensure_ascii=False, indent=2, sort_keys=True) + "\n", + ) + write_if_changed(str(api / "stats.json"), generate_stats(stats)) + write_if_changed( + str(api / "database.json"), + json.dumps(db, ensure_ascii=False, indent=2) + "\n", + ) + + schema_names = ( + "database.schema.json", "emulator.schema.json", "platform.schema.json", + "site-api-envelope.schema.json", "stats.schema.json", + ) + for schema_name in schema_names: + shutil.copy2(Path("schemas") / schema_name, schemas_dest / schema_name) + + file_rows = [] + for sha1, entry in sorted(db.get("files", {}).items()): + file_rows.append({ + "sha1": sha1, + "path": entry.get("path", ""), + "name": entry.get("name", ""), + "size": entry.get("size", 0), + "md5": entry.get("md5", ""), + "sha256": entry.get("sha256", ""), + "crc32": entry.get("crc32", ""), + "adler32": entry.get("adler32", ""), + "provenance_catalogs": ";".join( + sorted((entry.get("provenance") or {}).keys()) + ), + }) + write_if_changed( + str(downloads / "files.csv"), + _csv_document( + [ + "sha1", "path", "name", "size", "md5", "sha256", + "crc32", "adler32", "provenance_catalogs", + ], + file_rows, + ), + ) + write_if_changed( + str(downloads / "platform-files.csv"), + _csv_document( + [ + "platform_id", "platform", "system", "name", "destination", + "required", "region", "variant_group", "size", "sha1", + "sha256", "md5", "crc32", + ], + platform_files, + ), + ) + write_if_changed( + str(downloads / "cross-reference.csv"), + _csv_document( + [ + "platform_id", "platform", "profile_id", "emulator", "core", + "system", "classification", "type", "source", "upstream", + "profiled_commit", "file_count", + ], + cross_rows, + ), + ) + write_if_changed( + str(downloads / "gaps.csv"), + _csv_document( + [ + "layer", "platform_id", "platform", "emulator", "system", + "name", "status", "required", "in_repo", "reason", + "discrepancy", + ], + gap_rows, + ), + ) + _write_sqlite_export( + downloads / "retrobios.sqlite", db, platform_items, platform_files, + emulator_items, gap_rows, + ) + + assets = [ + (api / "database.json", "Content database", "application/json", "schemas/database.schema.json"), + (api / "platforms.json", "Platform contracts", "application/json", "schemas/site-api-envelope.schema.json"), + (api / "emulators.json", "Emulator profiles", "application/json", "schemas/site-api-envelope.schema.json"), + (api / "gaps.json", "Verification and coverage gaps", "application/json", "schemas/site-api-envelope.schema.json"), + (api / "stats.json", "Project statistics", "application/json", "schemas/stats.schema.json"), + (downloads / "files.csv", "File hashes CSV", "text/csv", None), + (downloads / "platform-files.csv", "Platform declarations CSV", "text/csv", None), + (downloads / "cross-reference.csv", "Cross-reference CSV", "text/csv", None), + (downloads / "gaps.csv", "Verification and coverage gaps CSV", "text/csv", None), + (downloads / "retrobios.sqlite", "SQLite snapshot", "application/vnd.sqlite3", None), + ] + catalog_items: list[dict] = [] + for path, title, media_type, schema in assets: + relative = path.relative_to(docs).as_posix() + item = { + "title": title, + "url": relative, + "media_type": media_type, + "bytes": path.stat().st_size, + "sha256": _sha256_path(path), + } + if schema: + item["schema"] = schema + catalog_items.append(item) + catalog = { + "schema_version": 1, + "generated_at": generated_at, + "kind": "catalog", + "count": len(catalog_items), + "items": catalog_items, + } + write_if_changed( + str(api / "catalog.json"), + json.dumps(catalog, ensure_ascii=False, indent=2, sort_keys=True) + "\n", + ) + return catalog_items + + +def generate_data_page(exports: list[dict]) -> str: + lines = [ + f"# Data & API - {SITE_NAME}", + "", + "RetroBIOS publishes the same metadata used by its verifier, pack builder " + "and website as versioned static files. No account or API key is required.", + "", + "[API catalog](api/v1/catalog.json){ .md-button .md-button--primary } " + "[SQLite snapshot](downloads/retrobios.sqlite){ .md-button }", + "", + "## Stable JSON endpoints", + "", + "| Dataset | Endpoint | Schema | Size |", + "|---------|----------|--------|-----:|", + ] + for item in exports: + if item["media_type"] != "application/json": + continue + schema = ( + f"[JSON Schema](api/v1/{item['schema']})" + if item.get("schema") else "-" + ) + lines.append( + f"| {item['title']} | [`{item['url']}`]({item['url']}) | " + f"{schema} | {_fmt_size(item['bytes'])} |" + ) + lines.extend([ + "", + "Every JSON document carries `schema_version`. Breaking changes use a new " + "URL prefix (`/api/v2/`); fields may only be added compatibly within v1.", + "", + "## Bulk downloads", + "", + "| Export | Format | Size | SHA256 |", + "|--------|--------|-----:|--------|", + ]) + for item in exports: + if item["media_type"] == "application/json": + continue + lines.append( + f"| [{item['title']}]({item['url']}) | `{item['media_type']}` | " + f"{_fmt_size(item['bytes'])} | `{item['sha256']}` |" + ) + lines.extend([ + "", + "The SQLite file contains indexed tables for content hashes, provenance " + "catalog matches, platform declarations, emulator profiles and current " + "gaps. It contains metadata only, never BIOS or firmware bytes.", + "", + "The gaps dataset carries both layers behind one `layer` column: " + "`platform` rows are anomalies against a platform's own BIOS list, " + "`emulator` rows are files a profiled core loads that no platform " + "declares. The two answer different questions and are not comparable " + "totals.", + "", + "## Semantics and limits", + "", + "Treat four questions separately: content identity (hashes), presence in " + "the collection, acceptance by a specific emulator, and catalog provenance. " + "One does not imply the others. In particular, presence, a matching dump " + "catalog, or emulator compatibility is not a statement about copyright, " + "ownership, or redistribution rights.", + "", + f"The data can be newer than the [latest published pack]({RELEASE_URL}); " + "pack publication is manual and only occurs after all release gates pass.", + "", + "See the [data model](wiki/data-model.md), [verification modes]" + "(wiki/verification-modes.md), and [methodology](wiki/architecture.md) " + "before interpreting aggregate counts.", + "", + ]) + return "\n".join(lines) + + +def _page_title(markdown: str, path: Path) -> str: + match = re.search(r"^#\s+(.+?)\s*$", markdown, flags=re.MULTILINE) + if match: + title = re.sub(r"<[^>]+>", "", match.group(1)) + return title.replace(f" - {SITE_NAME}", "").strip() + return path.stem.replace("-", " ").title() + + +def _browser_title(relative: Path, title: str) -> str: + """Return a concise, unique browser/search title for a generated page.""" + key = relative.as_posix() + index_titles = { + "index.md": SITE_NAME, + "platforms/index.md": "Platforms", + "systems/index.md": "Systems", + "emulators/index.md": "Emulators", + "wiki/index.md": "Guide and methodology", + } + if key in index_titles: + return index_titles[key] + if key.startswith("emulators/"): + return f"{title} emulator firmware" + if key.startswith("systems/"): + return f"{title} systems" + return title + + +def _plain_markdown(text: str) -> str: + text = re.sub(r"<[^>]+>", " ", text) + text = re.sub(r"!\[([^]]*)\]\([^)]*\)", r"\1", text) + text = re.sub(r"\[([^]]+)\]\([^)]*\)", r"\1", text) + text = re.sub(r"[`*_~]", "", text) + return re.sub(r"\s+", " ", text).strip() + + +def _page_description(markdown: str, relative: Path, title: str) -> str: + key = relative.as_posix() + specific = { + "index.md": "Source-traced BIOS and firmware metadata, platform verification, emulator profiles, gaps and reproducible retrogaming data exports.", + "data.md": "Versioned RetroBIOS JSON API, CSV exports, SQLite snapshot, schemas, checksums and data interpretation guidance.", + "cross-reference.md": "Cross-reference from retrogaming platforms to emulator cores, systems, upstream projects and profiled firmware files.", + "gaps.md": "Current RetroBIOS verification gaps, platform-to-emulator divergences, missing files and documented source limitations.", + "provenance.md": "Hash-based comparison of RetroBIOS metadata with No-Intro, Redump and TOSEC catalog snapshots.", + "which-pack.md": "One-line automatic RetroBIOS installation and platform-specific pack destinations, with verification and release caveats.", + } + if key in specific: + return specific[key] + if key.startswith("platforms/") and relative.stem != "index": + return f"{title}: declared BIOS contract, verification mode, coverage, destinations and emulator complement." + if key.startswith("emulators/") and relative.stem != "index": + return f"{title}: source-pinned emulator firmware profile with paths, hashes, requirements and validation behavior." + if key.startswith("systems/") and relative.stem != "index": + return f"{title}: indexed firmware files, hashes, variants, provenance and platform or emulator usage." + + paragraphs = re.split(r"\n\s*\n", markdown) + for paragraph in paragraphs: + stripped = paragraph.strip() + if ( + not stripped + or stripped.startswith(("#", "|", "```", "???", "<", "- ", "* ")) + ): + continue + plain = _plain_markdown(stripped) + if len(plain) >= 35: + return plain[:157].rstrip(" ,;:-") + ("..." if len(plain) > 157 else "") + return f"{title}: RetroBIOS source-traced retrogaming firmware reference." + + +def decorate_markdown_pages(docs: Path) -> None: + """Add per-page descriptions and machine-readable structured data.""" + for path in sorted(docs.rglob("*.md")): + relative = path.relative_to(docs) + if relative.parts and relative.parts[0] == "superpowers": + continue + markdown = path.read_text(encoding="utf-8") + if markdown.startswith("---\n") and "generated_by: retrobios-site" in markdown[:300]: + end = markdown.find("\n---\n", 4) + if end != -1: + markdown = markdown[end + 5:].lstrip("\n") + script_end = markdown.find("\n\n") + if markdown.startswith('\n\n"):] + elif markdown.startswith("---\n"): + continue + + title = _page_title(markdown, relative) + browser_title = _browser_title(relative, title) + description = _page_description(markdown, relative, title) + if relative.name == "index.md": + page_path = "" if len(relative.parts) == 1 else "/".join(relative.parts[:-1]) + "/" + else: + page_path = relative.with_suffix("").as_posix() + "/" + canonical = urllib.parse.urljoin(SITE_URL, page_path) + schema_type = "WebSite" if relative.as_posix() == "index.md" else "TechArticle" + if relative.as_posix() == "data.md": + schema_type = "Dataset" + structured = { + "@context": "https://schema.org", + "@type": schema_type, + "name": title, + "description": description, + "url": canonical, + "isPartOf": { + "@type": "WebSite", + "name": SITE_NAME, + "url": SITE_URL, + }, + } + structured_json = json.dumps( + structured, ensure_ascii=False, separators=(",", ":") + ).replace("", "<\\/") + front_matter = ( + "---\n" + "generated_by: retrobios-site\n" + f"title: {json.dumps(browser_title, ensure_ascii=False)}\n" + f"description: {json.dumps(description, ensure_ascii=False)}\n" + "---\n\n" + '\n\n" + ) + write_if_changed(str(path), front_matter + markdown) + + # Platform pages @@ -1367,7 +2201,6 @@ def generate_emulator_page( ("rom_path", "ROM path"), ("game_count", "Game count"), ("verification", "Checked by"), - ("source_ref", "Source ref"), ("analysis_date", "Analysis date"), ("analysis_commit", "Analysis commit"), ]: @@ -1378,6 +2211,10 @@ def generate_emulator_page( lines.append(f"| {label} | [{val}]({val}) |") else: lines.append(f"| {label} | {val} |") + if profile.get("source_ref"): + lines.append( + f"| Source ref | {_source_ref_markdown(profile, profile['source_ref'])} |" + ) lines.append("") lines.append("") lines.append("") @@ -1434,7 +2271,7 @@ def generate_emulator_page( # Notes if notes: - indented = notes.replace("\n", "\n ") + indented = _admonition_body(notes) lines.extend(['???+ note "Technical notes"', f" {indented}", ""]) if not files: @@ -1693,7 +2530,9 @@ def generate_emulator_page( for scope, checks in validation.items(): details.append(f"Validation ({scope}): {', '.join(checks)}") if source_ref: - details.append(f"Source: `{source_ref}`") + details.append( + f"Source: {_source_ref_markdown(profile, source_ref)}" + ) if platform_files: plats = sorted( p for p, names in platform_files.items() if fname in names @@ -1767,6 +2606,7 @@ def generate_gap_analysis( db: dict, data_names: set[str] | None = None, registry: dict | None = None, + gap_report: dict | None = None, ) -> str: """Generate a unified gap analysis page. @@ -1820,6 +2660,10 @@ def generate_gap_analysis( "", "Unified view of BIOS verification, file provenance, and coverage gaps.", "", + "[Download gaps CSV](downloads/gaps.csv){ .md-button } " + "[Open gaps API](api/v1/gaps.json){ .md-button } " + "[All data exports](data.md){ .md-button }", + "", '