perf: read yaml through the c loader

Loading the emulator profiles is the most expensive step of every
command here, and all forty call sites used the pure-Python scanner
while libyaml sat unused in the same wheel. One shared yaml_load picks
the C loader when pyyaml ships it: the 375 profiles parse in 0.18s
instead of 1.39s, and verify --platform retroarch drops from 2.17s to
0.73s. The loader class is the same restricted one safe_load uses.

es_bios.xml was parsed straight from the network while install.py
already refused a document declaring entities; both now share one
guard. Scrapers reach it through a single path bootstrap in the
package rather than two ad-hoc ones.
This commit is contained in:
Abdessamad Derraz committed 2026-08-11 00:55:16 +02:00
1 parent ab6a3bb26d
commit 3b8f2d75d5
18 files changed
+109 -42

No files matched your search

+10 -1
View File
@@ -10,9 +10,18 @@ from __future__ import annotations
import importlib
import pkgutil
import sys
from pathlib import Path
from .base_scraper import BaseScraper
# Scrapers run both as `python -m scripts.scraper.x` and as plain scripts, and
# they share helpers with the rest of scripts/. Putting that directory on the
# path once here is what lets every submodule say `from common import ...`
# instead of carrying its own bootstrap.
_SCRIPTS_DIR = str(Path(__file__).resolve().parent.parent)
if _SCRIPTS_DIR not in sys.path:
sys.path.insert(0, _SCRIPTS_DIR)
from .base_scraper import BaseScraper # noqa: E402
_scrapers: dict[str, type] = {}
+5 -3
View File
@@ -15,6 +15,8 @@ from typing import Any
import yaml
from common import yaml_load
_MAME_RELEASE_RE = re.compile(r"^0\.\d+")
@@ -294,7 +296,7 @@ def _diff_fbneo(
def _load_yaml(path: str) -> dict[str, Any]:
with open(path, encoding="utf-8") as f:
return yaml.safe_load(f) or {}
return yaml_load(f) or {}
def _load_json(path: str) -> dict[str, Any]:
@@ -496,7 +498,7 @@ def _patch_bios_entries(text: str, files: list[dict]) -> str:
def _append_new_entries(text: str, files: list[dict], original: str) -> str:
"""Append new bios_zip entries (system=None) that aren't in the original."""
# Parse original to get existing entry names (more reliable than text search)
existing_data = yaml.safe_load(original) or {}
existing_data = yaml_load(original) or {}
existing_names = {f["name"] for f in existing_data.get("files", [])}
new_entries = []
@@ -568,7 +570,7 @@ def _backup_and_write_fbneo(path: str, data: dict, hashes: dict) -> None:
patched = _patch_core_version(original, data.get("core_version", ""))
# Identify new ROM entries by comparing parsed data keys, not text search
existing_data = yaml.safe_load(original) or {}
existing_data = yaml_load(original) or {}
existing_keys = {
(f["archive"], f["name"])
for f in existing_data.get("files", [])
+2 -1
View File
@@ -9,6 +9,7 @@ import urllib.request
from abc import ABC, abstractmethod
from dataclasses import dataclass, field
from pathlib import Path
from common import yaml_load
@dataclass
@@ -277,7 +278,7 @@ def scraper_cli(
output_path = Path(args.output)
if output_path.exists():
with open(output_path) as f:
existing = yaml.safe_load(f) or {}
existing = yaml_load(f) or {}
# Preserve existing keys not generated by the scraper.
# Only keys present in the NEW config are considered scraper-generated.
# Everything else in the existing file is preserved.
+4 -2
View File
@@ -18,6 +18,8 @@ from pathlib import Path
import yaml
from common import yaml_load
from .base_scraper import BaseScraper, BiosRequirement
PLATFORM_NAME = "batocera"
@@ -203,7 +205,7 @@ class Scraper(BaseScraper):
raise ConnectionError(
f"Failed to fetch {CONFIGGEN_DEFAULTS_URL}: {e}"
) from e
data = yaml.safe_load(raw)
data = yaml_load(raw)
cores: set[str] = set()
standalone: set[str] = set()
for system, cfg in data.items():
@@ -386,7 +388,7 @@ class Scraper(BaseScraper):
)
if existing.exists():
with open(existing) as f:
old = yaml.safe_load(f) or {}
old = yaml_load(f) or {}
batocera_version = str(old.get("version", ""))
cores, standalone = self._fetch_cores()
-2
View File
@@ -54,8 +54,6 @@ STATUS_RANK = {
"Ideal": 4,
}
GAME_DATA_SYSTEMS = {"BSX", "Doom"}
GAME_DATA_FILES = {"VEC_Minestorm.vec"}
SYSTEM_ID_MAP: dict[str, str] = {
"32X": "sega-32x",
+3 -1
View File
@@ -19,6 +19,8 @@ from typing import Any
import yaml
from common import yaml_load
from scripts.scraper._hash_merge import compute_diff, merge_fbneo_profile
from scripts.scraper.fbneo_parser import parse_fbneo_source_tree
@@ -194,7 +196,7 @@ def _find_fbneo_profiles() -> list[Path]:
if path.name.endswith(".old.yml"):
continue
try:
data = yaml.safe_load(path.read_text(encoding="utf-8"))
data = yaml_load(path.read_text(encoding="utf-8"))
except (yaml.YAMLError, OSError):
continue
if not data or not isinstance(data, dict):
+3 -4
View File
@@ -15,6 +15,8 @@ from __future__ import annotations
import xml.etree.ElementTree as ET
from dataclasses import dataclass, field
from common import parse_untrusted_xml
@dataclass
class LogiqxRom:
@@ -46,10 +48,7 @@ def parse_logiqx(content: str | bytes) -> LogiqxDat:
rejected: DAT files never define them, and expanding entities
from untrusted packs opens entity-expansion attacks.
"""
haystack = content if isinstance(content, str) else content.decode("utf-8", "replace")
if "<!ENTITY" in haystack.upper():
raise ValueError("XML entity declarations are not allowed in DAT files")
root = ET.fromstring(content)
root = parse_untrusted_xml(content, "DAT files")
dat = LogiqxDat()
header = root.find("header")
+3 -1
View File
@@ -21,6 +21,8 @@ from typing import Any
import yaml
from common import yaml_load
from ._hash_merge import compute_diff, merge_mame_profile
from .mame_parser import parse_mame_source_tree
@@ -165,7 +167,7 @@ def _find_mame_profiles() -> list[Path]:
continue
try:
with open(path, encoding="utf-8") as f:
data = yaml.safe_load(f)
data = yaml_load(f)
if not isinstance(data, dict):
continue
upstream = data.get("upstream", "")
+5 -4
View File
@@ -16,7 +16,8 @@ Recalbox verification logic:
from __future__ import annotations
import sys
import xml.etree.ElementTree as ET
from common import parse_untrusted_xml
from .base_scraper import BaseScraper, BiosRequirement
@@ -109,7 +110,7 @@ class Scraper(BaseScraper):
def _fetch_cores(self) -> list[str]:
"""Extract unique core names from es_bios.xml bios elements."""
raw = self._fetch_raw()
root = ET.fromstring(raw)
root = parse_untrusted_xml(raw, "es_bios.xml")
cores: set[str] = set()
for bios_elem in root.findall(".//system/bios"):
raw_core = bios_elem.get("core", "").strip()
@@ -128,7 +129,7 @@ class Scraper(BaseScraper):
if not self.validate_format(raw):
raise ValueError("es_bios.xml format validation failed")
root = ET.fromstring(raw)
root = parse_untrusted_xml(raw, "es_bios.xml")
requirements = []
seen = set()
@@ -177,7 +178,7 @@ class Scraper(BaseScraper):
def fetch_full_requirements(self) -> list[dict]:
"""Parse es_bios.xml preserving all Recalbox-specific fields."""
raw = self._fetch_raw()
root = ET.fromstring(raw)
root = parse_untrusted_xml(raw, "es_bios.xml")
requirements = []
for system_elem in root.findall(".//system"):
@@ -19,6 +19,8 @@ from datetime import datetime, timezone
import yaml
from common import yaml_load
from . import BaseTargetScraper
PLATFORM_NAME = "batocera"
@@ -221,7 +223,7 @@ def _parse_es_systems(text: str) -> dict[str, list[str]]:
<core_name>: {requireAnyOf: [BR2_PACKAGE_FOO]}
"""
try:
data = yaml.safe_load(text)
data = yaml_load(text)
except yaml.YAMLError:
return {}
@@ -17,6 +17,8 @@ from datetime import datetime, timezone
import yaml
from common import yaml_load
from . import BaseTargetScraper
PLATFORM_NAME = "emudeck"
@@ -134,7 +136,7 @@ class Scraper(BaseTargetScraper):
if not os.path.exists(target_path):
return []
with open(target_path) as f:
data = yaml.safe_load(f) or {}
data = yaml_load(f) or {}
# Find a target matching the architecture
for tname, tinfo in data.get("targets", {}).items():
if tinfo.get("architecture") == arch: