From 6719aa415def7ef46753d0ced0b8a3f249e67b82 Mon Sep 17 00:00:00 2001 From: Abdessamad Derraz <3028866+Abdess@users.noreply.github.com> Date: Wed, 12 Aug 2026 03:11:47 +0200 Subject: [PATCH] feat: anchor prose citations beside source refs --- scripts/profile_sync.py | 787 +++++++++++++++++++++++++++++++++++-- scripts/upstream.py | 40 ++ tests/test_profile_sync.py | 609 ++++++++++++++++++++++++++++ tests/test_upstream.py | 55 ++- 4 files changed, 1452 insertions(+), 39 deletions(-) diff --git a/scripts/profile_sync.py b/scripts/profile_sync.py index b67206a1..702096d0 100644 --- a/scripts/profile_sync.py +++ b/scripts/profile_sync.py @@ -249,6 +249,189 @@ def _anchor_tokens(entry: dict) -> list[str]: return tokens +PROSE_CITE_RE = re.compile( + r"(?P[A-Za-z0-9_][\w./+-]*\.[A-Za-z]\w*):(?P\d+(?:-\d+)?)" +) +PROSE_CONT_RE = re.compile(r",(?P\d+(?:-\d+)?)(?![\w-])(?!\.\d)") +# A spaced continuation is accepted only for a range: `x.c:55-71, 80-147` +# follows the source_ref convention, while `x.c:100, 200 files` is prose. +PROSE_CONT_SPACED_RE = re.compile(r", +(?P\d+-\d+)(?![\w-])(?!\.\d)") +# Words that read as prose before a filename. Anything else in project +# position marks the citation external, the way `munt ROMInfo.cpp` does in +# a source_ref: the profile does not declare that repository. +PROSE_LINKING_WORDS = frozenset(( + "a", "an", "and", "are", "as", "at", "before", "both", "but", "by", + "each", "for", "from", "in", "into", "is", "it", "its", "no", "not", + "of", "on", "only", "or", "over", "per", "see", "so", "than", "that", + "the", "their", "then", "these", "this", "those", "to", "under", + "upstream", "uses", "via", "was", "when", "where", "while", "with", +)) +_PROJECT_WORD_RE = re.compile(r"[A-Za-z0-9][\w-]*") + + +@dataclass(frozen=True) +class Citation: + """One citation carried by a profile, wherever it is written. + + A structured citation lives under a source_ref key and keeps its entry + for the values it declares. A prose citation is a path:line run found + inside any other string scalar; its spans locate the path and every + range token inside the scalar line, so a recale can move the numbers + and leave the sentence around them alone. + """ + + field: str + kind: str + ref: str + entry: dict | None = None + holder: object = None + key: object = None + label: str = "" + line: int = 0 + path_span: tuple[int, int] = (0, 0) + spans: tuple = () + parts: tuple = () + + +def _walk_document(node, where: str = ""): + """Every citation carrier of the document, with its holder and field path. + + Yields ("ref", field, holder, key, value) for each source_ref key and + ("str", field, holder, key, value) for every other string scalar. One + walker builds every field path, so the paths that name a scalar today + still name it when another pass looks it up at another revision. + """ + if isinstance(node, dict): + for key, value in node.items(): + spot = f"{where}.{key}" if where else str(key) + if key == "source_ref": + yield "ref", spot, node, key, value + elif isinstance(value, str): + yield "str", spot, node, key, value + else: + yield from _walk_document(value, spot) + elif isinstance(node, list): + for index, item in enumerate(node): + segment = item.get("name") if isinstance(item, dict) else None + spot = f"{where}[{segment if segment is not None else index}]" + if isinstance(item, str): + yield "str", spot, node, index, item + else: + yield from _walk_document(item, spot) + + +def _prose_external(line: str, path_start: int, known=frozenset()) -> bool: + """True when the word before the path names an undeclared project. + + `munt ROMInfo.cpp:206-213` in prose follows the source_ref convention: + the leading word is the project, and no declared repository can confirm + the reference. The word counts as a project only when a list delimiter + or the line start precedes it, it reads as an identifier, it is not a + linking word, and it does not name a declared repository: in + `boot, in libretro.cpp:12` the candidate is prose, and in + `mednafen src/lynx/rom.cpp:55` it is the declared upstream itself. + """ + before = line[:path_start].rstrip() + if not before or before[-1] in ",:(": + return False + word_start = max(before.rfind(" "), before.rfind("\t"), before.rfind("(")) + 1 + word = before[word_start:] + if not _PROJECT_WORD_RE.fullmatch(word): + return False + if word.lower() in PROSE_LINKING_WORDS or word.lower() in known: + return False + lead = before[:word_start].rstrip() + return not lead or lead[-1] in ",:(" + + +def _prose_runs(text: str, known=frozenset()): + """Citation runs inside one prose scalar. + + A run is a path:range with its attached continuation ranges, written + without a space after the comma the way the corpus writes them: + `main.c:112,1624-1633` cites two ranges of one file, while `x.c:100, 200` + is a citation followed by prose. A token carrying :// is a URL, whose + host:port shape would otherwise pass for a citation. + """ + for number, line in enumerate(text.splitlines()): + consumed = 0 + for match in PROSE_CITE_RE.finditer(line): + if match.start() < consumed: + continue + token_start = max(line.rfind(" ", 0, match.start()), + line.rfind("\t", 0, match.start())) + 1 + token = line[token_start:].split()[0] if line[token_start:] else "" + if "://" in token: + continue + path = match.group("path") + path_start = match.start("path") + # A dotfile citation opens its token with the dot: `.gitlab-ci.yml` + # after a delimiter is the filename, while the dot of `boot.cfg` + # inside a word belongs to the word before it. + if ( + path_start > 0 + and line[path_start - 1] == "." + and (path_start == 1 or line[path_start - 2] in " \t(") + ): + path_start -= 1 + path = "." + path + if _prose_external(line, path_start, known): + consumed = match.end() + continue + path_span = (path_start, match.end("path")) + spans = [(match.start("range"), match.end("range"))] + end = match.end() + while True: + cont = PROSE_CONT_RE.match(line, end) + if cont is None: + cont = PROSE_CONT_SPACED_RE.match(line, end) + if cont is None: + break + spans.append((cont.start("range"), cont.end("range"))) + end = cont.end() + consumed = end + parts = [] + for lo, hi in spans: + rng = line[lo:hi] + a, _, b = rng.partition("-") + parts.append(RefPart(path, int(a), int(b or a), rng)) + yield ( + number, path_span, tuple(spans), tuple(parts), + line[path_start:end], + ) + + +def collect_citations(document: dict) -> list[Citation]: + """Every citation the document carries, in document order. + + A source_ref key anywhere in the document is a structured citation: + files[] holds most of them, data_directories[] carries some too. Every + other string scalar is scanned for prose runs. A citation is a property + of the text, not of a field name: anything this walk misses is a + location profile_sync cannot keep honest. + """ + known = frozenset( + repo.name.lower() + for _, _, url in declared_repositories(document) + if (repo := upstream.parse_repo(url)) is not None + ) + citations: list[Citation] = [] + for kind, field, holder, key, value in _walk_document(document): + if kind == "ref": + for label, ref in source_ref_values(value): + citations.append(Citation( + field=field, kind="ref", ref=ref, entry=holder, + holder=holder, key=key, label=label, + )) + continue + for line, path_span, spans, parts, run in _prose_runs(value, known): + citations.append(Citation( + field=field, kind="prose", ref=run, holder=holder, key=key, + line=line, path_span=path_span, spans=spans, parts=parts, + )) + return citations + + def worst_status(statuses) -> str: """Severity of an entry is the worst severity among its parts.""" worst = "ANCHORED" @@ -629,6 +812,8 @@ class EntryReport: source_ref: str status: str parts: list[PartResult] + kind: str = "ref" + field: str = "" @dataclass @@ -878,17 +1063,26 @@ def build_report( """Confront one profile with its upstream.""" report = ProfileReport(name=name, entries=[], counts={}) - refs = [ - ( - entry.get("name", "") + (f" [{label}]" if label else ""), - value, - _anchor_tokens(entry), - entry_hashes(entry), - ) - for entry in (profile.get("files") or []) - if isinstance(entry, dict) and entry.get("source_ref") - for label, value in source_ref_values(entry.get("source_ref")) - ] + # One citation surface for the whole document. A prose citation carries + # no declared value, so it anchors on content alone: the nudge and + # relocation heuristics stay off rather than guess from a profile-wide + # token pool. + refs = [] + for citation in collect_citations(profile): + if citation.kind == "ref": + entry = citation.entry + display = str(entry.get("name") or "") or citation.field + if citation.label: + display += f" [{citation.label}]" + refs.append(( + display, citation.ref, _anchor_tokens(entry), + entry_hashes(entry), split_source_ref(citation.ref), citation, + )) + else: + refs.append(( + citation.field, citation.ref, [], [], + list(citation.parts), citation, + )) if select_repo(profile) is None: declared = str(profile.get("source") or profile.get("upstream") or "") @@ -937,6 +1131,17 @@ def build_report( head, _, tail = path.partition("/") if tail and any(head == v.repo.name for v in views): candidates.append(tail) + if "/" not in path: + # A bare filename, the way prose cites files. The HEAD tree of + # each repository resolves it when exactly one path carries it. + for view in views: + _, tree = _context_for(view) + matches = [ + p for p in tree or [] + if posixpath.basename(p) == path + ] + if len(matches) == 1 and matches[0] not in candidates: + candidates.append(matches[0]) def score(lines) -> int: """How well a repository's cited line matches what the ref means. @@ -1090,28 +1295,30 @@ def build_report( self_check = bool(report.pinned_tag) or primary.pin == primary.head if self_check: staged = [ - (entry_name, ref, reconcile_self_check([ + (entry_name, ref, citation, reconcile_self_check([ verify_at_pin( part, fetch(PIN, part.path, part.start, tokens), tokens, hashes ) - for part in split_source_ref(ref) + for part in ref_parts ])) - for entry_name, ref, tokens, hashes in refs + for entry_name, ref, tokens, hashes, ref_parts, citation in refs ] else: staged = [ - (entry_name, ref, [ + (entry_name, ref, citation, [ anchor_across_views(part, tokens) - for part in split_source_ref(ref) + for part in ref_parts ]) - for entry_name, ref, tokens, hashes in refs + for entry_name, ref, tokens, hashes, ref_parts, citation in refs ] - shifts = dominant_shifts([p for _, _, parts in staged for p in parts]) - for entry_name, ref, parts in staged: + shifts = dominant_shifts([p for _, _, _, parts in staged for p in parts]) + for entry_name, ref, citation, parts in staged: parts = [resolve_by_shift(p, shifts) for p in parts] status = worst_status([p.status for p in parts]) - report.entries.append(EntryReport(entry_name, ref, status, parts)) + report.entries.append(EntryReport( + entry_name, ref, status, parts, citation.kind, citation.field + )) report.counts[status] = report.counts.get(status, 0) + 1 return report @@ -1507,6 +1714,168 @@ def _is_rewritable(ref: str) -> bool: ) +def _match_citation( + citations: list[Citation], pos: int, entry: EntryReport +) -> tuple[Citation | None, int]: + """First document citation at or after pos carrying the entry's ref. + + The pairing is by content and order, never by position alone: a + mode-keyed source_ref yields two report entries for one document line, + which is exactly the desynchronisation a positional pairing trips on. + """ + for index in range(pos, len(citations)): + citation = citations[index] + if citation.kind != entry.kind or citation.ref != entry.source_ref: + continue + if entry.kind == "prose" and entry.field and citation.field != entry.field: + continue + return citation, index + 1 + return None, pos + + +def _prose_moves( + entry: EntryReport, statuses +) -> dict[int, tuple[int, int, str | None]]: + """Recale targets per part of one prose run, empty when unsafe. + + A run cites one file: parts that disagree on where that file went, or a + rename that would leave untouched ranges pointing into the old file, + cannot be written without guessing. + """ + moves: dict[int, tuple[int, int, str | None]] = {} + renames = set() + for index, part in enumerate(entry.parts): + if part.status not in statuses or part.start is None: + continue + same_place = (part.start, part.end) == (part.part.start, part.part.end) + if same_place and not part.new_path: + continue + moves[index] = (part.start, part.end or part.start, part.new_path) + if part.new_path: + renames.add(part.new_path) + if len(renames) > 1: + return {} + if renames and len(moves) != len(entry.parts): + return {} + return moves + + +def _run_after_moves(parts, moves: dict[int, tuple[int, int, str | None]]) -> str: + """The prose run as it reads once its moved ranges are rewritten.""" + new_path = next( + (move[2] for move in moves.values() if move[2]), None + ) + ranges = [] + for index, part in enumerate(parts): + if index in moves: + start, end, _ = moves[index] + ranges.append(f"{start}" if end == start else f"{start}-{end}") + else: + ranges.append(part.raw) + return f"{new_path or parts[0].path}:{','.join(ranges)}" + + +def _apply_prose_edits( + text: str, jobs: list[tuple[Citation, dict]] +) -> tuple[str, list[str], list[str]]: + """Rewrite prose citation tokens in place, nothing else. + + Each run is located as the unique physical line carrying the scalar line + it sits on. Two candidate lines, or none, means the write cannot be + proved right, so the run is left alone and reported. All replacements of + one line are applied together, right to left, so the spans measured on + the original text stay valid. + """ + lines = text.splitlines() + trailing = "\n" if text.endswith("\n") else "" + file_edits: dict[int, list[tuple[int, int, str]]] = {} + value_edits: dict[tuple[int, object], dict[int, list]] = {} + holders: dict[tuple[int, object], tuple] = {} + applied: list[str] = [] + left: list[str] = [] + + for citation, moves in jobs: + value = str(citation.holder[citation.key]) + value_lines = value.splitlines() + if citation.line >= len(value_lines): + left.append(f"{citation.field}: {citation.ref} (scalar changed)") + continue + replacements = [] + for index, (start, end, _) in moves.items(): + lo, hi = citation.spans[index] + replacements.append( + (lo, hi, f"{start}" if end == start else f"{start}-{end}") + ) + new_path = next((m[2] for m in moves.values() if m[2]), None) + if new_path: + replacements.append((*citation.path_span, new_path)) + target = value_lines[citation.line] + hits = [i for i, line in enumerate(lines) if target and target in line] + if len(hits) == 1: + offset = lines[hits[0]].index(target) + file_edits.setdefault(hits[0], []).extend( + (offset + lo, offset + hi, new) for lo, hi, new in replacements + ) + else: + # A folded scalar has no physical line carrying its logical one. + # The run itself is a single word, which folding never breaks: + # when the whole file carries it exactly once, that occurrence + # is the citation, and the run is rewritten as one token. The + # match is delimited, or `x.c:481` would also hit inside a + # neighbouring `x.c:481-482`. + pattern = re.compile( + rf"(? " + f"{_run_after_moves(citation.parts, moves)}" + ) + + for number, replacements in file_edits.items(): + lines[number] = _replace_spans(lines[number], replacements) + for slot, edits in value_edits.items(): + holder, key = holders[slot] + value = str(holder[key]) + value_lines = value.splitlines() + for line_number, replacements in edits.items(): + value_lines[line_number] = _replace_spans( + value_lines[line_number], replacements + ) + holder[key] = "\n".join(value_lines) + ( + "\n" if value.endswith("\n") else "" + ) + return "\n".join(lines) + trailing, applied, left + + +def _replace_spans(line: str, replacements: list[tuple[int, int, str]]) -> str: + """Apply span replacements measured on the original line.""" + for lo, hi, new in sorted(replacements, reverse=True): + line = line[:lo] + new + line[hi:] + return line + + def rebase_refs( path: Path, report: ProfileReport, accept_changed: bool = False ) -> list[str]: @@ -1516,7 +1885,8 @@ def rebase_refs( are written against, so moving some refs to HEAD while others still describe the pinned revision would leave the profile self-contradictory, and the next comparison would read the pinned revision at line numbers - that only make sense at HEAD. + that only make sense at HEAD. Prose citations move with everything else: + only their located tokens are rewritten, the sentence stays. """ if report.pinned_tag: return [] @@ -1526,18 +1896,30 @@ def rebase_refs( return [] text = path.read_text(encoding="utf-8") document = yaml.safe_load(text) - carriers = [ - entry - for entry in (document.get("files") or []) - if isinstance(entry, dict) and entry.get("source_ref") - ] + citations = collect_citations(document) applied: list[str] = [] cursor = 0 + pos = 0 + prose_jobs: list[tuple[Citation, dict]] = [] - for position, entry in enumerate(report.entries): - if not _is_rewritable(entry.source_ref) or entry.name.endswith("]"): - # Rewriting would drop the author's annotations, or the refs live - # under a mode key rather than on a source_ref line. + for entry in report.entries: + citation, pos = _match_citation(citations, pos, entry) + if citation is None: + continue + if entry.kind == "prose": + moves = _prose_moves(entry, statuses) + if moves: + prose_jobs.append((citation, moves)) + continue + if citation.label or entry.name.endswith("]"): + # The ref lives under a mode key, not on a source_ref line. The + # entry name check also holds when the report and the document + # disagree: a stale report must not touch a line it never read. + continue + if not _is_rewritable(entry.source_ref): + # Rewriting would drop the author's annotations. Still consume + # the line so a later entry carrying the same ref cannot match it. + cursor = find_field_line(text, "source_ref", entry.source_ref, cursor) + 1 continue rendered = [] touched = False @@ -1550,8 +1932,6 @@ def rebase_refs( else: rendered.append(_original_part(part)) if not touched: - # Still consume this entry's line so a later entry carrying the - # same ref cannot match it. cursor = find_field_line(text, "source_ref", entry.source_ref, cursor) + 1 continue new_ref = ", ".join(rendered) @@ -1559,24 +1939,35 @@ def rebase_refs( text, "source_ref", entry.source_ref, new_ref, cursor ) cursor = index + 1 - carriers[position]["source_ref"] = new_ref + citation.holder[citation.key] = new_ref applied.append(f"{entry.source_ref} -> {new_ref}") + text, prose_applied, _ = _apply_prose_edits(text, prose_jobs) + applied.extend(prose_applied) if applied: apply_edit(path, text, document) return applied -def pending_recale(report: ProfileReport, accept_changed: bool = False) -> int: - """Parts that ought to move but sit in a ref the writer will not touch. +def pending_recale( + report: ProfileReport, accept_changed: bool = False, text: str | None = None +) -> int: + """Parts that ought to move but the writer has not moved. An annotated ref cannot be regenerated without losing its prose, so its parts stay on the pinned line numbers. Advancing the pin while they do would leave the profile describing two revisions at once. + + Prose runs are judged on the file as it stands: a run that should move + counts as pending until the document actually carries its recaled form, + so the pin cannot advance over prose that still describes the old + revision, whatever the reason it was not rewritten. """ movable = REBASE_STATUSES + (("CHANGED",) if accept_changed else ()) pending = 0 for entry in report.entries or []: + if entry.kind == "prose": + continue if _is_rewritable(entry.source_ref) and not entry.name.endswith("]"): continue pending += sum( @@ -1586,6 +1977,41 @@ def pending_recale(report: ProfileReport, accept_changed: bool = False) -> int: and part.start is not None and _rendered_part(part) != _original_part(part) ) + if text is None: + return pending + + document = yaml.safe_load(text) + pool: dict[tuple[str, str], int] = {} + for citation in collect_citations(document): + if citation.kind == "prose": + slot = (citation.field, citation.ref) + pool[slot] = pool.get(slot, 0) + 1 + for entry in report.entries or []: + if entry.kind != "prose": + continue + moves = _prose_moves(entry, movable) + if not moves: + # Either nothing has to move, or the run cannot be written + # without guessing; the second case still blocks the pin. + if any( + part.status in movable + and part.start is not None + and ( + (part.start, part.end) != (part.part.start, part.part.end) + or part.new_path + ) + for part in entry.parts + ): + pending += 1 + continue + rendered = _run_after_moves([p.part for p in entry.parts], moves) + if rendered == entry.source_ref: + continue + slot = (entry.field, rendered) + if pool.get(slot, 0) > 0: + pool[slot] -= 1 + continue + pending += 1 return pending @@ -1600,9 +2026,9 @@ def bump_commit( ) if any((report.counts or {}).get(s) for s in blocking): return False - if pending_recale(report, accept_changed): - return False text = path.read_text(encoding="utf-8") + if pending_recale(report, accept_changed, text): + return False document = yaml.safe_load(text) expected = dict(document) expected["source_commit"] = report.head @@ -1618,6 +2044,233 @@ def bump_commit( return True +def _git_history(path: Path) -> list[str]: + """Commits touching the profile, newest first, empty outside a repo.""" + result = subprocess.run( + ["git", "log", "--format=%H", "--", path.name], + cwd=path.parent, capture_output=True, text=True, check=False, + ) + if result.returncode != 0: + return [] + return result.stdout.split() + + +def _git_file_at(path: Path, sha: str) -> dict | None: + """The profile as it was at one commit, None when unreadable.""" + result = subprocess.run( + ["git", "show", f"{sha}:./{path.name}"], + cwd=path.parent, capture_output=True, text=True, check=False, + ) + if result.returncode != 0: + return None + try: + document = yaml.safe_load(result.stdout) + except yaml.YAMLError: + return None + return document if isinstance(document, dict) else None + + +def _scalar_values(document: dict) -> dict[str, str]: + """String scalars of the document, keyed by field path.""" + return { + field: value + for kind, field, _, _, value in _walk_document(document) + if kind == "str" + } + + +def _writing_pin( + revisions: list[tuple[str, dict]], intro_sha: str, field: str +) -> str | None: + """Pin recorded at the introducing commit, or first recorded after it. + + The backfill resolved missing pins from profiled_date, the writing time + of the profile, so the first value ever recorded stands for the scalars + that predate it. + """ + index = next( + (i for i, (sha, _) in enumerate(revisions) if sha == intro_sha), None + ) + if index is None: + return None + for position in range(index, -1, -1): + value = revisions[position][1].get(field) + if isinstance(value, str) and value: + return value + return None + + +def _resolve_at( + repo, sha: str, path: str, cache_dir: str, offline: bool +) -> tuple[str, object]: + """("found", (path, lines)) or ("unclear"|"missing", reason). + + Prose often cites a bare filename; the tree at the revision resolves it + when exactly one path carries that name. An unreadable or truncated + tree, or several paths carrying the name, proves nothing: those come + back "unclear", never "missing", because a verdict of absence taken on + a tree that could not be read would rot the citation with confidence. + """ + candidates = [path] + head, _, tail = path.partition("/") + if tail and head == repo.name: + candidates.append(tail) + for candidate in candidates: + lines = upstream.fetch_file(repo, sha, candidate, cache_dir, offline) + if lines is not None: + return "found", (candidate, lines) + if "/" not in path: + tree, truncated = upstream.list_tree(repo, sha, cache_dir, offline) + matches = [p for p in tree or [] if posixpath.basename(p) == path] + if len(matches) == 1: + lines = upstream.fetch_file( + repo, sha, matches[0], cache_dir, offline + ) + if lines is not None: + return "found", (matches[0], lines) + elif len(matches) > 1: + return "unclear", f"{len(matches)} paths carry the name" + elif truncated or not tree: + return "unclear", "tree unreadable at that revision" + return "missing", None + + +def _realign_part( + part: RefPart, pairs, cache_dir: str, offline: bool +) -> tuple[str, object] | None: + """One range, anchored from its writing revision to the current pin.""" + unclear = None + for repo, written, current in pairs: + state, payload = _resolve_at(repo, written, part.path, cache_dir, offline) + if state == "unclear": + unclear = payload + continue + if state == "missing": + continue + actual, old_lines = payload + new_lines = upstream.fetch_file( + repo, current, actual, cache_dir, offline + ) + if new_lines is None: + return "skip", f"{part.path} absent at the current pin" + anchored = anchor_block( + old_lines, new_lines, part.start, part.end or part.start + ) + if anchored.status == "ANCHORED": + return None + if anchored.status == "SHIFTED": + return "ok", (anchored.start, anchored.end) + return "skip", f"{part.path}:{part.start} {anchored.status.lower()}" + if unclear: + return "skip", f"{part.path}: {unclear}" + return "skip", f"{part.path} absent at the writing revision" + + +def realign_prose( + path: Path, cache_dir: str, offline: bool = False, dry_run: bool = False +) -> list[str]: + """Recale prose runs from the pin their text was written at. + + A prose run written under an older pin describes that revision, and + anchoring it from the current pin would faithfully track the wrong + content: the cited line holds someone else's code there. The writing pin + is read from the profile's own history, at the commit that introduced + the scalar's current text. + """ + text = path.read_text(encoding="utf-8") + document = yaml.safe_load(text) + if not isinstance(document, dict): + return [] + citations = [c for c in collect_citations(document) if c.kind == "prose"] + if not citations: + return [] + repos = [ + (field, upstream.parse_repo(str(document.get(field) or ""))) + for field in ("source", "upstream") + ] + repos = [(field, repo) for field, repo in repos if repo is not None] + if not repos: + return [] + history = _git_history(path) + if not history: + return [] + + current_values = _scalar_values(document) + fields = {citation.field for citation in citations} + alive = dict.fromkeys(fields, True) + intro: dict[str, str | None] = dict.fromkeys(fields) + revisions: list[tuple[str, dict]] = [] + for sha in history: + if not any(alive.values()): + break + past = _git_file_at(path, sha) + if past is None: + break + revisions.append((sha, past)) + values = _scalar_values(past) + for field in fields: + if not alive[field]: + continue + if values.get(field) == current_values.get(field): + intro[field] = sha + else: + alive[field] = False + + messages: list[str] = [] + jobs: list[tuple[Citation, dict]] = [] + for citation in citations: + intro_sha = intro.get(citation.field) + if intro_sha is None: + # The scalar is not committed yet: written now, under this pin. + continue + pairs = [] + for pin_field, repo in repos: + current = document.get(f"{pin_field}_commit") + if not isinstance(current, str) or not current: + continue + written = _writing_pin(revisions, intro_sha, f"{pin_field}_commit") + if written and written != current: + pairs.append((repo, written, current)) + if not pairs: + continue + moves: dict[int, tuple[int, int, str | None]] = {} + blocked: list[str] = [] + for index, part in enumerate(citation.parts): + outcome = _realign_part(part, pairs, cache_dir, offline) + if outcome is None: + continue + state, payload = outcome + if state == "ok": + start, end = payload + moves[index] = (start, end, None) + else: + blocked.append(payload) + if blocked: + # Half a run must not move: the untouched ranges would read as + # already realigned when they were never even located. + messages.extend( + f"read again: {citation.field}: {citation.ref} ({reason})" + for reason in blocked + ) + continue + if moves: + jobs.append((citation, moves)) + + if dry_run: + messages.extend( + f"would recale {citation.field}: {citation.ref} -> " + f"{_run_after_moves(citation.parts, moves)}" + for citation, moves in jobs + ) + return messages + new_text, applied, left = _apply_prose_edits(text, jobs) + messages.extend(applied) + messages.extend(f"left in place: {item}" for item in left) + if applied: + apply_edit(path, new_text, document) + return messages + + def emulators_dir_is_dirty(emulators_dir: str) -> bool: """True when the profile directory carries uncommitted changes.""" result = subprocess.run( @@ -1657,6 +2310,11 @@ def build_parser() -> argparse.ArgumentParser: parser.add_argument("--backfill-commits", action="store_true") parser.add_argument("--rebase-refs", action="store_true") parser.add_argument("--bump-commit", action="store_true") + parser.add_argument( + "--realign-prose", + action="store_true", + help="recale prose citations from the pin their text was written at", + ) parser.add_argument( "--accept-changed", action="store_true", @@ -1864,11 +2522,46 @@ def main() -> None: file=sys.stderr, ) raise SystemExit(1) + if args.realign_prose and ( + args.rebase_refs or args.bump_commit or args.backfill_commits + ): + print( + "--realign-prose runs alone: the report has to be rebuilt on the " + "realigned text before any other write.", + file=sys.stderr, + ) + raise SystemExit(1) + if args.realign_prose: + ignored = [ + flag for flag, on in ( + ("--json", args.as_json), + ("--markdown", args.markdown), + ("--fetch-plan", args.fetch_plan), + ("--triage", args.triage), + ("--changed-only", args.changed_only), + ("--check-version", args.check_version), + ("--detect-new-files", args.detect_new_files), + ("--watch-hashes", args.watch_hashes), + ("--full-diff", args.full_diff), + ("--tree-diff", args.tree_diff), + ("--accept-changed", args.accept_changed), + ) if on + ] + if ignored: + # A mode applies a flag or refuses it, never swallows it. + print( + f"--realign-prose does not apply {', '.join(ignored)}", + file=sys.stderr, + ) + raise SystemExit(1) profiles = load_emulator_profiles(args.emulators_dir, skip_aliases=False) selected = select_profiles(profiles, args) _check_quota(len(selected), args.offline) - writes = args.backfill_commits or args.rebase_refs or args.bump_commit + writes = ( + args.backfill_commits or args.rebase_refs or args.bump_commit + or args.realign_prose + ) if ( writes and not args.dry_run @@ -1882,6 +2575,24 @@ def main() -> None: ) raise SystemExit(1) + if args.realign_prose: + for name in selected: + profile_path = Path(args.emulators_dir) / f"{name}.yml" + if not profile_path.is_file(): + continue + try: + for line in realign_prose( + profile_path, args.cache_dir, args.offline, args.dry_run + ): + print(f"{name}: {line}") + except upstream.RateLimitError: + raise + except upstream.UpstreamError as exc: + print(f"{name}: {exc}", file=sys.stderr) + except YamlWriteError as exc: + print(f"{name}: write refused: {exc}", file=sys.stderr) + return + reports = [] for name, profile in selected.items(): try: diff --git a/scripts/upstream.py b/scripts/upstream.py index f7eeee57..ec047995 100644 --- a/scripts/upstream.py +++ b/scripts/upstream.py @@ -50,6 +50,11 @@ _HOSTS: dict[str, tuple[str, str, str]] = { "https://git.eden-emu.dev/api/v1", "https://git.eden-emu.dev", ), + "git.ryujinx.app": ( + "forgejo", + "https://git.ryujinx.app/api/v1", + "https://git.ryujinx.app", + ), } @@ -474,10 +479,45 @@ def _tree_url(repo: Repo, sha: str) -> str | None: return f"{repo.api_base}/repos/{repo.slug}/git/trees/{sha}?recursive=1" +TREE_PAGE_SIZE = 100 +MAX_TREE_PAGES = 50 + + +def _gitlab_tree( + repo: Repo, sha: str, cache_dir: str, offline: bool +) -> tuple[list[str], bool]: + """GitLab serves its tree in pages; each page is cached on its own URL. + + A page that cannot be read mid-walk reports the listing as truncated: + the paths already collected are real, but their absence proves nothing. + """ + paths: list[str] = [] + for page in range(1, MAX_TREE_PAGES + 1): + url = ( + f"{repo.api_base}/projects/{_project(repo)}/repository/tree" + f"?ref={sha}&recursive=true&per_page={TREE_PAGE_SIZE}&page={page}" + ) + payload = _api(url, cache_dir, offline) + if not isinstance(payload, list): + return paths, True + paths.extend( + entry["path"] + for entry in payload + if isinstance(entry, dict) + and entry.get("type") == "blob" + and entry.get("path") + ) + if len(payload) < TREE_PAGE_SIZE: + return paths, False + return paths, True + + def list_tree( repo: Repo, sha: str, cache_dir: str, offline: bool = False ) -> tuple[list[str], bool]: """Every blob path at one revision, and whether the forge truncated it.""" + if repo.family == "gitlab": + return _gitlab_tree(repo, sha, cache_dir, offline) url = _tree_url(repo, sha) if url is None: return [], True diff --git a/tests/test_profile_sync.py b/tests/test_profile_sync.py index 49731d4c..a9bd567f 100644 --- a/tests/test_profile_sync.py +++ b/tests/test_profile_sync.py @@ -6,6 +6,7 @@ import contextlib import io import json import os +import subprocess import sys import tempfile import unittest @@ -31,6 +32,7 @@ from profile_sync import ( build_report, bump_commit, check_version, + collect_citations, collect_tokens, declared_hashes, declared_names, @@ -1965,5 +1967,612 @@ class TestDriftScore(unittest.TestCase): self.assertEqual(drift_score(report, None, 0), 0) +class TestCollectCitations(unittest.TestCase): + def test_files_refs_in_document_order(self): + document = { + "files": [ + {"name": "a.bin", "source_ref": "a.c:1"}, + {"name": "b.bin"}, + {"name": "c.bin", "source_ref": "c.c:3"}, + ], + } + refs = [c for c in collect_citations(document) if c.kind == "ref"] + self.assertEqual([c.ref for c in refs], ["a.c:1", "c.c:3"]) + self.assertEqual(refs[0].field, "files[a.bin].source_ref") + self.assertIs(refs[0].entry, document["files"][0]) + + def test_data_directories_refs_are_citations_too(self): + document = { + "data_directories": [ + {"key": "sys", "source_ref": "loader.cpp:44"}, + ], + } + refs = collect_citations(document) + self.assertEqual(len(refs), 1) + self.assertEqual(refs[0].kind, "ref") + self.assertEqual(refs[0].ref, "loader.cpp:44") + + def test_mode_keyed_ref_keeps_its_label(self): + document = { + "files": [ + { + "name": "a.bin", + "source_ref": {"standalone": "a.c:1", "libretro": "b.c:2"}, + }, + ], + } + refs = collect_citations(document) + self.assertEqual( + [(c.label, c.ref) for c in refs], + [("standalone", "a.c:1"), ("libretro", "b.c:2")], + ) + + def test_notes_citation_with_continuation_run(self): + document = { + "notes": "The reset path (main.c:112,1624-1633) runs first.\n", + } + prose = collect_citations(document) + self.assertEqual(len(prose), 1) + self.assertEqual(prose[0].kind, "prose") + self.assertEqual(prose[0].field, "notes") + self.assertEqual(prose[0].ref, "main.c:112,1624-1633") + self.assertEqual( + [(p.path, p.start, p.end) for p in prose[0].parts], + [("main.c", 112, 112), ("main.c", 1624, 1633)], + ) + + def test_spans_locate_the_tokens_on_the_line(self): + document = {"notes": "See main.c:112,1624-1633 for the boot path.\n"} + citation = collect_citations(document)[0] + line = document["notes"].splitlines()[citation.line] + self.assertEqual(line[slice(*citation.path_span)], "main.c") + self.assertEqual( + [line[lo:hi] for lo, hi in citation.spans], ["112", "1624-1633"] + ) + + def test_comma_space_is_prose_not_a_continuation(self): + document = {"notes": "x.c:100, 200 files are read.\n"} + citation = collect_citations(document)[0] + self.assertEqual(citation.ref, "x.c:100") + + def test_a_url_port_is_not_a_citation(self): + document = {"notes": "Served from https://example.com:8080/path.\n"} + self.assertEqual(collect_citations(document), []) + + def test_version_and_ratio_text_is_not_a_citation(self): + document = {"notes": "Since v1.6.0:123 the ratio 16:9 applies.\n"} + self.assertEqual(collect_citations(document), []) + + def test_nested_note_and_exclusion_note_are_scanned(self): + document = { + "exclusion_note": "Loaded by hook.c:655 at boot.", + "files": [ + {"name": "a.bin", "note": "Table at data.rs:12-20."}, + ], + } + fields = {c.field for c in collect_citations(document)} + self.assertEqual(fields, {"exclusion_note", "files[a.bin].note"}) + + def test_string_inside_a_list_is_scanned(self): + document = { + "analysis": {"entries": ["checked in geo.c:234-243"]}, + } + citation = collect_citations(document)[0] + self.assertEqual(citation.field, "analysis.entries[0]") + self.assertEqual(citation.ref, "geo.c:234-243") + + def test_whole_value_pseudo_ref_is_a_prose_citation(self): + document = {"analysis": {"upstream_ref": "src/a.c:12-20"}} + citation = collect_citations(document)[0] + self.assertEqual(citation.kind, "prose") + self.assertEqual(citation.ref, "src/a.c:12-20") + + def test_two_runs_on_one_line_stay_separate(self): + document = {"notes": "hook.c:655-709 and hook.c:874-928 serve it.\n"} + refs = [c.ref for c in collect_citations(document)] + self.assertEqual(refs, ["hook.c:655-709", "hook.c:874-928"]) + + def test_trailing_comma_stays_prose(self): + document = {"notes": "boot (pif.c:252-260,\nthen the response).\n"} + citation = collect_citations(document)[0] + self.assertEqual(citation.ref, "pif.c:252-260") + + def test_continuation_ending_a_sentence_is_kept(self): + document = {"notes": "It reads b.c:2,8-9. Then it boots.\n"} + citation = collect_citations(document)[0] + self.assertEqual(citation.ref, "b.c:2,8-9") + + def test_decimal_number_is_not_a_continuation(self): + document = {"notes": "Set at x.c:1,2.5 percent of the frame.\n"} + citation = collect_citations(document)[0] + self.assertEqual(citation.ref, "x.c:1") + + def test_dotfile_keeps_its_leading_dot(self): + document = {"notes": "Built by CI (.gitlab-ci.yml:306) nightly.\n"} + citation = collect_citations(document)[0] + self.assertEqual(citation.ref, ".gitlab-ci.yml:306") + self.assertEqual(citation.parts[0].path, ".gitlab-ci.yml") + line = document["notes"].splitlines()[0] + self.assertEqual(line[slice(*citation.path_span)], ".gitlab-ci.yml") + + def test_sentence_dot_is_not_a_dotfile(self): + document = {"notes": "It runs on boot.data.c:12 is the table.\n"} + citation = collect_citations(document)[0] + self.assertEqual(citation.parts[0].path, "boot.data.c") + + def test_external_project_prefix_is_skipped(self): + document = { + "notes": "ref: mt32_model.cpp:36-97, munt ROMInfo.cpp:206-213\n" + } + refs = [c.ref for c in collect_citations(document)] + self.assertEqual(refs, ["mt32_model.cpp:36-97"]) + + def test_linking_word_is_not_a_project(self): + document = {"notes": "It boots, in libretro.cpp:12 as shown.\n"} + refs = [c.ref for c in collect_citations(document)] + self.assertEqual(refs, ["libretro.cpp:12"]) + + def test_spaced_range_continuation_is_part_of_the_run(self): + document = { + "notes": "ref: soundcanvas.cpp:55-71, 80-147, Ext plugin.cpp:23\n" + } + citations = collect_citations(document) + self.assertEqual(len(citations), 1) + self.assertEqual(citations[0].ref, "soundcanvas.cpp:55-71, 80-147") + self.assertEqual( + [(p.start, p.end) for p in citations[0].parts], + [(55, 71), (80, 147)], + ) + + def test_spaced_lone_number_stays_prose(self): + document = {"notes": "It reads x.c:100, 200 files at boot.\n"} + citation = collect_citations(document)[0] + self.assertEqual(citation.ref, "x.c:100") + + def test_declared_repository_name_is_not_external(self): + document = { + "upstream": "https://github.com/o/mednafen", + "notes": "ref: geo.c:1-2, mednafen src/lynx/rom.cpp:55-74\n", + } + refs = [c.ref for c in collect_citations(document)] + self.assertEqual(refs, ["geo.c:1-2", "src/lynx/rom.cpp:55-74"]) + + def test_punctuation_before_the_path_is_not_a_project(self): + document = {"notes": "loads it (-framefile); singe_utils.cpp:35-42\n"} + refs = [c.ref for c in collect_citations(document)] + self.assertEqual(refs, ["singe_utils.cpp:35-42"]) + + def test_the_word_upstream_is_prose(self): + document = {"notes": "same shape (upstream retro_host.c:196-198).\n"} + refs = [c.ref for c in collect_citations(document)] + self.assertEqual(refs, ["retro_host.c:196-198"]) + + +class TestCitationSurfaceGuard(unittest.TestCase): + def test_no_consumer_rederives_source_ref(self): + """collect_citations is the only reader of source_ref values. + + A consumer keying on the field name reopens the gap this closes: + citations outside its list rot while the pin advances. The walker + matches the key once; everything else goes through the collector. + """ + source = Path(profile_sync.__file__).read_text(encoding="utf-8") + self.assertNotIn('.get("source_ref")', source) + self.assertEqual(source.count('== "source_ref"'), 1) + + +PROSE_SAMPLE = '''emulator: Test +source: "https://github.com/o/n" +profiled_date: "2026-03-29" +source_commit: "pin" + +notes: | + The loader appends dc/ to the system directory (a.c:10) and reads + both ranges (b.c:2,8-9) on boot. + +files: + - name: "a.bin" + source_ref: "a.c:10-12" +''' + + +class TestRebaseProse(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.path = Path(self.tmp.name) / "p.yml" + self.path.write_text(PROSE_SAMPLE, encoding="utf-8") + + def tearDown(self): + self.tmp.cleanup() + + def _report(self, entries, counts=None): + return ProfileReport( + name="p", repo="o/n", pin="pin", head="head", + entries=entries, counts=counts or {}, + ) + + def _prose_entry(self, ref, parts, field="notes"): + status = worst_status([p.status for p in parts]) + return EntryReport(field, ref, status, parts, "prose", field) + + def test_shifted_prose_run_moves_and_the_sentence_stays(self): + part = PartResult(RefPart("a.c", 10, 10, "10"), "SHIFTED", None, 14, 14, []) + applied = rebase_refs( + self.path, self._report([self._prose_entry("a.c:10", [part])]) + ) + self.assertEqual(applied, ["notes: a.c:10 -> a.c:14"]) + text = self.path.read_text() + self.assertIn("system directory (a.c:14) and reads", text) + self.assertEqual( + yaml.safe_load(text)["notes"].count("a.c:14"), 1 + ) + + def test_continuation_ranges_move_together(self): + parts = [ + PartResult(RefPart("b.c", 2, 2, "2"), "SHIFTED", None, 5, 5, []), + PartResult(RefPart("b.c", 8, 9, "8-9"), "SHIFTED", None, 11, 12, []), + ] + applied = rebase_refs( + self.path, self._report([self._prose_entry("b.c:2,8-9", parts)]) + ) + self.assertEqual(applied, ["notes: b.c:2,8-9 -> b.c:5,11-12"]) + self.assertIn("both ranges (b.c:5,11-12) on boot", self.path.read_text()) + + def test_prose_and_structured_move_in_one_pass(self): + prose = PartResult(RefPart("a.c", 10, 10, "10"), "SHIFTED", None, 14, 14, []) + structured = PartResult(RefPart("a.c", 10, 12), "SHIFTED", None, 14, 16, []) + applied = rebase_refs( + self.path, + self._report([ + self._prose_entry("a.c:10", [prose]), + EntryReport("a.bin", "a.c:10-12", "SHIFTED", [structured]), + ]), + ) + self.assertEqual(len(applied), 2) + text = self.path.read_text() + self.assertIn('source_ref: "a.c:14-16"', text) + self.assertIn("(a.c:14)", text) + + def test_review_status_blocks_prose_too(self): + part = PartResult(RefPart("a.c", 10, 10, "10"), "SHIFTED", None, 14, 14, []) + report = self._report( + [self._prose_entry("a.c:10", [part])], counts={"GONE": 1} + ) + self.assertEqual(rebase_refs(self.path, report), []) + self.assertIn("(a.c:10)", self.path.read_text()) + + def test_ambiguous_sentence_location_is_left_alone(self): + text = PROSE_SAMPLE.replace( + "files:", + 'quirks: |\n The loader appends dc/ to the system directory ' + '(a.c:10) and reads\n nothing else.\nfiles:', + ) + self.path.write_text(text, encoding="utf-8") + part = PartResult(RefPart("a.c", 10, 10, "10"), "SHIFTED", None, 14, 14, []) + applied = rebase_refs( + self.path, self._report([self._prose_entry("a.c:10", [part])]) + ) + self.assertEqual(applied, []) + self.assertEqual(self.path.read_text(), text) + + def test_folded_scalar_moves_through_its_run_token(self): + text = PROSE_SAMPLE.replace( + "files:", + "quirk_note: >\n Loaded beside the card\n (q.c:7). Checked" + " later.\nfiles:", + ) + self.path.write_text(text, encoding="utf-8") + part = PartResult(RefPart("q.c", 7, 7, "7"), "SHIFTED", None, 9, 9, []) + applied = rebase_refs( + self.path, + self._report([self._prose_entry("q.c:7", [part], "quirk_note")]), + ) + self.assertEqual(applied, ["quirk_note: q.c:7 -> q.c:9"]) + content = self.path.read_text() + self.assertIn("(q.c:9). Checked", content) + self.assertIn( + "(q.c:9).", yaml.safe_load(content)["quirk_note"] + ) + + def test_parts_disagreeing_on_the_new_file_do_not_move(self): + parts = [ + PartResult(RefPart("b.c", 2, 2, "2"), "RENAMED", "src/b.c", 5, 5, []), + PartResult(RefPart("b.c", 8, 9, "8-9"), "RENAMED", "old/b.c", 11, 12, []), + ] + applied = rebase_refs( + self.path, self._report([self._prose_entry("b.c:2,8-9", parts)]) + ) + self.assertEqual(applied, []) + + def test_renamed_prose_path_is_rewritten(self): + part = PartResult( + RefPart("a.c", 10, 10, "10"), "RENAMED", "src/a.c", 14, 14, [] + ) + applied = rebase_refs( + self.path, self._report([self._prose_entry("a.c:10", [part])]) + ) + self.assertEqual(applied, ["notes: a.c:10 -> src/a.c:14"]) + self.assertIn("(src/a.c:14)", self.path.read_text()) + + +class TestPendingProse(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.path = Path(self.tmp.name) / "p.yml" + self.path.write_text(PROSE_SAMPLE, encoding="utf-8") + + def tearDown(self): + self.tmp.cleanup() + + def _report(self, parts): + status = worst_status([p.status for p in parts]) + entry = EntryReport("notes", "a.c:10", status, parts, "prose", "notes") + return ProfileReport( + name="p", repo="o/n", pin="pin", head="newhead", + entries=[entry], counts={status: 1}, + ) + + def test_moved_prose_still_in_the_file_blocks_the_pin(self): + part = PartResult(RefPart("a.c", 10, 10, "10"), "SHIFTED", None, 14, 14, []) + report = self._report([part]) + text = self.path.read_text() + self.assertEqual(profile_sync.pending_recale(report, text=text), 1) + self.assertFalse(bump_commit(self.path, report)) + self.assertEqual( + yaml.safe_load(self.path.read_text())["source_commit"], "pin" + ) + + def test_recaled_prose_lets_the_pin_advance(self): + part = PartResult(RefPart("a.c", 10, 10, "10"), "SHIFTED", None, 14, 14, []) + report = self._report([part]) + rebase_refs(self.path, report) + self.assertTrue(bump_commit(self.path, report)) + self.assertEqual( + yaml.safe_load(self.path.read_text())["source_commit"], "newhead" + ) + + def test_anchored_prose_never_blocks(self): + part = PartResult(RefPart("a.c", 10, 10, "10"), "ANCHORED", None, None, None, []) + report = self._report([part]) + self.assertEqual( + profile_sync.pending_recale(report, text=self.path.read_text()), 0 + ) + self.assertTrue(bump_commit(self.path, report)) + + def test_unwritable_prose_blocks_the_pin(self): + text = PROSE_SAMPLE.replace( + "files:", + 'quirks: |\n The loader appends dc/ to the system directory ' + '(a.c:10) and reads\n nothing else.\nfiles:', + ) + self.path.write_text(text, encoding="utf-8") + part = PartResult(RefPart("a.c", 10, 10, "10"), "SHIFTED", None, 14, 14, []) + report = self._report([part]) + self.assertEqual(rebase_refs(self.path, report), []) + self.assertFalse(bump_commit(self.path, report)) + + +class TestRealignProse(unittest.TestCase): + """A prose scalar written under an older pin realigns from that pin.""" + + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.dir = Path(self.tmp.name) + self.path = self.dir / "p.yml" + self.files: dict[tuple[str, str], list[str]] = {} + self._orig = ( + profile_sync.upstream.fetch_file, + profile_sync.upstream.list_tree, + ) + profile_sync.upstream.fetch_file = ( + lambda repo, sha, path, cache_dir, offline=False: self.files.get( + (sha, path) + ) + ) + profile_sync.upstream.list_tree = ( + lambda repo, sha, cache_dir, offline=False: ( + sorted({p for _, p in self.files}), False + ) + ) + + def tearDown(self): + ( + profile_sync.upstream.fetch_file, + profile_sync.upstream.list_tree, + ) = self._orig + self.tmp.cleanup() + + def _git(self, *argv): + subprocess.run( + ["git", "-c", "commit.gpgsign=false", *argv], + cwd=self.dir, capture_output=True, check=True, + ) + + def _commit(self, text, message): + self.path.write_text(text, encoding="utf-8") + self._git("add", "p.yml") + self._git("commit", "-m", message) + + def _repo_with_advanced_pin(self): + self._git("init", "-q") + self._git("config", "user.email", "t@t") + self._git("config", "user.name", "t") + written = PROSE_SAMPLE.replace('"pin"', '"oldpin"') + self._commit(written, "profile") + self._commit(PROSE_SAMPLE, "advance pin") + + def test_run_realigns_from_the_writing_pin(self): + self._repo_with_advanced_pin() + # At the writing pin line 10 is the subject; at the current pin the + # same content sits on line 14. Anchoring from the current pin would + # track whatever occupies line 10 there instead. + self.files[("oldpin", "a.c")] = ["x"] * 9 + ["subject"] + self.files[("pin", "a.c")] = ["x"] * 13 + ["subject"] + self.files[("oldpin", "b.c")] = ["p", "two", "q", "r", "s", "t", "u", "eight", "nine"] + self.files[("pin", "b.c")] = ["p", "two", "q", "r", "s", "t", "u", "eight", "nine"] + messages = profile_sync.realign_prose(self.path, self.tmp.name) + self.assertIn("notes: a.c:10 -> a.c:14", messages) + text = self.path.read_text() + self.assertIn("(a.c:14)", text) + self.assertIn("(b.c:2,8-9)", text) + self.assertEqual( + yaml.safe_load(text)["source_commit"], "pin", + ) + + def test_dry_run_writes_nothing(self): + self._repo_with_advanced_pin() + self.files[("oldpin", "a.c")] = ["x"] * 9 + ["subject"] + self.files[("pin", "a.c")] = ["x"] * 13 + ["subject"] + self.files[("oldpin", "b.c")] = list("pqrstuvwx") + self.files[("pin", "b.c")] = list("pqrstuvwx") + messages = profile_sync.realign_prose( + self.path, self.tmp.name, dry_run=True + ) + self.assertTrue(any("would recale" in m for m in messages)) + self.assertIn("(a.c:10)", self.path.read_text()) + + def test_scalar_written_at_the_current_pin_is_left_alone(self): + self._git("init", "-q") + self._git("config", "user.email", "t@t") + self._git("config", "user.name", "t") + self._commit(PROSE_SAMPLE, "profile") + self.files[("pin", "a.c")] = ["x"] * 9 + ["subject"] + self.assertEqual( + profile_sync.realign_prose(self.path, self.tmp.name), [] + ) + + def test_range_lost_between_the_pins_is_reported_not_moved(self): + self._repo_with_advanced_pin() + self.files[("oldpin", "a.c")] = ["x"] * 9 + ["subject"] + self.files[("pin", "a.c")] = ["y"] * 20 + self.files[("oldpin", "b.c")] = list("pqrstuvwx") + self.files[("pin", "b.c")] = list("pqrstuvwx") + messages = profile_sync.realign_prose(self.path, self.tmp.name) + self.assertTrue(any("read again" in m for m in messages)) + self.assertIn("(a.c:10)", self.path.read_text()) + + def test_unreadable_tree_is_not_an_absence(self): + self._git("init", "-q") + self._git("config", "user.email", "t@t") + self._git("config", "user.name", "t") + base = PROSE_SAMPLE.replace("(a.c:10)", "(deep.c:10)") + self._commit(base.replace('"pin"', '"oldpin"'), "profile") + self._commit(base, "advance pin") + profile_sync.upstream.list_tree = ( + lambda repo, sha, cache_dir, offline=False: ([], True) + ) + self.files[("oldpin", "b.c")] = list("pqrstuvwx") + self.files[("pin", "b.c")] = list("pqrstuvwx") + messages = profile_sync.realign_prose(self.path, self.tmp.name) + self.assertTrue( + any("tree unreadable" in m for m in messages), messages + ) + self.assertFalse(any("absent at the writing" in m for m in messages)) + + def test_several_paths_carrying_the_name_stay_unclear(self): + self._git("init", "-q") + self._git("config", "user.email", "t@t") + self._git("config", "user.name", "t") + base = PROSE_SAMPLE.replace("(a.c:10)", "(deep.c:10)") + self._commit(base.replace('"pin"', '"oldpin"'), "profile") + self._commit(base, "advance pin") + self.files[("oldpin", "src/deep.c")] = ["x"] + self.files[("oldpin", "contrib/deep.c")] = ["y"] + self.files[("oldpin", "b.c")] = list("pqrstuvwx") + self.files[("pin", "b.c")] = list("pqrstuvwx") + messages = profile_sync.realign_prose(self.path, self.tmp.name) + self.assertTrue( + any("2 paths carry the name" in m for m in messages), messages + ) + + def test_bare_filename_resolves_through_the_tree(self): + self._git("init", "-q") + self._git("config", "user.email", "t@t") + self._git("config", "user.name", "t") + base = PROSE_SAMPLE.replace("(a.c:10)", "(deep.c:10)") + self._commit(base.replace('"pin"', '"oldpin"'), "profile") + self._commit(base, "advance pin") + self.files[("oldpin", "src/deep.c")] = ["x"] * 9 + ["subject"] + self.files[("pin", "src/deep.c")] = ["x"] * 13 + ["subject"] + self.files[("oldpin", "b.c")] = list("pqrstuvwx") + self.files[("pin", "b.c")] = list("pqrstuvwx") + messages = profile_sync.realign_prose(self.path, self.tmp.name) + self.assertIn("notes: deep.c:10 -> deep.c:14", messages) + self.assertIn("(deep.c:14)", self.path.read_text()) + + +class TestRealignFlagMatrix(unittest.TestCase): + """--realign-prose applies a flag or refuses it, never swallows it.""" + + def _run(self, *extra): + argv = ["profile_sync.py", "--emulator", "x", "--realign-prose", *extra] + stderr = io.StringIO() + saved = sys.argv + try: + sys.argv = argv + with contextlib.redirect_stderr(stderr): + with self.assertRaises(SystemExit) as caught: + profile_sync.main() + finally: + sys.argv = saved + return caught.exception.code, stderr.getvalue() + + def test_output_flags_are_refused(self): + for flag in ( + "--json", "--markdown", "--fetch-plan", "--triage", + "--changed-only", "--check-version", "--detect-new-files", + "--watch-hashes", "--full-diff", "--tree-diff", + "--accept-changed", + ): + code, err = self._run(flag) + self.assertEqual(code, 1, flag) + self.assertIn("does not apply", err, flag) + + def test_write_flags_are_refused(self): + for flag in ("--rebase-refs", "--bump-commit", "--backfill-commits"): + code, err = self._run(flag) + self.assertEqual(code, 1, flag) + self.assertIn("runs alone", err, flag) + + +class TestBuildReportProse(TestBuildReport): + """Prose citations run through the same anchoring as source_refs.""" + + def test_notes_citation_is_anchored_and_counted(self): + profile = self._profile(["a.c:2"]) + profile["notes"] = "The loader reads it (a.c:2).\n" + self.files[("pinsha", "a.c")] = ["x", "hit", "y"] + self.files[("headsha", "a.c")] = ["pad", "x", "hit", "y"] + report = build_report("test", profile, self.dir) + self.assertEqual(report.counts["SHIFTED"], 2) + prose = [e for e in report.entries if e.kind == "prose"] + self.assertEqual(len(prose), 1) + self.assertEqual(prose[0].name, "notes") + self.assertEqual(prose[0].field, "notes") + self.assertEqual(prose[0].parts[0].start, 3) + + def test_gone_prose_citation_needs_review(self): + profile = self._profile(["a.c:2"]) + profile["notes"] = "Written by lost.c:9 at boot.\n" + self.files[("pinsha", "a.c")] = ["x", "hit"] + self.files[("headsha", "a.c")] = ["x", "hit"] + report = build_report("test", profile, self.dir) + self.assertEqual(report.counts.get("GONE"), 1) + self.assertEqual(report.needs_review(), 1) + + def test_prose_only_profile_is_not_skipped(self): + profile = { + "emulator": "T", + "source": "https://github.com/o/n", + "profiled_date": "2026-03-29", + "notes": "Boot path in a.c:2.\n", + } + self.files[("pinsha", "a.c")] = ["x", "hit"] + self.files[("headsha", "a.c")] = ["x", "hit"] + report = build_report("test", profile, self.dir) + self.assertIsNone(report.skipped) + self.assertEqual(report.entries[0].kind, "prose") + + if __name__ == "__main__": unittest.main() diff --git a/tests/test_upstream.py b/tests/test_upstream.py index 6537b88e..5637d42f 100644 --- a/tests/test_upstream.py +++ b/tests/test_upstream.py @@ -59,6 +59,7 @@ class TestParseRepo(unittest.TestCase): for url in ( "https://git.citron-emu.org/citron/emu", "https://git.eden-emu.dev/eden-emu/eden", + "https://git.ryujinx.app/projects/Kenji-NX", ): self.assertEqual(parse_repo(url).family, "forgejo") @@ -139,7 +140,7 @@ class TestTokenScope(unittest.TestCase): self.assertNotIn("Authorization", h) def test_withheld_from_forgejo_instances(self): - for host in ("git.citron-emu.org", "git.eden-emu.dev"): + for host in ("git.citron-emu.org", "git.eden-emu.dev", "git.ryujinx.app"): h = upstream._headers(f"https://{host}/api/v1/repos/o/n/commits") self.assertNotIn("Authorization", h, host) @@ -476,5 +477,57 @@ class TestCompare(unittest.TestCase): ) +class TestGitlabTree(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory() + self.dir = self.tmp.name + self.repo = upstream.parse_repo("https://gitlab.com/g/p") + self.pages: dict[int, object] = {} + self._orig = (upstream._http_json, upstream._http_text) + + def fake(url): + page = int(url.rsplit("page=", 1)[1]) + payload = self.pages.get(page) + if payload == "boom": + raise upstream.UpstreamError(url) + return payload + + upstream._http_json = fake + + def tearDown(self): + upstream._http_json, upstream._http_text = self._orig + self.tmp.cleanup() + + def _blobs(self, *names): + return [{"type": "blob", "path": name} for name in names] + + def test_single_page_is_complete(self): + self.pages[1] = self._blobs("src/a.c", "src/b.c") + [ + {"type": "tree", "path": "src"} + ] + paths, truncated = upstream.list_tree(self.repo, "sha", self.dir) + self.assertEqual(paths, ["src/a.c", "src/b.c"]) + self.assertFalse(truncated) + + def test_pages_accumulate_until_a_short_one(self): + self.pages[1] = self._blobs( + *[f"f{i}.c" for i in range(upstream.TREE_PAGE_SIZE)] + ) + self.pages[2] = self._blobs("last.c") + paths, truncated = upstream.list_tree(self.repo, "sha", self.dir) + self.assertEqual(len(paths), upstream.TREE_PAGE_SIZE + 1) + self.assertIn("last.c", paths) + self.assertFalse(truncated) + + def test_unreadable_page_reports_truncation(self): + self.pages[1] = self._blobs( + *[f"f{i}.c" for i in range(upstream.TREE_PAGE_SIZE)] + ) + self.pages[2] = None + paths, truncated = upstream.list_tree(self.repo, "sha", self.dir) + self.assertEqual(len(paths), upstream.TREE_PAGE_SIZE) + self.assertTrue(truncated) + + if __name__ == "__main__": unittest.main()