diff --git a/scripts/notebook_tools/check_accent_restoration_invariants.py b/scripts/notebook_tools/check_accent_restoration_invariants.py new file mode 100644 index 0000000000..bf910e2d3a --- /dev/null +++ b/scripts/notebook_tools/check_accent_restoration_invariants.py @@ -0,0 +1,190 @@ +#!/usr/bin/env python3 +"""Invariant check for accent-restoration tranches (ruling #16638, 2026-10-03). + +A restoration tranche must change NOTHING but accents. Reviewing #18814, the +coordinator found an artefact neither the review nor the dossier had caught: +33 mid-sentence capitalisations across 17 cells ("la Même convention", +"l'Inférence bayesienne", "le second Paramètre"). No ACCENT_PAIRS entry maps a +lowercase form to a capitalised one, so the tooling was not the cause -- the +tranche *procedure* was. The ruling therefore asks for a per-cell organ: + + strip_accents(base) == strip_accents(head) for every markdown cell. + +strip_accents drops combining marks but NOT case, so the equality is exact on +everything except accents: capitalisation, rewording, added/removed words, +renumbering -- all of it breaks the invariant and is reported. Code cells must +be byte-identical (a prose tranche never touches them), and the cell count and +each cell type must not move. The comparison base is the tranche's own parent +commit (--base-sha), never a distant main: renumberings landed between the +tranche and a later main are NOT tranche changes and must not poison the +check. + +Validated by its false negatives, never by its hits: the #18814 defect +("la meme convention" -> "la Même convention") MUST be flagged, a clean +restoration ("theoreme" -> "théorème") MUST pass, and a code-cell edit MUST +be reported even though it produces no markdown diff. + +Usage: python check_accent_restoration_invariants.py + --base-sha [--json] + [--fail-on-findings] + +Output: one line per finding `KIND cell #N (excerpt)`, then a total. +--fail-on-findings exits 2 when any finding exists. Exit 1 is reserved for +tool failure (git show failed, notebook unreadable) -- a vacuous crash is +never a clean check. +""" + +import argparse +import json +import subprocess +import sys +import unicodedata +from pathlib import Path + +# Sibling import (organ-first): the accent-stripping function is the detector's, +# never a copy. If the two organs ever diverge on what an accent is, they +# diverge together. +sys.path.insert(0, str(Path(__file__).resolve().parent)) +from detect_markdown_deaccent import _strip_accents # noqa: E402 + + +def _cell_source(cell: dict) -> str: + """Join a cell source given as list-of-lines or plain string.""" + src = cell.get("source", "") + if isinstance(src, list): + return "".join(src) + return str(src) + + +def _first_diff_context(a: str, b: str, span: int = 30) -> tuple[str, str]: + """Excerpt of both strings around their first divergent position.""" + limit = min(len(a), len(b)) + i = 0 + while i < limit and a[i] == b[i]: + i += 1 + lo = max(0, i - span) + return a[lo : i + span], b[lo : i + span] + + +def _load_base_notebook(notebook: Path, base_sha: str) -> dict: + """Read the notebook as it stands at base_sha via `git show`.""" + top = subprocess.run( + ["git", "rev-parse", "--show-toplevel"], + capture_output=True, + text=True, + encoding="utf-8", + errors="replace", + check=True, + ).stdout.strip() + rel = notebook.resolve().relative_to(Path(top)).as_posix() + raw = subprocess.run( + ["git", "-C", top, "show", f"{base_sha}:{rel}"], + capture_output=True, + text=True, + encoding="utf-8", + check=True, + ).stdout + return json.loads(raw) + + +def _cell_invariant_violations(base_nb: dict, head_nb: dict) -> list[dict]: + """Findings for every cell-level change an accent tranche must not make. + + Markdown cells are compared through strip_accents (case-preserving), code + cells byte-for-byte; cell count and cell types must not move. Findings are + ordered by cell index, kind by kind. + """ + findings: list[dict] = [] + base_cells = base_nb.get("cells", []) + head_cells = head_nb.get("cells", []) + if len(base_cells) != len(head_cells): + findings.append( + { + "kind": "CELL_COUNT_CHANGED", + "cell": None, + "detail": f"base {len(base_cells)} cells vs head {len(head_cells)} cells", + } + ) + for i, (b, h) in enumerate(zip(base_cells, head_cells)): + b_type = b.get("cell_type", "") + h_type = h.get("cell_type", "") + if b_type != h_type: + findings.append( + { + "kind": "TYPE_CHANGED", + "cell": i, + "detail": f"base {b_type} vs head {h_type}", + } + ) + continue + b_src = _cell_source(b) + h_src = _cell_source(h) + if b_type == "markdown": + if _strip_accents(b_src) != _strip_accents(h_src): + base_ctx, head_ctx = _first_diff_context( + _strip_accents(b_src), _strip_accents(h_src) + ) + findings.append( + { + "kind": "MARKDOWN_INVARIANT", + "cell": i, + "detail": f"base ...{base_ctx!r} vs head ...{head_ctx!r}", + } + ) + else: + if b_src != h_src: + base_ctx, head_ctx = _first_diff_context(b_src, h_src) + findings.append( + { + "kind": "CODE_MODIFIED", + "cell": i, + "detail": f"base ...{base_ctx!r} vs head ...{head_ctx!r}", + } + ) + return findings + + +def main(argv: list[str] | None = None) -> int: + parser = argparse.ArgumentParser( + description="Check that an accent-restoration tranche changed nothing but accents." + ) + parser.add_argument("notebook", type=Path, help="notebook on the tranche head") + parser.add_argument( + "--base-sha", + required=True, + help="commit (sha or ref) holding the pre-tranche notebook -- typically the tranche parent", + ) + parser.add_argument("--json", action="store_true", help="emit a JSON report") + parser.add_argument( + "--fail-on-findings", action="store_true", help="exit 2 when any finding exists" + ) + args = parser.parse_args(argv) + + try: + head_nb = json.loads(args.notebook.read_text(encoding="utf-8")) + base_nb = _load_base_notebook(args.notebook, args.base_sha) + except (OSError, json.JSONDecodeError, subprocess.CalledProcessError) as exc: + print(f"ERROR: cannot load notebook or base ({exc})", file=sys.stderr) + return 1 + + findings = _cell_invariant_violations(base_nb, head_nb) + report = { + "notebook": str(args.notebook), + "base_sha": args.base_sha, + "total": len(findings), + "findings": findings, + } + if args.json: + print(json.dumps(report, ensure_ascii=False, indent=2)) + else: + for f in findings: + cell = "cell #" + str(f["cell"]) if f["cell"] is not None else "cells" + print(f"{f['kind']} {cell} ({f['detail']})") + print(f"TOTAL {report['total']} finding(s) -- {args.notebook}") + if findings and args.fail_on_findings: + return 2 + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/tests/test_check_accent_restoration_invariants.py b/scripts/tests/test_check_accent_restoration_invariants.py new file mode 100644 index 0000000000..edf32a58bd --- /dev/null +++ b/scripts/tests/test_check_accent_restoration_invariants.py @@ -0,0 +1,161 @@ +"""Tests for check_accent_restoration_invariants (ruling #16638 point 1). + +Validated by false negatives: the #18814 capitalisation defect MUST be +flagged, a clean restoration MUST pass, a code edit MUST be reported. +""" + +import json +import subprocess as subprocess_mod +import sys +from pathlib import Path + +import pytest + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "notebook_tools")) +import check_accent_restoration_invariants as cri # noqa: E402 + + +def _nb(cells: list[dict]) -> dict: + return {"cells": cells} + + +def _md(src: str) -> dict: + return {"cell_type": "markdown", "source": [src]} + + +def _code(src: str) -> dict: + return {"cell_type": "code", "source": [src]} + + +class TestCellInvariantViolations: + def test_clean_restoration_passes(self): + """theoreme -> théorème: accents only, no finding.""" + base = _nb([_md("Le theoreme de Bayes s'applique ici."), _code("x = 1")]) + head = _nb([_md("Le théorème de Bayes s'applique ici."), _code("x = 1")]) + assert cri._cell_invariant_violations(base, head) == [] + + def test_18814_capitalisation_defect_is_flagged(self): + """The founding defect: 'la meme convention' -> 'la Même convention'.""" + base = _nb([_md("On suit la meme convention que precedemment.")]) + head = _nb([_md("On suit la Même convention que precedemment.")]) + findings = cri._cell_invariant_violations(base, head) + assert len(findings) == 1 + assert findings[0]["kind"] == "MARKDOWN_INVARIANT" + assert findings[0]["cell"] == 0 + + def test_case_preserved_by_strip_means_case_change_breaks_invariant(self): + """Lowercase->capital mid-word, no accent involved at all.""" + base = _nb([_md("le second parametre vaut 2")]) + head = _nb([_md("le second Parametre vaut 2")]) + findings = cri._cell_invariant_violations(base, head) + assert [f["kind"] for f in findings] == ["MARKDOWN_INVARIANT"] + + def test_rewording_is_flagged(self): + """A real word change (not accents) breaks the invariant.""" + base = _nb([_md("Le modele est simple")]) + head = _nb([_md("Le modèle est très simple")]) + assert [f["kind"] for f in cri._cell_invariant_violations(base, head)] == [ + "MARKDOWN_INVARIANT" + ] + + def test_code_cell_edit_is_flagged(self): + """A code edit produces no markdown diff but must be reported.""" + base = _nb([_md("texte"), _code("resultat = 1")]) + head = _nb([_md("texte"), _code("resultat = 2")]) + findings = cri._cell_invariant_violations(base, head) + assert [f["kind"] for f in findings] == ["CODE_MODIFIED"] + assert findings[0]["cell"] == 1 + + def test_cell_count_change_is_flagged(self): + base = _nb([_md("a")]) + head = _nb([_md("a"), _md("b")]) + findings = cri._cell_invariant_violations(base, head) + assert [f["kind"] for f in findings] == ["CELL_COUNT_CHANGED"] + + def test_cell_type_change_is_flagged(self): + base = _nb([_md("theoreme")]) + head = _nb([_code("theoreme")]) + assert [f["kind"] for f in cri._cell_invariant_violations(base, head)] == [ + "TYPE_CHANGED" + ] + + def test_list_and_string_sources_both_handled(self): + """nbformat source is list-of-lines; some tools emit a plain string.""" + base = _nb([{"cell_type": "markdown", "source": "theoreme ok"}]) + head = _nb([{"cell_type": "markdown", "source": ["théorème ok"]}]) + assert cri._cell_invariant_violations(base, head) == [] + + +class TestLoadBaseNotebook: + def test_git_show_roundtrip(self, tmp_path, monkeypatch): + """base notebook is read via git show sha:relpath from the repo root.""" + nb = {"cells": [{"cell_type": "markdown", "source": ["theoreme"]}]} + + def fake_run(cmd, **kwargs): + if cmd[:3] == ["git", "rev-parse", "--show-toplevel"]: + return subprocess_mod.CompletedProcess(cmd, 0, stdout=str(tmp_path)) + assert cmd[:4] == ["git", "-C", str(tmp_path), "show"] + assert cmd[4] == "abc123:dir/nb.ipynb" + return subprocess_mod.CompletedProcess( + cmd, 0, stdout=json.dumps(nb, ensure_ascii=False) + ) + + monkeypatch.setattr(subprocess_mod, "run", fake_run) + nb_path = tmp_path / "dir" / "nb.ipynb" + loaded = cri._load_base_notebook(nb_path, "abc123") + assert loaded == nb + + def test_git_show_failure_raises(self, tmp_path, monkeypatch): + """A missing sha at the base path surfaces as CalledProcessError.""" + + def fake_run(cmd, **kwargs): + if cmd[:3] == ["git", "rev-parse", "--show-toplevel"]: + return subprocess_mod.CompletedProcess(cmd, 0, stdout=str(tmp_path)) + raise subprocess_mod.CalledProcessError(128, cmd) + + monkeypatch.setattr(subprocess_mod, "run", fake_run) + with pytest.raises(subprocess_mod.CalledProcessError): + cri._load_base_notebook(tmp_path / "nb.ipynb", "deadbeef") + + +class TestMain: + def _write_nb(self, tmp_path, cells: list[dict]) -> Path: + path = tmp_path / "nb.ipynb" + path.write_text(json.dumps(_nb(cells), ensure_ascii=False), encoding="utf-8") + return path + + def test_clean_trancho_exits_zero(self, tmp_path, monkeypatch, capsys): + path = self._write_nb(tmp_path, [_md("Le théorème tient.")]) + monkeypatch.setattr( + cri, "_load_base_notebook", lambda p, sha: _nb([_md("Le theoreme tient.")]) + ) + rc = cri.main([str(path), "--base-sha", "abc", "--fail-on-findings"]) + assert rc == 0 + assert "TOTAL 0" in capsys.readouterr().out + + def test_finding_exits_two(self, tmp_path, monkeypatch, capsys): + """The #18814 defect, end to end through main.""" + path = self._write_nb(tmp_path, [_md("la Même convention")]) + monkeypatch.setattr( + cri, "_load_base_notebook", lambda p, sha: _nb([_md("la meme convention")]) + ) + rc = cri.main([str(path), "--base-sha", "abc", "--fail-on-findings"]) + assert rc == 2 + out = capsys.readouterr().out + assert "MARKDOWN_INVARIANT cell #0" in out + + def test_json_report_shape(self, tmp_path, monkeypatch, capsys): + path = self._write_nb(tmp_path, [_md("le Parametre")]) + monkeypatch.setattr( + cri, "_load_base_notebook", lambda p, sha: _nb([_md("le parametre")]) + ) + rc = cri.main([str(path), "--base-sha", "abc", "--json"]) + assert rc == 0 + report = json.loads(capsys.readouterr().out) + assert report["total"] == 1 + assert report["findings"][0]["kind"] == "MARKDOWN_INVARIANT" + assert report["base_sha"] == "abc" + + def test_missing_file_exits_one(self, tmp_path): + rc = cri.main([str(tmp_path / "nope.ipynb"), "--base-sha", "abc"]) + assert rc == 1