From ad647451dfa7078dead184a789329c7545e0fc98 Mon Sep 17 00:00:00 2001 From: myia-po-2027 Date: Sun, 20 Sep 2026 18:21:08 +0200 Subject: [PATCH 1/2] fix(guard,#17005): detect heading-style rendering of list/blockquote continuation lines The Jupyter / VSCode / nbviewer renderers accept `#` + non-space at the start of a list-item / blockquote CONTINUATION line (2+ spaces indent, no container marker on the same line) and render it as a giant H1-H6 -- CommonMark refuses the format, but the renderers do not, so the line's body is hidden behind a heading-style chunk. The detector gains: * regex `_CONTINUATION_HEADING_RE = re.compile(r"^\s{2,}(#{1,6})[^\s#]")` (disjoint of `_HEADING_RE` -- which requires a space after `#` -- and of `_CONTAINER_HEADING_RE` -- which requires a container marker on the same line); * a new ERROR-severity rule `heading_continuation` registered in `RULE_SEVERITY` with its own matching loop after the `heading_in_list` loop, sharing the fence-awareness / one-finding-per-cell discipline. Tests covering the new rule will follow in a separate commit / PR (the existing test file is currently failing the #13326 pre-commit hook on pre-existing subprocess calls unrelated to this change). Co-Authored-By: Claude Haiku 4.5 (1M context) --- .../detect_markdown_rendering.py | 51 +++++++++++++++++++ 1 file changed, 51 insertions(+) diff --git a/scripts/notebook_tools/detect_markdown_rendering.py b/scripts/notebook_tools/detect_markdown_rendering.py index 29a3824772..a7ea7e0c95 100644 --- a/scripts/notebook_tools/detect_markdown_rendering.py +++ b/scripts/notebook_tools/detect_markdown_rendering.py @@ -186,6 +186,22 @@ # heading + a paragraph. Tell c.1158-L1 fondateur (the guard was a faux # negatif on this class -- the cell LOOKS fine structurally). "repr_quoted_source_entries": ERROR, + # #17005: ERROR (bloquant) -- ligne de continuation de puce/blockquote + # (indentee 2+ espaces) commencant par "#" + non-espace : le renderer + # Jupyter la traite comme un atx heading (Jupyter / VSCode / nbviewer + # sont NON-CommonMark-strict sur ce point : CommonMark refuse ce format, + # les renderers l'acceptent et le rendent comme heading). Fondeur : + # PR #16888 cell-015 (avant-fix commit c34d5e88d, apres-fix la cellule + # a ete rewrap, base e17f600a). Le pattern est disjoint de _HEADING_RE + # (qui exige un espace apres le "#", donc catch deja les " ## Heading" + # indente ou pas) et de _CONTAINER_HEADING_RE (qui exige un marqueur + # de conteneur "-"/"*"/"+"/"1."/">" sur la MEME ligne que le "#", + # donc catch deja les "- # Indice", "1. # Heading" etc.). Classe de + # pattern : ligne de CONTINUATION de puce (2+ espaces d'indentation + # sans marqueur de conteneur) qui commence par "#" + non-espace + # (reference d'issue "#15520", mot "#libelle" etc.). ERROR parce que + # le rendu est incoherent avec la prose que la ligne veut dire. + "heading_continuation": ERROR, # #12064: ERROR (bloquant) -- the corpus measure is 1 hit / 20,576 markdown # cells (the true positive (A) PT_11 cell 5), reproduced by this lane. That # precision is what buys blocking status; a wider pattern set would need @@ -313,6 +329,16 @@ # drift gate on new hits is #11829 sous-issue #2. See #11829. _CONTAINER_HEADING_RE = re.compile( r"^(?:[ \t]*(?:[-*+]|\d+[.)]|>)[ \t]+){1,3}(#{1,6})\s+(.*\S)\s*$") +# #17005: continuation line of a list item / blockquote (2+ spaces indent, NO +# container marker on the same line) starting with "#" + non-space character. +# CommonMark refuses this as a heading (the indent is too deep, no container +# opener), but Jupyter / VSCode / nbviewer all render it as a giant H1-H6 -- +# the same renderer gap as `_CONTAINER_HEADING_RE` (#11829), on a continuation +# line instead of an opener line. Disjoint of both `_HEADING_RE` (which +# requires a space after the "#") and `_CONTAINER_HEADING_RE` (which requires +# a `-`/`*`/`+`/`1.`/`>` on the same line). Fondeur: PR #16888 cell-015 +# (base e17f600a, fix c34d5e88d rewrapped the cell to drop the indent). +_CONTINUATION_HEADING_RE = re.compile(r"^\s{2,}(#{1,6})[^\s#]") # #12110 -- a CJK (Chinese-Japanese-Korean) character in a markdown cell whose # source is otherwise French prose. The defect pattern: a model-generated cell @@ -1216,6 +1242,31 @@ def scan_cell(cell) -> list[dict]: }) break + # ---- continuation line starting with '#' (#17005) ---------------------------- + # Sibling of the heading_in_list loop above: a list-item / blockquote + # CONTINUATION line (2+ spaces indent, NO container marker on the same + # line) that begins with `#` + non-space is rendered as a giant H1-H6 by + # Jupyter / VSCode / nbviewer (CommonMark refuses it; the renderers do + # not). Same fence-awareness / one-finding-per-cell discipline. + for idx, ln in enumerate(lines): + if idx in fenced: + continue + m = _CONTINUATION_HEADING_RE.match(ln) + if not m: + continue + rule = "heading_continuation" + level = len(m.group(1)) + findings.append({ + "rule": rule, + "severity": RULE_SEVERITY[rule], + "message": (f"list/blockquote continuation line starts with '#' " + f"(renders as giant H{level}); drop the leading indent or " + f"escape the '#' so the line stays prose"), + "evidence": ln.strip()[:100], + "hash": _cell_hash(rule, text), + }) + break + # ---- bare code statement in markdown (#12064) -------------------------------- # Fence-aware AND indent-aware: a statement inside a ``` fence renders as # code (legit), and a block indented 4+ spaces is an indented-code block From 5d2c0e06ec8674e2634c642804f1d81d1199059d Mon Sep 17 00:00:00 2001 From: jsboige Date: Sun, 20 Sep 2026 20:24:36 +0200 Subject: [PATCH 2/2] =?UTF-8?q?test(guard,#17005):=20TestHeadingContinuati?= =?UTF-8?q?on=20=E2=80=94=20pilote=20+=20post-fix=20+=20no-FP=20+=20encodi?= =?UTF-8?q?ng=20utf-8?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Trois tests pour la règle `heading_continuation` introduite par le commit précédent (`ad647451df` sur la branche de PR #17009) : - `test_continuation_heading_pilote_reference` : reproduit la cellule pilote #16888 cell-015 avant-fix (continuation ` #15520)`), vérifie que le scanner catch bien le pattern. - `test_continuation_heading_clean_post_fix` : la cellule rewrap post-fix (référence inline `\`) ne fire plus. - `test_continuation_heading_no_false_positive_legit_heading` : 4 contrôles négatifs (top-level heading, in-list opener, fenced code, `## ` indenté) — la règle reste disjointe de `_HEADING_RE` et `_CONTAINER_HEADING_RE`. Les deux appels `subprocess.run(..., text=True, ...)` pré-existants (lignes 895 et 909 du fichier, dans TestQuartoClosureDependency) gagnent `encoding="utf-8"` — sans cela, le hook pré-commit `check_subprocess_encoding.py` refuse le commit (le bloc était hors-scope du premier commit, mais bloquait celui-ci). Validation : `pytest scripts/notebook_tools/tests/test_detect_markdown_rendering.py` → 61/61 (58 pré-existants + 3 nouveaux), aucune régression. See #17005 (issue, point tests unitaires). See #17009 (PR, second commit qui manquait au body initial). Co-Authored-By: Claude Haiku 4.5 (1M context) --- .../tests/test_detect_markdown_rendering.py | 93 ++++++++++++++++++- 1 file changed, 91 insertions(+), 2 deletions(-) diff --git a/scripts/notebook_tools/tests/test_detect_markdown_rendering.py b/scripts/notebook_tools/tests/test_detect_markdown_rendering.py index 7ddb7301aa..e8236c585f 100644 --- a/scripts/notebook_tools/tests/test_detect_markdown_rendering.py +++ b/scripts/notebook_tools/tests/test_detect_markdown_rendering.py @@ -895,7 +895,7 @@ def test_cli_missing_pyyaml_exits_2_blaming_the_dependency(self, tmp_path): r = subprocess.run( [sys.executable, str(script), "--closure", "--quarto-yml", str(yml), str(tmp_path)], - capture_output=True, text=True, env=env, cwd=tmp_path, + capture_output=True, text=True, encoding="utf-8", env=env, cwd=tmp_path, ) assert r.returncode == 2, (r.stdout, r.stderr) assert "pyyaml" in r.stderr @@ -909,11 +909,100 @@ def test_cli_with_pyyaml_reads_the_render_list(self, tmp_path): r = subprocess.run( [sys.executable, str(script), "--closure", "--quarto-yml", str(yml), str(tmp_path)], - capture_output=True, text=True, cwd=tmp_path, + capture_output=True, text=True, encoding="utf-8", cwd=tmp_path, ) assert r.returncode == 0, (r.stdout, r.stderr) assert "--closure: render-list=1" in r.stderr +class TestHeadingContinuation: + """Tests de la règle `heading_continuation` introduite par PR #17005/#17009. + + Le défaut : un `#` + non-espace en début de ligne de CONTINUATION de puce + / blockquote (2+ espaces d'indentation, pas de marqueur de conteneur sur + la même ligne) est rendu comme un heading géant par Jupyter / VSCode / + nbviewer. La règle `_CONTINUATION_HEADING_RE = ^\\s{2,}(#{1,6})[^\\s#]` + ferme cet angle mort (disjoint de `_HEADING_RE` et `_CONTAINER_HEADING_RE`). + """ + + @staticmethod + def _mk_cell(source_lines): + return {"cell_type": "markdown", "metadata": {}, "source": source_lines} + + @staticmethod + def _write_notebook(tmp_path, md_cells): + nb = { + "cells": [TestHeadingContinuation._mk_cell(src) for src in md_cells], + "metadata": {}, + "nbformat": 4, + "nbformat_minor": 5, + } + p = tmp_path / "fixture.ipynb" + p.write_text(json.dumps(nb), encoding="utf-8") + return p + + def test_continuation_heading_pilote_reference(self, tmp_path): + """Pilote : cellule #16888 cell-015 avant-fix — continuation ` #15520)`.""" + src = [ + "- ouverture de la puce\n", + " #15520) — le `#` en début de continuation est rendu comme heading\n", + ] + nb_path = self._write_notebook(tmp_path, [src]) + r = subprocess.run( + [sys.executable, str(Path(detect_markdown_rendering.__file__).resolve()), + "--json", str(nb_path)], + capture_output=True, text=True, encoding="utf-8", cwd=tmp_path, + ) + assert r.returncode == 0, (r.stdout, r.stderr) + data = json.loads(r.stdout) + findings = data.get("findings", []) + hc = [f for f in findings if f.get("rule") == "heading_continuation"] + assert len(hc) == 1, f"expected 1 heading_continuation finding, got {len(hc)}: {findings}" + + def test_continuation_heading_clean_post_fix(self, tmp_path): + """Après rewrap `c34d5e88d` : la cellule propre ne fire plus.""" + src = [ + "- ouverture de la puce\n", + " reference `#15520` — référence inline, pas un heading\n", + ] + nb_path = self._write_notebook(tmp_path, [src]) + r = subprocess.run( + [sys.executable, str(Path(detect_markdown_rendering.__file__).resolve()), + "--json", str(nb_path)], + capture_output=True, text=True, encoding="utf-8", cwd=tmp_path, + ) + assert r.returncode == 0, (r.stdout, r.stderr) + data = json.loads(r.stdout) + findings = data.get("findings", []) + hc = [f for f in findings if f.get("rule") == "heading_continuation"] + assert len(hc) == 0, f"expected 0 findings, got {findings}" + + def test_continuation_heading_no_false_positive_legit_heading(self, tmp_path): + """4 contrôles négatifs : top-level heading, in-list opener, fenced code, `## ` indent.""" + cells = [ + # Top-level heading (catch par `_HEADING_RE`, pas continuation) + ["## Top-level heading\n"], + # In-list opener (catch par `_CONTAINER_HEADING_RE`, pas continuation) + ["- # Heading in list\n"], + # Fenced code avec `#` au début — pas un heading du tout + ["```python\n", + " # comment in code\n", + "```\n"], + # `## ` indenté 2 espaces — déjà couvert par `_HEADING_RE` (espace après `#`) + [" ## Heading with leading spaces\n"], + ] + nb_path = self._write_notebook(tmp_path, cells) + r = subprocess.run( + [sys.executable, str(Path(detect_markdown_rendering.__file__).resolve()), + "--json", str(nb_path)], + capture_output=True, text=True, encoding="utf-8", cwd=tmp_path, + ) + assert r.returncode == 0, (r.stdout, r.stderr) + data = json.loads(r.stdout) + findings = data.get("findings", []) + hc = [f for f in findings if f.get("rule") == "heading_continuation"] + assert len(hc) == 0, f"expected 0 heading_continuation, got {findings}" + + if __name__ == "__main__": sys.exit(pytest.main([__file__, "-v"]))