Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
51 changes: 51 additions & 0 deletions scripts/notebook_tools/detect_markdown_rendering.py
Original file line number Diff line number Diff line change
Expand Up @@ -186,6 +186,22 @@
# heading + a paragraph. Tell c.1158-L1 fondateur (the guard was a faux
# negatif on this class -- the cell LOOKS fine structurally).
"repr_quoted_source_entries": ERROR,
# #17005: ERROR (bloquant) -- ligne de continuation de puce/blockquote
# (indentee 2+ espaces) commencant par "#" + non-espace : le renderer
# Jupyter la traite comme un atx heading (Jupyter / VSCode / nbviewer
# sont NON-CommonMark-strict sur ce point : CommonMark refuse ce format,
# les renderers l'acceptent et le rendent comme heading). Fondeur :
# PR #16888 cell-015 (avant-fix commit c34d5e88d, apres-fix la cellule
# a ete rewrap, base e17f600a). Le pattern est disjoint de _HEADING_RE
# (qui exige un espace apres le "#", donc catch deja les " ## Heading"
# indente ou pas) et de _CONTAINER_HEADING_RE (qui exige un marqueur
# de conteneur "-"/"*"/"+"/"1."/">" sur la MEME ligne que le "#",
# donc catch deja les "- # Indice", "1. # Heading" etc.). Classe de
# pattern : ligne de CONTINUATION de puce (2+ espaces d'indentation
# sans marqueur de conteneur) qui commence par "#" + non-espace
# (reference d'issue "#15520", mot "#libelle" etc.). ERROR parce que
# le rendu est incoherent avec la prose que la ligne veut dire.
"heading_continuation": ERROR,
# #12064: ERROR (bloquant) -- the corpus measure is 1 hit / 20,576 markdown
# cells (the true positive (A) PT_11 cell 5), reproduced by this lane. That
# precision is what buys blocking status; a wider pattern set would need
Expand Down Expand Up @@ -313,6 +329,16 @@
# drift gate on new hits is #11829 sous-issue #2. See #11829.
_CONTAINER_HEADING_RE = re.compile(
r"^(?:[ \t]*(?:[-*+]|\d+[.)]|>)[ \t]+){1,3}(#{1,6})\s+(.*\S)\s*$")
# #17005: continuation line of a list item / blockquote (2+ spaces indent, NO
# container marker on the same line) starting with "#" + non-space character.
# CommonMark refuses this as a heading (the indent is too deep, no container
# opener), but Jupyter / VSCode / nbviewer all render it as a giant H1-H6 --
# the same renderer gap as `_CONTAINER_HEADING_RE` (#11829), on a continuation
# line instead of an opener line. Disjoint of both `_HEADING_RE` (which
# requires a space after the "#") and `_CONTAINER_HEADING_RE` (which requires
# a `-`/`*`/`+`/`1.`/`>` on the same line). Fondeur: PR #16888 cell-015
# (base e17f600a, fix c34d5e88d rewrapped the cell to drop the indent).
_CONTINUATION_HEADING_RE = re.compile(r"^\s{2,}(#{1,6})[^\s#]")

# #12110 -- a CJK (Chinese-Japanese-Korean) character in a markdown cell whose
# source is otherwise French prose. The defect pattern: a model-generated cell
Expand Down Expand Up @@ -1216,6 +1242,31 @@ def scan_cell(cell) -> list[dict]:
})
break

# ---- continuation line starting with '#' (#17005) ----------------------------
# Sibling of the heading_in_list loop above: a list-item / blockquote
# CONTINUATION line (2+ spaces indent, NO container marker on the same
# line) that begins with `#` + non-space is rendered as a giant H1-H6 by
# Jupyter / VSCode / nbviewer (CommonMark refuses it; the renderers do
# not). Same fence-awareness / one-finding-per-cell discipline.
for idx, ln in enumerate(lines):
if idx in fenced:
continue
m = _CONTINUATION_HEADING_RE.match(ln)
if not m:
continue
rule = "heading_continuation"
level = len(m.group(1))
findings.append({
"rule": rule,
"severity": RULE_SEVERITY[rule],
"message": (f"list/blockquote continuation line starts with '#' "
f"(renders as giant H{level}); drop the leading indent or "
f"escape the '#' so the line stays prose"),
"evidence": ln.strip()[:100],
"hash": _cell_hash(rule, text),
})
break

# ---- bare code statement in markdown (#12064) --------------------------------
# Fence-aware AND indent-aware: a statement inside a ``` fence renders as
# code (legit), and a block indented 4+ spaces is an indented-code block
Expand Down
93 changes: 91 additions & 2 deletions scripts/notebook_tools/tests/test_detect_markdown_rendering.py
Original file line number Diff line number Diff line change
Expand Up @@ -895,7 +895,7 @@ def test_cli_missing_pyyaml_exits_2_blaming_the_dependency(self, tmp_path):
r = subprocess.run(
[sys.executable, str(script), "--closure", "--quarto-yml", str(yml),
str(tmp_path)],
capture_output=True, text=True, env=env, cwd=tmp_path,
capture_output=True, text=True, encoding="utf-8", env=env, cwd=tmp_path,
)
assert r.returncode == 2, (r.stdout, r.stderr)
assert "pyyaml" in r.stderr
Expand All @@ -909,11 +909,100 @@ def test_cli_with_pyyaml_reads_the_render_list(self, tmp_path):
r = subprocess.run(
[sys.executable, str(script), "--closure", "--quarto-yml", str(yml),
str(tmp_path)],
capture_output=True, text=True, cwd=tmp_path,
capture_output=True, text=True, encoding="utf-8", cwd=tmp_path,
)
assert r.returncode == 0, (r.stdout, r.stderr)
assert "--closure: render-list=1" in r.stderr


class TestHeadingContinuation:
"""Tests de la règle `heading_continuation` introduite par PR #17005/#17009.

Le défaut : un `#` + non-espace en début de ligne de CONTINUATION de puce
/ blockquote (2+ espaces d'indentation, pas de marqueur de conteneur sur
la même ligne) est rendu comme un heading géant par Jupyter / VSCode /
nbviewer. La règle `_CONTINUATION_HEADING_RE = ^\\s{2,}(#{1,6})[^\\s#]`
ferme cet angle mort (disjoint de `_HEADING_RE` et `_CONTAINER_HEADING_RE`).
"""

@staticmethod
def _mk_cell(source_lines):
return {"cell_type": "markdown", "metadata": {}, "source": source_lines}

@staticmethod
def _write_notebook(tmp_path, md_cells):
nb = {
"cells": [TestHeadingContinuation._mk_cell(src) for src in md_cells],
"metadata": {},
"nbformat": 4,
"nbformat_minor": 5,
}
p = tmp_path / "fixture.ipynb"
p.write_text(json.dumps(nb), encoding="utf-8")
return p

def test_continuation_heading_pilote_reference(self, tmp_path):
"""Pilote : cellule #16888 cell-015 avant-fix — continuation ` #15520)`."""
src = [
"- ouverture de la puce\n",
" #15520) — le `#` en début de continuation est rendu comme heading\n",
]
nb_path = self._write_notebook(tmp_path, [src])
r = subprocess.run(
[sys.executable, str(Path(detect_markdown_rendering.__file__).resolve()),
"--json", str(nb_path)],
capture_output=True, text=True, encoding="utf-8", cwd=tmp_path,
)
assert r.returncode == 0, (r.stdout, r.stderr)
data = json.loads(r.stdout)
findings = data.get("findings", [])
hc = [f for f in findings if f.get("rule") == "heading_continuation"]
assert len(hc) == 1, f"expected 1 heading_continuation finding, got {len(hc)}: {findings}"

def test_continuation_heading_clean_post_fix(self, tmp_path):
"""Après rewrap `c34d5e88d` : la cellule propre ne fire plus."""
src = [
"- ouverture de la puce\n",
" reference `#15520` — référence inline, pas un heading\n",
]
nb_path = self._write_notebook(tmp_path, [src])
r = subprocess.run(
[sys.executable, str(Path(detect_markdown_rendering.__file__).resolve()),
"--json", str(nb_path)],
capture_output=True, text=True, encoding="utf-8", cwd=tmp_path,
)
assert r.returncode == 0, (r.stdout, r.stderr)
data = json.loads(r.stdout)
findings = data.get("findings", [])
hc = [f for f in findings if f.get("rule") == "heading_continuation"]
assert len(hc) == 0, f"expected 0 findings, got {findings}"

def test_continuation_heading_no_false_positive_legit_heading(self, tmp_path):
"""4 contrôles négatifs : top-level heading, in-list opener, fenced code, `## ` indent."""
cells = [
# Top-level heading (catch par `_HEADING_RE`, pas continuation)
["## Top-level heading\n"],
# In-list opener (catch par `_CONTAINER_HEADING_RE`, pas continuation)
["- # Heading in list\n"],
# Fenced code avec `#` au début — pas un heading du tout
["```python\n",
" # comment in code\n",
"```\n"],
# `## ` indenté 2 espaces — déjà couvert par `_HEADING_RE` (espace après `#`)
[" ## Heading with leading spaces\n"],
]
nb_path = self._write_notebook(tmp_path, cells)
r = subprocess.run(
[sys.executable, str(Path(detect_markdown_rendering.__file__).resolve()),
"--json", str(nb_path)],
capture_output=True, text=True, encoding="utf-8", cwd=tmp_path,
)
assert r.returncode == 0, (r.stdout, r.stderr)
data = json.loads(r.stdout)
findings = data.get("findings", [])
hc = [f for f in findings if f.get("rule") == "heading_continuation"]
assert len(hc) == 0, f"expected 0 heading_continuation, got {findings}"


if __name__ == "__main__":
sys.exit(pytest.main([__file__, "-v"]))
Loading