From 14f921385b3f2222dc1232f97687f232f2f309c1 Mon Sep 17 00:00:00 2001 From: jsboige Date: Thu, 24 Sep 2026 16:58:05 +0200 Subject: [PATCH] fix(translation,#17677): valider l'invariant du pivot hash_ == src_hash Nouveau verdict PIVOT_HASH_MISMATCH dans check_translation_sync.py (T2). La boucle TARGET_LANGS saute la langue pivot (fr) et SRC_DRIFT ne compare que src_hash a la source notebook : hash_fr n'etait lu par aucun check. Un resync manuel aux colonnes decalees (#17649, revert a0693034f2) passait tous les checks au vert sur une ligne incoherente. Check ROW-INTERNE, place avant le chargement du notebook (une ligne corrompue reste signalee meme en ORPHAN_ROW) : - hash_ != src_hash -> invariant de construction viole - sinon cell_hash(text_) != hash_ -> texte pivot incoherent avec son hash declare 8 tests nouveaux (55/55 verts). Recette T1 bornee par un roundtrip extract -> check : 0 faux positif. Passage sur les CSV reels : 90 lignes signalees (3 hash decale, 87 text_fr stale vs hash) - advisory en CI (--check non-bloquant, translation-drift.yml), tri en suivi. Closes #17677 Co-Authored-By: Claude Sonnet 5 --- scripts/tests/test_translation_sync.py | 124 ++++++++++++++++++ scripts/translation/check_translation_sync.py | 38 +++++- 2 files changed, 161 insertions(+), 1 deletion(-) diff --git a/scripts/tests/test_translation_sync.py b/scripts/tests/test_translation_sync.py index 3e3860935c..280720761b 100644 --- a/scripts/tests/test_translation_sync.py +++ b/scripts/tests/test_translation_sync.py @@ -575,3 +575,127 @@ def test_check_csv_fr_contam_lazy_text_load_no_false_positive(tmp_path): ]) anomalies = t2.check_csv(csv, repo) assert all(a["verdict"] != "FR_CONTAM" for a in anomalies) + + +# --------------------------------------------------------------------------- # +# PIVOT_HASH_MISMATCH (#17677) # +# # +# Invariant de construction du pivot (T1) : hash_ == src_hash == # +# cell_hash(text_). Aucun verdict ne couvrait hash_fr (SRC_DRIFT # +# compare src_hash a la source ; la boucle TARGET_LANGS saute le pivot). # +# Cas fondateur #17649 : un resync manuel avait ecrit dans hash_fr le hash # +# du text_en de la meme ligne (colonnes decalees) et passait tous les checks. # +# --------------------------------------------------------------------------- # + + +def test_pivot_mismatch_shifted_columns_founding_case(tmp_path): + """hash_fr porte le hash du text_en (decalage d'un cran, forme #17649). + + src_hash est CORRECT (== hash de la source notebook) -> SRC_DRIFT vert ; + seul le champ que personne ne lisait etait faux. C'est exactement la + matrice de l'incident fondateur. + """ + repo = tmp_path + src_text = "## 1. Exploration des donnees sectorielles" + en_text = "## 1. Sector Data Exploration" + _write_notebook(repo / "S" / "Foo.ipynb", [_make_cell("c1", "markdown", src_text)]) + csv = _write_csv(tmp_path / "drift.csv", [ + _row("S/Foo.ipynb", "c1", t2.cell_hash(src_text), # src_hash correct + **{"text_fr": src_text, + "hash_fr": t2.cell_hash(en_text), # decale : hash du text_en + "text_en": en_text, + "hash_en": t2.cell_hash("old english text")}) + ]) + anomalies = t2.check_csv(csv, repo) + piv = [a for a in anomalies if a["verdict"] == "PIVOT_HASH_MISMATCH"] + assert len(piv) == 1 + assert "hash_fr" in piv[0]["detail"] + # La source est in sync : pas de SRC_DRIFT qui viendrait brouiller le signal. + assert all(a["verdict"] != "SRC_DRIFT" for a in anomalies) + + +def test_pivot_invariant_respected_no_anomaly(tmp_path): + """hash_fr == src_hash == cell_hash(text_fr) -> aucune anomalie pivot.""" + repo = tmp_path + src_text = "Texte pivot coherent" + _write_notebook(repo / "S" / "Foo.ipynb", [_make_cell("c1", "markdown", src_text)]) + csv = _write_csv(tmp_path / "drift.csv", [ + _row("S/Foo.ipynb", "c1", t2.cell_hash(src_text), + **{"text_fr": src_text, "hash_fr": t2.cell_hash(src_text)}) + ]) + anomalies = t2.check_csv(csv, repo) + assert all(a["verdict"] != "PIVOT_HASH_MISMATCH" for a in anomalies) + + +def test_pivot_text_incoherent_with_declared_hash(tmp_path): + """Les deux hash coincident entre eux mais text_fr ne hash pas vers eux + (texte pivot edite sans mise a jour des hashes) -> PIVOT_HASH_MISMATCH.""" + repo = tmp_path + _write_notebook(repo / "S" / "Foo.ipynb", [_make_cell("c1", "markdown", "source actuelle")]) + stale = "ancien texte pivot jamais reflechi dans les hashes" + csv = _write_csv(tmp_path / "drift.csv", [ + _row("S/Foo.ipynb", "c1", t2.cell_hash("hash fantome"), + **{"text_fr": stale, "hash_fr": t2.cell_hash("hash fantome")}) + ]) + anomalies = t2.check_csv(csv, repo) + piv = [a for a in anomalies if a["verdict"] == "PIVOT_HASH_MISMATCH"] + assert len(piv) == 1 + assert f"cell_hash(text_fr)" in piv[0]["detail"] + + +def test_pivot_check_is_row_internal_notebook_absent(tmp_path): + """La verification pivot ne depend PAS du notebook : une ligne corrompue + dont le notebook est absent reste signalee (coexiste avec ORPHAN_ROW).""" + repo = tmp_path # aucun notebook ecrit + csv = _write_csv(tmp_path / "drift.csv", [ + _row("S/Foo.ipynb", "c1", t2.cell_hash("source"), + **{"text_fr": "source", "hash_fr": t2.cell_hash("autre chose")}) + ]) + anomalies = t2.check_csv(csv, repo) + verdicts = [a["verdict"] for a in anomalies] + assert "PIVOT_HASH_MISMATCH" in verdicts + assert "ORPHAN_ROW" in verdicts + + +def test_pivot_empty_hash_pre_t3_no_anomaly(tmp_path): + """hash_fr vide (etat pre-T1 legacy) -> lenient, pas d'anomalie pivot.""" + repo = tmp_path + src_text = "cellule sans hash depose" + _write_notebook(repo / "S" / "Foo.ipynb", [_make_cell("c1", "markdown", src_text)]) + csv = _write_csv(tmp_path / "drift.csv", [ + _row("S/Foo.ipynb", "c1", t2.cell_hash(src_text), + **{"text_fr": src_text}) # hash_fr reste vide + ]) + anomalies = t2.check_csv(csv, repo) + assert all(a["verdict"] != "PIVOT_HASH_MISMATCH" for a in anomalies) + + +def test_pivot_mismatch_coexists_with_src_drift(tmp_path): + """Une ligne peut etre a la fois SRC_DRIFT (source bougee) et pivot-corrompue : + les deux verdicts sont complementaires, pas exclusifs.""" + repo = tmp_path + _write_notebook(repo / "S" / "Foo.ipynb", [_make_cell("c1", "markdown", "source nouvelle")]) + csv = _write_csv(tmp_path / "drift.csv", [ + _row("S/Foo.ipynb", "c1", t2.cell_hash("source ancienne"), + **{"text_fr": "source ancienne", + "hash_fr": t2.cell_hash("decale encore")}) # != src_hash + ]) + anomalies = t2.check_csv(csv, repo) + verdicts = [a["verdict"] for a in anomalies] + assert "SRC_DRIFT" in verdicts + assert "PIVOT_HASH_MISMATCH" in verdicts + + +def test_roundtrip_extract_then_check_pivot_clean(tmp_path): + """Un CSV fraichement extrait par T1 ne produit AUCUN PIVOT_HASH_MISMATCH : + l'invariant T1 (hash_{src_lang} == src_hash == cell_hash(text)) tient par + construction, le nouveau verdict ne doit pas le contredire.""" + repo = tmp_path + nb = _write_notebook(repo / "S" / "Foo.ipynb", [ + _make_cell("c1", "markdown", "# Titre avec accents éàü"), + _make_cell("c2", "code", "print('bonjour')"), + ]) + rows = t1.extract_notebook(nb, repo, src_lang="fr") + csv = _write_csv(tmp_path / "drift.csv", rows) + anomalies = t2.check_csv(csv, repo) + assert all(a["verdict"] != "PIVOT_HASH_MISMATCH" for a in anomalies) diff --git a/scripts/translation/check_translation_sync.py b/scripts/translation/check_translation_sync.py index c40ace0acb..15e1a1c038 100644 --- a/scripts/translation/check_translation_sync.py +++ b/scripts/translation/check_translation_sync.py @@ -22,6 +22,14 @@ (non traduite, francais leaké dans la colonne en/es/pt/...). Miroir Argumentum multilingual-drift-audit.py : val == fr_val, garde len >= 4 (#6949 harmonisation 5-classes, 4e classe). + PIVOT_HASH_MISMATCH l'invariant de construction du pivot est violé : + hash_ != src_hash, ou cell_hash(text_) + != hash_. extract_cells_to_csv.py (T1) pose les + trois égaux par construction ; un verdict était nécessaire + car SRC_DRIFT compare src_hash à la source notebook et la + boucle TARGET_LANGS saute le pivot : hash_fr n'était lu par + personne (#17677, mesuré sur #17649 — un resync manuel aux + colonnes décalées passait tous les checks au vert). Note taxonomie Argumentum (#6949) : la 5e classe COGNATE (noms propres / faux-amis legitiment repetes, kind == "name", informationnelle — hors @@ -184,6 +192,35 @@ def check_csv(csv_path: Path, repo_root: Path) -> list[dict]: if not nb_rel or not cell_id: continue + csv_src_hash = row.get("src_hash", "") + + # PIVOT_HASH_MISMATCH (#17677) : invariant de construction du pivot. + # T1 pose row[hash_{src_lang}] = row["src_hash"] et + # row[text_{src_lang}] = text (avec src_hash = cell_hash(text)) : les + # trois doivent coïncider. Vérification ROW-INTERNE, placée AVANT le + # chargement du notebook : elle ne dépend ni de sa présence ni de sa + # lisibilité (une ligne corrompue reste signalée même en ORPHAN_ROW). + # Sans elle, hash_fr n'était lu par aucun verdict — SRC_DRIFT compare + # src_hash à la source, et la boucle TARGET_LANGS saute le pivot. + row_lang = row.get("src_lang", "") or PIVOT_LANG + csv_pivot_hash = row.get(f"hash_{row_lang}", "") or "" + pivot_text = row.get(f"text_{row_lang}", "") or "" + if csv_pivot_hash and csv_src_hash and csv_pivot_hash != csv_src_hash: + anomalies.append( + {"csv": str(csv_path), "notebook": nb_rel, "cell_id": cell_id, + "verdict": "PIVOT_HASH_MISMATCH", + "detail": f"hash_{row_lang}={csv_pivot_hash} != src_hash={csv_src_hash} " + f"(invariant de construction violé — resync manuel décalé ?)"} + ) + elif pivot_text and csv_pivot_hash and cell_hash(pivot_text) != csv_pivot_hash: + anomalies.append( + {"csv": str(csv_path), "notebook": nb_rel, "cell_id": cell_id, + "verdict": "PIVOT_HASH_MISMATCH", + "detail": f"cell_hash(text_{row_lang})={cell_hash(pivot_text)} " + f"!= hash_{row_lang}={csv_pivot_hash} (texte pivot incohérent " + f"avec son hash déclaré)"} + ) + # Notebook source (cache par chemin). if nb_rel not in source_cache: source_cache[nb_rel] = load_notebook_cells(repo_root / nb_rel) @@ -203,7 +240,6 @@ def check_csv(csv_path: Path, repo_root: Path) -> list[dict]: # SRC_DRIFT : le source a-t-il bougé depuis la dernière synchro ? current_src = src_cells[cell_id] - csv_src_hash = row.get("src_hash", "") if csv_src_hash and current_src != csv_src_hash: anomalies.append( {"csv": str(csv_path), "notebook": nb_rel, "cell_id": cell_id,