Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
124 changes: 124 additions & 0 deletions scripts/tests/test_translation_sync.py
Original file line number Diff line number Diff line change
Expand Up @@ -575,3 +575,127 @@ def test_check_csv_fr_contam_lazy_text_load_no_false_positive(tmp_path):
])
anomalies = t2.check_csv(csv, repo)
assert all(a["verdict"] != "FR_CONTAM" for a in anomalies)


# --------------------------------------------------------------------------- #
# PIVOT_HASH_MISMATCH (#17677) #
# #
# Invariant de construction du pivot (T1) : hash_<src_lang> == src_hash == #
# cell_hash(text_<src_lang>). Aucun verdict ne couvrait hash_fr (SRC_DRIFT #
# compare src_hash a la source ; la boucle TARGET_LANGS saute le pivot). #
# Cas fondateur #17649 : un resync manuel avait ecrit dans hash_fr le hash #
# du text_en de la meme ligne (colonnes decalees) et passait tous les checks. #
# --------------------------------------------------------------------------- #


def test_pivot_mismatch_shifted_columns_founding_case(tmp_path):
"""hash_fr porte le hash du text_en (decalage d'un cran, forme #17649).

src_hash est CORRECT (== hash de la source notebook) -> SRC_DRIFT vert ;
seul le champ que personne ne lisait etait faux. C'est exactement la
matrice de l'incident fondateur.
"""
repo = tmp_path
src_text = "## 1. Exploration des donnees sectorielles"
en_text = "## 1. Sector Data Exploration"
_write_notebook(repo / "S" / "Foo.ipynb", [_make_cell("c1", "markdown", src_text)])
csv = _write_csv(tmp_path / "drift.csv", [
_row("S/Foo.ipynb", "c1", t2.cell_hash(src_text), # src_hash correct
**{"text_fr": src_text,
"hash_fr": t2.cell_hash(en_text), # decale : hash du text_en
"text_en": en_text,
"hash_en": t2.cell_hash("old english text")})
])
anomalies = t2.check_csv(csv, repo)
piv = [a for a in anomalies if a["verdict"] == "PIVOT_HASH_MISMATCH"]
assert len(piv) == 1
assert "hash_fr" in piv[0]["detail"]
# La source est in sync : pas de SRC_DRIFT qui viendrait brouiller le signal.
assert all(a["verdict"] != "SRC_DRIFT" for a in anomalies)


def test_pivot_invariant_respected_no_anomaly(tmp_path):
"""hash_fr == src_hash == cell_hash(text_fr) -> aucune anomalie pivot."""
repo = tmp_path
src_text = "Texte pivot coherent"
_write_notebook(repo / "S" / "Foo.ipynb", [_make_cell("c1", "markdown", src_text)])
csv = _write_csv(tmp_path / "drift.csv", [
_row("S/Foo.ipynb", "c1", t2.cell_hash(src_text),
**{"text_fr": src_text, "hash_fr": t2.cell_hash(src_text)})
])
anomalies = t2.check_csv(csv, repo)
assert all(a["verdict"] != "PIVOT_HASH_MISMATCH" for a in anomalies)


def test_pivot_text_incoherent_with_declared_hash(tmp_path):
"""Les deux hash coincident entre eux mais text_fr ne hash pas vers eux
(texte pivot edite sans mise a jour des hashes) -> PIVOT_HASH_MISMATCH."""
repo = tmp_path
_write_notebook(repo / "S" / "Foo.ipynb", [_make_cell("c1", "markdown", "source actuelle")])
stale = "ancien texte pivot jamais reflechi dans les hashes"
csv = _write_csv(tmp_path / "drift.csv", [
_row("S/Foo.ipynb", "c1", t2.cell_hash("hash fantome"),
**{"text_fr": stale, "hash_fr": t2.cell_hash("hash fantome")})
])
anomalies = t2.check_csv(csv, repo)
piv = [a for a in anomalies if a["verdict"] == "PIVOT_HASH_MISMATCH"]
assert len(piv) == 1
assert f"cell_hash(text_fr)" in piv[0]["detail"]


def test_pivot_check_is_row_internal_notebook_absent(tmp_path):
"""La verification pivot ne depend PAS du notebook : une ligne corrompue
dont le notebook est absent reste signalee (coexiste avec ORPHAN_ROW)."""
repo = tmp_path # aucun notebook ecrit
csv = _write_csv(tmp_path / "drift.csv", [
_row("S/Foo.ipynb", "c1", t2.cell_hash("source"),
**{"text_fr": "source", "hash_fr": t2.cell_hash("autre chose")})
])
anomalies = t2.check_csv(csv, repo)
verdicts = [a["verdict"] for a in anomalies]
assert "PIVOT_HASH_MISMATCH" in verdicts
assert "ORPHAN_ROW" in verdicts


def test_pivot_empty_hash_pre_t3_no_anomaly(tmp_path):
"""hash_fr vide (etat pre-T1 legacy) -> lenient, pas d'anomalie pivot."""
repo = tmp_path
src_text = "cellule sans hash depose"
_write_notebook(repo / "S" / "Foo.ipynb", [_make_cell("c1", "markdown", src_text)])
csv = _write_csv(tmp_path / "drift.csv", [
_row("S/Foo.ipynb", "c1", t2.cell_hash(src_text),
**{"text_fr": src_text}) # hash_fr reste vide
])
anomalies = t2.check_csv(csv, repo)
assert all(a["verdict"] != "PIVOT_HASH_MISMATCH" for a in anomalies)


def test_pivot_mismatch_coexists_with_src_drift(tmp_path):
"""Une ligne peut etre a la fois SRC_DRIFT (source bougee) et pivot-corrompue :
les deux verdicts sont complementaires, pas exclusifs."""
repo = tmp_path
_write_notebook(repo / "S" / "Foo.ipynb", [_make_cell("c1", "markdown", "source nouvelle")])
csv = _write_csv(tmp_path / "drift.csv", [
_row("S/Foo.ipynb", "c1", t2.cell_hash("source ancienne"),
**{"text_fr": "source ancienne",
"hash_fr": t2.cell_hash("decale encore")}) # != src_hash
])
anomalies = t2.check_csv(csv, repo)
verdicts = [a["verdict"] for a in anomalies]
assert "SRC_DRIFT" in verdicts
assert "PIVOT_HASH_MISMATCH" in verdicts


def test_roundtrip_extract_then_check_pivot_clean(tmp_path):
"""Un CSV fraichement extrait par T1 ne produit AUCUN PIVOT_HASH_MISMATCH :
l'invariant T1 (hash_{src_lang} == src_hash == cell_hash(text)) tient par
construction, le nouveau verdict ne doit pas le contredire."""
repo = tmp_path
nb = _write_notebook(repo / "S" / "Foo.ipynb", [
_make_cell("c1", "markdown", "# Titre avec accents éàü"),
_make_cell("c2", "code", "print('bonjour')"),
])
rows = t1.extract_notebook(nb, repo, src_lang="fr")
csv = _write_csv(tmp_path / "drift.csv", rows)
anomalies = t2.check_csv(csv, repo)
assert all(a["verdict"] != "PIVOT_HASH_MISMATCH" for a in anomalies)
38 changes: 37 additions & 1 deletion scripts/translation/check_translation_sync.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,14 @@
(non traduite, francais leaké dans la colonne en/es/pt/...).
Miroir Argumentum multilingual-drift-audit.py : val == fr_val,
garde len >= 4 (#6949 harmonisation 5-classes, 4e classe).
PIVOT_HASH_MISMATCH l'invariant de construction du pivot est violé :
hash_<src_lang> != src_hash, ou cell_hash(text_<src_lang>)
!= hash_<src_lang>. extract_cells_to_csv.py (T1) pose les
trois égaux par construction ; un verdict était nécessaire
car SRC_DRIFT compare src_hash à la source notebook et la
boucle TARGET_LANGS saute le pivot : hash_fr n'était lu par
personne (#17677, mesuré sur #17649 — un resync manuel aux
colonnes décalées passait tous les checks au vert).

Note taxonomie Argumentum (#6949) : la 5e classe COGNATE (noms propres /
faux-amis legitiment repetes, kind == "name", informationnelle — hors
Expand Down Expand Up @@ -184,6 +192,35 @@ def check_csv(csv_path: Path, repo_root: Path) -> list[dict]:
if not nb_rel or not cell_id:
continue

csv_src_hash = row.get("src_hash", "")

# PIVOT_HASH_MISMATCH (#17677) : invariant de construction du pivot.
# T1 pose row[hash_{src_lang}] = row["src_hash"] et
# row[text_{src_lang}] = text (avec src_hash = cell_hash(text)) : les
# trois doivent coïncider. Vérification ROW-INTERNE, placée AVANT le
# chargement du notebook : elle ne dépend ni de sa présence ni de sa
# lisibilité (une ligne corrompue reste signalée même en ORPHAN_ROW).
# Sans elle, hash_fr n'était lu par aucun verdict — SRC_DRIFT compare
# src_hash à la source, et la boucle TARGET_LANGS saute le pivot.
row_lang = row.get("src_lang", "") or PIVOT_LANG
csv_pivot_hash = row.get(f"hash_{row_lang}", "") or ""
pivot_text = row.get(f"text_{row_lang}", "") or ""
if csv_pivot_hash and csv_src_hash and csv_pivot_hash != csv_src_hash:
anomalies.append(
{"csv": str(csv_path), "notebook": nb_rel, "cell_id": cell_id,
"verdict": "PIVOT_HASH_MISMATCH",
"detail": f"hash_{row_lang}={csv_pivot_hash} != src_hash={csv_src_hash} "
f"(invariant de construction violé — resync manuel décalé ?)"}
)
elif pivot_text and csv_pivot_hash and cell_hash(pivot_text) != csv_pivot_hash:
anomalies.append(
{"csv": str(csv_path), "notebook": nb_rel, "cell_id": cell_id,
"verdict": "PIVOT_HASH_MISMATCH",
"detail": f"cell_hash(text_{row_lang})={cell_hash(pivot_text)} "
f"!= hash_{row_lang}={csv_pivot_hash} (texte pivot incohérent "
f"avec son hash déclaré)"}
)

# Notebook source (cache par chemin).
if nb_rel not in source_cache:
source_cache[nb_rel] = load_notebook_cells(repo_root / nb_rel)
Expand All @@ -203,7 +240,6 @@ def check_csv(csv_path: Path, repo_root: Path) -> list[dict]:

# SRC_DRIFT : le source a-t-il bougé depuis la dernière synchro ?
current_src = src_cells[cell_id]
csv_src_hash = row.get("src_hash", "")
if csv_src_hash and current_src != csv_src_hash:
anomalies.append(
{"csv": str(csv_path), "notebook": nb_rel, "cell_id": cell_id,
Expand Down
Loading