diff --git a/.github/workflows/pr-gate-sweep-health-advisory.yml b/.github/workflows/pr-gate-sweep-health-advisory.yml index b2465f4376..f7b2b4b0b4 100644 --- a/.github/workflows/pr-gate-sweep-health-advisory.yml +++ b/.github/workflows/pr-gate-sweep-health-advisory.yml @@ -40,6 +40,31 @@ name: PR gate sweep health advisory # (default Actions short-circuit) -- the observer went half-blind exactly when # it was red. Both probes now always run (continue-on-error) and a final # aggregate step carries the red: one red probe must not silence the other. +# +# 2026-09-28 (#18292): both probes are honest but each carried a false positive +# on the `push` leg, and together they made this advisory red on 37 of main's +# 45 commits that day -- an observer red on almost every commit stops being +# read, which is the same failure as being silent (the docstring of +# scripts/ci/check_scheduler_liveness.py names it: "a permanently red organ on +# a chronic condition gets ignored"). +# +# - cadence probe: a DEAD of the `schedule` delivery is CHRONIC since +# 2026-09-09 (#15332 -- 30/60-min crons served every ~4h30). The `push` leg +# now passes `--dead-exit warning`: the verdict is unchanged and NO +# threshold moves, only the severity printed on that leg (::warning::, +# green step). `schedule` and `workflow_dispatch` keep the red -- those are +# the legs where a dead delivery is the thing being observed. +# - sweep-age probe: on `push`, this job and the sweep start from the SAME +# commit, so the probe read the PREVIOUS merge's success while its own +# commit's sweep was still running (run 36461754953 vs sweep 36461754984, +# same head_sha). It now asks first whether a sweep run carries this +# commit and is alive; if so the age has nothing to judge. When no live +# run carries it (rc 3) or the probe fails (rc 4), the age criterion below +# still decides -- the race guard cannot mask a sweep that is truly dead +# for this commit. +# +# Both fixes are the same shape: keep the measurement, stop recopying it as a +# red onto a leg where it is not an incident. Neither weakens a gate. on: schedule: @@ -81,6 +106,7 @@ jobs: env: GH_TOKEN: ${{ github.token }} REPO: ${{ github.repository }} + EVENT_NAME: ${{ github.event_name }} run: | set -uo pipefail # Mesure la cadence SERVIE par organe (declaree vs observee) et rougit @@ -91,7 +117,22 @@ jobs: # d'agregation, qui porte le rouge de CE step et de la sonde d'age -- # sans cela, un echec ici SKIPPERAIT la sonde d'age (short-circuit # Actions, repare sur dispatch ai-01 2026-09-10). - python scripts/ci/check_scheduler_liveness.py --repo "$REPO" + # + # 2026-09-28 (#18292) : sur `push`, un DEAD de livraison 'schedule' + # est un etat CHRONIQUE (#15332 : crons a 30/60 min servis toutes les + # ~4 h 30 depuis le 09/09), pas un incident. Recopie en rouge sur + # cette jambe, il a rougi 37 des 45 commits de main du 28/09 sans + # qu'aucun ne soit casse -- et un rouge sur presque chaque commit + # masque les vrais. La jambe `push` le NOMME donc en avertissement + # (::warning::, step vert) ; les jambes `schedule` et + # `workflow_dispatch` gardent le rouge. Aucun seuil ne bouge et le + # verdict de l'organe est inchange : seule la severite imprimee ici + # change. + if [ "${EVENT_NAME}" = "push" ]; then + python scripts/ci/check_scheduler_liveness.py --repo "$REPO" --dead-exit warning + else + python scripts/ci/check_scheduler_liveness.py --repo "$REPO" + fi - name: Check age of last successful sweep run id: sweep_age @@ -99,10 +140,30 @@ jobs: continue-on-error: true env: GH_TOKEN: ${{ github.token }} + REPO: ${{ github.repository }} + EVENT_NAME: ${{ github.event_name }} STALE_AFTER_MINUTES: "60" run: | set -uo pipefail + # 2026-09-28 (#18292) : sur `push`, la jambe du sweep part du MEME + # commit que celle-ci. Cette sonde cherche le dernier succes TERMINE, + # donc tant que le sweep de ce commit tourne elle lit celui du merge + # precedent -- et rougit un commit dont le sweep est en cours + # (instance : run 36461754953, sweep 36461754984, meme head_sha). + # Ce n'est pas une cadence morte, c'est une course : quand un run du + # sweep porte ce commit et est vivant (ou vient de conclure + # `success`), l'age n'a rien a juger. rc 3 (aucun run vivant) et rc 4 + # (sonde en echec) retombent sur le critere d'age ci-dessous -- le + # mode ne remplace pas la sonde, il lui retire un faux positif ; il ne + # peut donc pas masquer un sweep vraiment mort pour ce commit. + if [ "${EVENT_NAME}" = "push" ]; then + if python scripts/ci/check_scheduler_liveness.py --repo "$REPO" --sweep-alive-for-sha "${GITHUB_SHA}"; then + echo "[sweep-health] sweep vivant pour ce commit -- age non juge (course, #18292)" + exit 0 + fi + fi + # Most recent *successful* run of the sweep, with its creation time. # Filtering to conclusion=success keeps a transient failure from # tripping the alarm as long as a success happened inside the window. diff --git a/scripts/ci/check_scheduler_liveness.py b/scripts/ci/check_scheduler_liveness.py index 6183ae2e83..7bd231ca18 100644 --- a/scripts/ci/check_scheduler_liveness.py +++ b/scripts/ci/check_scheduler_liveness.py @@ -36,13 +36,39 @@ - ``OK`` : last scheduled run within 2x the declared interval - ``LATE`` : within 2x-4x (degraded delivery; visible, not red) - ``DEAD`` : beyond max(4x declared, 90 min) -- or no scheduled run at all - in history. THIS is what reddens the step (exit 1). + in history. THIS is what reddens the step (exit 1) -- sauf en + mode ``--dead-exit warning``, ou il est nomme et laisse le + step vert (cf. le paragraphe suivant). - ``UNKNOWN`` : the gh probe itself failed for that organ -- named, never silently folded into OK or DEAD. Control against a dead instrument: if EVERY organ returns UNKNOWN or zero runs, the probe -- not the scheduler -- is the suspect, and the organ exits 1 with ``INSTRUMENT_UNKNOWN`` rather than reporting a fleet-wide green. + +## Severite d'un DEAD : deux modes, jamais un seuil deplace (#18292) + +Mesure du 2026-09-28 : la livraison ``schedule`` de GitHub sert les crons a 30 et +60 min toutes les ~4 h 30 depuis le 09/09 (#15332), donc un DEAD est CHRONIQUE +et non un incident. Recopie en rouge sur la jambe ``push``, il rougissait **37 +des 45 commits** de ``main`` du jour, sans qu'aucun de ces commits ne soit +cassee -- et un rouge sur presque chaque commit masque les vrais rouges, ce que +la docstring ci-dessus dit vouloir eviter. + +``--dead-exit warning`` nomme le DEAD (``::warning::``) et laisse le step vert. +Le **verdict est inchange** (``evaluate`` ne bouge pas) et aucun seuil ne monte : +seule la severite imprimee change, et elle se choisit **par jambe** dans le +workflow -- ``push`` en avertissement, ``schedule`` et ``workflow_dispatch`` +en rouge. ``INSTRUMENT_UNKNOWN`` garde son rouge dans les deux modes : c'est la +mesure qui est en panne, pas la cadence qu'on excuse. + +## Sweep du meme push (#18292) + +``sweep_run_alive_for_sha`` repond a une course, pas a une cadence : la jambe +``push`` de l'advisory et celle du sweep partent du meme commit, donc la sonde +d'age -- qui cherche le dernier succes TERMINE -- trouve celui du merge +precedent et rougit un commit dont le sweep tourne encore (instance : run +``36461754953``, sweep ``36461754984``, meme ``head_sha``). """ from __future__ import annotations @@ -190,7 +216,9 @@ def probe(repo: str, workflow: str) -> tuple[list[datetime] | None, str]: "--json", "createdAt", ] try: - out = subprocess.run(cmd, capture_output=True, text=True, timeout=120) + out = subprocess.run( + cmd, capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=120 + ) except (OSError, subprocess.SubprocessError) as exc: return None, f"sonde gh indisponible: {type(exc).__name__}" if out.returncode != 0: @@ -202,12 +230,118 @@ def probe(repo: str, workflow: str) -> tuple[list[datetime] | None, str]: return None, f"payload illisible: {type(exc).__name__}" +# --- Sweep du meme push (#18292) ------------------------------------------- + +SWEEP_WORKFLOW = "pr-gate-stale-sweep.yml" + +# Etats de file d'Actions qui rendent (ou vont rendre) le service. `completed` +# est traite a part : il ne vaut service que sur `success`. +LIVE_RUN_STATES = ("queued", "in_progress", "waiting", "pending", "requested") + + +def sweep_run_alive_for_sha(runs: list[dict], sha: str) -> tuple[bool, str]: + """Un run du sweep porte-t-il CE commit, et rend-il (ou va-t-il rendre) le service ? + + Vivant = ``queued``/``in_progress`` (le service va etre rendu) ou + ``completed``+``success`` (il vient de l'etre). Un run ``completed`` non + reussi ne compte PAS : un sweep en echec ne vaut pas service rendu, et la + sonde d'age doit garder le droit de rougir. Plusieurs runs peuvent porter le + meme ``head_sha`` (relances) : le premier vivant suffit. + """ + for run in runs: + if (run.get("headSha") or "") != sha: + continue + status = (run.get("status") or "").lower() + conclusion = (run.get("conclusion") or "").lower() + if status in LIVE_RUN_STATES: + return True, status + if status == "completed" and conclusion == "success": + return True, "completed/success" + return False, "" + + +def probe_sweep_runs(repo: str) -> tuple[list[dict] | None, str]: + """(runs, error). runs est None quand la sonde elle-meme a echoue.""" + cmd = [ + "gh", "run", "list", "--repo", repo, "--workflow", SWEEP_WORKFLOW, + "--limit", str(SAMPLE_LIMIT), "--json", "headSha,status,conclusion", + ] + try: + out = subprocess.run( + cmd, capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=120 + ) + except (OSError, subprocess.SubprocessError) as exc: + return None, f"sonde gh indisponible: {type(exc).__name__}" + if out.returncode != 0: + err = (out.stderr or "").strip().splitlines() + return None, f"gh rc={out.returncode}: {err[0] if err else 'sans message'}" + try: + runs = json.loads(out.stdout or "[]") + except (ValueError, TypeError) as exc: + return None, f"payload illisible: {type(exc).__name__}" + if not isinstance(runs, list): + return None, "payload illisible: liste de runs attendue" + return runs, "" + + +def sweep_alive_verdict(repo: str, sha: str) -> int: + """0 = sweep vivant pour ce commit, 3 = pas vivant, 4 = sonde en echec. + + 3 et 4 laissent la jambe retomber sur le critere d'age : le mode ne remplace + pas la sonde d'age, il lui retire un faux positif de course. + """ + runs, error = probe_sweep_runs(repo) + if runs is None: + print( + f"[sweep-health] sonde des runs de {SWEEP_WORKFLOW} en echec ({error}) " + "-- repli sur le critere d'age", + file=sys.stderr, + ) + return 4 + alive, state = sweep_run_alive_for_sha(runs, sha) + if alive: + print( + f"[sweep-health] un run de {SWEEP_WORKFLOW} porte ce commit ({state}) " + "-- sweep vivant, l'age n'est pas juge" + ) + return 0 + print( + f"[sweep-health] aucun run vivant de {SWEEP_WORKFLOW} pour ce commit " + "-- repli sur le critere d'age" + ) + return 3 + + def main(argv: list[str] | None = None) -> int: ap = argparse.ArgumentParser(description="Scheduler liveness (#15332)") ap.add_argument("--repo", default=os.environ.get("REPO", "jsboige/CoursIA")) ap.add_argument("--json", action="store_true", help="sortie machine") + ap.add_argument( + "--dead-exit", + choices=("error", "warning"), + default="error", + help=( + "severite d'un verdict DEAD : 'error' (defaut, exit 1) ou 'warning' " + "(::warning:: nomme, exit 0). Ne deplace AUCUN seuil et ne change " + "aucun verdict -- separe seulement le rouge de la jambe `push` de " + "celui des jambes `schedule`/`workflow_dispatch` (#18292)." + ), + ) + ap.add_argument( + "--sweep-alive-for-sha", + metavar="SHA", + default=None, + help=( + "mode sonde de course (#18292) : 0 si un run de " + f"{SWEEP_WORKFLOW} porte ce commit et est vivant, 3 sinon, " + "4 si la sonde echoue" + ), + ) args = ap.parse_args(argv) + if args.sweep_alive_for_sha: + return sweep_alive_verdict(args.repo, args.sweep_alive_for_sha) + now = datetime.now(timezone.utc) results: list[Liveness] = [] for organ in REGISTRY: @@ -263,12 +397,22 @@ def main(argv: list[str] | None = None) -> int: ) return 1 + dead_marker = "::warning::" if args.dead_exit == "warning" else "::error::" for r in dead: print( - f"::error::[scheduler-liveness] {r.organ.workflow}: {r.detail or 'dernier run planifie trop ancien'} " + f"{dead_marker}[scheduler-liveness] {r.organ.workflow}: {r.detail or 'dernier run planifie trop ancien'} " f"-- cadence declaree {r.organ.declared_min:.0f} min ({r.organ.note}). Voir #15332.", file=sys.stderr, ) + if dead and args.dead_exit == "warning": + print( + f"::warning::[scheduler-liveness] {len(dead)} verdict(s) DEAD rendus en " + "avertissement sur cette jambe : la livraison 'schedule' est un etat " + "CHRONIQUE (#15332), et un rouge sur presque chaque commit masque les " + "vrais rouges. Le rouge reste porte par les jambes 'schedule' et " + "'workflow_dispatch' (#18292).", + file=sys.stderr, + ) for r in unknown: print( f"::warning::[scheduler-liveness] {r.organ.workflow}: sonde en echec ({r.detail}) " @@ -282,7 +426,9 @@ def main(argv: list[str] | None = None) -> int: file=sys.stderr, ) - return 1 if dead else 0 + if dead and args.dead_exit == "error": + return 1 + return 0 if __name__ == "__main__": diff --git a/scripts/tests/test_check_scheduler_liveness.py b/scripts/tests/test_check_scheduler_liveness.py index 8e1a89468b..8c39341f0f 100644 --- a/scripts/tests/test_check_scheduler_liveness.py +++ b/scripts/tests/test_check_scheduler_liveness.py @@ -182,3 +182,164 @@ def test_cli_json_surface(monkeypatch, capsys): assert sl.main(["--repo", "x/y", "--json"]) == 0 payload = json.loads(capsys.readouterr().out) assert payload["dead"] == 0 and len(payload["organs"]) == len(sl.REGISTRY) + + +# --- severite d'un DEAD (#18292) --------------------------------------------- +# +# Mesure du 2026-09-28: un DEAD de livraison 'schedule' recopie en rouge sur la +# jambe `push` a rougi 37 des 45 commits de main du jour, sans qu'aucun ne soit +# casse. La jambe `push` passe donc `--dead-exit warning`. Le VERDICT ne bouge +# pas et aucun seuil ne monte -- seule la severite imprimee change. + +def _one_dead_fleet(monkeypatch): + """Six organs, exactly one DEAD (the hourly sweep at 10 h).""" + mapping = {o.workflow: (fresh(), "") for o in sl.REGISTRY} + mapping[HOURLY.workflow] = (fresh(10), "") + _stub_probe(mapping, monkeypatch) + + +def test_cli_dead_default_stays_error_and_exit_1(monkeypatch, capsys): + _one_dead_fleet(monkeypatch) + assert sl.main(["--repo", "x/y"]) == 1 + err = capsys.readouterr().err + assert "::error::[scheduler-liveness] pr-gate-stale-sweep.yml" in err + + +def test_cli_dead_exit_warning_names_it_green(monkeypatch, capsys): + _one_dead_fleet(monkeypatch) + assert sl.main(["--repo", "x/y", "--dead-exit", "warning"]) == 0 + err = capsys.readouterr().err + # Nomme, pas tu: le DEAD est toujours imprime, seul son marqueur change. + assert "::warning::[scheduler-liveness] pr-gate-stale-sweep.yml" in err + assert "::error::[scheduler-liveness] pr-gate-stale-sweep.yml" not in err + assert "avertissement sur cette jambe" in err + + +def test_severity_does_not_move_a_threshold(): + # Le litmus anti-gaming: --dead-exit ne touche ni les seuils ni le verdict. + assert sl.DEAD_FACTOR == 4 and sl.DEAD_FLOOR_MIN == 90.0 + assert sl.verdict_for(241.0, 60.0) == "DEAD" + assert sl.verdict_for(239.0, 60.0) == "LATE" + + +def test_instrument_unknown_keeps_its_red_in_warning_mode(monkeypatch, capsys): + # C'est la MESURE qui est en panne, pas la cadence qu'on excuse. + mapping = {o.workflow: (None, "sonde gh indisponible: OSError") for o in sl.REGISTRY} + _stub_probe(mapping, monkeypatch) + assert sl.main(["--repo", "x/y", "--dead-exit", "warning"]) == 1 + assert "INSTRUMENT_UNKNOWN" in capsys.readouterr().err + + +# --- sweep du meme push (#18292) --------------------------------------------- +# +# Course, pas cadence: sur `push`, cette jambe et le sweep partent du MEME +# commit. La sonde d'age cherche le dernier succes TERMINE et lisait donc celui +# du merge precedent (instance fondatrice: run 36461754953 vs sweep 36461754984, +# meme head_sha). + +SHA = "36461754953a" * 2 + + +def _run(sha, status, conclusion=""): + return {"headSha": sha, "status": status, "conclusion": conclusion} + + +def test_sweep_alive_covers_the_founding_race(): + runs = [_run("76461754984b", "completed", "success"), _run(SHA, "in_progress")] + assert sl.sweep_run_alive_for_sha(runs, SHA) == (True, "in_progress") + + +def test_sweep_queued_and_success_are_alive(): + assert sl.sweep_run_alive_for_sha([_run(SHA, "queued")], SHA) == (True, "queued") + assert sl.sweep_run_alive_for_sha( + [_run(SHA, "completed", "success")], SHA + ) == (True, "completed/success") + + +def test_sweep_failure_for_this_sha_is_not_service(): + # Un sweep en echec ne vaut pas service rendu: la sonde d'age garde le + # droit de rougir. + assert sl.sweep_run_alive_for_sha([_run(SHA, "completed", "failure")], SHA) == (False, "") + + +def test_sweep_other_sha_does_not_excuse_this_commit(): + # Le coeur du fix: un run vivant pour un AUTRE commit ne dit rien de celui-ci. + assert sl.sweep_run_alive_for_sha([_run("deadbeef" * 5, "in_progress")], SHA) == (False, "") + + +def test_sweep_empty_history_and_missing_sha_are_not_alive(): + assert sl.sweep_run_alive_for_sha([], SHA) == (False, "") + assert sl.sweep_run_alive_for_sha([{"status": "in_progress"}], SHA) == (False, "") + + +def test_sweep_rerun_of_same_sha_first_live_run_wins(): + runs = [ + _run(SHA, "completed", "cancelled"), + _run("deadbeef" * 5, "in_progress"), + _run(SHA, "in_progress"), + ] + assert sl.sweep_run_alive_for_sha(runs, SHA) == (True, "in_progress") + + +def test_sweep_probe_payload_must_be_a_list(monkeypatch): + class R: + returncode = 0 + stdout = '{"headSha": "x"}' + stderr = "" + monkeypatch.setattr(sl.subprocess, "run", lambda *a, **k: R()) + runs, error = sl.probe_sweep_runs("jsboige/CoursIA") + assert runs is None and "liste de runs attendue" in error + + +def test_sweep_probe_failure_is_named(monkeypatch): + class R: + returncode = 1 + stdout = "" + stderr = "gh: rate limit exceeded" + monkeypatch.setattr(sl.subprocess, "run", lambda *a, **k: R()) + runs, error = sl.probe_sweep_runs("jsboige/CoursIA") + assert runs is None and "rc=1" in error + + +def _stub_sweep_probe(payload, monkeypatch): + def fake(repo): + return payload + monkeypatch.setattr(sl, "probe_sweep_runs", fake) + + +def test_cli_sweep_alive_positive_control_exits_0(monkeypatch, capsys): + # Controle positif d'acceptance: rejouer la situation du run 36461754953 + # (sweep du meme head_sha en cours) doit rendre une sonde d'age verte. + _stub_probe_never_called(monkeypatch) + _stub_sweep_probe(([_run(SHA, "in_progress")], ""), monkeypatch) + assert sl.main(["--repo", "x/y", "--sweep-alive-for-sha", SHA]) == 0 + assert "sweep vivant" in capsys.readouterr().out + + +def test_cli_sweep_alive_negative_control_exits_3(monkeypatch, capsys): + # Controle negatif: aucun run du sweep pour ce commit -> repli sur l'age, + # donc rc 3 (et non 0): le mode ne peut pas masquer un sweep mort. + _stub_sweep_probe(([_run("deadbeef" * 5, "completed", "success")], ""), monkeypatch) + assert sl.main(["--repo", "x/y", "--sweep-alive-for-sha", SHA]) == 3 + assert "repli sur le critere d'age" in capsys.readouterr().out + + +def test_cli_sweep_alive_probe_failure_exits_4(monkeypatch, capsys): + _stub_sweep_probe((None, "gh rc=1: rate limit exceeded"), monkeypatch) + assert sl.main(["--repo", "x/y", "--sweep-alive-for-sha", SHA]) == 4 + assert "repli sur le critere d'age" in capsys.readouterr().err + + +def test_cli_sweep_alive_short_circuits_the_cadence_probe(monkeypatch): + _stub_probe_never_called(monkeypatch) + _stub_sweep_probe(([], ""), monkeypatch) + assert sl.main(["--repo", "x/y", "--sweep-alive-for-sha", SHA]) == 3 + + +def _stub_probe_never_called(monkeypatch): + """Fails loudly if the race mode still walks the cadence registry.""" + + def boom(repo, workflow): + raise AssertionError("--sweep-alive-for-sha ne doit pas sonder la cadence") + + monkeypatch.setattr(sl, "probe", boom)