Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
80 changes: 26 additions & 54 deletions .github/workflows/pr-gate-rerun.yml
Original file line number Diff line number Diff line change
Expand Up @@ -77,17 +77,30 @@ jobs:
resolve:
name: Resolve the re-aggregation route (manual)
# Q35 etape 2 (#17397) : agregateur pur -- resolve/re-run du gate, aucun
# secret, docker, GPU ni toolchain locale. Re-agreger decide si une PR est
# mergeable : tant que ce travail tourne sur le parc local, il se degrade
# exactement quand la file qu'il doit debloquer grossit (saturation du
# 2026-09-22). La garde same-repo ci-dessous est conservee telle quelle --
# seul le runner change, aucune semantic de declenchement.
# secret, docker, GPU ni toolchain locale (python stdlib seul, #17680).
# Re-agreger decide si une PR est mergeable : tant que ce travail tourne
# sur le parc local, il se degrade exactement quand la file qu'il doit
# debloquer grossit (saturation du 2026-09-22). La garde same-repo
# ci-dessous est conservee telle quelle -- seul le runner change, aucune
# semantic de declenchement.
runs-on: ubuntu-latest
if: github.event.pull_request.head.repo.full_name == null || github.event.pull_request.head.repo.full_name == github.repository
outputs:
action: ${{ steps.resolve.outputs.action }}
run_id: ${{ steps.resolve.outputs.run_id }}
steps:
# #17680 : la route vit dans scripts/ci/pr_gate_route.py (stdlib pur) --
# testable en unitaire, horloge incluse. Le bash inline ne pouvait pas
# distinguer une tentative queued MORTE d'une saine : statut identique,
# decision identique, impasse complete (cf #17099, 19h30 queued).
- uses: actions/checkout@v4
with:
sparse-checkout: |
scripts/ci/pr_gate_route.py
sparse-checkout-cone-mode: false
- uses: actions/setup-python@v5
with:
python-version: '3.11'
- name: Resolve route
id: resolve
env:
Expand All @@ -103,55 +116,14 @@ jobs:
echo "action=skip" >> "$GITHUB_OUTPUT"
exit 0
fi
# Find the original pr-gate.yml run for THIS head SHA. event=pull_request
# only (pr-gate.yml also accepts workflow_dispatch, whose payload's
# --sha target is not the PR head). Most recently CREATED, per #11519:
# a rerun replays the frozen payload of its original event, so an older
# run for the same SHA is fine (same SHA, fresh API reads) but a run
# for an older SHA would re-aggregate a stale head.
RUN_JSON=$(gh api "repos/${REPO}/actions/runs?head_sha=${HEAD_SHA}&event=pull_request&per_page=100" \
--jq '[.workflow_runs[] | select(.name == "PR gate")] | sort_by(.created_at) | last')
if [ -z "$RUN_JSON" ] || [ "$RUN_JSON" = "null" ]; then
# #16624: the OLD message here claimed "the gate will run on its
# own" -- FALSE for the retarget population. A PR retargeted onto
# main (edited, not in the pre-#16624 default types) never fired
# the gate and NOTHING will fire it: no push, no synchronize, and
# a close/reopen performed by a bot token triggers no run
# (GITHUB_TOKEN anti-recursion). The two honest paths are below.
EXISTING=$(gh api "repos/${REPO}/commits/${HEAD_SHA}/check-runs?per_page=100" \
--jq '[.check_runs[] | select(.name == "PR gate")] | length' 2>/dev/null || echo 1)
if [ "$EXISTING" != "0" ]; then
echo "[pr-gate] a 'PR gate' check-run already exists on ${HEAD_SHA} though no pull_request run does -- refusing to POST beside it (#11519 twin), skip"
echo "action=skip" >> "$GITHUB_OUTPUT"
exit 0
fi
echo "[pr-gate] no pull_request PR gate run and no 'PR gate' check-run for ${HEAD_SHA} -- gate ABSENT (#16624): aggregating the head and POSTing the verdict"
echo "action=aggregate_absent" >> "$GITHUB_OUTPUT"
exit 0
fi
RUN_ID=$(echo "$RUN_JSON" | jq -r '.id')
STATUS=$(echo "$RUN_JSON" | jq -r '.status')
CONCLUSION=$(echo "$RUN_JSON" | jq -r '.conclusion')
echo "[pr-gate] target run ${RUN_ID}: status=${STATUS} conclusion=${CONCLUSION}"
echo "run_id=${RUN_ID}" >> "$GITHUB_OUTPUT"
# Already running/queued: it will aggregate the current check state
# by itself -- rerunning now is impossible (API rejects it) and
# useless. This is also the loop bound for near-simultaneous guard
# completions (#11519: "borner la boucle").
if [ "$STATUS" != "completed" ]; then
echo "[pr-gate] run in flight (${STATUS}) -- it will see the fresh guard verdict, skip"
echo "action=skip" >> "$GITHUB_OUTPUT"
exit 0
fi
# Green already: this leg is not what blocks the PR. No rerun.
if [ "$CONCLUSION" = "success" ]; then
echo "[pr-gate] run already green -- nothing to rescue, skip"
echo "action=skip" >> "$GITHUB_OUTPUT"
exit 0
fi
# Full rerun (NOT --failed): the gate is a single job, and a full
# rerun replays the aggregation end-to-end against live check state.
echo "action=rerun" >> "$GITHUB_OUTPUT"
# Route decision, cancel-and-wait for dead queued attempts included:
# no run found -> twin guard / gate-absent (#11519/#16624); queued
# beyond --stale-hours -> cancel + wait completed + rerun (#17680);
# fresh in-flight -> skip; completed -> rerun unless green.
python scripts/ci/pr_gate_route.py \
--repo "$REPO" \
--sha "$HEAD_SHA" \
--pr-number "$PR_NUMBER"

rerun:
name: Re-run the original PR gate (manual)
Expand Down
21 changes: 17 additions & 4 deletions .github/workflows/pr-gate-stale-sweep.yml
Original file line number Diff line number Diff line change
Expand Up @@ -976,10 +976,23 @@ jobs:
echo "[stale-sweep] (dry-run) would re-run #$NUM ($SHA) via run $RUN_ID"
else
echo "[stale-sweep] re-running gate for #$NUM ($SHA) -- run $RUN_ID"
# `|| echo`: one PR whose re-run is refused must not abort the loop
# nor redden a sweep whose other re-runs landed.
gh run rerun "$RUN_ID" --repo "$REPO" \
|| echo "[stale-sweep] re-run refused for #$NUM (non-fatal)"
# A refused re-run must not abort the loop nor redden a sweep
# whose other re-runs landed. #17680: the refusal is the
# dead-queued signature (the API rejects re-running a
# non-completed run) -- the run sits `queued` with an empty job
# list forever (#17099: 19h30). Dispatch the manual harness:
# its route (scripts/ci/pr_gate_route.py) cancels a stale
# queued attempt, waits for `completed`, then re-runs. Harmless
# on the other refusal cause (flipped to in_progress between
# read and rerun): the route sees a fresh in-flight run and
# skips. The harness is concurrency-grouped per PR, so repeated
# refusals cannot stampede.
if ! gh run rerun "$RUN_ID" --repo "$REPO"; then
echo "[stale-sweep] re-run refused for #$NUM (run $RUN_ID not completed) -- dispatching PR gate (re-aggregate) to revive it (#17680)"
gh workflow run pr-gate-rerun.yml --repo "$REPO" \
-f pr_number="$NUM" -f head_sha="$SHA" \
|| echo "[stale-sweep] revive dispatch refused for #$NUM (non-fatal)"
fi
fi
if [ "${RANK:-1}" = "0" ]; then m=$((m + 1)); else i=$((i + 1)); fi
done < /tmp/candidates.txt
Expand Down
Loading
Loading