Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
240 changes: 145 additions & 95 deletions scripts/check_adjoint_prevalidation.py
Original file line number Diff line number Diff line change
Expand Up @@ -103,7 +103,7 @@
import subprocess
import sys
from dataclasses import dataclass
from typing import Any
from typing import Any, Sequence

try:
import gh_identity
Expand Down Expand Up @@ -266,10 +266,21 @@ class Dossier:
created_at: str = ""


API_USAGE: dict[str, int] = {"rest": 0, "graphql": 0}


def gh_json(args: list[str]) -> Any:
proc = subprocess.run(
["gh", *args], capture_output=True, text=True, encoding="utf-8"
)
# Compteur REST/GraphQL (#17315) : tout l'organe paie ici, la mesure aussi.
# GraphQL (`api graphql`) consomme le bucket partage de la flotte ; tout le
# reste (REST `api`, `pr view`, ...) paie le bucket core. Publie par
# ``--json`` (cle ``api_usage``) et sur stderr en fin d'invocation.
if len(args) >= 2 and args[0] == "api" and args[1] == "graphql":
API_USAGE["graphql"] += 1
else:
API_USAGE["rest"] += 1
if proc.returncode != 0:
raise RuntimeError(proc.stderr.strip() or "gh command failed")
return json.loads(proc.stdout)
Expand Down Expand Up @@ -608,8 +619,11 @@ def _fingerprint_payload(
"threads": snapshot.get("threads") or [],
}
if include_checks:
# Diagnostic-only branch (#17315): reads the REST check-runs of the
# head (present in every snapshot) instead of the retired GraphQL
# rollup. Never hashed into a stamp since #16957.
payload["checks"] = sorted(
snapshot.get("statusCheckRollup") or [],
snapshot.get("checkRuns") or [],
key=lambda row: json.dumps(
row, sort_keys=True, separators=(",", ":")
),
Expand Down Expand Up @@ -698,30 +712,6 @@ def surfaces_fingerprint(
)


def legacy_surfaces_fingerprint(
snapshot: dict[str, Any],
comment_limit: int | None = None,
neutral_after: str | None = None,
) -> str:
"""Pre-#16957 stamp algorithm: the same payload PLUS the check rollup.

Kept so dossiers stamped before #16957 -- whose hash embedded the check
state -- remain verifiable for as long as that state is byte-identical.
Once any check concludes, a legacy stamp stops matching; recovery is one
mechanical re-stamp (--template recomputes every mechanical field, no
re-reading of surfaces), after which no check conclusion can ever expire
the dossier again. A SHA-256 over data that has since changed cannot be
re-derived, which is why zero-touch recovery of raced legacy stamps is
not offered. Drop this function when no open dossier carries a legacy
stamp.
"""
return _digest(
_fingerprint_payload(
snapshot, comment_limit, neutral_after, True, bot_forms_18637=False
)
)


def pre18637_surfaces_fingerprint(
snapshot: dict[str, Any],
comment_limit: int | None = None,
Expand Down Expand Up @@ -1189,20 +1179,22 @@ def validate_dossier(dossier: Dossier, snapshot: dict[str, Any]) -> list[str]:
if not re.fullmatch(r"[0-9a-f]{64}", f.get("surfaces-sha256", "")):
errors.append("surfaces-sha256 must be a lowercase SHA-256")
# Dual acceptance (#16957): a stamp matches the post-fix fingerprint
# (discussion surfaces only) or the legacy one (which also embedded the
# check rollup). Both certify every discussion surface; the legacy digest
# (discussion surfaces only) or the pre-#18637 one (bot comments hashed
# whole). Both certify every discussion surface; the pre-#18637 digest
# is strictly more fields, so accepting either weakens nothing.
# The pre-#16957 legacy algorithm (which also embedded the check rollup,
# and with it the last GraphQL consumer of the snapshot) is retired
# (#17315): its two carriers (#16950, #16891) are merged, and an eventual
# residual legacy stamp now degrades to this refusal -- fail-closed, one
# mechanical --template re-stamp recovers it.
live_fingerprint = surfaces_fingerprint(
snapshot, dossier.comment_index, dossier.created_at
)
legacy_fingerprint = legacy_surfaces_fingerprint(
snapshot, dossier.comment_index, dossier.created_at
)
pre18637_fingerprint = pre18637_surfaces_fingerprint(
snapshot, dossier.comment_index, dossier.created_at
)
if f.get("surfaces-sha256") not in {
live_fingerprint, legacy_fingerprint, pre18637_fingerprint
live_fingerprint, pre18637_fingerprint
}:
# #16931 defaut 3 (mesure 16928) : deux hachages opaques sont
# inexploitables — la lane refabrique le dossier EN AVEUGLE. Le refus
Expand Down Expand Up @@ -1445,24 +1437,36 @@ def build_result(
return result


def review_threads(pr: int) -> list[dict[str, Any]]:
query = """
query($owner:String!,$repo:String!,$number:Int!,$cursor:String){
repository(owner:$owner,name:$repo){
pullRequest(number:$number){
reviewThreads(first:100,after:$cursor){
nodes{
id isResolved isOutdated path line
comments(first:100){
totalCount
nodes{id body createdAt author{login}}
}
}
pageInfo{hasNextPage endCursor}
}
}
}
}"""
REVIEW_THREADS_BATCH = 8

_REVIEW_THREADS_FIELDS = (
"reviewThreads(first:100){"
"nodes{id isResolved isOutdated path line "
"comments(first:100){totalCount "
"nodes{id body createdAt author{login}}}} "
"pageInfo{hasNextPage endCursor}}"
)


def _check_inline_pagination(nodes: list[dict[str, Any]]) -> None:
for thread in nodes:
inline = thread.get("comments") or {}
if inline.get("totalCount", 0) > len(inline.get("nodes") or []):
raise RuntimeError(
"inline thread has more than 100 comments; complete pagination required"
)


def _review_threads_paginated(pr: int) -> list[dict[str, Any]]:
"""Per-PR cursor pagination -- the fallback when one PR fills a page."""
query = f"""
query($owner:String!,$repo:String!,$number:Int!,$cursor:String){{
repository(owner:$owner,name:$repo){{
pullRequest(number:$number){{
{_REVIEW_THREADS_FIELDS}
}}
}}
}}"""
owner, repo = REPO.split("/", 1)
cursor: str | None = None
threads: list[dict[str, Any]] = []
Expand All @@ -1477,19 +1481,64 @@ def review_threads(pr: int) -> list[dict[str, Any]]:
data = gh_json(args)
connection = data["data"]["repository"]["pullRequest"]["reviewThreads"]
nodes = connection.get("nodes") or []
for thread in nodes:
inline = thread.get("comments") or {}
if inline.get("totalCount", 0) > len(inline.get("nodes") or []):
raise RuntimeError(
"inline thread has more than 100 comments; complete pagination required"
)
_check_inline_pagination(nodes)
threads.extend(nodes)
page = connection["pageInfo"]
if not page["hasNextPage"]:
return threads
cursor = page["endCursor"]


def review_threads_bulk(prs: Sequence[int]) -> dict[int, list[dict[str, Any]]]:
"""Review threads of N pull requests in ONE GraphQL query per batch.

#17315: mass passes (gate sweeps, B.0 probes over the fleet) paid one
paginated query per pull request for the third B.0 surface; aliases
collapse N pulls into a single operation per batch, ~N/8 times fewer.
A pull request whose threads overflow the first page (rare) falls back
to per-PR pagination for that PR alone.
"""
results: dict[int, list[dict[str, Any]]] = {}
owner, repo = REPO.split("/", 1)
remaining = list(dict.fromkeys(prs))
while remaining:
batch = remaining[:REVIEW_THREADS_BATCH]
del remaining[:REVIEW_THREADS_BATCH]
aliases = [f"p{i}" for i in range(len(batch))]
defs = "".join(f",${alias}:Int!" for alias in aliases)
body = "".join(
f"{alias}: repository(owner:$owner,name:$repo)"
f"{{pullRequest(number:${alias}){{{_REVIEW_THREADS_FIELDS}}}}}"
for alias in aliases
)
query = f"query($owner:String!,$repo:String!{defs}){{{body}}}"
args = [
"api", "graphql", "-f", f"query={query}",
"-F", f"owner={owner}", "-F", f"repo={repo}",
]
for alias, pr in zip(aliases, batch):
args.extend(["-F", f"{alias}={pr}"])
data = gh_json(args)
for alias, pr in zip(aliases, batch):
# L'alias nomme le `repository`; la connexion vit sous `pullRequest`.
# Mesure 2026-10-08 : lire `[alias]["reviewThreads"]` leve un KeyError
# -- le format de reponse se verifie sur un appel LIVE, un fixture
# ecrit depuis la meme lecture reproduit l'erreur a l'identique.
connection = data["data"][alias]["pullRequest"]["reviewThreads"]
nodes = connection.get("nodes") or []
_check_inline_pagination(nodes)
if connection["pageInfo"]["hasNextPage"]:
results[pr] = _review_threads_paginated(pr)
else:
results[pr] = nodes
return results


def review_threads(pr: int) -> list[dict[str, Any]]:
"""Single-PR convenience over the aliased bulk fetch: one operation."""
return review_threads_bulk([pr])[pr]


def _issue_comments(pr: int) -> list[dict[str, Any]]:
rows = gh_json([
"api", f"repos/{REPO}/issues/{pr}/comments", "--paginate",
Expand Down Expand Up @@ -1549,16 +1598,17 @@ def _head_check_runs(head_sha: str) -> list[dict[str, Any]]:
page += 1


def _pr_metadata(pr: int, *, with_rollup: bool) -> dict[str, Any]:
# Scalar fields come from REST (`repos/.../pulls/N`) so the shared GraphQL
# quota only pays for the check rollup below. Keys keep the exact shape
# `gh pr view --json` produced, so fingerprints and the identity bracket
# stay byte-compatible with dossiers stamped before this change.
def _pr_metadata(pr: int) -> dict[str, Any]:
# Scalar fields come from REST (`repos/.../pulls/N`) -- the whole snapshot
# bracket no longer touches GraphQL (#17315): the check rollup it used to
# carry served only the retired legacy fingerprint. Keys keep the exact
# shape `gh pr view --json` produced, so fingerprints and the identity
# bracket stay byte-compatible with dossiers stamped before this change.
row = gh_json(["api", f"repos/{REPO}/pulls/{pr}"])
if not isinstance(row, dict):
raise RuntimeError("pull request response is not an object")
state = row.get("state") or ""
data: dict[str, Any] = {
return {
"number": row.get("number"),
"title": row.get("title"),
"body": row.get("body") or "",
Expand All @@ -1578,46 +1628,34 @@ def _pr_metadata(pr: int, *, with_rollup: bool) -> dict[str, Any]:
"additions": row.get("additions"),
"deletions": row.get("deletions"),
}
if with_rollup:
# The check rollup has no REST equivalent, so it stays on GraphQL
# as a single-field query instead of the former twelve-field one.
rollup = gh_json([
"pr", "view", str(pr), "--repo", REPO,
"--json", "statusCheckRollup",
])
if not isinstance(rollup, dict):
raise RuntimeError("pull request response is not an object")
data["statusCheckRollup"] = rollup.get("statusCheckRollup")
return data


def _metadata_identity(data: dict[str, Any]) -> str:
normalized = dict(data)
normalized["statusCheckRollup"] = sorted(
data.get("statusCheckRollup") or [],
key=lambda row: json.dumps(row, sort_keys=True, separators=(",", ":")),
)
return json.dumps(
normalized, ensure_ascii=False, sort_keys=True, separators=(",", ":")
data, ensure_ascii=False, sort_keys=True, separators=(",", ":")
)


def load_snapshot(pr: int) -> dict[str, Any]:
before = _pr_metadata(pr, with_rollup=True)
before = _pr_metadata(pr)
snapshot = dict(before)
snapshot["comments"] = _issue_comments(pr)
snapshot["reviews"] = _reviews(pr)
snapshot["threads"] = review_threads(pr)
# Fetched inside the before/after bracket: a check concluding during the
# read bumps updatedAt and aborts the snapshot (transient UNKNOWN, the
# caller retries), so the claim verification below never reads a state
# that was already stale when captured. The B.0 probe (`probe_b0`) is NOT
# in this bracket: it runs after, and only on a READY dossier. A remark
# posted between the snapshot and the probe therefore makes the organ
# contradict a `b0: clear` claim -- a conservative refusal, which a rerun
# names as a changed discussion surface.
# Fetched inside the before/after bracket, keyed on the head read BEFORE
# it: a push landing during the read moves headRefOid, the second metadata
# read disagrees, and the snapshot is refused (transient UNKNOWN, the
# caller retries) rather than pairing the old head's check-runs with the
# new one. Since #17315 the bracket covers PR scalar fields only -- the
# rollup that used to carry check state is gone, so a check changing
# conclusion mid-read is not itself a bracket event; the claim
# verification below compares against the live check-runs. The B.0 probe
# (`probe_b0`) is NOT in this bracket: it runs after, and only on a READY
# dossier. A remark posted between the snapshot and the probe therefore
# makes the organ contradict a `b0: clear` claim -- a conservative
# refusal, which a rerun names as a changed discussion surface.
snapshot["checkRuns"] = _head_check_runs(snapshot["headRefOid"])
after = _pr_metadata(pr, with_rollup=True)
after = _pr_metadata(pr)
if _metadata_identity(before) != _metadata_identity(after):
raise RuntimeError("pull request changed while prevalidation snapshot was read")
return snapshot
Expand Down Expand Up @@ -1945,6 +1983,9 @@ def main() -> int:
result["ready"] = False
result["verdict"] = "FROZEN"
result["frozen"] = frozen_reason
# #17315 : le cout de l'invocation est publie avec le verdict -- le
# compteur distingue les deux buckets (REST core vs GraphQL partage).
result["api_usage"] = dict(API_USAGE)
if args.json:
print(json.dumps(result, ensure_ascii=False))
elif frozen_reason is not None:
Expand Down Expand Up @@ -1975,12 +2016,21 @@ def main() -> int:
for error in errors:
print(f" - {error}")
if frozen_reason is not None:
return EXIT_BLOCKED_WITH_SUBSTANCE
if ready:
return EXIT_READY
if verdict == VERDICT_BLOCKED:
return EXIT_BLOCKED_WITH_SUBSTANCE
return EXIT_NO_DOSSIER
code = EXIT_BLOCKED_WITH_SUBSTANCE
elif ready:
code = EXIT_READY
elif verdict == VERDICT_BLOCKED:
code = EXIT_BLOCKED_WITH_SUBSTANCE
else:
code = EXIT_NO_DOSSIER
# #17315 : chiffres publies sur stderr (stdout reste le verdict parse par
# les appelants) -- une lane secretaire peut agreger le cout par invocation.
print(
f"API usage: {API_USAGE['rest']} REST, "
f"{API_USAGE['graphql']} GraphQL operation(s)",
file=sys.stderr,
)
return code


if __name__ == "__main__":
Expand Down
Loading
Loading