Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
66 changes: 46 additions & 20 deletions scripts/epic_body_staleness.py
Original file line number Diff line number Diff line change
Expand Up @@ -27,7 +27,8 @@

EPIC_TITLE_RE = re.compile(r"\bepic\b", re.IGNORECASE)
ISSUE_REF_RE = re.compile(r"(?<![\w])#(\d+)\b")
OPEN_ISSUE_LIMIT = 500
OPEN_ISSUE_PROBE_START = 500
OPEN_ISSUE_PROBE_CEILING = 20000
SEARCH_RESULT_CAP = 1000
MERGED_SLICE_DAYS = 3
MAX_LOOKBACK_DAYS = 3650
Expand Down Expand Up @@ -294,26 +295,51 @@ def _default_repo() -> str:
return result.stdout.strip() or "jsboige/CoursIA"


def list_open_epics(repo: str) -> list[Epic]:
rows = _gh_json(
[
"issue",
"list",
"--repo",
repo,
"--state",
"open",
"--limit",
str(OPEN_ISSUE_LIMIT),
"--json",
"number,title,body,labels",
]
) or []
if len(rows) >= OPEN_ISSUE_LIMIT:
raise RuntimeError(
f"open-issue corpus reached its {OPEN_ISSUE_LIMIT}-issue fetch limit"
def _fetch_open_issues(repo: str) -> list[dict]:
"""Return every open issue, proving the corpus is exhausted and not capped.

A single ``--limit`` both bounds the corpus and hides the bound: asking for
exactly N rows cannot tell "this repository has N open issues" apart from
"the fetch stopped at N". The probe therefore raises the request until a
reply comes back **shorter** than requested -- the only observable that
proves exhaustion -- and refuses at the ceiling rather than reporting a
truncated corpus as a complete one.

The ceiling is not decoration. The corpus outgrew a fixed 500 and the
analyzer then raised on every run, measuring nothing at all: a fail-closed
guard whose bound sits below the data is an off switch, not a guard.
"""
limit = OPEN_ISSUE_PROBE_START
while True:
payload = _gh_json(
[
"issue",
"list",
"--repo",
repo,
"--state",
"open",
"--limit",
str(limit),
"--json",
"number,title,body,labels",
]
)
return [Epic.from_gh_dict(row) for row in rows if is_epic(row)]
if not isinstance(payload, list):
raise RuntimeError(
f"open-issue fetch returned {type(payload).__name__}, not a list"
)
if len(payload) < limit:
return payload
if limit >= OPEN_ISSUE_PROBE_CEILING:
raise RuntimeError(
f"open-issue corpus still full at its {limit}-issue fetch ceiling"
)
limit = min(limit * 2, OPEN_ISSUE_PROBE_CEILING)


def list_open_epics(repo: str) -> list[Epic]:
return [Epic.from_gh_dict(row) for row in _fetch_open_issues(repo) if is_epic(row)]


def _merged_pr_slice(repo: str, since: date, until: date) -> list[dict]:
Expand Down
77 changes: 77 additions & 0 deletions scripts/tests/test_epic_body_staleness.py
Original file line number Diff line number Diff line change
Expand Up @@ -292,3 +292,80 @@ def test_cli_rejects_nonpositive_limit(capsys):

assert exc.value.code == 2
assert "--pr-limit must be positive" in capsys.readouterr().err


def test_open_issue_probe_widens_past_a_full_reply(monkeypatch):
"""Regression: a corpus larger than the first probe used to mean no measure.

At 506 open issues the analyzer raised on every run and reported nothing.
A reply that fills the request is not exhaustion, so the probe widens.
"""
asked = []

def fake_gh(args):
limit = int(args[args.index("--limit") + 1])
asked.append(limit)
count = limit if limit <= 500 else 506
return [
{"number": n, "title": f"issue {n}", "labels": [], "body": ""}
for n in range(count)
]

monkeypatch.setattr(_MODULE, "_gh_json", fake_gh)

rows = _MODULE._fetch_open_issues("example/repo")

assert asked == [_MODULE.OPEN_ISSUE_PROBE_START, 1000]
assert len(rows) == 506


def test_open_issue_probe_stops_at_a_short_reply(monkeypatch):
asked = []

def fake_gh(args):
asked.append(int(args[args.index("--limit") + 1]))
return [{"number": 1, "title": "issue", "labels": [], "body": ""}]

monkeypatch.setattr(_MODULE, "_gh_json", fake_gh)

assert len(_MODULE._fetch_open_issues("example/repo")) == 1
assert asked == [_MODULE.OPEN_ISSUE_PROBE_START]


def test_open_issue_probe_refuses_at_the_ceiling(monkeypatch):
"""Every reply full: refuse loudly instead of calling the corpus complete."""

def fake_gh(args):
limit = int(args[args.index("--limit") + 1])
return [
{"number": n, "title": f"issue {n}", "labels": [], "body": ""}
for n in range(limit)
]

monkeypatch.setattr(_MODULE, "_gh_json", fake_gh)

with pytest.raises(RuntimeError) as exc:
_MODULE._fetch_open_issues("example/repo")

assert str(_MODULE.OPEN_ISSUE_PROBE_CEILING) in str(exc.value)


def test_open_issue_probe_rejects_a_payload_that_is_not_a_list(monkeypatch):
"""An unread corpus is not an empty one: `null` must not read as zero issues."""
monkeypatch.setattr(_MODULE, "_gh_json", lambda args: None)

with pytest.raises(RuntimeError) as exc:
_MODULE._fetch_open_issues("example/repo")

assert "NoneType" in str(exc.value)


def test_list_open_epics_keeps_only_epics(monkeypatch):
rows = [
{"number": 1, "title": "[EPIC] X", "labels": [], "body": ""},
{"number": 2, "title": "plain issue", "labels": [], "body": ""},
{"number": 3, "title": "Tracker", "labels": [{"name": "EPIC"}], "body": ""},
]
monkeypatch.setattr(_MODULE, "_gh_json", lambda args: rows)

assert [epic.number for epic in _MODULE.list_open_epics("example/repo")] == [1, 3]
Loading