diff --git a/scripts/roosync_archive_backfill.py b/scripts/roosync_archive_backfill.py index d1d07b57da..6991aeeff0 100644 --- a/scripts/roosync_archive_backfill.py +++ b/scripts/roosync_archive_backfill.py @@ -24,11 +24,18 @@ LLM — pose juste un tag `markForBackfill: true` dans le frontmatter, le backfill reel reste a la charge d'un operateur humain ou d'un script dedie branchant vLLM) - -**Hors scope** : rejouer la condensation LLM elle-meme (instrument RooSync -proprietaire), toucher au dashboard vivant, modifier le format d'archive -existant. Cet outil est **un detecteur + un marqueur**, pas un moteur de -resume. +- `--summarize` : passe de rattrapage bornee — pour chaque archive fallback + de la fenetre (plus recentes d'abord), demande un resume a un endpoint + OpenAI-compatible (`--base-url`, defaut vLLM:5002 ; sur lane Ollama : + `--base-url http://localhost:11434/v1 --model `), insere le bloc + resume AVANT les messages verbatim, bascule `llmGenerated: true` / + `fallbackTruncation: false` et ajoute une ligne de provenance + `backfilledAt`/`backfillModel`. S'arrete au premier echec LLM (jamais + dans le chemin d'un append : outil autonome, l'auto-condensation n'est + pas touchee). `--dry-run` liste les candidates sans appel LLM. + +**Hors scope** : toucher au dashboard vivant, modifier le format d'archive +existant, la stabilisation vLLM (deja portee par les watchdogs, cf #8889). **§ SOTA** (cf. `.claude/rules/sota-not-workaround.md`) : l'outil s'appuie sur le format verbatim existant (pas de workaround degrade), aucune reinvention du frontmatter, @@ -43,6 +50,7 @@ import re import sys from dataclasses import asdict, dataclass, field +from datetime import datetime, timedelta, timezone from pathlib import Path from typing import Iterable, Iterator @@ -266,6 +274,194 @@ def cmd_backfill(target: Path, archives: list[ArchiveInfo]) -> int: return 0 +# --- Resume LLM : passe de rattrapage (--summarize, #8889) --------------------- + +DEFAULT_LLM_BASE_URL = "http://localhost:5002/v1" +DEFAULT_LLM_MODEL = "qwen3.6-35b-a3b" # condenseur RooSync (body #8889) +DEFAULT_LLM_TIMEOUT_S = 120 +DEFAULT_MAX_CHARS = 24000 + +# Titre du bloc insere — sert aussi de garde d'idempotence in-file. +SUMMARY_BLOCK_MARKER = "## Résumé (backfill) des" + +SYSTEM_PROMPT = ( + "Tu résumes une fenêtre d'archive du canal de coordination RooSync " + "(cluster d'agents IA). Réponds en markdown français, sans titre de " + "niveau 1, en trois sections : « Thèmes principaux », « Actions et " + "résultats », « En attente » (une puce par point). Commence directement " + "par la première section : pas de préambule, tu ne t'adresses pas au " + "lecteur. Sois factuel et concret : cite les noms de lanes, PRs, issues " + "et machines quand ils apparaissent. Pas de commentaire sur la tâche " + "elle-même." +) + + +class LlmUnavailableError(RuntimeError): + """LLM de resume injoignable ou reponse inutilisable : la passe s'arrete.""" + + +def request_summary(text: str, *, base_url: str, model: str, + timeout: int = DEFAULT_LLM_TIMEOUT_S, + max_chars: int = DEFAULT_MAX_CHARS) -> str: + """Demande un resume a un endpoint OpenAI-compatible (vLLM, Ollama). + + Stdlib uniquement (urllib) : l'outil doit tourner hors venv sur les lanes. + Le contenu est borne a `max_chars` (le resume de rattrapage n'a pas besoin + du contexte integral, cf claim #8889 2026-10-01). + """ + import urllib.error + import urllib.request + + content = text[:max_chars] + if len(text) > max_chars: + content += "\n\n[... tronqué pour le résumé ...]" + payload = { + "model": model, + "messages": [ + {"role": "system", "content": SYSTEM_PROMPT}, + {"role": "user", "content": content}, + ], + "temperature": 0.2, + "max_tokens": 700, + } + req = urllib.request.Request( + f"{base_url.rstrip('/')}/chat/completions", + data=json.dumps(payload).encode("utf-8"), + headers={"Content-Type": "application/json"}, + method="POST", + ) + try: + with urllib.request.urlopen(req, timeout=timeout) as resp: + body = json.loads(resp.read().decode("utf-8")) + except (urllib.error.URLError, TimeoutError, OSError) as exc: + raise LlmUnavailableError(f"LLM injoignable ({base_url}): {exc}") from exc + except json.JSONDecodeError as exc: + raise LlmUnavailableError(f"réponse LLM non-JSON ({base_url}): {exc}") from exc + try: + summary = str(body["choices"][0]["message"]["content"]).strip() + except (KeyError, IndexError, TypeError) as exc: + raise LlmUnavailableError(f"réponse LLM sans contenu exploitable: {str(body)[:200]}") from exc + if not summary: + raise LlmUnavailableError("réponse LLM vide") + return summary + + +def apply_summary_to_archive(text: str, summary: str, *, model: str, + now_iso: str) -> str: + """Applique le rattrapage sur le contenu d'une archive fallback. + + - Frontmatter : `llmGenerated` false->true, `fallbackTruncation` true->false, + plus une ligne de provenance `backfilledAt` / `backfillModel`. + - Corps : insertion du bloc resume AVANT le premier separateur `---` qui + suit l'en-tete — les messages verbatim ne bougent pas d'un octet + (critere d'acceptance #8889). + """ + if SUMMARY_BLOCK_MARKER in text: + raise ValueError("archive déjà porteuse d'un résumé backfill") + + def _flip(m: re.Match) -> str: + lines = [] + for line in m.group("body").splitlines(): + stripped = line.strip() + if stripped == "llmGenerated: false": + lines.append("llmGenerated: true") + elif stripped == "fallbackTruncation: true": + lines.append("fallbackTruncation: false") + else: + lines.append(line) + lines.append(f"backfilledAt: '{now_iso}'") + lines.append(f"backfillModel: {model}") + return "---\n" + "\n".join(lines) + "\n---\n" + + new_text = FRONTMATTER_RE.sub(_flip, text, count=1) + if new_text == text: + raise ValueError("frontmatter introuvable ou déjà transformé") + + fm = parse_frontmatter(text) + n = fm.message_count if fm.message_count is not None else "?" + block = ( + f"{SUMMARY_BLOCK_MARKER} {n} messages archivés\n\n" + f"{summary}\n\n" + f"> Résumé généré rétroactivement le {now_iso} par `{model}` via " + f"`scripts/roosync_archive_backfill.py --summarize` (fenêtre " + f"initialement archivée sans résumé, cf #8889)." + ) + + fm_match = FRONTMATTER_RE.match(new_text) + head, rest = new_text[:fm_match.end()], new_text[fm_match.end():] + sep = rest.find("\n---\n") + if sep == -1: + return head + rest.rstrip("\n") + "\n\n" + block + "\n" + return head + rest[:sep] + "\n\n" + block + rest[sep:] + + +def cmd_summarize(archives: list[ArchiveInfo], *, root: Path, limit: int, + days: int | None, base_url: str, model: str, timeout: int, + max_chars: int, dry_run: bool, + now: datetime | None = None) -> int: + """Passe de rattrapage bornee : resume les archives fallback les plus + recentes, s'arrete au premier echec LLM, idempotente par construction + (les archives traitees sortent du predicat de detection).""" + now = now or datetime.now(timezone.utc) + if days is not None: + cutoff = now - timedelta(days=days) + + def _in_window(a: ArchiveInfo) -> bool: + try: + ts = datetime.strptime(a.iso, "%Y-%m-%dT%H-%M-%S").replace(tzinfo=timezone.utc) + except ValueError: + return True # ISO illisible : ne pas exclure silencieusement + return ts >= cutoff + + archives = [a for a in archives if _in_window(a)] + ordered = sorted(archives, key=lambda a: a.iso, reverse=True)[:limit] + + if dry_run: + for a in ordered: + print(f"[dry-run] {a.iso} {a.workspace} ({a.message_count} msg, {a.size_bytes} o) {a.path}") + print(f"[dry-run] {len(ordered)} candidates / {len(archives)} fallback " + f"detectees (limit={limit}, days={days}, base_url={base_url})") + return 0 + + processed = 0 + for a in ordered: + try: + text = a.path.read_text(encoding="utf-8", errors="replace") + except OSError as exc: + print(f"[summarize] SKIP lecture impossible {a.path.name}: {exc}") + continue + if SUMMARY_BLOCK_MARKER in text: + print(f"[summarize] SKIP déjà pourvue {a.path.name}") + continue + print(f"[summarize] {a.iso} {a.workspace} ({a.message_count} msg) ...", flush=True) + try: + summary = request_summary(text, base_url=base_url, model=model, + timeout=timeout, max_chars=max_chars) + except LlmUnavailableError as exc: + print(f"[summarize] ARRÊT au premier échec LLM : {exc}") + if processed: + print(f"[summarize] PARTIEL : {processed} résumée(s), la passe " + f"reprendra proprement (idempotente).") + return 1 + print("[summarize] Aucune archive touchée. Relancer quand le LLM " + f"répond (base_url={base_url}).") + return 2 + try: + new_text = apply_summary_to_archive( + text, summary, model=model, + now_iso=now.strftime("%Y-%m-%dT%H:%M:%SZ"), + ) + except ValueError as exc: + print(f"[summarize] SKIP {a.path.name}: {exc}") + continue + a.path.write_text(new_text, encoding="utf-8") + processed += 1 + print(f"[summarize] OK {a.path.name} ({len(summary)} car. de résumé)") + print(f"[summarize] Passe terminée : {processed} résumée(s) / " + f"{len(ordered)} candidates / {len(archives)} fallback dans la fenêtre.") + return 0 + + # --- Main --------------------------------------------------------------------- def main() -> int: @@ -281,6 +477,22 @@ def main() -> int: help="Statistiques agregees par dashboard + JSON") group.add_argument("--backfill", type=Path, metavar="ARCHIVE_PATH", help="Marquer une archive precise pour backfill manuel") + group.add_argument("--summarize", action="store_true", + help="Passe de rattrapage LLM bornee (resume + bascule frontmatter)") + parser.add_argument("--limit", type=int, default=10, + help="Nombre max d'archives resumees par passe (--summarize)") + parser.add_argument("--days", type=int, default=30, + help="Fenetre en jours, plus recentes seulement (--summarize)") + parser.add_argument("--base-url", default=DEFAULT_LLM_BASE_URL, + help="Endpoint OpenAI-compatible du LLM de resume (defaut : vLLM:5002)") + parser.add_argument("--model", default=DEFAULT_LLM_MODEL, + help="Modele de resume (defaut : condenseur RooSync)") + parser.add_argument("--timeout", type=int, default=DEFAULT_LLM_TIMEOUT_S, + help="Timeout HTTP par resume, en secondes") + parser.add_argument("--max-chars", type=int, default=DEFAULT_MAX_CHARS, + help="Borne de contexte envoye au LLM par archive") + parser.add_argument("--dry-run", action="store_true", + help="Avec --summarize : lister les candidates sans appel LLM ni ecriture") parser.add_argument("--output-dir", type=Path, default=Path("scripts/results/roosync_archive_backfill"), help="Repertoire de sortie pour les rapports JSON") @@ -295,6 +507,12 @@ def main() -> int: return cmd_report(archives, args.output_dir) if args.backfill: return cmd_backfill(args.backfill, archives) + if args.summarize: + return cmd_summarize( + archives, root=args.root, limit=args.limit, days=args.days, + base_url=args.base_url, model=args.model, timeout=args.timeout, + max_chars=args.max_chars, dry_run=args.dry_run, + ) return 1 # unreachable (mutually_exclusive_group required) diff --git a/scripts/tests/test_roosync_archive_backfill.py b/scripts/tests/test_roosync_archive_backfill.py index cf6a433138..e792a376f3 100644 --- a/scripts/tests/test_roosync_archive_backfill.py +++ b/scripts/tests/test_roosync_archive_backfill.py @@ -6,6 +6,8 @@ - detect_fallback_archives : detection sur arborescence synthetique - cmd_list : sortie formatee - cmd_backfill : marquage idempotent +- cmd_summarize : passe de rattrapage LLM (bloc resume, flags, idempotence, + arret au premier echec, fenetre jours, dry-run) """ from __future__ import annotations @@ -216,5 +218,164 @@ def test_cmd_backfill_idempotent(tmp_path: Path): assert "DEJA MARQUEE" in result.stdout +# --- Tests cmd_summarize (passe de rattrapage LLM, #8889) ---------------------- + +FALLBACK_BODY = ( + "# Archive (fallback): x\n\n" + "Archived: 2026-09-01T10:00:00.000Z\n" + "Messages: 5\n" + "Method: Truncation fallback (LLM unavailable)\n\n" + "[FALLBACK TRUNCATION] 5 messages archived without LLM summary. Circuit breaker failures: 1.\n\n" + "---\n\n" + "### [2026-09-01T09:59:00.000Z] lane|ws\n\nPremier message verbatim.\n\n---\n\n" + "### [2026-09-01T09:59:30.000Z] lane|ws\n\nSecond message verbatim.\n" +) +SUMMARY_STUB = "### Thèmes principaux\n- Test du bloc résumé.\n" + + +def _run_summarize(root: Path, **overrides): + """Invoque cmd_summarize sur root avec les overrides donnes.""" + from roosync_archive_backfill import cmd_summarize, detect_fallback_archives + kwargs = dict( + root=root, limit=5, days=None, base_url="http://invalid.test", + model="stub-model", timeout=1, max_chars=1000, dry_run=False, + ) + kwargs.update(overrides) + return cmd_summarize(detect_fallback_archives(root), **kwargs) + + +def _messages_region(text: str) -> str: + """Region messages verbatim : tout a partir du premier en-tete de message.""" + idx = text.find("### [") + assert idx != -1, f"pas de message dans l'archive: {text!r}" + return text[idx:] + + +def test_summarize_flips_flags_and_inserts_block(tmp_path, monkeypatch): + """Une archive traitee : flags bascules, bloc resume insere AVANT les + messages, messages verbatim intacts octet par octet (acceptance #8889).""" + monkeypatch.setattr("roosync_archive_backfill.request_summary", + lambda text, **kw: SUMMARY_STUB) + root = tmp_path / "archive" + target = _make_archive(root, "workspace-x-2026-09-01T10-00-00-fallback.md", + messages=5, body=FALLBACK_BODY) + before = target.read_text(encoding="utf-8") + + assert _run_summarize(root) == 0 + + after = target.read_text(encoding="utf-8") + assert "llmGenerated: true" in after + assert "fallbackTruncation: false" in after + assert "llmGenerated: false" not in after + assert "## Résumé (backfill) des 5 messages archivés" in after + assert SUMMARY_STUB in after + assert "backfilledAt: '20" in after + assert "backfillModel: stub-model" in after + # Critere acceptance : le bloc resume precede les messages, qui ne bougent pas. + assert after.find("## Résumé (backfill) des") < after.find("### [2026-09-01T09:59:00") + assert _messages_region(before) == _messages_region(after) + + +def test_summarize_idempotent_second_pass_noop(tmp_path, monkeypatch): + """Deuxieme passe : les archives traitees sortent du predicat de detection, + aucun fichier ne bouge (acceptance #8889).""" + monkeypatch.setattr("roosync_archive_backfill.request_summary", + lambda text, **kw: SUMMARY_STUB) + root = tmp_path / "archive" + target = _make_archive(root, "workspace-x-2026-09-01T10-00-00-fallback.md", + messages=5, body=FALLBACK_BODY) + assert _run_summarize(root) == 0 + after_first = target.read_text(encoding="utf-8") + + # La detection ne voit plus l'archive ; la passe ne touche rien. + from roosync_archive_backfill import detect_fallback_archives + assert detect_fallback_archives(root) == [] + assert _run_summarize(root) == 0 + assert target.read_text(encoding="utf-8") == after_first + + +def test_summarize_llm_unreachable_before_any_work(tmp_path, monkeypatch): + """LLM injoignable des le debut : exit 2, aucune archive touchee.""" + from roosync_archive_backfill import LlmUnavailableError + def _boom(text, **kw): + raise LlmUnavailableError("connection refused") + monkeypatch.setattr("roosync_archive_backfill.request_summary", _boom) + root = tmp_path / "archive" + target = _make_archive(root, "workspace-x-2026-09-01T10-00-00-fallback.md", + messages=5, body=FALLBACK_BODY) + before = target.read_text(encoding="utf-8") + assert _run_summarize(root) == 2 + assert target.read_text(encoding="utf-8") == before + + +def test_summarize_stops_at_first_failure_mid_pass(tmp_path, monkeypatch): + """Echec LLM en cours de passe : arret immediat (exit 1), la plus recente + est traitee, l'autre reste intacte — on ne martele pas un service en boot.""" + from roosync_archive_backfill import LlmUnavailableError + calls = [] + def _flaky(text, **kw): + if calls: + raise LlmUnavailableError("wedge") + calls.append(1) + return SUMMARY_STUB + monkeypatch.setattr("roosync_archive_backfill.request_summary", _flaky) + root = tmp_path / "archive" + older = _make_archive(root, "workspace-x-2026-09-01T10-00-00-fallback.md", + messages=5, body=FALLBACK_BODY) + newer = _make_archive(root, "workspace-x-2026-09-01T11-00-00-fallback.md", + messages=5, body=FALLBACK_BODY) + # Ordre : plus recentes d'abord -> newer traitee, older non atteinte. + assert _run_summarize(root) == 1 + assert "## Résumé (backfill) des" in newer.read_text(encoding="utf-8") + assert "## Résumé (backfill) des" not in older.read_text(encoding="utf-8") + + +def test_summarize_days_window_excludes_old_archives(tmp_path, monkeypatch): + """Fenetre --days : une archive plus vieille que la fenetre n'est pas + candidate (borne suggeree par le body #8889).""" + from datetime import datetime, timezone + monkeypatch.setattr("roosync_archive_backfill.request_summary", + lambda text, **kw: SUMMARY_STUB) + root = tmp_path / "archive" + _make_archive(root, "workspace-x-2026-09-01T10-00-00-fallback.md", + messages=5, body=FALLBACK_BODY) + # now = 2026-10-04, days=30 -> cutoff 2026-09-04 : archive du 09-01 exclue. + rc = _run_summarize(root, days=30, dry_run=True, + now=datetime(2026, 10, 4, tzinfo=timezone.utc)) + assert rc == 0 + # dry-run sans candidate : aucune ecriture, exit 0. + + +def test_summarize_dry_run_never_calls_llm_nor_writes(tmp_path, monkeypatch, capsys): + """--dry-run : liste les candidates sans appel LLM ni ecriture.""" + def _must_not_run(text, **kw): + raise AssertionError("request_summary ne doit pas etre appele en dry-run") + monkeypatch.setattr("roosync_archive_backfill.request_summary", _must_not_run) + root = tmp_path / "archive" + target = _make_archive(root, "workspace-x-2026-09-01T10-00-00-fallback.md", + messages=5, body=FALLBACK_BODY) + before = target.read_text(encoding="utf-8") + assert _run_summarize(root, dry_run=True) == 0 + out = capsys.readouterr().out + assert "1 candidates" in out + assert "workspace-x" in out + assert target.read_text(encoding="utf-8") == before + + +def test_apply_summary_without_body_separator_appends(tmp_path): + """Archive sans separateur --- dans le corps : le bloc est appendu, pas perdu.""" + from roosync_archive_backfill import apply_summary_to_archive + text = ( + "---\nmessageCount: 3\nllmGenerated: false\nfallbackTruncation: true\n---\n" + "# Archive (fallback): x\n\n[FALLBACK TRUNCATION] 3 messages.\n" + ) + result = apply_summary_to_archive(text, SUMMARY_STUB, model="m", + now_iso="2026-10-04T00:00:00Z") + assert "## Résumé (backfill) des 3 messages archivés" in result + assert "llmGenerated: true" in result + # Le bloc se retrouve apres l'en-tete, en fin de corps. + assert result.rstrip().endswith("cf #8889).") + + if __name__ == "__main__": sys.exit(pytest.main([__file__, "-v"]))