From 83af484491842dfd662ca676fb1a70c2419befd5 Mon Sep 17 00:00:00 2001 From: Jean-Baptiste Alleaume Date: Wed, 9 Sep 2026 12:12:35 +0000 Subject: [PATCH 1/6] feat(pi): add pi extension porting Claude hooks to pi events Build a self-contained pi extension (pi-extension/) that ports the Claude Code autoresearch plugin to pi's extension event system, keeping the same spirit as the claude and opencode integrations. Guardrails (hooks -> pi events), all fail-open: PreToolUse (Bash) -> tool_call (bash) scout/privacy/dangerous-cmd block PreToolUse (Read/Write) -> tool_call (read/...) scout/privacy block UserPromptSubmit -> before_agent_start iteration-context + dev-rules inject UserPromptSubmit -> input simplify-gate (block/warn shipping verbs) SessionStart -> session_start session-init SessionEnd -> session_shutdown stop-notify Commands (14) are exposed as pi prompt templates (/autoresearch, /autoresearch_debug, ...) and the dispatcher as a pi skill (/skill:autoresearch), contributed via resources_discover. The AR_DISABLE_* per-hook env flags are preserved. Verified end-to-end: tsc --strict passes (0 errors), 22/22 logic tests pass, extension loads under pi's jiti loader, and session-init/iteration-context/ scout-block/dangerous-cmd-block/simplify-gate/stop-notify all fire correctly (confirmed in ~/.pi/agent/hooks/.logs/). AGENTS.md documents the pi install path, agent-specific notes, and repo layout. --- AGENTS.md | 35 ++ pi-extension/README.md | 143 ++++++ pi-extension/package.json | 28 ++ pi-extension/prompts/autoresearch.md | 110 +++++ pi-extension/prompts/autoresearch_debug.md | 97 ++++ pi-extension/prompts/autoresearch_evals.md | 118 +++++ pi-extension/prompts/autoresearch_fix.md | 102 ++++ pi-extension/prompts/autoresearch_improve.md | 116 +++++ pi-extension/prompts/autoresearch_learn.md | 136 ++++++ pi-extension/prompts/autoresearch_plan.md | 95 ++++ pi-extension/prompts/autoresearch_predict.md | 94 ++++ pi-extension/prompts/autoresearch_probe.md | 116 +++++ pi-extension/prompts/autoresearch_reason.md | 109 +++++ .../prompts/autoresearch_regression.md | 110 +++++ pi-extension/prompts/autoresearch_scenario.md | 106 +++++ pi-extension/prompts/autoresearch_security.md | 101 ++++ pi-extension/prompts/autoresearch_ship.md | 120 +++++ pi-extension/skills/autoresearch/SKILL.md | 107 +++++ .../references/orchestrator-routing.md | 89 ++++ .../references/predict-personas.md | 65 +++ .../references/reason-judge-protocol.md | 87 ++++ .../references/security-checklist.md | 76 +++ .../autoresearch/scripts/orchestrate.sh | 448 ++++++++++++++++++ .../autoresearch/scripts/score-regression.sh | 205 ++++++++ pi-extension/src/guardrails.ts | 219 +++++++++ pi-extension/src/hooks/dangerous-cmd-block.ts | 102 ++++ pi-extension/src/hooks/dev-rules-reminder.ts | 44 ++ pi-extension/src/hooks/iteration-context.ts | 109 +++++ pi-extension/src/hooks/privacy-block.ts | 81 ++++ pi-extension/src/hooks/scout-block.ts | 61 +++ pi-extension/src/hooks/simplify-gate.ts | 118 +++++ pi-extension/src/hooks/stop-notify.ts | 145 ++++++ pi-extension/src/index.ts | 43 ++ pi-extension/src/lib/ignore.ts | 83 ++++ pi-extension/src/lib/paths.ts | 245 ++++++++++ pi-extension/src/lib/session-state.ts | 129 +++++ pi-extension/src/lib/shell.ts | 289 +++++++++++ pi-extension/src/lib/tsv.ts | 77 +++ 38 files changed, 4558 insertions(+) create mode 100644 pi-extension/README.md create mode 100644 pi-extension/package.json create mode 100644 pi-extension/prompts/autoresearch.md create mode 100644 pi-extension/prompts/autoresearch_debug.md create mode 100644 pi-extension/prompts/autoresearch_evals.md create mode 100644 pi-extension/prompts/autoresearch_fix.md create mode 100644 pi-extension/prompts/autoresearch_improve.md create mode 100644 pi-extension/prompts/autoresearch_learn.md create mode 100644 pi-extension/prompts/autoresearch_plan.md create mode 100644 pi-extension/prompts/autoresearch_predict.md create mode 100644 pi-extension/prompts/autoresearch_probe.md create mode 100644 pi-extension/prompts/autoresearch_reason.md create mode 100644 pi-extension/prompts/autoresearch_regression.md create mode 100644 pi-extension/prompts/autoresearch_scenario.md create mode 100644 pi-extension/prompts/autoresearch_security.md create mode 100644 pi-extension/prompts/autoresearch_ship.md create mode 100644 pi-extension/skills/autoresearch/SKILL.md create mode 100644 pi-extension/skills/autoresearch/references/orchestrator-routing.md create mode 100644 pi-extension/skills/autoresearch/references/predict-personas.md create mode 100644 pi-extension/skills/autoresearch/references/reason-judge-protocol.md create mode 100644 pi-extension/skills/autoresearch/references/security-checklist.md create mode 100755 pi-extension/skills/autoresearch/scripts/orchestrate.sh create mode 100755 pi-extension/skills/autoresearch/scripts/score-regression.sh create mode 100644 pi-extension/src/guardrails.ts create mode 100644 pi-extension/src/hooks/dangerous-cmd-block.ts create mode 100644 pi-extension/src/hooks/dev-rules-reminder.ts create mode 100644 pi-extension/src/hooks/iteration-context.ts create mode 100644 pi-extension/src/hooks/privacy-block.ts create mode 100644 pi-extension/src/hooks/scout-block.ts create mode 100644 pi-extension/src/hooks/simplify-gate.ts create mode 100644 pi-extension/src/hooks/stop-notify.ts create mode 100644 pi-extension/src/index.ts create mode 100644 pi-extension/src/lib/ignore.ts create mode 100644 pi-extension/src/lib/paths.ts create mode 100644 pi-extension/src/lib/session-state.ts create mode 100644 pi-extension/src/lib/shell.ts create mode 100644 pi-extension/src/lib/tsv.ts diff --git a/AGENTS.md b/AGENTS.md index 2bb2f911..4fdec5dc 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -31,6 +31,25 @@ cd autoresearch Invoke via the `$autoresearch` mention syntax: `$autoresearch [flags]`. +### pi (extension) + +```bash +pi install git:github.com/uditgoenka/autoresearch +``` + +Then enable the `pi-extension` package as an extension in `~/.pi/agent/settings.json`: + +```json +{ "extensions": ["@uditgoenka/autoresearch-pi"] } +``` + +Or try it without installing: `pi -e ./pi-extension`. Invoke commands as pi +prompt templates: `/autoresearch`, `/autoresearch_debug`, `/autoresearch_fix`, … +The safety guardrails (scout/privacy/dangerous-cmd block, iteration context, +simplify gate, stop-notify) are ported from the Claude hooks to pi's extension +events. See [`pi-extension/README.md`](pi-extension/README.md) for the full +hook→event mapping and per-hook disable flags. + ### Manual (any agent) Copy the skill files into your agent's skill directory: @@ -45,6 +64,9 @@ cp autoresearch/claude-plugin/commands/autoresearch.md .claude/commands/autorese # Codex cp -r autoresearch/plugins/autoresearch ~/.agents/plugins/autoresearch + +# pi (extension) +cp -r autoresearch/pi-extension ~/.pi/agent/extensions/autoresearch-pi ``` --- @@ -298,6 +320,15 @@ iteration commit metric delta status description - Plugin files: `plugins/autoresearch/` with `skills/` - Command contracts live in each command file under `plugins/autoresearch/skills/autoresearch/` +### pi + +- Commands are pi **prompt templates**: `/autoresearch`, `/autoresearch_debug`, `/autoresearch_fix`, … (underscore names) +- The dispatcher is a pi **skill**: `/skill:autoresearch` +- Interactive setup uses `ctx.ui` (confirm/input/select) via the extension when context is missing +- Extension files: `pi-extension/src/` (guardrails) + `pi-extension/skills/autoresearch/` + `pi-extension/prompts/` +- Safety guardrails are ported from the Claude hooks to pi extension events (`tool_call`, `before_agent_start`, `input`, `session_start`, `session_shutdown`) +- Per-hook disable flags: `AR_DISABLE_*` env vars (see `pi-extension/README.md`) + ### Other Agents (OpenCode, Gemini CLI, etc.) - Read this file for the command surface and configuration contract @@ -319,6 +350,10 @@ autoresearch/ ├── claude-plugin/ ← Claude Code distribution package │ ├── skills/autoresearch/SKILL.md ← Main skill + references/ │ └── commands/autoresearch/ ← Subcommand registrations +├── pi-extension/ ← pi (coding agent) extension package +│ ├── src/ ← Guardrails ported from Claude hooks → pi events +│ ├── skills/autoresearch/SKILL.md ← Skill dispatcher + references/ +│ └── prompts/ ← 14 command prompt templates └── plugins/autoresearch/ ← Codex distribution package └── skills/autoresearch/SKILL.md ← Codex skill router + references/ ``` diff --git a/pi-extension/README.md b/pi-extension/README.md new file mode 100644 index 00000000..83678db7 --- /dev/null +++ b/pi-extension/README.md @@ -0,0 +1,143 @@ +# Autoresearch for pi + +Autonomous goal-directed iteration for [pi](https://github.com/earendil-works/pi-coding-agent) — a port of the Claude Code autoresearch plugin to pi's extension system. + +One metric, constrained scope, fast verification, automatic rollback, git as memory. Works on **any** domain — code, content, marketing, sales, DevOps — anything with a measurable metric. + +**Core loop:** Modify → Verify → Keep/Discard → Repeat. + +This package keeps the same spirit as the [Claude Code](../claude-plugin) and [OpenCode](../.opencode) integrations: the 14 commands are markdown protocols (here exposed as pi **prompt templates** + a **skill**), and the safety guardrails are ported from the Claude hooks to pi's **extension events**. + +--- + +## What's in the box + +``` +pi-extension/ +├── package.json # pi package manifest (extension + skill + prompts) +├── src/ +│ ├── index.ts # entry: registers guardrails + resources_discover +│ ├── guardrails.ts # wires hooks → pi events +│ ├── lib/ +│ │ ├── shell.ts # shell command tokenizer (port of ar-hook-utils) +│ │ ├── ignore.ts # .ckignore / gitignore matcher (port of ignore.cjs) +│ │ ├── session-state.ts # per-session state in $TMPDIR + bounded hook log +│ │ ├── tsv.ts # iteration-results TSV reader +│ │ └── paths.ts # sensitive-file + .ckignore path helpers +│ └── hooks/ +│ ├── dangerous-cmd-block.ts # blocks force-push, hard reset, rm -rf, … +│ ├── scout-block.ts # blocks access to .ckignore'd paths +│ ├── privacy-block.ts # confirms sensitive-file access +│ ├── iteration-context.ts # injects active iteration TSV every 5th prompt +│ ├── dev-rules-reminder.ts # injects dev context (plan/standards) +│ ├── simplify-gate.ts # blocks/warns shipping verbs with too many LOC +│ └── stop-notify.ts # terminal notification + webhook on shutdown +├── skills/autoresearch/ # SKILL.md + references + orchestrator scripts +└── prompts/ # 14 command prompt templates (/autoresearch, /autoresearch_debug, …) +``` + +## Hook → event mapping + +| Claude Code hook | pi event | autoresearch hook(s) | +|---|---|---| +| `PreToolUse` (Bash) | `tool_call` (`bash`) | scout-block, privacy-block, dangerous-cmd-block | +| `PreToolUse` (Read/Write/Edit) | `tool_call` (`read`/`write`/`edit`) | scout-block, privacy-block | +| `UserPromptSubmit` | `before_agent_start` | iteration-context, dev-rules-reminder (inject message) | +| `UserPromptSubmit` | `input` | simplify-gate (block/warn shipping verbs) | +| `SessionStart` | `session_start` | session-init (persist project/branch state) | +| `SessionEnd` | `session_shutdown` | stop-notify (notify + webhook + cleanup) | + +Every hook **fails open** — a guardrail malfunction never blocks legitimate work. + +> **Note on subagent context:** Claude's `SubagentStart` hook injected iteration state into subagents. pi subagents inherit the loaded autoresearch skill, so the active-iteration context already flows via the skill + `iteration-context` on the parent. There is no direct subagent-spawn event in pi's extension API, so the `subagent-context` hook is not ported. + +## Commands + +The 14 autoresearch commands are exposed as pi **prompt templates** (markdown protocols, invoked via `/name`) and the dispatcher is a pi **skill** (`/skill:autoresearch`): + +| Command | Does | Default Iterations | +|---|---|---| +| `/autoresearch` | Iterate against a metric: modify → verify → keep/discard | 25 | +| `/autoresearch_plan` | Convert a goal into validated Scope, Metric, Verify config | N/A | +| `/autoresearch_debug` | Hunt bugs: hypothesize → test → falsify → repeat | 15 | +| `/autoresearch_fix` | Crush errors one-by-one until zero remain | 20 | +| `/autoresearch_security` | STRIDE + OWASP audit with red-team personas | 15 | +| `/autoresearch_ship` | Ship through 8 phases: checklist → dry-run → deploy → verify | N/A | +| `/autoresearch_scenario` | Generate edge cases across 12 dimensions | 20 | +| `/autoresearch_predict` | 5 expert personas debate before implementation | N/A | +| `/autoresearch_learn` | Scout codebase → generate docs or wiki → validate → fix loop | 10 | +| `/autoresearch_reason` | Adversarial debate with blind judges until convergence | 8 | +| `/autoresearch_probe` | 8 personas interrogate requirements until saturation | 15 | +| `/autoresearch_improve` | Research ICP challenges, discover improvements, generate PRDs | 15 | +| `/autoresearch_evals` | Analyze iteration results: trends, plateaus, regressions | N/A | +| `/autoresearch_regression` | Regression stability gate: baseline vs candidate | N/A | + +## Install + +### From this repo (git) + +```bash +pi install git:github.com/uditgoenka/autoresearch +``` + +Then enable the `pi-extension` package as an extension in your settings (`~/.pi/agent/settings.json` for user scope, or `.pi/settings.json` for project scope): + +```json +{ + "extensions": ["@uditgoenka/autoresearch-pi"] +} +``` + +### Try it without installing + +```bash +pi -e ./pi-extension +``` + +### Manual (any pi setup) + +Copy this directory into your extensions folder: + +```bash +cp -r pi-extension ~/.pi/agent/extensions/autoresearch-pi +``` + +## Usage + +``` +/autoresearch +Goal: Increase test coverage from 72% to 90% +Scope: src/**/*.test.ts, src/**/*.ts +Metric: coverage % (higher is better) +Verify: npm test -- --coverage | grep "All files" +Iterations: 50 +``` + +Don't know what metric to use? + +``` +/autoresearch_plan +Goal: Make the API respond faster +``` + +## Configuration + +Per-hook disable flags (same as the Claude plugin): + +| Env var | Disables | +|---|---| +| `AR_DISABLE_SCOUT_BLOCK` | .ckignore path blocking | +| `AR_DISABLE_PRIVACY_BLOCK` | sensitive-file confirmation | +| `AR_DISABLE_DANGEROUS_CMD_BLOCK` | destructive-command blocking | +| `AR_DISABLE_ITERATION_CONTEXT` | iteration TSV injection | +| `AR_DISABLE_DEV_RULES_REMINDER` | dev-context injection | +| `AR_DISABLE_SIMPLIFY_GATE` | LOC shipping gate | +| `AR_DISABLE_SESSION_INIT` | session-state init | +| `AR_DISABLE_STOP_NOTIFY` | session-end notification | +| `AR_NOTIFY_WEBHOOK` | optional webhook URL for session-end notifications | + +Runtime logs (bounded metadata only — no paths, commands, or secrets) are written to `~/.pi/agent/hooks/.logs//hook-log.jsonl`. + +## License + +MIT — see [LICENSE](../LICENSE). diff --git a/pi-extension/package.json b/pi-extension/package.json new file mode 100644 index 00000000..6f526de1 --- /dev/null +++ b/pi-extension/package.json @@ -0,0 +1,28 @@ +{ + "name": "@uditgoenka/autoresearch-pi", + "version": "2.2.2", + "description": "Autoresearch for pi — autonomous goal-directed iteration. Ports the Claude Code hooks (safety guardrails + iteration context injection) to pi's extension event system, and contributes the autoresearch skill + 14 command prompt templates.", + "license": "MIT", + "type": "module", + "author": { + "name": "Udit Goenka", + "url": "https://github.com/uditgoenka" + }, + "repository": { + "type": "git", + "url": "git+https://github.com/uditgoenka/autoresearch.git" + }, + "keywords": [ + "pi", + "pi-coding-agent", + "autoresearch", + "autonomous", + "iteration", + "agent" + ], + "pi": { + "extensions": ["./src/index.ts"], + "skills": ["./skills/autoresearch/SKILL.md"], + "prompts": ["./prompts"] + } +} diff --git a/pi-extension/prompts/autoresearch.md b/pi-extension/prompts/autoresearch.md new file mode 100644 index 00000000..7eb1cf12 --- /dev/null +++ b/pi-extension/prompts/autoresearch.md @@ -0,0 +1,110 @@ +--- +name: autoresearch +description: "Autonomous iteration loop: modify, verify, keep/discard against any metric" +argument-hint: "[Goal: ] [Scope: ] [Metric: ] [Verify: ] [Guard: ] [Iterations: N] [--evals]" +--- + +EXECUTE IMMEDIATELY — do not deliberate before reading this protocol. + +## Parse Arguments + +Extract from $ARGUMENTS: +- `Goal:` — what to improve +- `Scope:` or `--scope` — file globs +- `Metric:` — what to measure +- `Direction:` — higher_is_better (default) or lower_is_better +- `Verify:` — shell command that outputs a number +- `Guard:` — optional safety command (must always pass) +- `Iterations:` or `--iterations` — integer N for bounded mode (default: 25). "unlimited" for unbounded. +- `--evals` — enable mid-loop checkpoints +- `--evals-interval N` — checkpoint frequency override +- `--chain ` — comma-separated downstream commands + +## Setup (if required context missing) + +If Goal, Scope, Metric, or Verify missing → use request_user_input (single batched call): + Q1 (Goal): "What do you want to improve?" + Q2 (Scope): "Which files?" — suggest globs from project + Q3 (Metric+Verify): "How to measure? Provide a shell command that outputs a number" + Q4 (Guard): "Safety command that must always pass?" — options: test cmd, build cmd, skip +If ALL provided inline → skip setup, proceed directly. + +## Precondition Checks + +1. Verify git repo exists (`git rev-parse --git-dir`) +2. Check clean working tree (`git status --porcelain`) — warn if dirty +3. Check for stale lock files, detached HEAD +4. If Guard set → run Guard to establish guard baseline +5. Fail fast on any critical issue. Warn on non-critical. + +## Verify Safety Screen + +Before first dry-run, screen Verify command for: rm -rf, fork bombs, curl|sh, embedded credentials, outbound writes. Block dangerous commands. + +## Establish Baseline (Iteration 0) + +1. Run Verify command → extract numeric metric +2. Record as iteration 0 in TSV: `0\t{timestamp}\t{commit}\t{metric}\t0.0\t{guard}\t-\tbaseline\tinitial state` +3. Create output directory: `autoresearch/loop-{YYMMDD}-{HHMM}/` +4. Write TSV header: `# metric_direction: {direction}\niteration\ttimestamp\tcommit\tmetric\tdelta\tguard\tguard-metric\tstatus\tdescription` + +## Iteration Loop + +For each iteration (1 to max_iterations, or unbounded): + +### Phase 1: Review (read git history as memory) +- Read last 10-20 lines of results TSV +- Run `git log --oneline -20` — see what worked/failed +- If last iteration was "keep" → run `git diff HEAD~1` to see what improved metric +- Identify: what worked, what failed, what's untried + +### Phase 2: Modify +- Based on review, make ONE focused change to improve the metric +- Change must be atomic — one logical unit of work + +### Phase 3: Commit +- Stage and commit with `experiment: {description}` prefix +- Record commit SHA + +### Phase 4: Verify +- Run Verify command → extract new metric value +- Calculate delta from previous iteration +- Metric improved (correct direction) → candidate for keep + +### Phase 5: Guard (if configured) +- Run Guard command. If fails → revert regardless of metric improvement + +### Phase 6: Decide +- **keep** — metric improved, guard passed → commit stays +- **discard** — metric worsened → `git revert HEAD --no-edit` +- **crash** — verify/guard command failed → `git revert HEAD --no-edit` +- **no-op** — no change made this iteration +- **hook-blocked** — git hook blocked the commit +- **metric-error** — verify output not a valid number → `git revert HEAD --no-edit` + +### Phase 7: Log +Append row to TSV: iteration, timestamp, commit/-, metric, delta, guard status, guard-metric, status, description + +### Eval Checkpoint +If --evals: check if current_iteration % interval == 0 → run checkpoint analysis. + +### Bounded Check +If bounded: current_iteration >= max_iterations → exit loop, print summary. + +## Summary (after loop ends) + +Print: total iterations, kept/discarded counts, starting metric → final metric, improvement %, top 3 most effective changes. + +## Eval Checkpoint (--evals flag) + +If --evals present: +- Compute interval: floor(max_iterations / 3), min 1. Fixed 10 if unbounded. Override: --evals-interval N. +- Every {interval} iterations, pause and analyze current results TSV. +- Print: `--- Eval Checkpoint (iterations {X}-{Y}) ---\nMetric: {start} → {end} ({delta}) | Kept: {n}/{total} | Trend: {up/flat/down}\n{one-line recommendation}\n---` +- If plateau 3+ checkpoints → recommend early stop. +- At loop end → full evals summary to evals-summary.md in output directory. + +## Chain Handoff + +After completion, write handoff.json to output directory: version "2.1.0", source "loop", timestamp, status (COMPLETE|USER_INTERRUPT|BOUNDED|ERROR), results_tsv path, findings[], config{goal, scope, metric, direction, verify}. +Invoke next target in --chain order. Propagate --evals flag. diff --git a/pi-extension/prompts/autoresearch_debug.md b/pi-extension/prompts/autoresearch_debug.md new file mode 100644 index 00000000..7c481e7a --- /dev/null +++ b/pi-extension/prompts/autoresearch_debug.md @@ -0,0 +1,97 @@ +--- +name: autoresearch_debug +description: "Hunt bugs with scientific method: hypothesize, test, falsify, repeat" +argument-hint: "[Scope: ] [Symptom: ] [Iterations: N] [--fix] [--evals]" +--- + +EXECUTE IMMEDIATELY. + +## Parse Arguments + +Extract from $ARGUMENTS: +- `Scope:` or `--scope` — file globs to investigate +- `Symptom:` or `--symptom` — error message or behavior description +- `Iterations:` or `--iterations` — default 15. "unlimited" for unbounded. +- `--fix` — shorthand for `--chain fix` +- `--severity` — filter: critical, high, medium, low +- `--technique` — force specific technique +- `--evals`, `--evals-interval N`, `--chain` + +## Setup (if required context missing) + +If Scope and Symptom both missing: +1. Auto-scan: run tests, lint, typecheck to detect existing failures +2. request_user_input (single batch): + Q1 (Issue): "What's the problem?" — hunt all bugs, specific error, failing tests, CI failure, performance + Q2 (Scope): "Which files?" — suggested globs + entire codebase + Q3 (Depth): "How deep?" — quick (5), standard (15), deep (30+), unlimited + Q4 (After): "When bugs found?" — report only, find and fix (--chain fix), chain to other, ask each time +If all provided → skip. + +## Investigation Techniques + +| Technique | When to Use | +|---|---| +| Binary search | Know when it worked, find when it broke | +| Differential | Compare working vs broken state | +| Minimal reproduction | Simplify to smallest failing case | +| Trace | Follow execution path through code | +| Pattern search | Grep for known anti-patterns | +| Working backwards | Start from error, trace to root cause | + +## Establish Baseline (before loop) + +1. Auto-scan for failures if no symptom provided +2. Create output directory: `autoresearch/debug-{YYMMDD}-{HHMM}/` +3. TSV header: `# metric_direction: higher_is_better\niteration\ttimestamp\thypothesis\tstatus\ttechnique\tevidence\tfile_line` +4. Metric = cumulative confirmed findings count + +## Iteration Loop + +### Phase 1: Review Context +- Read results TSV (past findings) +- Assess: what's been tested, what vectors remain +- If no hypotheses left → early stop + +### Phase 2: Hypothesize +- Form ONE specific, falsifiable hypothesis +- Format: "I hypothesize that {X} because {evidence}. Test by {Y}." +- Hypothesis must be testable and different from all previous + +### Phase 3: Investigate +- Apply appropriate technique for this hypothesis +- Read relevant code, run targeted tests, check logs +- Collect evidence (file:line references required) + +### Phase 4: Classify +- **confirmed** — hypothesis correct, bug found with evidence +- **disproven** — hypothesis wrong, evidence against it +- **inconclusive** — can't prove or disprove, needs different approach + +### Phase 5: Log +Append to TSV: iteration, timestamp, hypothesis, status, technique, evidence, file_line + +### Eval Checkpoint +If --evals: check if current_iteration % interval == 0 → run checkpoint analysis. + +### Bounded Check +If bounded: current_iteration >= max_iterations → exit loop, print summary. + +## Summary + +Print: total hypotheses tested, confirmed/disproven/inconclusive counts, all confirmed bugs with severity and file:line. + +## Eval Checkpoint (--evals flag) + +If --evals present: +- Compute interval: floor(max_iterations / 3), min 1. Fixed 10 if unbounded. Override: --evals-interval N. +- Every {interval} iterations, pause and analyze current results TSV. +- Print: `--- Eval Checkpoint (iterations {X}-{Y}) ---\nFindings: {confirmed} confirmed | Trend: {up/flat/down}\n{one-line recommendation}\n---` +- If plateau 3+ checkpoints (no new confirmed) → recommend early stop. +- At loop end → full evals summary to evals-summary.md in output directory. + +## Chain Handoff + +After completion, write handoff.json to output directory: version "2.1.0", source "debug", timestamp, status (COMPLETE|USER_INTERRUPT|BOUNDED|ERROR), results_tsv path, findings = confirmed bugs with severity + file:line, config{scope, symptom}. +If --fix flag → chain to fix automatically. +Invoke next target in --chain order. Propagate --evals flag. diff --git a/pi-extension/prompts/autoresearch_evals.md b/pi-extension/prompts/autoresearch_evals.md new file mode 100644 index 00000000..c0f940ca --- /dev/null +++ b/pi-extension/prompts/autoresearch_evals.md @@ -0,0 +1,118 @@ +--- +name: autoresearch_evals +description: "Analyze iteration results: trends, plateaus, regressions, recommendations" +argument-hint: "[path/to/results.tsv] [--format text|json|md]" +--- + +EXECUTE IMMEDIATELY. + +## Parse Arguments + +Extract from $ARGUMENTS: +- Positional path to a specific TSV file +- `--format` — output format: text (default console), json, md (markdown file) +- `--compare ` — (v2.2.0 placeholder, not yet implemented) + +## Input Discovery + +1. If path provided → use that TSV directly +2. If no path → scan current directory + `autoresearch/*/` for `*-results.tsv` files +3. If multiple found → request_user_input: "Which results to analyze?" — list found files +4. If none found → request_user_input: "Provide path to results TSV" +5. Also scan project root for v2.0.03 legacy TSV files (backward compat) + +## Parse TSV + +1. Read line 1: extract `# metric_direction: higher_is_better|lower_is_better` comment + - If missing → infer from column names (metric/error_count → guess, or ask user) +2. Read line 2: header row → detect available columns +3. Read remaining lines: data rows +4. Handle missing `timestamp` column gracefully (v2.0.03 compat) + +## Column Detection & Analysis + +Activate analysis based on columns present in header: + +| Column | Analysis | +|---|---| +| `metric` | Trend direction, plateau detection (3+ flat iterations), diminishing returns, biggest single-iteration jumps | +| `delta` | Per-iteration efficiency, cumulative improvement, effort-to-gain ratio | +| `status` | Keep/discard rate, crash frequency, success streaks, failure clusters, longest winning streak | +| `guard` + `guard-metric` | Guard failure rate, metric-improved-but-guard-failed analysis | +| `severity` | Severity distribution (critical/high/medium/low/info), critical discovery rate per iteration | +| `hypothesis` + `status` | Confirmation rate, investigation efficiency, most productive techniques | +| `commit` | File hotspot analysis (cross-ref with `git diff` for kept commits), change size correlation | +| `technique` | Technique effectiveness ranking | +| `dimension` | Dimension coverage completeness (X/12) | +| `candidate_label` + `judge_verdict` | Convergence speed, oscillation count | +| `error_type` | Error category distribution, fix rate per category | +| `classification` | New vs extension vs duplicate ratio, saturation curve | +| `convergence_count` | Convergence trajectory | + +Unknown columns: report presence but skip analysis. Forward-compatible with future subcommands. + +## Report Structure + +``` +## Evals Summary — {subcommand} ({N} iterations) + +### Key Metrics +- Total iterations: N | Kept: X | Reverted: Y | Revert rate: Z% +- Starting metric: A | Final metric: B | Improvement: C% + +### Trend Analysis +- Metric progression: [description of trajectory] +- Plateau detected at iteration N (metric stable for M iterations) +- Biggest win: iteration X (+delta, description) +- Biggest loss: iteration Y (-delta, description) +- Diminishing returns: [after iteration N, average delta dropped below threshold] + +### Patterns +- What types of changes succeeded: [extracted from descriptions of kept iterations] +- What types of changes failed: [extracted from descriptions of discarded iterations] +- File hotspots: [files changed most in kept iterations, if commit data available] +- Technique effectiveness: [ranked by confirmation rate, if technique column present] + +### Recommendation +- [continue / stop / change strategy — based on trend, plateau, revert rate] +- [specific actionable suggestion based on pattern analysis] +``` + +## Output + +- Console: structured report (30-50 lines) +- If `--format md` → write `evals-summary.md` in same directory as input TSV +- If `--format json` → write `evals-summary.json` with structured data + +## Mid-Loop Checkpoint Protocol (for --evals flag in other commands) + +This section documents the checkpoint protocol that looping commands embed: + +- **Adaptive interval:** `floor(max_iterations / 3)`, minimum 1. Fixed 10 for unbounded. Override: `--evals-interval N`. +- **Checkpoint format (5 lines max):** + ``` + --- Eval Checkpoint (iterations {X}-{Y}) --- + Metric: {start} → {end} ({delta}) | Kept: {n}/{total} | Trend: {up/flat/down} + {one-line recommendation} + --- + ``` +- **Early stop recommendation:** if plateau detected for 3+ consecutive checkpoints +- **Final summary:** at loop end, produce full evals report to console + evals-summary.md + +### Adaptive Interval Examples + +| Subcommand | Default Iterations | Interval | Checkpoints At | +|---|---|---|---| +| reason | 8 | 2 | 2, 4, 6, 8 | +| learn | 10 | 3 | 3, 6, 9, final | +| debug/security | 15 | 5 | 5, 10, 15 | +| fix/scenario | 20 | 6 | 6, 12, 18, final | +| core | 25 | 8 | 8, 16, 24, final | +| unbounded | unlimited | 10 | every 10 | + +## Backward Compatibility + +- v2.0.03 TSV files: column names preserved, `timestamp` absence handled gracefully +- Fuzzy column matching: `metric_value` → `metric`, `error_count` → `metric` +- Files in project root (not `autoresearch/` subdirectory) → discovered during scan +- v2.0.03 status values all supported: baseline, keep, keep (reworked), discard, crash, no-op, hook-blocked, metric-error diff --git a/pi-extension/prompts/autoresearch_fix.md b/pi-extension/prompts/autoresearch_fix.md new file mode 100644 index 00000000..4ffb7c1e --- /dev/null +++ b/pi-extension/prompts/autoresearch_fix.md @@ -0,0 +1,102 @@ +--- +name: autoresearch_fix +description: "Crush errors one-by-one until zero remain: tests, types, lint, build" +argument-hint: "[Target: ] [Scope: ] [Guard: ] [Iterations: N] [--evals] [--from-debug]" +--- + +EXECUTE IMMEDIATELY. + +## Parse Arguments + +Extract from $ARGUMENTS: +- `Target:` or `--target` — command that shows errors (e.g., `npm test`, `tsc --noEmit`) +- `Scope:` or `--scope` — file globs to modify +- `Guard:` or `--guard` — safety command (must always pass) +- `Iterations:` or `--iterations` — default 20. "unlimited" for unbounded. +- `--from-debug` — read handoff.json from previous debug run +- `--category` — filter: test, type, lint, build +- `--evals`, `--evals-interval N`, `--chain` + +## Setup (if required context missing) + +If Target and Scope both missing: +1. Auto-detect failures: run test suite, type checker, linter, build +2. Present results via request_user_input (single batched call): + Q1 (Fix What): "Found [N] test failures, [M] type errors, [K] lint errors. Fix what?" — everything, only tests, only types, only lint + Q2 (Guard): "Safety command that must always pass?" — npm test, tsc, npm run build, skip + Q3 (Scope): "Which files can I modify?" — suggested globs from error locations + all + Q4 (Launch): "Ready?" — fix until zero, fix with limit, cancel +If all provided → skip setup. +If --from-debug → read handoff.json for scope and findings. + +## Precondition Checks + +Verify: git repo exists, clean working tree, no lock files, no detached HEAD. Fail fast on critical issues. + +## Establish Baseline (Iteration 0) + +1. Run Target command → count errors (metric = error count, direction = lower_is_better) +2. Record baseline in TSV +3. Create output directory: `autoresearch/fix-{YYMMDD}-{HHMM}/` +4. TSV header: `# metric_direction: lower_is_better\niteration\ttimestamp\terror_type\terror_fixed\tcommit\tmetric\tdelta\tguard\tstatus\tdescription` + +## Iteration Loop (until zero errors or max_iterations) + +### Phase 1: Review +- Read results TSV + git log +- Run Target to get current error list +- If error count == 0 → exit loop (SUCCESS) + +### Phase 2: Prioritize +Order: crash/fatal → test failures → type errors → lint → warnings. +Within category: easiest first (single-file fixes before cross-file). + +### Phase 3: Fix ONE Thing +- Pick the highest-priority error +- Make ONE focused fix (atomic — addresses exactly one error) +- Record error type and which error was fixed + +### Phase 4: Commit +- Stage and commit: `experiment: fix {error_type} — {description}` + +### Phase 5: Verify +- Run Target → count errors → compute delta +- Expected: error count decreased by 1 or more + +### Phase 6: Guard +- If Guard set → run Guard. If fails → revert. + +### Phase 7: Decide +- **keep** — error count decreased AND guard passes +- **keep (reworked)** — fix needed adjustment, second attempt worked +- **discard** — error count same/increased → `git revert HEAD --no-edit` +- **crash** — target/guard command failed → revert +- **hook-blocked** — git hook blocked the commit +- **metric-error** — target output not parseable → revert + +### Phase 8: Log +Append row: iteration, timestamp, error_type, error_fixed, commit/-, metric (error count), delta, guard, status, description + +### Eval Checkpoint +If --evals: check if current_iteration % interval == 0 → run checkpoint analysis. + +### Bounded Check +If bounded: current_iteration >= max_iterations → exit loop, print summary. + +## Summary + +Print: total errors fixed, remaining errors, error types distribution, fix success rate. + +## Eval Checkpoint (--evals flag) + +If --evals present: +- Compute interval: floor(max_iterations / 3), min 1. Fixed 10 if unbounded. Override: --evals-interval N. +- Every {interval} iterations, pause and analyze current results TSV. +- Print: `--- Eval Checkpoint (iterations {X}-{Y}) ---\nErrors: {start} → {end} ({delta}) | Kept: {n}/{total} | Trend: {up/flat/down}\n{one-line recommendation}\n---` +- If plateau 3+ checkpoints → recommend early stop. +- At loop end → full evals summary to evals-summary.md in output directory. + +## Chain Handoff + +After completion, write handoff.json to output directory: version "2.1.0", source "fix", timestamp, status (COMPLETE|USER_INTERRUPT|BOUNDED|ERROR), results_tsv path, findings = unfixed errors, config{target, scope, guard}. +Invoke next target in --chain order. Propagate --evals flag. diff --git a/pi-extension/prompts/autoresearch_improve.md b/pi-extension/prompts/autoresearch_improve.md new file mode 100644 index 00000000..7b3860c6 --- /dev/null +++ b/pi-extension/prompts/autoresearch_improve.md @@ -0,0 +1,116 @@ +--- +name: autoresearch_improve +description: "Research ICP challenges, discover improvements, generate PRDs" +argument-hint: "[Goal: ] [--icp ] [--discover] [--no-discover] [--seeds ] [--depth shallow|standard|deep] [Iterations: N] [--evals]" +--- + +EXECUTE IMMEDIATELY. + +## Parse Arguments + +Extract from $ARGUMENTS: +- `Goal:` — product area to improve (or full $ARGUMENTS if no keyword) +- `--icp` or `ICP:` — ideal customer profile description +- `--discover` — force inline codebase scan even when context exists +- `--no-discover` — skip auto-discover, warn instead +- `--seeds ` — override default research category seeds +- `--depth` — shallow (5 iterations), standard (15), deep (30) +- `--features` — comma-separated feature names to pre-select for PRD generation +- `Iterations:` or `--iterations` — default 15. "unlimited" for unbounded. +- `--evals`, `--evals-interval N` + +If upstream `handoff.json` exists in CWD → read it. Map source findings to default seed categories: +- probe → ICP challenges, UX & experience +- predict → Competitor gaps, Revenue & growth +- debug/security → Competitor gaps, ICP challenges +- Override with `--seeds`. + +## Setup (if Goal or ICP missing) + +request_user_input (single batch): + Q1 (Goal): "What product area to improve?" — open text + Q2 (ICP): "Who is your ideal customer?" — open text describing target buyer/user + Q3 (Pain points): "Top 3 pain points your customers face?" — open text + Q4 (Competitors): "Key competitors?" — open text, or "skip" + Q5 (Depth): "How deep?" — shallow (5 iterations, quick scan), standard (15, recommended), deep (30+, exhaustive) +If all provided inline → skip. + +## Phase 1: Product Context + +Resolve product context (priority chain): +1. Learn summary (`autoresearch/learn-*/summary.md`, most recent) → read it +2. README.md (≥500 chars, non-boilerplate) → extract product description +3. `package.json` / `pyproject.toml` / `Cargo.toml` description (≥10 chars) → use it +4. If ALL above absent AND NOT `--no-discover` → auto-discover: scan 10 key files (manifest, routes, models, config), cap 1500 tokens +5. If `--discover` → force scan regardless of above +6. If nothing found → warn: "No product context. Run `$autoresearch learn --mode summarize` for better results." + +## Phase 2: Research Loop + +Create output directory: `autoresearch/improve-{YYMMDD}-{HHMM}/` +TSV header: `# metric_direction: higher_is_better` +Columns: `iteration|timestamp|category|research_question|status|source|insight_problem|insight_mechanism|confidence|classification` + +**5 research categories:** +1. ICP challenges — pain points, jobs-to-be-done, unmet needs +2. Competitor gaps — weaknesses, missing features, technical differentiators +3. Market trends — timing signals, emerging patterns, regulatory shifts +4. UX & experience — interaction models, onboarding, retention mechanics +5. Revenue & growth — pricing, acquisition, monetization, upsell/expansion + +**Iteration protocol:** +- Reserve first 5 iterations: one per category (forced breadth) +- Remaining iterations: target categories with richest signal +- Per iteration: form research question → WebSearch → synthesize → normalize to canonical insight schema → classify (new/extension/duplicate) → tag confidence (HIGH: 3+ sources, MEDIUM: 2, LOW: 1) → cross-check against codebase → log +- **Saturation:** net-new insights < 2 for 3 consecutive non-reserved iterations → SATURATED, exit loop +- Hard ceiling (Iterations flag) as infinite-loop guard + +**Insight schema:** `{problem: 10-word canonical form, affected_persona: ICP segment, proposed_mechanism: how to address, expected_outcome: what success looks like}` +**Classification:** New = novel {problem, persona} pair. Extension = same pair, different mechanism. Duplicate = same pair + mechanism → skip. + +### Eval Checkpoint +If --evals: check if current_iteration % interval == 0 → run checkpoint. +Print: `--- Eval Checkpoint (iterations {X}-{Y}) ---\nInsights: {total} (+{new}) | Categories: {covered}/5 | Saturation: {window}/3\n{recommendation}\n---` + +## Phase 3: Feature Ranking + Selection + +1. **ICP binary gate** — filter insights not serving the stated ICP +2. **3-tier bucketing** — Must-have / Nice-to-have / Moonshot +3. **Pairwise ranking** within Must-have tier only (cap 7-10 items) +4. **2-sentence rationale** per item citing research evidence +5. **Confidence indicator** per item (HIGH / MEDIUM / LOW) + +Write `improvement-plan.md` with full tiered ranking. + +request_user_input (multi-select): present tiered list, user selects which features become PRDs. +If `--features` provided → pre-select matching items, still show for confirmation. + +## Phase 4: PRD Generation + +Per selected feature, write `prd-{feature-slug}.md`: +- Top disclaimer: "Auto-generated from research findings. DECISION NEEDED items and LOW-confidence sections require your judgment." +- Problem statement (from research evidence chain) +- User stories (from ICP + persona data) +- Requirements (functional + non-functional, MoSCoW from tier) +- Acceptance criteria +- Technical approach (from codebase context, framed as "suggested starting points") +- Risks + confidence (evidence tiers: primary = codebase, secondary = web research) +- Success metrics +- `DECISION NEEDED` markers for unresolvable tradeoffs +- `Open Questions` section + +Write `research-findings.md` — all insights with citations + confidence. +Write `summary.md` — overview, research stats, category coverage, saturation status. + +## Summary + +Print: total iterations, insights discovered (new/extension), categories covered, saturation status, PRDs generated, output directory path. + +## Eval Summary (--evals flag) + +If --evals: write `evals-summary.md` to output directory with full analysis. + +## Handoff + +Write `handoff.json`: version "2.1.0", source "improve", timestamp, status (COMPLETE|SATURATED|USER_INTERRUPT|BOUNDED|ERROR), results_tsv path, findings = improvements with tier + confidence + prd_path, config{goal, icp, depth, categories_explored, insights_total, prds_generated}. +Improve is a terminal emitter — no downstream chain invocation. diff --git a/pi-extension/prompts/autoresearch_learn.md b/pi-extension/prompts/autoresearch_learn.md new file mode 100644 index 00000000..96e319c3 --- /dev/null +++ b/pi-extension/prompts/autoresearch_learn.md @@ -0,0 +1,136 @@ +--- +name: autoresearch_learn +description: "Scout codebase and auto-generate docs — or a navigable wiki knowledge base — with validation-fix loop" +argument-hint: "[Mode: ] [Scope: ] [Iterations: N] [--depth ] [--modules ] [--force] [--evals]" +--- + +EXECUTE IMMEDIATELY. + +## Parse Arguments + +Extract from $ARGUMENTS: +- `Mode:` or `--mode` — init (create from scratch), update (refresh existing), check (validate), summarize (brief overview), wiki (navigable knowledge base) +- `Scope:` or `--scope` — file globs to document +- `Depth:` or `--depth` — overview, standard, comprehensive +- `--file ` — specific file to document +- `--scan` — force fresh codebase scout +- `--topics` — comma-separated focus topics +- `--modules ` — wiki mode: comma-separated module names/paths overriding auto-detection +- `--force` — wiki mode: regenerate all pages from scratch, ignore existing manifest +- `--no-fix` — validate only, don't auto-fix issues +- `--format` — markdown (default), json, rst +- `Iterations:` or `--iterations` — default 10. "unlimited" for unbounded. +- `--evals`, `--evals-interval N`, `--chain`, `--` + +## Setup (if Mode or Scope missing) + +request_user_input (single batch): + Q1 (Mode): "What to do?" — init (generate docs), update (refresh), check (validate), summarize (overview), wiki (knowledge base) + Q2 (Scope): "Which files?" — suggested globs + entire codebase + Q3 (Depth): "How detailed?" — overview only, standard, comprehensive + Q4 (Topics): "Focus on?" — architecture, API, database, testing, all +If all provided → skip. + +## Establish Baseline + +1. Scout codebase: file tree, imports/exports, existing docs +2. Identify documentation gaps (undocumented files, outdated docs, missing READMEs) +3. Create output directory: `autoresearch/learn-{YYMMDD}-{HHMM}/` +4. TSV header: `# metric_direction: higher_is_better\niteration\ttimestamp\tfile_documented\tvalidation_status\tissues_found\tissues_fixed\tdescription` +5. Metric = files with valid documentation (higher is better) + +## Summarize Mode (no loop) + +If mode == summarize: +- One-shot: scan codebase → produce structured summary +- Write summary.md to output directory +- Skip iteration loop entirely + +## Wiki Mode (no per-file loop) + +If mode == wiki: reuse Scout (Phase 1) + Analyze output, then generate a navigable `wiki/` knowledge base. Skip the init/update/check loop. Metric = `pages_generated / pages_planned × 100` (from manifest); size target 300 lines/page. + +### Module Discovery (priority order) +1. `--modules` flag (explicit override, always wins; every path must resolve inside project root — reject escapes) +2. Monorepo workspaces (`workspaces` in package.json, Cargo workspace members, `pnpm-workspace.yaml`) +3. Per-directory project files (`pyproject.toml`, `Cargo.toml`, `go.mod`, `*.csproj`) +4. Heuristic: dirs with 3+ source files (code extensions only — .ts/.py/.go/.rs/.java/.rb/.swift/.kt/.c/.cpp/.cs; tests count, config/markdown don't); nested dirs roll up to nearest module ancestor + +Cap 10 modules. If >10, group by top-level dir; if a group has >5 sub-modules, expand and take 10 largest by file count. + +### Plan (write-ahead) +1. `mkdir -p wiki/modules/`; append `wiki-manifest.json` to `.gitignore` if absent +2. Write `wiki-manifest.json` BEFORE generating — `{version:"1", generated_at, generation_status:"in_progress", modules_detected:[…], pages_planned:N, pages:{ "wiki/architecture.md":{status:"pending",type:"architecture"}, "wiki/modules/.md":{…"module"}, "wiki/glossary.md":{…}, "wiki/onboarding.md":{…}, "wiki/index.md":{…} }}` +3. Write stub `wiki/index.md` listing every planned page as `[pending]` (navigation survives interruption) +4. Resume: if valid manifest exists → `--force` deletes it and regenerates all, else skip `"generated"` pages and only do `"pending"`. Corrupted manifest (invalid JSON, or missing `version`/`pages`) without `--force` → error directing user to `--force`. + +### Generate (priority-first, one agent call per page, bounded context) +1. `architecture.md` — system overview from scout context. Up to 5 Mermaid diagrams, 3 types only (`graph TD`, `sequenceDiagram`, `classDiagram`); include one canonical example of each in the prompt; pick by signal. +2. Module pages (alphabetical) — per-module agent gets: (a) file listing, (b) first 50 lines of ≤10 key files (entry points → largest → alphabetical), (c) Phase 2 overview. Required sections: Overview, Key Files; optional: Patterns/API/Dependencies/Getting Started. +3. `glossary.md` — domain terms from class names, exports, types, comments; filter language keywords + stdlib; soft cap ~60-80, prioritize terms in 3+ files. +4. `onboarding.md` — reading order, env setup, first-contribution workflow, gotchas. Sources: dir structure, README/docs, manifests, entry-point sampling, `git log --since='6 months ago'` directory frequency (skip with note if not a git repo or >10s). +5. `index.md` — final pass: replace stub with real page descriptions + reading order. + +### Per-page contract +- `generated_by: autoresearch` in YAML frontmatter +- ~300 lines/page (soft); Mermaid ≤15 nodes/diagram; forward-only cross-links, ≤10 per page + +### Safety +- **Secrets (2-layer):** (1) prompt instructs "summarize config, never include verbatim values from .env/credentials or strings matching key/secret/token/password; extract env var *names* not *values*"; (2) post-gen, `grep -rlE '(AKIA[0-9A-Z]{16}|sk-[a-zA-Z0-9]{20,}|ghp_[a-zA-Z0-9]{36}|password\s*[:=]\s*\S+|mongodb(\+srv)?://\S+|postgres(ql)?://\S+)' wiki/` and warn (non-blocking) in the report. +- **Name collision:** before overwriting a page, check for `generated_by: autoresearch` frontmatter; if absent (user-created) skip with warning. `--force` overrides. + +### Finish +After each page is written, flip its manifest entry `pending`→`generated` (interrupt-safe). When all done, set `generation_status:"complete"`. Then run Phase 3 (Validate) with wiki path swap: replace `docs/` with `wiki/` (`ls wiki/*.md wiki/modules/*.md 2>/dev/null`), use maxLoc 300. Output: `✓ Wiki: [N] modules, [M] pages generated`. + +## Iteration Loop (init/update/check modes) + +### Phase 1: Scout +- Scan for documentation gaps +- Prioritize: no docs → outdated docs → incomplete docs +- If no gaps remain → early stop (SUCCESS) + +### Phase 2: Generate/Update +- Pick highest-priority gap +- Write or update documentation for ONE file/module +- Follow project conventions for doc format and location + +### Phase 3: Validate +- Check generated docs against code: descriptions accurate? Examples valid? Links work? +- Run doc linters if available +- Record: validation_status (pass/fail), issues found + +### Phase 4: Fix (unless --no-fix) +- If validation finds issues → fix the doc +- Commit clean doc: `docs: document {file/module}` + +### Phase 5: Log +Append to TSV: iteration, timestamp, file_documented, validation_status, issues_found, issues_fixed, description + +### Eval Checkpoint +If --evals: check if current_iteration % interval == 0 → run checkpoint. + +### Bounded Check +If bounded: current_iteration >= max_iterations → exit loop. + +## Output + +- `learn-results.tsv` +- `summary.md` — documentation overview +- `validation-report.md` — issues found/fixed + +## Summary + +Print: files documented, validation pass rate, issues found/fixed, remaining gaps. + +## Eval Checkpoint (--evals flag) + +If --evals present: +- Compute interval: floor(max_iterations / 3), min 1. Fixed 10 if unbounded. +- Print: `--- Eval Checkpoint (iterations {X}-{Y}) ---\nDocs written: {n} | Validation: {pass}/{total} | Gaps remaining: {m}\n{recommendation}\n---` +- If 3+ checkpoints with no new docs → recommend early stop. +- At loop end → full evals summary to evals-summary.md. + +## Chain Handoff + +After completion, write handoff.json: version "2.1.0", source "learn", timestamp, status, results_tsv path, findings = documentation gaps remaining, config{mode, scope, depth}. +Invoke next target in --chain order. Propagate --evals flag. diff --git a/pi-extension/prompts/autoresearch_plan.md b/pi-extension/prompts/autoresearch_plan.md new file mode 100644 index 00000000..a7cfe510 --- /dev/null +++ b/pi-extension/prompts/autoresearch_plan.md @@ -0,0 +1,95 @@ +--- +name: autoresearch_plan +description: "Convert a goal into validated Scope, Metric, Direction, Verify config" +argument-hint: "[Goal: ] [--chain ]" +--- + +EXECUTE IMMEDIATELY. + +## Parse Arguments + +Extract from $ARGUMENTS: +- `Goal:` — text after keyword, or full $ARGUMENTS if no keyword +- `--chain ` — comma-separated downstream commands +- `--` — chain shorthand + +Remaining text = goal description. + +## Setup (if Goal missing) + +request_user_input (single batch): + Q1 (Goal): "What do you want to achieve?" — open text + Q2 (Type): "What kind of goal?" — improve a metric, fix errors, audit security, explore edge cases, document code, ship something +If Goal provided → skip. + +## Phase 1: Analyze Goal + +Parse the goal to determine: +- Is it measurable? (metric-driven vs subjective) +- What's the natural scope? (files, modules, entire codebase) +- What subcommand fits best? (core loop, fix, debug, security, etc.) + +## Phase 2: Derive Scope + +1. Scan project structure +2. Identify files relevant to the goal +3. Propose file globs +4. If ambiguous → ask user to confirm + +## Phase 3: Derive Metric + Direction + +For metric-driven goals: +- Identify what to measure (test coverage, error count, bundle size, latency, etc.) +- Determine direction: higher_is_better or lower_is_better +- Propose metric name and description + +For subjective goals: +- Suggest proxy metrics where possible +- Or recommend $autoresearch reason for non-measurable goals + +## Phase 4: Derive Verify Command + +1. Identify how to extract the metric as a number from a shell command +2. Propose Verify command (e.g., `npm test -- --coverage | grep "All files" | awk '{print $10}'`) +3. **Safety screen:** check proposed command for rm -rf, fork bombs, curl|sh, credentials +4. Dry-run the Verify command → confirm it outputs a valid number +5. If dry-run fails → adjust command and retry + +## Phase 5: Derive Guard (optional) + +Propose a Guard command if applicable: +- Test suite: `npm test` / `pytest` / `go test ./...` +- Type check: `tsc --noEmit` / `mypy` +- Build: `npm run build` +- None if not applicable + +## Phase 6: Suggest Iterations + +Based on goal complexity: +- Simple metric improvement → 10-15 +- Moderate refactoring → 20-25 +- Complex multi-file changes → 30+ +- Recommend bounded default, mention `Iterations: unlimited` option + +## Phase 7: Present Config + +Output a ready-to-run autoresearch config block: + +``` +$autoresearch +Goal: {derived goal} +Scope: {derived globs} +Metric: {derived metric} +Direction: {higher_is_better|lower_is_better} +Verify: {derived command} +Guard: {derived guard or omit} +Iterations: {suggested count} +``` + +Ask user: "Run this config now, or adjust?" + +## Chain Handoff + +If --chain set: +- Write handoff.json: version "2.1.0", source "plan", timestamp, status COMPLETE, config = derived config block +- Invoke next target with the derived config diff --git a/pi-extension/prompts/autoresearch_predict.md b/pi-extension/prompts/autoresearch_predict.md new file mode 100644 index 00000000..8ed34e54 --- /dev/null +++ b/pi-extension/prompts/autoresearch_predict.md @@ -0,0 +1,94 @@ +--- +name: autoresearch_predict +description: "5 expert personas debate proposed changes before implementation" +argument-hint: "[Scope: ] [Goal: ] [--depth shallow|standard|deep] [--adversarial] [--chain ]" +--- + +EXECUTE IMMEDIATELY. + +## Parse Arguments + +Extract from $ARGUMENTS: +- `Scope:` or `--scope` — file globs to analyze +- `Goal:` or `--goal` — focus area for analysis +- `Depth:` or `--depth` — shallow (3 personas, 1 round), standard (5, 2), deep (8, 3) +- `--personas N` — override persona count (3-8) +- `--rounds N` — override debate rounds (1-3) +- `--adversarial` — use hostile reviewer personas instead of default +- `--budget N` — max findings across all personas (default 40) +- `--fail-on ` — CI gate: exit non-zero if findings at/above threshold +- `--incremental` — reuse existing knowledge files, update only changed files +- `--chain`, `--` + +Remaining text not matching flags = goal description. + +## Setup (if Scope or Goal missing) + +request_user_input (single batch): + Q1 (Scope): "Which files to analyze?" — suggested globs + entire codebase + Q2 (Goal): "What should personas focus on?" — code quality, security, performance, architecture, all + Q3 (Depth): "How deep?" — shallow (3 personas, 1 round), standard (5, 2 — recommended), deep (8, 3) + Q4 (Chain): "After analysis, chain to?" — debug, security, fix, ship, scenario, no chain +If all provided → skip. + +## Phase 1: Reconnaissance + +Scan all in-scope files. Build structured knowledge: +- File inventory with purpose annotations +- Dependency graph (imports/exports) +- API surface (routes, handlers, types) +- Data flow (inputs → processing → outputs → storage) +- Existing test coverage map + +## Phase 2: Persona Generation + +Load `references/predict-personas.md` for persona definitions. + +**Default set (5):** Architect, Security Analyst, Performance Engineer, Reliability Engineer, Devil's Advocate. +**Adversarial set (--adversarial):** Breaker, Cheater, Scaler, Newbie, Malicious Insider. + +Each persona receives: task description + codebase knowledge + their specific evaluation criteria. +Personas are isolated — no shared context between them. + +## Phase 3: Independent Analysis + +Each persona analyzes the codebase independently: +- Read relevant code through their lens +- Produce findings with: title, severity, confidence (0-100%), file:line, recommendation +- Max findings per persona: budget / persona_count + +## Phase 4: Debate (per round) + +For each debate round: +1. Present all personas' findings to each other +2. Each persona can: challenge findings, raise new issues, change confidence +3. Cross-examination: personas must respond to challenges with evidence +4. No persona can dismiss without counter-evidence + +## Phase 5: Consensus + +Synthesizer aggregates all findings: +1. Deduplicate (same file:line + same issue = merge, keep highest severity) +2. Resolve conflicts (if personas disagree, note dissent) +3. **Anti-herd check:** if all personas agree on everything, synthesizer MUST find at least 1 counter-argument +4. Rank by: severity × average confidence × persona agreement count + +## Phase 6: Report + +Create output directory: `autoresearch/predict-{YYMMDD}-{HHMM}/` + +Write: +- `summary.md` — top findings, consensus view, risk assessment +- `debate.md` — full persona analysis + debate transcript +- Per-persona sections with individual findings + +Print to console: top 10 findings ranked by severity × confidence. + +## Phase 7: CI Gate + +If `--fail-on` set: check findings against threshold. Exit non-zero if exceeded. + +## Chain Handoff + +Write handoff.json: version "2.1.0", source "predict", timestamp, status (COMPLETE|ERROR), findings = consensus findings with severity + confidence + file:line, config{scope, goal, depth}. +Invoke next target in --chain order. diff --git a/pi-extension/prompts/autoresearch_probe.md b/pi-extension/prompts/autoresearch_probe.md new file mode 100644 index 00000000..1cbfce6b --- /dev/null +++ b/pi-extension/prompts/autoresearch_probe.md @@ -0,0 +1,116 @@ +--- +name: autoresearch_probe +description: "8 personas interrogate requirements until constraints saturate" +argument-hint: "[Topic: ] [Scope: ] [--depth shallow|standard|deep] [--personas N] [--mode interactive|autonomous] [Iterations: N] [--evals]" +--- + +EXECUTE IMMEDIATELY. + +## Parse Arguments + +Extract from $ARGUMENTS: +- `Topic:` — strip keyword, remaining text is topic (or full $ARGUMENTS if no keyword) +- `Scope:` or `--scope` — file globs for codebase grounding +- `Depth:` or `--depth` — shallow (5 rounds), standard (15), deep (30) +- `--personas N` or `Personas:` — active persona count (3-8, default 6) +- `--saturation-threshold N` — net-new constraints/round below which counts toward saturation (default 2) +- `--mode` or `Mode:` — interactive (default, uses request_user_input) or autonomous (self-answers from codebase) +- `--adversarial` — rotate hostile personas to front +- `Iterations:` or `--iterations` — default 15 rounds. "unlimited" for unbounded. +- `--evals`, `--evals-interval N`, `--chain`, `--` + +## Setup (if Topic missing) + +request_user_input (single batch): + Q1 (Topic): "What to probe?" — open text describing feature, requirement, or design + Q2 (Scope): "Which files for context?" — suggested globs + entire codebase + Q3 (Depth): "How deep?" — shallow (5 rounds), standard (15), deep (30), unlimited + Q4 (Mode): "How to answer persona questions?" — interactive (you answer), autonomous (agent infers from code) +If all provided → skip. + +## 8 Personas + +| # | Persona | Focus | +|---|---|---| +| 1 | Domain Expert | Business rules, domain constraints, terminology | +| 2 | End User | Usability, expectations, error recovery | +| 3 | Skeptic | Assumptions that might be wrong | +| 4 | Edge-Case Hunter | Boundary conditions, rare scenarios | +| 5 | Ops Engineer | Deployment, monitoring, scaling, failure modes | +| 6 | Security Reviewer | Attack vectors, data protection, auth | +| 7 | Contradiction Finder | Conflicts between requirements | +| 8 | Scope Guardian | Feature creep, unnecessary complexity | + +If --adversarial: rotate Skeptic + Contradiction Finder + Edge-Case Hunter to front. + +## Phase 1: Seed + +- Parse topic into initial constraint set +- Read codebase context (if --scope provided) +- Initialize constraint registry (empty) + +## Round Loop + +### Phase 2: Persona Activation +- Select 2-3 personas for this round (rotate through all 8) +- Each persona generates 3-5 probing questions from their perspective + +### Phase 3: Codebase Grounding +- Check questions against existing code for evidence +- Annotate questions with: relevant file:line, existing behavior, gaps + +### Phase 4: Answer Capture +- **Interactive mode:** present questions via request_user_input, collect answers +- **Autonomous mode:** infer answers from codebase context, label confidence (high/medium/low) + +### Phase 5: Constraint Extraction +- Parse answers into atomic constraints +- Each constraint: id, source persona, description, confidence, evidence +- Deduplicate against existing registry + +### Phase 6: Cross-Check +- Check new constraints against existing for conflicts +- Flag contradictions for resolution (interactive → ask user, autonomous → note uncertainty) + +### Phase 7: Saturation Check +- Count net-new constraints this round +- If net-new < saturation_threshold for 3 consecutive rounds → SATURATED, exit loop +- Track: total constraints, new this round, saturation window + +### Phase 8: Log +Append to output: round number, personas active, questions asked, constraints extracted, net-new count + +### Eval Checkpoint +If --evals: check if current_round % interval == 0 → run checkpoint. + +### Bounded Check +If bounded: current_round >= max_iterations → exit loop. + +## Phase 9: Synthesize & Output + +Create output directory: `autoresearch/probe-{YYMMDD}-{HHMM}/` + +1. Write `constraints.md` — full constraint registry organized by category +2. Write `conflicts.md` — unresolved contradictions +3. Generate ready-to-run autoresearch config: + - Derived Goal, Scope, Metric, Verify from constraints + - Include as code block in summary.md + +Print: total rounds, constraints found, saturation status, unresolved conflicts. + +## Summary + +Print: total rounds, total constraints, net-new trend, saturation status, top 5 most impactful constraints. + +## Eval Checkpoint (--evals flag) + +If --evals present: +- Compute interval: floor(max_iterations / 3), min 1. Fixed 10 if unbounded. +- Print: `--- Eval Checkpoint (rounds {X}-{Y}) ---\nConstraints: {total} (+{new}) | Saturation: {window_count}/3\n{recommendation}\n---` +- If saturated 3+ checkpoints → recommend early stop. +- At loop end → full evals summary to evals-summary.md. + +## Chain Handoff + +Write handoff.json: version "2.1.0", source "probe", timestamp, status (COMPLETE|SATURATED|USER_INTERRUPT|BOUNDED|ERROR), findings = constraints, config = derived autoresearch config. +Invoke next target in --chain order. Propagate --evals flag. diff --git a/pi-extension/prompts/autoresearch_reason.md b/pi-extension/prompts/autoresearch_reason.md new file mode 100644 index 00000000..d17dc90f --- /dev/null +++ b/pi-extension/prompts/autoresearch_reason.md @@ -0,0 +1,109 @@ +--- +name: autoresearch_reason +description: "Adversarial debate with blind judges until convergence" +argument-hint: "[Task: ] [Domain: ] [--mode convergent|creative|debate] [--judges N] [Iterations: N] [--evals]" +--- + +EXECUTE IMMEDIATELY. + +## Parse Arguments + +Extract from $ARGUMENTS: +- `Task:` — question, proposal, design, argument, or claim to refine +- `Domain:` or `--domain` — software, product, business, security, research, content +- `Mode:` or `--mode` — convergent (default), creative, debate +- `--judges N` or `Judges:` — blind judge count (3 default, 5 thorough, 7 deep) +- `--convergence N` or `Convergence:` — stop when incumbent wins N consecutive rounds (default 3) +- `Iterations:` or `--iterations` — default 8. "unlimited" for unbounded. +- `--judge-personas` — custom judge persona overrides +- `--no-synthesis` — skip synthesis, pure debate only +- `--temperature` — generation temperature hint +- `--evals`, `--evals-interval N`, `--chain`, `--` + +Remaining text not matching flags = task description. + +## Setup (if Task or Domain missing) + +request_user_input (single batch): + Q1 (Task): "What should be reasoned about?" — open text + Q2 (Domain): "What domain?" — software architecture, product strategy, business decision, security, research, content + Q3 (Mode): "Refinement mode?" — convergent (stop when winner repeats), creative (never auto-stop), debate (no synthesis) + Q4 (Judges): "How many blind judges?" — 3 (default), 5 (thorough), 7 (deep) +If all provided → skip. + +## Setup Phase + +1. Load `references/reason-judge-protocol.md` for judge and convergence specs +2. Parse domain → select domain-specific judge criteria +3. Create output directory: `autoresearch/reason-{YYMMDD}-{HHMM}/` +4. TSV header: `round\ttimestamp\tcandidate_label\tjudge_verdict\tconvergence_count\tdescription` +5. Initialize: incumbent = null, convergence_count = 0 + +## Round Loop + +### Phase 1: Generate-A +- If round 1: Author-A generates first candidate from task description +- If round N>1: incumbent is Author-A's candidate +- Cold-start: Author-A sees ONLY task description + domain context + +### Phase 2: Critic +- Critic receives candidate-A (cold-start, no shared session) +- MUST find at least 3 specific weaknesses +- MUST suggest what a superior candidate would do differently +- Role is purely adversarial — never compliment + +### Phase 3: Generate-B +- Author-B receives: task + candidate-A + critique (cold-start) +- Produces candidate-B addressing critique while preserving A's strengths + +### Phase 4: Synthesize (unless --no-synthesis or debate mode) +- Synthesizer receives: task + A + B (cold-start) +- Produces hybrid candidate-AB merging best of both + +### Phase 5: Blind Judge Panel +- Each judge receives 3 candidates with RANDOMIZED labels (Label-X, Label-Y, Label-Z) +- Judges evaluate independently on domain-specific criteria +- Each produces ranking + one-paragraph justification +- Verdict: majority vote. Tie → synthesized candidate wins. + +### Phase 6: Convergence Check +- If winner == incumbent → convergence_count++ +- If winner != incumbent → convergence_count = 1, winner becomes incumbent +- **Convergent mode**: convergence_count >= N → CONVERGED, stop +- **Creative mode**: never auto-stop +- **Debate mode**: same as convergent, no synthesis + +### Phase 7: Oscillation Guard +If incumbent changed 5+ times in last 8 rounds → recommend early stop (not converging). + +### Phase 8: Log +Append to TSV: round, timestamp, winning candidate label, judge verdict, convergence_count, description + +### Eval Checkpoint +If --evals: check if current_round % interval == 0 → run checkpoint. + +### Bounded Check +If bounded: current_round >= max_iterations → exit loop. + +## Output + +- `reason-results.tsv` — per-round results +- `lineage.md` — full history of candidates + critiques + judge reasoning +- `summary.md` — final winner, convergence trajectory, key insights + +## Summary + +Print: total rounds, convergence status, final winner summary, judge agreement rate. + +## Eval Checkpoint (--evals flag) + +If --evals present: +- Compute interval: floor(max_iterations / 3), min 1. Fixed 10 if unbounded. +- Print: `--- Eval Checkpoint (rounds {X}-{Y}) ---\nIncumbent: {label} | Convergence: {count}/{target} | Oscillations: {n}\n{recommendation}\n---` +- If oscillation detected 3+ checkpoints → recommend early stop. +- At loop end → full evals summary to evals-summary.md. + +## Chain Handoff + +After completion, write handoff.json: version "2.1.0", source "reason", timestamp, status (COMPLETE|CONVERGED|USER_INTERRUPT|BOUNDED|ERROR), results_tsv path, findings = [{id, type: "recommendation", summary: winner description}], config{task, domain, mode}. +Invoke next target in --chain order. Propagate --evals flag. diff --git a/pi-extension/prompts/autoresearch_regression.md b/pi-extension/prompts/autoresearch_regression.md new file mode 100644 index 00000000..1a97ea7d --- /dev/null +++ b/pi-extension/prompts/autoresearch_regression.md @@ -0,0 +1,110 @@ +--- +name: autoresearch_regression +description: "Layered regression stability gate: capture baseline behavior on the base ref, diff the candidate, verdict STABLE/UNSTABLE before you push" +argument-hint: "[Base: ] [Scope: ] [--select auto|full|affected] [--samples N] [--noise-band %] [--matrix] [--max-runs N] [--baseline-cache] [Baseline: ] [--probe|--no-probe|--probe deep] [--predict --reason --debug --fix --fix-cycles N --evals --evals-interval N --chain]" +--- + +EXECUTE IMMEDIATELY. + +A regression is a **green→red transition ONLY**. The gate orchestrates the project's OWN test/bench/snapshot/migrate commands (it is a protocol, not a bundled framework), captures baseline behavior in an isolated git worktree, re-runs the candidate, and reports a tiered ship/no-ship verdict. + +## Parse Arguments + +Extract from $ARGUMENTS: +- `Base:` or `--base` — base ref to diff against. Default: `git merge-base HEAD main` (else `main`/`master`). +- `Scope:` or `--scope` — file globs limiting the change surface. +- `--select auto|full|affected` — test selection (default `auto`). `auto` = use the detected affected-test mapper if available, else FULL suite. Never a silent subset. +- `--samples N` — SCORE samples/side (default 7). `--noise-band %` — perf tolerance (default 5%). +- `--matrix` — opt-in matrix axis (OFF by default). `--max-runs N` — ceiling (default 200). +- `--baseline-cache` (default on) — reuse `baseline//` by SHA. `Baseline: ` — bypass capture. +- `--probe` (default) / `--probe deep` / `--no-probe`. +- `--predict --reason --debug --fix --fix-cycles N --evals --evals-interval N --chain ` and `--` shorthand. +- `Iterations:` — repeat-axis count for `--select`/repeat sweeps. + +## Setup / Probe-on-launch + +1. Auto-detect per-dimension verify commands: `package.json` scripts, `Makefile`, `nx`, migrate config, bench/snapshot/size scripts. +2. request_user_input (single batch) to confirm detected commands + base ref + which dimensions to run. +3. **Auto-skip probe** when CI / no-TTY / `--mode autonomous` / complete-config / chained-handoff — log the inferred config instead of asking. + +## Classification Phase (first-class, before any differential) + +Establish the baseline green-set per dimension, then tag each unit. Match by **test-id first, then path**. + +| State | Meaning | Gated? | +|---|---|---| +| `regression-eligible` | green on baseline | YES — only green→red counts | +| `pre-existing` | red→red (already failing) | no — excluded | +| `new-coverage` | absent→red (brand-new test) | no — new coverage, ungated | +| `flaky` | nondeterministic on baseline | no — routed to flakiness SCORE | +| `baseline-unavailable` | dimension never green | no — advisory only | + +**Core invariant: red→red, absent→red, and flake→red are NOT regressions.** Run flakiness N× on **both** baseline and candidate; a candidate failure inside the baseline flake-envelope routes to flakiness SCORE, never to a regression. `5/5 green ≠ non-flaky` — detection probability is `1−(1−p)^n` (≈23% at p=5%, n=5); print it. + +## Baseline Capture + +`git worktree add --detach ` (detached SHA — avoids "branch already checked out" when Base==HEAD) → `baseline//`; `--baseline-cache` reuses by SHA. Then per worktree: `git submodule update --init` + dependency install (lockfile is SHA-pinned so the cache stays sound). Per-dimension **setup tiers**: api-contract = file-diff, no build; functional / integration-e2e / data-migration = full env. On completion or crash: `git worktree remove` + `git worktree prune`. Warn on concurrent index-lock contention. `Baseline: ` bypasses capture. + +## Dimension Registry (8) + +| Dim | Tier | Compare | Key params | +|---|---|---|---| +| functional | HARD | baseline green-set vs candidate; new fail = regression | test cmd, globs | +| api-contract | HARD | schema/exports diff → breaking? | schema cmd, breaking ruleset | +| data-migration | HARD | default: up applies clean + idempotent re-apply + app boots/schema valid; schema/rowcount roundtrip opt-in | migrate cmds, fixture, allowlisted DB | +| integration-e2e | HARD | e2e green-set diff | e2e cmd | +| flakiness | SCORE | run N× on baseline + candidate, count nondeterministic | runs (def 5), flake-threshold | +| performance | SCORE | K **independent-process** samples/side, Mann-Whitney U AND effect beyond `max(noise-band%, k·stdev)`, report median delta | bench cmd, samples=7, noise-band=5%, k=2 | +| resource | SCORE | mem/bundle/size delta vs budget | size cmd, budget | +| visual-ui | SCORE | containerize render; default `maxDiffPixelRatio` + AA-detection; SSIM = per-page escalation | snapshot cmd, diff-threshold, mask regions | + +`--select auto` mapper (`jest --findRelatedTests` fed the changed-file list; `nx affected` project-graph) is **best-effort static-import** — blind to dynamic/runtime/global-setup couplings. The report names the mapper + its blind-spot caveat; a HARD STABLE earned on an affected subset prints "run `--select full` for high-stakes". FULL suite is the correctness default. + +- **performance independence:** each sample = an independent process launch (warmups discarded), never an in-process iteration — autocorrelation/GC/thermal otherwise violate Mann-Whitney's independence assumption. At n=7 the test detects only ≳1σ regressions; raise `--samples` for tight gates. +- **data-migration guard:** opt-in. Before any migration the DB URL MUST pass an **anchored** allowlist — host is exactly `localhost` / `127.0.0.1` / a container or service hostname, OR the database name carries a `_test` / `_ci` suffix. A bare substring (e.g. `test` inside `latest`, `ci` inside `precision`) does **not** qualify. Anything else is refused — ephemeral only, never dev/prod — and even an allowlisted URL requires explicit user confirm before applying. Missing/absent down-migration = forward-only advisory, **never a finding**. + +## Differential Loop (per dim × axis × run) + +Run candidate verify vs baseline metric → compute `regressed` bool + 0-100 `subscore`. Axes: `diff` (default), `repeat N×`, `full`, `matrix` (opt-in). Log one TSV row per cell. + +**--max-runs ceiling:** projected = dims × axes × samples × matrix-cells; if > `--max-runs` (default 200) → warn + require confirm (CI default = abort with message). + +## Verdict + +- Any HARD `regressed=true` with `classification=regression-eligible` → **UNSTABLE** (green→red hard-blocks). +- Else `stability_score = Σ(weight × dim_subscore)` over SCORE dims that ran (flakiness .30 / performance .30 / resource .20 / visual .20, renormalized over present dims). **STABLE iff ≥ 95** (`REG_THRESHOLD`/weights overridable). +- Print the **score math** (per-dim contribution table) + declare **dims-ran vs UNAVAILABLE** — an UNAVAILABLE dimension is always listed, never silently passed. + +Backed by `scripts/score-regression.sh verdict ` relative to the installed Autoresearch skill directory, never the caller's working directory (exit 0 STABLE / 1 UNSTABLE). + +## Hunter (root cause) + +On a confirmed HARD regression, auto-engage. Bisect (reuse `debug`) ONLY when the failing case passes a **3/3 reproducibility gate**. SCORE / non-deterministic regressions → differential root-cause + optional `--reason` / `--predict`, no bisect. Non-reproducible → "manual triage" finding. + +## --fix Re-gate + +`--fix` repairs blocking regressions, max **3 cycles** (`--fix-cycles N`). Each cycle MUST strictly shrink the blocking-set else STOP "fix not converging". Intermediate re-gate scopes to failing+touched dims; the final cycle runs the full battery. No HARD-gate bypass. + +## Output + +`autoresearch/regression-{YYMMDD}-{HHMM}/` → `regression-results.tsv`, `stability-report.md`, `dimensions/.md`, `baseline/`, `evals-summary.md` (if `--evals`), `handoff.json`. + +TSV header: `# metric_direction: higher_is_better` then +`iteration\ttimestamp\tdimension\taxis\ttier\tclassification\tbaseline\tcandidate\tdelta\tregressed\tsubscore\tseverity\tstatus\tfile_line\tdescription`. + +## Eval Checkpoint (--evals flag) + +Interval = floor(max_runs / 3), min 1 (fixed 10 if unbounded); override `--evals-interval N`. Every interval, analyze the results TSV; print trend (up/flat/down) + one-line recommendation. Plateau 3+ checkpoints → recommend early stop. At end → full summary to `evals-summary.md`. + +## Chain Handoff + +Write `handoff.json` to the output directory: version "2.1.0", source "regression", timestamp, +`status` ∈ family enum {COMPLETE, CONVERGED, SATURATED, BOUNDED, USER_INTERRUPT, ERROR} (backward-compat with evals/ship consumers), +`verdict` ∈ {STABLE, UNSTABLE, BASELINE_UNAVAILABLE} + `regression_state` ∈ {REGRESSION_FOUND, REGRESSION_FIXED, none} — `ship` reads `verdict` for the deploy-gate, +`results_tsv` path, `findings` = blocking regressions (dim, severity, file_line, classification), `config`{base, scope, dims, axes, verdict-math}. + +If `--fix` → chain to fix automatically. Invoke next `--chain` target in order; propagate `--evals`. Canonical combo: `--predict --evals --fix --ship` = predict → gate → (hunter on HARD) → fix(≤3) → re-gate → ship iff STABLE (deploy still needs explicit approval). + +## Safety + +Verify-command screen (no `rm -rf` / `curl|sh`); worktree cleanup + prune on crash; data-migration refuses any non-allowlisted DB URL; probe auto-skips non-interactively; chained `ship` never auto-deploys. diff --git a/pi-extension/prompts/autoresearch_scenario.md b/pi-extension/prompts/autoresearch_scenario.md new file mode 100644 index 00000000..6569bbe7 --- /dev/null +++ b/pi-extension/prompts/autoresearch_scenario.md @@ -0,0 +1,106 @@ +--- +name: autoresearch_scenario +description: "Generate edge cases across 12 dimensions from a seed scenario" +argument-hint: "[Scenario: ] [Domain: ] [Scope: ] [Iterations: N] [--depth ] [--focus ] [--evals]" +--- + +EXECUTE IMMEDIATELY. + +## Parse Arguments + +Extract from $ARGUMENTS: +- `Scenario:` — seed scenario description (or full $ARGUMENTS text if no keyword) +- `Domain:` or `--domain` — web, mobile, API, CLI, data pipeline, infrastructure +- `Scope:` or `--scope` — file globs for codebase context +- `Focus:` or `--focus` — specific dimension to prioritize +- `--depth` — shallow (10), standard (20), deep (40+) +- `--format` — markdown (default), json, gherkin +- `Iterations:` or `--iterations` — default 20. "unlimited" for unbounded. +- `--evals`, `--evals-interval N`, `--chain`, `--` + +## Setup (if Scenario or Domain missing) + +request_user_input (single batch): + Q1 (Scenario): "Describe the feature/flow to explore" + Q2 (Domain): "What domain?" — web app, mobile app, API, CLI, data pipeline, infrastructure + Q3 (Scope): "Which files for context?" — suggested globs + entire codebase + Q4 (Depth): "How deep?" — quick (10), standard (20), deep (40+), unlimited +If all provided → skip. + +## 12 Dimensions + +| # | Dimension | Explores | +|---|---|---| +| 1 | Happy path | Normal successful flows | +| 2 | Validation | Input boundaries, types, formats | +| 3 | Permissions | Auth, roles, access control | +| 4 | Concurrency | Race conditions, deadlocks, ordering | +| 5 | State | Invalid transitions, corruption | +| 6 | Scale | High volume, large data, many users | +| 7 | Failure | Network errors, timeouts, partial failures | +| 8 | Security | Injection, abuse, bypass attempts | +| 9 | Integration | Third-party failures, API contract violations | +| 10 | Data | Null, empty, unicode, injection, overflow | +| 11 | UX | Confusion, misuse, accessibility | +| 12 | Recovery | Retry, rollback, idempotency | + +## Establish Baseline + +1. Read seed scenario + codebase context +2. Create output directory: `autoresearch/scenario-{YYMMDD}-{HHMM}/` +3. TSV header: `iteration\ttimestamp\tscenario\tdimension\tclassification\tseverity\tdescription` +4. No metric_direction comment (exploration, not optimization) + +## Iteration Loop + +### Phase 1: Review +- Read results TSV, check dimension coverage +- Identify underexplored dimensions +- If --focus → prioritize that dimension + +### Phase 2: Generate +- Pick next dimension (round-robin, or priority if --focus) +- Generate 3-5 specific scenarios for this dimension +- Each: title, dimension, classification, severity, description + +### Phase 3: Classify +- **new** — genuinely novel edge case +- **extension** — builds on previously found scenario +- **duplicate** — already covered (skip, don't log) + +### Phase 4: Log +Append new/extension scenarios to TSV. Skip duplicates. +Severity: critical/high/medium/low. + +### Phase 5: Saturation Check +If 3 consecutive iterations produce only duplicates → dimension saturated, move to next. +If ALL dimensions saturated → early stop. + +### Eval Checkpoint +If --evals: check if current_iteration % interval == 0 → run checkpoint. + +### Bounded Check +If bounded: current_iteration >= max_iterations → exit loop. + +## Output + +- Write `scenarios.md` (organized by dimension, severity-ranked within each) +- Write `edge-cases.md` (flat severity-ranked list) +- `scenario-results.tsv` + +## Summary + +Print: total scenarios (new/extension/duplicate), dimension coverage (X/12 explored), severity distribution. + +## Eval Checkpoint (--evals flag) + +If --evals present: +- Compute interval: floor(max_iterations / 3), min 1. Fixed 10 if unbounded. +- Print: `--- Eval Checkpoint (iterations {X}-{Y}) ---\nNew scenarios: {n} | Dimensions covered: {x}/12 | Saturation: {status}\n{recommendation}\n---` +- If 3+ checkpoints with mostly duplicates → recommend early stop. +- At loop end → full evals summary to evals-summary.md. + +## Chain Handoff + +After completion, write handoff.json: version "2.1.0", source "scenario", timestamp, status, results_tsv path, findings = scenarios by severity, config{scenario, domain, scope}. +Invoke next target in --chain order. Propagate --evals flag. diff --git a/pi-extension/prompts/autoresearch_security.md b/pi-extension/prompts/autoresearch_security.md new file mode 100644 index 00000000..625be5c0 --- /dev/null +++ b/pi-extension/prompts/autoresearch_security.md @@ -0,0 +1,101 @@ +--- +name: autoresearch_security +description: "STRIDE + OWASP security audit with red-team adversarial personas" +argument-hint: "[Scope: ] [Focus: ] [Iterations: N] [--diff] [--fix] [--fail-on ] [--evals]" +--- + +EXECUTE IMMEDIATELY. + +## Parse Arguments + +Extract from $ARGUMENTS: +- `Scope:` or `--scope` — file globs to audit +- `Focus:` — specific area (auth, API, data handling, etc.) +- `Depth:` or `--depth` — quick (5 iterations), standard (15), deep (30+) +- `Iterations:` or `--iterations` — default 15. "unlimited" for unbounded. +- `--diff` — delta mode: only audit files changed since last audit +- `--fix` — after audit, auto-fix Critical/High findings (chains to fix) +- `--fail-on ` — exit non-zero if findings at/above threshold (CI gate) +- `--evals`, `--evals-interval N`, `--chain`, `--` + +## Setup (if required context missing) + +If Scope missing and no --diff: +1. Scan codebase for tech stack, frameworks, API routes +2. request_user_input (single batch): + Q1 (Scope): "What to audit?" — entire codebase, API + middleware, auth, external-facing + Q2 (Depth): "How thorough?" — quick (5), standard (15), deep (30+), unlimited + Q3 (Action): "What to do with findings?" — report only, report + auto-fix, report + CI gate +If all provided → skip. + +## Setup Phase (once, before loop) + +1. **Reconnaissance** — scan: package.json/requirements.txt (deps), .env.example (secrets), Dockerfile (infra), API route files (attack surface), auth/middleware (trust boundaries), DB schemas (data assets), CI/CD configs (supply chain) +2. **Asset Identification** — catalog data stores, auth systems, external services, user inputs +3. **Trust Boundary Mapping** — browser↔server, public↔authenticated, user↔admin, CI↔prod +4. **STRIDE Threat Model** — generate threats per category. Load `references/security-checklist.md` for checklist. +5. **Attack Surface Map** — entry points, data flows, abuse paths +6. **Baseline** — count known issues, initialize coverage tracking + +Create output directory: `autoresearch/security-{YYMMDD}-{HHMM}/` +Write: overview.md, threat-model.md, attack-surface-map.md +TSV header: `# metric_direction: higher_is_better\niteration\ttimestamp\tfinding\tseverity\towasp\tstride\tevidence\tfile_line` + +## Iteration Loop + +### Phase 1: Review +- Read results TSV + coverage tracking +- Identify untested attack vectors from threat model +- Prioritize: untested OWASP categories → untested STRIDE → depth on existing + +### Phase 2: Attack +- Adopt red-team persona for this vector (rotate: Security Adversary, Supply Chain, Insider Threat, Infra Attacker) +- Deep-dive into relevant code with adversarial mindset +- Look for: code paths, input handling, auth checks, data flows + +### Phase 3: Validate +- Construct proof: file:line + specific attack scenario +- Every finding MUST have code evidence — no theoretical fluff +- Classify severity: Critical/High/Medium/Low/Info +- Map to OWASP (A01-A10) and STRIDE (S/T/R/I/D/E) + +### Phase 4: Log +- Append finding to TSV +- Update coverage tracking +- Print coverage every 5 iterations: + `OWASP: [A01✓ A02✓ A03✗ ...] X/10 | STRIDE: [S✓ T✓ R✗ ...] Y/6 | Score: Z` + +### Composite Metric +`score = (owasp_tested/10)*50 + (stride_tested/6)*30 + min(findings, 20)` + +### Eval Checkpoint +If --evals: check if current_iteration % interval == 0 → run checkpoint. + +### Bounded Check +If bounded: current_iteration >= max_iterations → exit loop. + +## After Loop + +1. Write `findings.md` (severity-ranked) +2. Write `owasp-coverage.md` +3. Write `recommendations.md` +4. If `--fix` → chain to fix with Critical/High findings +5. If `--fail-on` → check findings against threshold, exit non-zero if exceeded + +## Summary + +Print: total findings by severity, OWASP coverage X/10, STRIDE coverage Y/6, composite score. + +## Eval Checkpoint (--evals flag) + +If --evals present: +- Compute interval: floor(max_iterations / 3), min 1. Fixed 10 if unbounded. Override: --evals-interval N. +- Every {interval} iterations, analyze results TSV. +- Print: `--- Eval Checkpoint (iterations {X}-{Y}) ---\nScore: {start} → {end} | New findings: {n} | Coverage: OWASP {x}/10, STRIDE {y}/6\n{recommendation}\n---` +- If no new findings 3+ checkpoints → recommend early stop. +- At loop end → full evals summary to evals-summary.md. + +## Chain Handoff + +After completion, write handoff.json to output directory: version "2.1.0", source "security", timestamp, status (COMPLETE|USER_INTERRUPT|BOUNDED|ERROR), results_tsv path, findings = all findings with severity + OWASP + STRIDE + file:line, config{scope, focus, depth}. +Invoke next target in --chain order. Propagate --evals flag. diff --git a/pi-extension/prompts/autoresearch_ship.md b/pi-extension/prompts/autoresearch_ship.md new file mode 100644 index 00000000..78d35594 --- /dev/null +++ b/pi-extension/prompts/autoresearch_ship.md @@ -0,0 +1,120 @@ +--- +name: autoresearch_ship +description: "Ship anything through 8 phases: checklist, dry-run, deploy, verify" +argument-hint: "[Target: ] [--type ] [--dry-run] [--auto] [--force] [--rollback] [--checklist-only] [--monitor N]" +--- + +EXECUTE IMMEDIATELY. + +## Parse Arguments + +Extract from $ARGUMENTS: +- `Target:` or `--target` — what to ship (path, PR, artifact, deployment) +- `--type ` — override auto-detection: code-pr, code-release, deployment, content, docs, package, config +- `--dry-run` — validate everything but don't ship +- `--auto` — auto-approve if no errors found +- `--force` — skip non-critical items (blockers still enforced) +- `--rollback` — undo last ship action +- `--monitor N` — post-ship monitoring for N minutes +- `--checklist-only` — only generate checklist, don't execute +- `--chain`, `--` + +Remaining text = description of what to ship. + +## Setup (if Target or Type unclear) + +1. Auto-detect ship type from context: + - Has uncommitted changes or PR → code-pr + - Has version bump / changelog → code-release + - Has Dockerfile / deploy config → deployment + - Has markdown / content files → content + - Has package.json version change → package +2. If still unclear → request_user_input (single batch): + Q1 (What): "What are you shipping?" — code PR, release, deployment, content, docs, package + Q2 (Target): "Specific target?" — current branch, specific PR, specific path + Q3 (Mode): "How to ship?" — full workflow, dry-run only, checklist only +If all clear → skip. + +## Phase 1: Identify + +- Determine ship type (auto-detected or --type override) +- Identify target artifact(s) +- Map to domain-specific checklist + +## Phase 2: Inventory + +Gather everything that will be shipped: +- Files changed (git diff) +- Dependencies affected +- Config changes +- Migration files +- Breaking changes + +## Phase 3: Checklist + +Generate domain-specific checklist: + +**Code PR:** tests pass, types check, lint clean, no secrets, PR description, reviewers assigned +**Release:** version bumped, changelog updated, migration tested, rollback plan +**Deployment:** env vars set, health checks configured, rollback ready, monitoring active +**Content:** links valid, images optimized, SEO metadata, spell check +**Package:** version bumped, README updated, breaking changes documented, CI green + +If `--checklist-only` → output checklist and stop. + +## Phase 4: Prepare + +Execute pre-ship tasks: +- Run test suite +- Run type checker +- Run linter +- Check for secrets in diff +- Validate configs +- Flag blockers (must-fix) vs warnings (can-ship-with) + +If blockers found → STOP, report blockers, ask user to fix. + +## Phase 5: Dry-Run + +If `--dry-run` or always before actual ship: +- Simulate the ship action without executing +- Report what WOULD happen +- If `--dry-run` → stop here + +## Phase 6: Ship + +**REQUIRES EXPLICIT USER APPROVAL** (unless --auto with zero errors). + +Execute the ship action: +- Code PR: create/update PR, request reviewers +- Release: tag, build, publish +- Deployment: deploy to target environment +- Content: publish to CMS/platform + +## Phase 7: Verify + +Post-ship verification: +- Confirm artifact is live/accessible +- Run smoke tests if available +- Check monitoring for errors +- If `--monitor N` → watch for N minutes + +## Phase 8: Log + +Create output directory: `autoresearch/ship-{YYMMDD}-{HHMM}/` +Write: +- `checklist.md` — completed checklist with pass/fail per item +- `summary.md` — what was shipped, verification results +- `ship-log.tsv` — phase-by-phase log + +## Rollback + +If `--rollback`: +- Identify last ship action from most recent ship log +- Reverse it (revert PR, unpublish, rollback deployment) +- Verify rollback succeeded + +## Chain Handoff + +Write handoff.json: version "2.1.0", source "ship", timestamp, status (COMPLETE|DRY_RUN|ROLLBACK|ERROR), findings = blockers/warnings found during prep. +Invoke next target in --chain order. diff --git a/pi-extension/skills/autoresearch/SKILL.md b/pi-extension/skills/autoresearch/SKILL.md new file mode 100644 index 00000000..22f2056a --- /dev/null +++ b/pi-extension/skills/autoresearch/SKILL.md @@ -0,0 +1,107 @@ +--- +name: autoresearch +description: "Autonomous iteration loop: modify, verify, keep/discard against any metric" +version: 2.2.2 +--- + +# Autoresearch — Autonomous Goal-directed Iteration + +## Safety Invariants (all subcommands) +- Never push, publish, or deploy without explicit user approval. +- Bounded by default. Override with `Iterations: unlimited`. +- All results logged to `autoresearch/{subcommand}-{YYMMDD}-{HHMM}/` directory. +- Chain handoff via `handoff.json`. Evals reads `*-results.tsv`. + +## Dispatch (bare `$autoresearch`) + +Parse the invocation in this order: + +| Condition | Mode | +|---|---| +| `Metric:` or `Verify:` present | **Classic** — existing metric loop, unchanged | +| Free-form natural-language goal, no metric/verify | **Orchestrator** — see Orchestrator section | +| Nothing | **Setup wizard** — interactive config builder | +| `--classic` flag | Force Classic regardless of goal text | +| `--auto` flag | Force Orchestrator regardless of goal text | + +Print a banner on every invocation: `[autoresearch] mode: classic | orchestrator | wizard`. + +## Subcommands + +| Command | Does | Default Iterations | +|---|---|---| +| `$autoresearch` | Iterate against a metric: modify → verify → keep/discard | 25 | +| `$autoresearch plan` | Convert a goal into validated Scope, Metric, Verify config | N/A | +| `$autoresearch debug` | Hunt bugs: hypothesize → test → falsify → repeat | 15 | +| `$autoresearch fix` | Crush errors one-by-one until zero remain | 20 | +| `$autoresearch security` | STRIDE + OWASP audit with red-team personas | 15 | +| `$autoresearch ship` | Ship through 8 phases: checklist → dry-run → deploy → verify | N/A | +| `$autoresearch scenario` | Generate edge cases across 12 dimensions | 20 | +| `$autoresearch predict` | 5 expert personas debate before implementation | N/A | +| `$autoresearch learn` | Scout codebase → generate docs or wiki → validate → fix loop | 10 | +| `$autoresearch reason` | Adversarial debate with blind judges until convergence | 8 | +| `$autoresearch probe` | 8 personas interrogate requirements until saturation | 15 | +| `$autoresearch improve` | Research ICP challenges, discover improvements, generate PRDs | 15 | +| `$autoresearch evals` | Analyze iteration results: trends, plateaus, regressions | N/A | +| `$autoresearch regression` | Regression stability gate: baseline vs candidate, verdict STABLE/UNSTABLE | N/A | + +## Universal Flags + +| Flag | Applies To | Purpose | +|---|---|---| +| `Iterations: N` | All looping | Set iteration count | +| `Iterations: unlimited` | All looping | Opt-in unbounded | +| `--evals` | All looping | Mid-loop checkpoints + final summary | +| `--evals-interval N` | All looping | Override checkpoint frequency | +| `--chain ` | All | Sequential handoff after completion | +| `--` | All | Shorthand for `--chain ` | +| `--dry-run` | Orchestrator | Print derived config + planned pipeline; no execution | +| `--max-cycles N` | Orchestrator | Hard ceiling on orchestration cycles (default 50) | +| `--classic` | Bare `$autoresearch` | Force Classic metric-loop mode | +| `--auto` | Bare `$autoresearch` | Force Orchestrator mode | + +## Orchestrator + +Activated when a plain-language goal is given without `Metric:`/`Verify:`. Classifies the goal into a **Goal archetype** — see `references/orchestrator-routing.md` for the archetype table and router decision table. + +Resolve every `scripts/...` path below relative to this installed skill directory, never relative to the caller's working directory. + +**Two modes based on archetype:** +- **Orchestration loop** — predicate-bearing archetypes (ship-ready, optimize-metric, fix-broken, harden, build-feature, explore). Goal has a mechanical Success predicate; the loop runs until that predicate is met. +- **Single-pass dispatch** — subjective/terminal archetypes (document, what-to-build, decide-design). Routes once to the fitting subcommand (learn / improve / reason), lets it self-terminate, then reports. No loop, no Plateau, no ship gate. + +### Orchestration Loop Steps + +Backed by `scripts/orchestrate.sh` (deterministic seam — all routing logic lives there). Subcommands exposed: `classify`, `next-hop`, `units`, `plateau`, `screen-cmd`, `verdict`, `validate-state`, `screen-state-predicate`. + +1. **Classify** — `scripts/orchestrate.sh classify ""` → archetype label + mode. +2. **Derive predicate** — reuse `plan` logic to produce a concrete Success predicate: exact shell command + expected output. For `optimize-metric`, run the full plan/wizard derivation internally. +3. **Confirm** — ONE `request_user_input` showing: archetype, mode, concrete predicate (command + expected output), terminal choice (stop-at-verified vs proceed-to-ship). Misclassifications are caught here, not mid-run. +4. **Round-0 dry-run** — prove the predicate command runs and returns a value; safety-screen every derived command via `screen-cmd`; print projected cycle budget. Stop here if `--dry-run`. +5. **Loop** until predicate satisfied: + a. Assess state via cheap signals (last `handoff.json`, regression verdict, error count) + affected-test verify. + b. `scripts/orchestrate.sh next-hop orchestrator-state.json` → next subcommand. + c. Run subcommand (its own bounded inner loop). + d. Record per-hop outcome ∈ {progressed, no-op, failed, blocked}. + e. Fold hop's `handoff.json` into `orchestrator-state.json`. + f. `scripts/orchestrate.sh units` → recompute **Units remaining**. +6. **Stop conditions** (checked after each hop): + - Predicate met → ship gate (only if ship is in the pipeline) else `CONVERGED`. + - `scripts/orchestrate.sh plateau orchestrator-state.json` → true → stop + report `PLATEAU`. + - Cycles > ceiling (default 50, override `--max-cycles N`) → stop + report `CEILING`. + - Hop outcome `blocked`/`failed` with no alternative route → checkpoint + stop + report `BLOCKED`. + +### Orchestrator State + +`orchestrator-state.json` — orchestrator-owned, additive. Tracks: goal, archetype, predicate, terminal-choice, `units_remaining` history, cycle count, per-hop pipeline log with outcomes, current incumbent. Each hop's `handoff.json` is unchanged (single-hop bridge); the orchestrator reads it and folds it in. Two clearly-owned state objects, no overlap. + +### Orchestrator Safety Invariants + +- **Never auto-approve ship/deploy/push.** The orchestrator never passes `--auto` to `ship`; deploy always requires explicit user approval. +- **Data-migration behind anchored DB-URL allowlist.** Reuses regression's allowlist — host must be `localhost`/`127.0.0.1`/container hostname, or database name carries `_test`/`_ci` suffix. Bare substring match does not qualify. Anything else refused. +- **screen-cmd on every derived command** — run before the loop starts AND on every command read from a persisted state file on resume. Persisted commands are never trusted; resume re-screens the pinned predicate via `screen-state-predicate` and refuses on `refuse`. +- **No un-screened commands mid-loop.** The autonomous loop cannot introduce new shell commands that bypass `screen-cmd`. +- **Predicate pinned, not re-derived.** Round-0 writes the derived Success predicate verbatim into `orchestrator-state.json`; every cycle and every resume reuses that exact string so "done" is reproducible across runs. +- **Validate the ledger before routing.** `validate-state` gates `orchestrator-state.json` (required fields + coarse types); a malformed ledger is not trusted to route from. +- **Independent verify before convergence.** High-impact changes accepted on the working signal set `pending_verify`; `next-hop` routes to a `verify` hop (held-out / adversarial check) before `DONE` or ship. The verify hop never auto-approves ship. +- **Unknown-units cycles excluded from Plateau counter.** A cycle where `units` returns `unknown` (e.g. runner crash) is not counted as zero-progress; repeated `unknown` routes to `BLOCKED`. diff --git a/pi-extension/skills/autoresearch/references/orchestrator-routing.md b/pi-extension/skills/autoresearch/references/orchestrator-routing.md new file mode 100644 index 00000000..aee04d5f --- /dev/null +++ b/pi-extension/skills/autoresearch/references/orchestrator-routing.md @@ -0,0 +1,89 @@ +# Orchestrator Routing + +## Goal Archetypes + +| Archetype | Trigger Keywords | Mode | Preset Pipeline | +|---|---|---|---| +| `ship-ready` | ship, release, deploy, publish, production-ready, merge | loop | probe, debug, fix, regression, ship | +| `optimize-metric` | improve, optimize, increase, reduce, faster, smaller, coverage, score | loop | plan, (classic loop), evals | +| `fix-broken` | fix, broken, failing, error, crash, bug, can't run, tests fail | loop | debug, fix, regression | +| `harden` | security, vulnerability, audit, OWASP, CVE, harden, lock down | loop | security, fix, security | +| `build-feature` | build, add, implement, create, new feature, acceptance test | loop | (acceptance-test derive), debug, fix, regression | +| `explore` | understand, explore, investigate, what does, how does, edge cases | loop | probe, scenario, plan | +| `document` | document, wiki, generate docs, explain codebase, write guide | dispatch | learn | +| `what-to-build` | what should I build, ideas, improvements, PRD, roadmap | dispatch | improve | +| `decide-design` | which approach, compare options, design decision, architecture choice | dispatch | reason | + +Keyword matching is fuzzy — partial matches and synonyms qualify. When a goal matches multiple archetypes, prefer the more specific one (fix-broken over explore; ship-ready over fix-broken if "ship" is explicit). When ambiguous, show the top two candidates in the upfront confirm and let the user choose. + +## Router Decision Table + +The `next-hop` subcommand of `scripts/orchestrate.sh` reads `orchestrator-state.json` and applies these rules in order. First match wins. + +| State Signal | Source | Next Hop | +|---|---|---| +| `errors > 0` in last handoff | handoff.json `findings` | `fix` | +| regression verdict `UNSTABLE` | handoff.json `verdict` | `regression` | +| `untested_gaps` flagged | handoff.json or units output | `debug` | +| `pending_verify` true | orchestrator-state.json | `verify` (fresh independent acceptance check) | +| predicate met | Success predicate command exit/output | `DONE` (exit loop) | +| hop outcome `blocked` or `failed`, no retry route | orchestrator-state.json | `BLOCKED` (checkpoint + stop) | +| plateau detected | `scripts/orchestrate.sh plateau` | `PLATEAU` (stop + report) | +| archetype pipeline has remaining steps | preset pipeline sequence | next preset step | +| all preset steps exhausted, predicate not met | — | `regression` (convergence re-check) | + +State signals are cheap reads — last `handoff.json` plus the regression verdict field and error count. No re-run of the full suite just to route. + +## Independent Verify & Overfit Guard + +The orchestrator must not optimize and accept against the same signal — that lets a +change game its own metric. For `optimize-metric` and `build-feature`, the acceptance +check runs on a **held-out** set (a fresh scenario set or holdout assertions), separate +from the `units` signal used to choose the change. When a high-impact change is accepted +on the working signal, the orchestrator sets `pending_verify` in `orchestrator-state.json`; +`next-hop` then routes to a **verify** hop (dispatched to `reason` or `predict` as an +independent adversarial check) before declaring `DONE` or shipping. The verify hop is +advisory input to convergence — it never auto-approves ship, which stays human-gated. + +## Two-Mode Split + +**Orchestration loop** — used when the goal has an external, mechanical Success predicate: a shell command that returns a value the orchestrator can compare across cycles. Progress is objective (Units remaining falls), plateau is well-defined, and the loop terminates on convergence or a safety backstop. Archetypes: ship-ready, optimize-metric, fix-broken, harden, build-feature, explore. + +**Single-pass dispatch** — used when no mechanical predicate exists. The goal is subjective or the subcommand is internally-converging (reason runs its own adversarial loop) or a one-shot terminal emitter (learn, improve produce a document and stop). The orchestrator routes once, the subcommand self-terminates, and the orchestrator reports the result. No Units remaining, no Plateau counter, no ship gate. Archetypes: document, what-to-build, decide-design. + +The criterion is: "Can the orchestrator independently verify done without re-running the subcommand?" If yes → loop. If no → dispatch. + +## Build-Feature: TDD Ladder + +The `build-feature` archetype has no pre-existing metric, so progress is reframed as `green-assertion-count` (monotone integer, higher-is-better). A change that turns a red sub-test green is kept; a change that regresses a green sub-test is reverted. A floor-guard prevents reverting scaffolding commits that compile and add no new failures but pass zero new tests. Large net-new scope (greenfield with no existing test suite) is detected and the orchestrator advises handing off to a dedicated build command rather than grinding cycles. + +## Preset Pipelines (Reference) + +| Archetype | Step 1 | Step 2 | Step 3 | Step 4 | Step 5 | +|---|---|---|---|---|---| +| ship-ready | probe | debug | fix | regression | ship | +| optimize-metric | plan | (classic loop) | holdout-verify | evals | — | +| fix-broken | debug | fix | regression | — | — | +| harden | security | fix | security | — | — | +| build-feature | (acceptance-test derive) | debug | fix | regression | — | +| explore | probe | scenario | plan | — | — | +| document | learn | — | — | — | — | +| what-to-build | improve | — | — | — | — | +| decide-design | reason | — | — | — | — | + +Presets are starting pipelines. The router adapts per cycle from observed state — it may skip, repeat, or reorder steps based on the decision table above. The preset is a prior, not a fixed schedule. + +## Glossary + +Terms used consistently across this file, SKILL.md, and orchestrator-state.json. Definitions live in CONTEXT.md. + +| Term | Short meaning | +|---|---| +| Goal archetype | Classification of the user's natural-language goal into one of the 9 categories above | +| Success predicate | Exact shell command + expected output that defines "done" for Orchestration loop goals | +| Units remaining | Scalar measure of open gaps (failing tests, errors, metric delta); lower-is-better; computed by `scripts/orchestrate.sh units` | +| Plateau | Units remaining flat or worse for N consecutive computed cycles (default 5); oscillation that nets zero also qualifies | +| Orchestration loop | The cycle-bounded assess→route→run→record loop used for predicate-bearing archetypes | +| Single-pass dispatch | One-shot routing to a self-terminating subcommand; no loop, Plateau, ceiling, or ship gate | +| Independent verify hop | A `verify` routing step (reason/predict) that checks an accepted high-impact change against a fresh signal before DONE/ship; gated by `pending_verify` | +| Holdout-verify | Acceptance check run on a held-out set, separate from the `units` signal used to choose the change, to prevent overfitting the metric | diff --git a/pi-extension/skills/autoresearch/references/predict-personas.md b/pi-extension/skills/autoresearch/references/predict-personas.md new file mode 100644 index 00000000..383e2379 --- /dev/null +++ b/pi-extension/skills/autoresearch/references/predict-personas.md @@ -0,0 +1,65 @@ +# Predict Personas + +## Default Persona Set (5 personas) + +### 1. Software Architect +- **Focus:** System design, component boundaries, data flow, scalability +- **Questions:** Does this design scale? Are boundaries clean? Is coupling minimized? Will this survive 10x growth? +- **Evidence required:** file:line citations, dependency graphs, coupling metrics +- **Red flags:** God classes, circular dependencies, leaky abstractions, shared mutable state + +### 2. Security Analyst +- **Focus:** Attack surfaces, auth/authz, data protection, injection vectors +- **Questions:** Can this be exploited? Are trust boundaries enforced? Is data sanitized? Are secrets protected? +- **Evidence required:** file:line citations, attack scenarios, data flow through trust boundaries +- **Red flags:** Raw SQL, missing authz, hardcoded secrets, unsanitized user input + +### 3. Performance Engineer +- **Focus:** Latency, throughput, resource usage, algorithmic complexity +- **Questions:** Will this be fast enough? What's the worst case? Where are the bottlenecks? Is caching effective? +- **Evidence required:** file:line citations, complexity analysis, resource estimates +- **Red flags:** N+1 queries, unbounded loops, missing indexes, synchronous I/O in hot paths + +### 4. Reliability Engineer +- **Focus:** Error handling, failure modes, observability, recovery +- **Questions:** What happens when this fails? Can we detect it? Can we recover? Is it observable? +- **Evidence required:** file:line citations, failure scenarios, recovery paths +- **Red flags:** Swallowed errors, missing retries, no circuit breakers, silent failures + +### 5. Devil's Advocate +- **Focus:** Assumptions, edge cases, hidden complexity, maintainability +- **Questions:** What assumptions are wrong? What's the simplest thing that breaks this? Is this over-engineered? +- **Evidence required:** Concrete counter-examples, edge case scenarios +- **Red flags:** Happy-path-only design, untested assumptions, complexity without justification + +## Adversarial Persona Set (activated with --adversarial) + +Replace default personas with hostile reviewers: +1. **The Breaker** — tries to crash/corrupt the system +2. **The Cheater** — finds ways to bypass rules and abuse features +3. **The Scaler** — imagines 1000x load and finds what breaks +4. **The Newbie** — misuses every API and expects it to work +5. **The Malicious Insider** — has credentials, wants to exfiltrate + +## Debate Protocol + +1. Each persona analyzes independently (no shared context between personas) +2. Findings reported with confidence score (0-100%) +3. Cross-examination: personas challenge each other's findings +4. Synthesizer aggregates, removes duplicates, resolves conflicts +5. Anti-herd check: if all personas agree, synthesizer must find at least 1 counter-argument +6. Final consensus: ranked findings with persona attribution + +## Output Format + +Each persona produces: +``` +### [Persona Name] — [N findings] +| # | Finding | Severity | Confidence | File:Line | Recommendation | +``` + +Synthesizer produces: +``` +### Consensus — [N findings after dedup] +| # | Finding | Severity | Agreement | Source Personas | Action | +``` diff --git a/pi-extension/skills/autoresearch/references/reason-judge-protocol.md b/pi-extension/skills/autoresearch/references/reason-judge-protocol.md new file mode 100644 index 00000000..7f5e7061 --- /dev/null +++ b/pi-extension/skills/autoresearch/references/reason-judge-protocol.md @@ -0,0 +1,87 @@ +# Reason Judge Protocol + +## Adversarial Refinement Loop + +``` +Round N: + 1. Author-A generates candidate (or incumbent from previous round) + 2. Critic attacks candidate — MUST find weaknesses (forced adversarial) + 3. Author-B reads task + candidate-A + critique → produces candidate-B + 4. Synthesizer reads A + B → produces hybrid candidate-AB + 5. Judge panel receives 3 candidates with randomized labels → picks winner + 6. Winner becomes incumbent for round N+1 +``` + +## Agent Isolation Rules + +- Each agent (Author-A, Critic, Author-B, Synthesizer, Judges) runs COLD START +- No shared session state between agents — prevents sycophancy +- Agents receive ONLY: task description + relevant candidate(s) + critique +- Judges receive candidates with randomized labels (Label-X, Label-Y, Label-Z) +- Judges MUST compare and rank — "all are good" is not a valid verdict + +## Critic Protocol + +The critic MUST: +1. Identify at least 3 specific weaknesses in the candidate +2. Provide concrete evidence for each weakness +3. Suggest what a superior candidate would do differently +4. Rate candidate on domain-specific criteria (1-10 scale) +5. Never compliment the candidate — role is purely adversarial + +## Judge Protocol + +Each judge receives: +- Task description (identical for all judges) +- 3 candidates with randomized labels (Label-X, Label-Y, Label-Z) +- Evaluation criteria relevant to the domain + +Each judge MUST: +1. Evaluate each candidate independently on all criteria +2. Produce a ranking (1st, 2nd, 3rd) with reasoning +3. Select a winner with one-paragraph justification +4. Label randomization prevents position bias + +Verdict: majority vote. Tie → synthesized candidate (Label-Z) wins. + +## Convergence Detection + +| Mode | Stop Condition | +|---|---| +| Convergent (default) | Same incumbent wins N consecutive rounds (default N=3) | +| Creative | Never auto-stops; runs until iteration limit | +| Debate | Same as convergent but no synthesis step | + +## Oscillation Guard + +If the incumbent changes more than 5 times in the last 8 rounds → recommend early stop. The candidates are not converging — further rounds waste context. + +## Domain-Specific Judge Criteria + +| Domain | Criteria | +|---|---| +| Software architecture | Scalability, maintainability, performance, security, simplicity | +| Product strategy | Market fit, feasibility, differentiation, risk, timeline | +| Business decision | ROI, risk, alignment, resource requirements, reversibility | +| Security approach | Coverage, false positive rate, practicality, compliance | +| Research hypothesis | Testability, novelty, evidence support, explanatory power | +| Content/writing | Clarity, accuracy, engagement, completeness, actionability | + +## Output Files + +| File | Content | +|---|---| +| `reason-results.tsv` | Per-round: round, candidate_label, judge_verdict, convergence_count, description | +| `lineage.md` | Full history of all candidates + critiques + judge reasoning | +| `summary.md` | Final winner, convergence trajectory, key insights | +| `handoff.json` | Chain handoff with winner as primary finding | + +## TSV Schema + +``` +round timestamp candidate_label judge_verdict convergence_count description +1 2026-05-19T00:00:00Z Candidate-A winner 1 Event sourcing with CQRS +2 2026-05-19T00:05:00Z Candidate-AB winner 1 Hybrid: event sourcing for writes, read projections +3 2026-05-19T00:10:00Z Candidate-AB winner 2 Refined hybrid with materialized views +4 2026-05-19T00:15:00Z Candidate-AB winner 3 CONVERGED — same approach refined +``` diff --git a/pi-extension/skills/autoresearch/references/security-checklist.md b/pi-extension/skills/autoresearch/references/security-checklist.md new file mode 100644 index 00000000..c1dedecf --- /dev/null +++ b/pi-extension/skills/autoresearch/references/security-checklist.md @@ -0,0 +1,76 @@ +# Security Audit Checklist + +## STRIDE Threat Categories + +| Category | Threat | Look For | +|---|---|---| +| Spoofing | Identity impersonation | Weak auth, token prediction, session fixation | +| Tampering | Data modification | Unvalidated input, missing integrity checks, SQL injection | +| Repudiation | Deniable actions | Missing audit logs, unsigned transactions | +| Info Disclosure | Data leaks | Error messages with stack traces, verbose logging, exposed env vars | +| Denial of Service | Availability attacks | Unbounded queries, missing rate limits, regex DoS | +| Elevation of Privilege | Unauthorized access | Missing authz checks, IDOR, privilege escalation paths | + +## OWASP Top 10 (2021) Checklist + +| # | Category | Key Checks | +|---|---|---| +| A01 | Broken Access Control | IDOR, missing function-level authz, CORS misconfiguration, path traversal | +| A02 | Cryptographic Failures | Plaintext secrets, weak algorithms, missing TLS, hardcoded keys | +| A03 | Injection | SQL, NoSQL, OS command, LDAP, XSS (stored/reflected/DOM) | +| A04 | Insecure Design | Missing threat model, no rate limiting, no abuse prevention | +| A05 | Security Misconfiguration | Default credentials, unnecessary features enabled, missing headers | +| A06 | Vulnerable Components | Known CVEs in dependencies, outdated packages, unmaintained libs | +| A07 | Auth Failures | Credential stuffing, brute force, weak passwords, missing MFA | +| A08 | Data Integrity Failures | Unsigned updates, insecure deserialization, CI/CD poisoning | +| A09 | Logging Failures | Missing security events, insufficient monitoring, no alerting | +| A10 | SSRF | Unvalidated URLs, internal service access, cloud metadata exposure | + +## Red-Team Personas + +| Persona | Focus | Mindset | +|---|---|---| +| Security Adversary | Auth, crypto, injection | External attacker with browser + Burp Suite | +| Supply Chain Attacker | Dependencies, CI/CD, build pipeline | Compromise through third-party code | +| Insider Threat | Data access, privilege abuse, exfiltration | Authenticated user with malicious intent | +| Infrastructure Attacker | Network, cloud config, containers | Target infrastructure misconfigurations | + +## Severity Classification + +| Severity | Criteria | Examples | +|---|---|---| +| Critical | Remote exploitation, no auth required, data breach | RCE, SQL injection, auth bypass | +| High | Requires some access, significant impact | Stored XSS, IDOR, privilege escalation | +| Medium | Limited impact or requires interaction | CSRF, reflected XSS, info disclosure | +| Low | Minimal impact, informational | Missing headers, verbose errors | +| Info | Best practice recommendation | Hardening suggestions, defense in depth | + +## Composite Metric Formula + +``` +score = (owasp_categories_tested / 10) * 50 + + (stride_categories_tested / 6) * 30 + + min(unique_findings, 20) +``` + +Higher is better. Perfect score = 100 (all OWASP tested + all STRIDE tested + 20 findings). + +## Coverage Tracking + +Print coverage summary every 5 iterations: +``` +OWASP: [A01✓ A02✓ A03✗ A04✗ A05✓ A06✗ A07✓ A08✗ A09✗ A10✗] 4/10 +STRIDE: [S✓ T✓ R✗ I✓ D✗ E✗] 3/6 +Score: 48.3 | Findings: 7 +``` + +## Finding Format + +Every finding requires: +1. **Title** — one-line summary +2. **Severity** — Critical/High/Medium/Low/Info +3. **OWASP** — A01-A10 category +4. **STRIDE** — S/T/R/I/D/E category +5. **Evidence** — file:line + attack scenario (no theoretical fluff) +6. **Reproduction** — steps to trigger +7. **Mitigation** — concrete fix recommendation diff --git a/pi-extension/skills/autoresearch/scripts/orchestrate.sh b/pi-extension/skills/autoresearch/scripts/orchestrate.sh new file mode 100755 index 00000000..c414fd5b --- /dev/null +++ b/pi-extension/skills/autoresearch/scripts/orchestrate.sh @@ -0,0 +1,448 @@ +#!/usr/bin/env bash +# orchestrate.sh — deterministic seam for the autoresearch orchestrator loop. +# +# classify → Goal archetype label (keyword heuristics) +# next-hop → Next subcommand from router decision table +# units → Units-remaining scalar (lower_is_better) +# plateau → Exit 0 if last N computed values are flat-or-worse +# screen-cmd → "ok" exit 0 | "refuse" exit 1 safety gate +# verdict → CONVERGED|PLATEAU|CEILING|BLOCKED + ship-gate +# +# All subcommands are pure and CI-usable via exit codes. +set -uo pipefail + +# --------------------------------------------------------------------------- +# classify: map a goal string to one of the 9 Goal archetype labels. +# Priority order matters: higher-stakes archetypes checked first so that +# "fix and add the broken feature" → fix-broken, not build-feature. +# --------------------------------------------------------------------------- +classify() { + local goal="${1:?usage: classify }" + local g + g=$(printf '%s' "$goal" | tr '[:upper:]' '[:lower:]') + + # Security/hardening — above build because "secure" is higher stakes than "add" + if printf '%s' "$g" | grep -qE '(secure|harden|vuln)'; then + echo "harden"; return 0 + fi + + # Broken/bugfix + if printf '%s' "$g" | grep -qE '(fix|bug|broken)'; then + echo "fix-broken"; return 0 + fi + + # Ship/release/deploy + if printf '%s' "$g" | grep -qE '(ship|release|deploy)'; then + echo "ship-ready"; return 0 + fi + + # Product direction — requires a "what …" question so bare "next"/"build" in a + # build-feature goal (e.g. "build the next-gen parser") doesn't mis-route here. + if printf '%s' "$g" | grep -qE '(what.*build|what.*next)'; then + echo "what-to-build"; return 0 + fi + + # Build/implement/add — "feature" alone is insufficient; any of these words qualify + if printf '%s' "$g" | grep -qE '(build|implement|add)'; then + echo "build-feature"; return 0 + fi + + # Metric optimization + if printf '%s' "$g" | grep -qE '(faster|smaller|reduce|optimize|coverage)'; then + echo "optimize-metric"; return 0 + fi + + # Documentation + if printf '%s' "$g" | grep -qE '(document|docs)'; then + echo "document"; return 0 + fi + + # Design decision — "should we" / "decide" / "approach" + if printf '%s' "$g" | grep -qE '(should we|decide|approach)'; then + echo "decide-design"; return 0 + fi + + # Default: open-ended investigation + echo "explore" +} + +# --------------------------------------------------------------------------- +# next-hop: cheap router over fields in a state JSON file. +# Decision order: errors → regression → untested gaps → ship/DONE. +# --------------------------------------------------------------------------- +next-hop() { + local state_file="${1:?usage: next-hop }" + if [[ ! -f "$state_file" ]]; then + echo "ERROR: missing state file" >&2; return 2 + fi + + # Parse with sed/grep — no jq dependency (score-regression.sh doesn't use jq) + local errors regression gaps archetype + errors=$(grep -o '"errors_remaining"[[:space:]]*:[[:space:]]*[0-9]*' "$state_file" \ + | grep -o '[0-9]*$') + regression=$(grep -o '"regression_verdict"[[:space:]]*:[[:space:]]*"[^"]*"' "$state_file" \ + | grep -o '"[^"]*"$' | tr -d '"') + gaps=$(grep -o '"untested_gaps"[[:space:]]*:[[:space:]]*[0-9]*' "$state_file" \ + | grep -o '[0-9]*$') + archetype=$(grep -o '"archetype"[[:space:]]*:[[:space:]]*"[^"]*"' "$state_file" \ + | grep -o '"[^"]*"$' | tr -d '"') + + # Optional: pending_verify gates an independent acceptance check before DONE/ship. + # Absent (or false) → routing is identical to prior behavior. + local pending + pending=$(grep -o '"pending_verify"[[:space:]]*:[[:space:]]*[a-z]*' "$state_file" \ + | grep -o '[a-z]*$') + + # Guard: missing required fields + if [[ -z "$errors" || -z "$regression" || -z "$gaps" ]]; then + echo "ERROR: malformed state file" >&2; return 2 + fi + + if [[ "$errors" -gt 0 ]]; then + echo "fix"; return 0 + fi + + if [[ "$regression" == "UNSTABLE" ]]; then + echo "regression"; return 0 + fi + + if [[ "$gaps" -gt 0 ]]; then + echo "debug"; return 0 + fi + + # Gaps clear but an accepted high-impact change still needs a fresh, independent + # acceptance check (separate from the signal used to choose it) → verify first. + if [[ "$pending" == "true" ]]; then + echo "verify"; return 0 + fi + + # All clear: ship if archetype has ship in the pipeline, else DONE + if [[ "$archetype" == "ship-ready" ]]; then + echo "ship"; return 0 + fi + + echo "DONE" +} + +# --------------------------------------------------------------------------- +# units: compute Units-remaining scalar from a results JSON file. +# Formula: failing_tests + open_hard_regressions + (metric_delta / metric_target) +# Prints "unknown" and exits 2 when inputs are missing or uncomputable. +# --------------------------------------------------------------------------- +units() { + local results_file="${1:?usage: units }" + if [[ ! -f "$results_file" ]]; then + echo "unknown"; return 2 + fi + + local ft regressions delta target + ft=$(grep -o '"failing_tests"[[:space:]]*:[[:space:]]*[0-9.]*' "$results_file" \ + | grep -o '[0-9.]*$') + regressions=$(grep -o '"open_hard_regressions"[[:space:]]*:[[:space:]]*[0-9.]*' "$results_file" \ + | grep -o '[0-9.]*$') + delta=$(grep -o '"metric_delta"[[:space:]]*:[[:space:]]*[0-9.]*' "$results_file" \ + | grep -o '[0-9.]*$') + target=$(grep -o '"metric_target"[[:space:]]*:[[:space:]]*[0-9.]*' "$results_file" \ + | grep -o '[0-9.]*$') + + if [[ -z "$ft" || -z "$regressions" || -z "$delta" || -z "$target" ]]; then + echo "unknown"; return 2 + fi + + # Integer check: metric_target must be non-zero to avoid divide-by-zero + if [[ "$target" == "0" || "$target" == "0.0" ]]; then + echo "unknown"; return 2 + fi + + # awk handles floating point; strip trailing .0 for clean integer output + awk -v ft="$ft" -v r="$regressions" -v d="$delta" -v t="$target" ' + BEGIN { + val = ft + r + (d / t) + # Strip unnecessary trailing zeros (e.g. 4.500 → 4.5, 0.000 → 0) + if (val == int(val)) printf "%d\n", val + else printf "%g\n", val + } + ' +} + +# --------------------------------------------------------------------------- +# plateau: read newline list of unit values; determine if progress has stalled. +# Skips interleaved "unknown" cycles; N=5 consecutive trailing unknowns = BLOCKED. +# Exit 0 = plateau (no net progress); exit 1 = still improving; exit 3 = BLOCKED. +# --------------------------------------------------------------------------- +plateau() { + local history_file="${1:?usage: plateau }" + local n=5 + + if [[ ! -f "$history_file" ]]; then + echo "BLOCKED"; return 3 + fi + + awk -v n="$n" ' + { + line = $0 + gsub(/^[[:space:]]+|[[:space:]]+$/, "", line) + if (line == "unknown") { trailing_unknown++; next } # crash/uncomputable cycle + if (line ~ /^[0-9]/) { vals[++count] = line + 0; trailing_unknown = 0 } + } + END { + # A runner stuck emitting "unknown" must not read as progress: n consecutive + # trailing unknowns (or no computed value at all) → BLOCKED, not "improving". + if (trailing_unknown >= n) { print "BLOCKED"; exit 3 } + if (count == 0) { print "BLOCKED"; exit 3 } + + # Need at least n computed values before a plateau call. + if (count < n) { exit 1 } + + # Net progress over the window = last value strictly below the first + # (lower_is_better). Any oscillation that nets flat-or-worse is a plateau, + # so a thrashing loop stops instead of running to the ceiling. + start = count - n + 1 + if (vals[count] < vals[start]) { exit 1 } # net improvement → still working + exit 0 # flat or worse → plateau + } + ' "$history_file" +} + +# --------------------------------------------------------------------------- +# screen-cmd: safety gate for shell strings before execution. +# Prints "ok" / "refuse". Anchored DB-host allowlist: only localhost, +# 127.0.0.1, or a plain hostname (no dots) with a _test or _ci dbname suffix. +# Bare substring "test" inside words like "latest" or "precision" must NOT qualify. +# --------------------------------------------------------------------------- +screen-cmd() { + local cmd="${1:?usage: screen-cmd }" + + # rm with recursive AND force, in any flag arrangement: bundled (-rf/-Rf/-fr), + # separate (-r -f), or long (--recursive --force). Both flags must be present. + # The optional path prefix catches path-qualified invocations (/bin/rm, ./rm, + # /usr/local/bin/rm) that a bare command-name anchor would miss. + if printf '%s' "$cmd" | grep -qE '(^|[[:space:]])([^[:space:]]*/)?rm([[:space:]]|$)'; then + local rm_rec=0 rm_force=0 + printf '%s' "$cmd" | grep -qE -- '(^|[[:space:]])-[a-zA-Z]*[rR]|--recursive' && rm_rec=1 + printf '%s' "$cmd" | grep -qE -- '(^|[[:space:]])-[a-zA-Z]*[fF]|--force' && rm_force=1 + if [[ "$rm_rec" -eq 1 && "$rm_force" -eq 1 ]]; then + echo "refuse"; return 1 + fi + fi + + # curl/wget piped to an interpreter (sh/bash/zsh/dash/fish/ksh/python/perl/ruby/ + # node/php), including a path-qualified one (| /bin/bash). Enumerated interpreters + # rather than "refuse any curl pipe" so a legitimate derived predicate that pipes + # curl output to a parser (jq/grep/awk) is not falsely refused. + if printf '%s' "$cmd" | grep -qE '(curl|wget)[^|]*\|[[:space:]]*([^[:space:]]*/)?(sh|bash|zsh|dash|fish|ksh|python[0-9.]*|perl|ruby|node|php)([[:space:]]|$)'; then + echo "refuse"; return 1 + fi + + # curl/wget routed through xargs into an interpreter. The xargs wrapper sidesteps the + # direct pipe matcher above, so a remote payload still reaches a shell. + if printf '%s' "$cmd" | grep -qE '(curl|wget)[^|]*\|.*xargs.*[[:space:]]([^[:space:]]*/)?(sh|bash|zsh|dash|ksh|python[0-9.]*|perl|ruby|node|php)([[:space:]]|$)'; then + echo "refuse"; return 1 + fi + + # Output piped to netcat exfiltrates data off-host. + if printf '%s' "$cmd" | grep -qE '\|[[:space:]]*([^[:space:]]*/)?(nc|ncat|netcat)([[:space:]]|$)'; then + echo "refuse"; return 1 + fi + + # Raw block-device write — dd target or shell redirect onto a disk device wipes it. + # Scoped to real device families (incl. SD/eMMC mmcblk, mdadm md, device-mapper dm-) + # so dd/redirect to /dev/null or a regular file stays ok. + if printf '%s' "$cmd" | grep -qE '(of=|>[[:space:]]*)/dev/(sd|hd|vd|nvme|disk|mapper|loop|xvd|mmcblk|md|dm-)'; then + echo "refuse"; return 1 + fi + + # Filesystem format destroys everything on a partition. Optional path prefix catches a + # path-qualified invocation (/sbin/mkfs.ext4) that a bare-name anchor would miss. + if printf '%s' "$cmd" | grep -qE '(^|[[:space:]])([^[:space:]]*/)?(mkfs|mke2fs)'; then + echo "refuse"; return 1 + fi + + # find ... -delete mass-removes matched files. Both tokens required so a plain find + # search (no -delete) is not refused; optional path prefix catches /usr/bin/find. + if printf '%s' "$cmd" | grep -qE '(^|[[:space:]])([^[:space:]]*/)?find([[:space:]]|$)' \ + && printf '%s' "$cmd" | grep -qE '[[:space:]]-delete([[:space:]]|$)'; then + echo "refuse"; return 1 + fi + + # shred overwrites then unlinks — irrecoverable. + if printf '%s' "$cmd" | grep -qE '(^|[[:space:]])([^[:space:]]*/)?shred([[:space:]]|$)'; then + echo "refuse"; return 1 + fi + + # truncate to zero size destroys file contents in place. Non-zero sizes are allowed. + # Optional path prefix catches /usr/bin/truncate; size matcher covers -s 0, -s0, + # --size 0, and --size=0. + if printf '%s' "$cmd" | grep -qE '(^|[[:space:]])([^[:space:]]*/)?truncate([[:space:]]|$)' \ + && printf '%s' "$cmd" | grep -qE '(-s[[:space:]]*0|--size[[:space:]]*=?[[:space:]]*0)([[:space:]]|$)'; then + echo "refuse"; return 1 + fi + + # Recursive chmod to a zero mode locks an entire tree out of access. Scoped to the + # zero lock-out (000/00/0 octal short forms) so ordinary recursive permission changes + # are not refused; optional path prefix catches /bin/chmod. + if printf '%s' "$cmd" | grep -qE '(^|[[:space:]])([^[:space:]]*/)?chmod([[:space:]]|$)' \ + && printf '%s' "$cmd" | grep -qE '(-R|--recursive)([[:space:]]|$)' \ + && printf '%s' "$cmd" | grep -qE '(^|[[:space:]])(000|00|0)([[:space:]]|$)'; then + echo "refuse"; return 1 + fi + + # Fork bomb pattern + if printf '%s' "$cmd" | grep -qF ':(){ :|:'; then + echo "refuse"; return 1 + fi + if printf '%s' "$cmd" | grep -qE ':\(\)\{'; then + echo "refuse"; return 1 + fi + + # AWS credential patterns (key IDs start with AKIA, secret keys are 40-char base64) + if printf '%s' "$cmd" | grep -qE 'AKIA[0-9A-Z]{16}'; then + echo "refuse"; return 1 + fi + + # PASSWORD= credential pattern + if printf '%s' "$cmd" | grep -qE 'PASSWORD[[:space:]]*='; then + echo "refuse"; return 1 + fi + + # Private key headers — pattern starts with dashes so pass -- to avoid flag misparse + if printf '%s' "$cmd" | grep -qE -- 'BEGIN (RSA |EC |OPENSSH |DSA )?PRIVATE KEY'; then + echo "refuse"; return 1 + fi + + # Database URL safety: extract host and dbname from postgres:// or postgresql:// URIs + # Pattern: postgres(ql)://user:pass@HOST/DBNAME or postgres(ql)://HOST/DBNAME + if printf '%s' "$cmd" | grep -qE 'postgres(ql)?://'; then + # Extract the host portion (after @ or after ://) + local db_host db_name + db_host=$(printf '%s' "$cmd" \ + | grep -oE 'postgres(ql)?://[^[:space:]]+' \ + | sed -E 's|postgres(ql)?://([^@]+@)?([^/:]+)[:/].*|\3|') + db_name=$(printf '%s' "$cmd" \ + | grep -oE 'postgres(ql)?://[^[:space:]]+' \ + | sed -E 's|postgres(ql)?://[^/]*/([^?[:space:]]+).*|\2|') + + # Allowed hosts: localhost, 127.0.0.1, or a single-label hostname (no dots = container) + local host_ok=0 + if [[ "$db_host" == "localhost" || "$db_host" == "127.0.0.1" ]]; then + host_ok=1 + elif printf '%s' "$db_host" | grep -qvE '\.'; then + # No dots = plain container hostname → allowed + host_ok=1 + fi + + if [[ "$host_ok" -eq 0 ]]; then + # Non-allowlisted host: dbname must end with _test or _ci (anchored suffix, not substring) + if printf '%s' "$db_name" | grep -qE '_test$|_ci$'; then + echo "ok"; return 0 + fi + echo "refuse"; return 1 + fi + fi + + echo "ok"; return 0 +} + +# --------------------------------------------------------------------------- +# verdict: synthesize a convergence verdict from state JSON. +# Reads: units, plateau, ceiling fields. Prints verdict + ship-gate line. +# Exit 0 = CONVERGED; exit 1 = not converged; exit 2 = error. +# --------------------------------------------------------------------------- +verdict() { + local state_file="${1:?usage: verdict }" + if [[ ! -f "$state_file" ]]; then + echo "BLOCKED"; echo "ship=no"; return 2 + fi + + local units_val plateau_val ceiling_val + units_val=$(grep -o '"units"[[:space:]]*:[[:space:]]*[0-9.]*' "$state_file" \ + | grep -o '[0-9.]*$') + plateau_val=$(grep -o '"plateau"[[:space:]]*:[[:space:]]*[a-z]*' "$state_file" \ + | grep -o '[a-z]*$') + ceiling_val=$(grep -o '"ceiling"[[:space:]]*:[[:space:]]*[a-z]*' "$state_file" \ + | grep -o '[a-z]*$') + + if [[ -z "$units_val" ]]; then + echo "BLOCKED"; echo "ship=no"; return 2 + fi + + if [[ "$plateau_val" == "true" ]]; then + echo "PLATEAU"; echo "ship=no"; return 1 + fi + + if [[ "$ceiling_val" == "true" ]]; then + echo "CEILING"; echo "ship=no"; return 1 + fi + + # units==0 with no plateau/ceiling → converged + if awk -v u="$units_val" 'BEGIN { exit (u == 0 ? 0 : 1) }'; then + echo "CONVERGED"; echo "ship=yes"; return 0 + fi + + # units > 0, no plateau/ceiling signal yet → still running + echo "BLOCKED"; echo "ship=no"; return 1 +} + +# --------------------------------------------------------------------------- +# validate-state: schema gate for orchestrator-state.json. The ledger is the +# loop's evidence trail; a malformed one must not be trusted to route from. +# Prints "valid" exit 0 | "invalid" exit 2. Node is a checked installation +# prerequisite, so use its JSON parser rather than approximating JSON with grep. +# --------------------------------------------------------------------------- +validate-state() { + local state_file="${1:?usage: validate-state }" + if [[ ! -f "$state_file" ]]; then + echo "invalid"; return 2 + fi + + if ! node -e ' + const fs = require("fs"); + const state = JSON.parse(fs.readFileSync(process.argv[1], "utf8")); + const strings = ["goal", "archetype", "predicate", "terminal_choice"]; + if (!strings.every((key) => typeof state[key] === "string" && state[key].length > 0)) process.exit(1); + if (!Number.isInteger(state.cycle) || state.cycle < 0) process.exit(1); + if (!Array.isArray(state.units_remaining) || !Array.isArray(state.pipeline_log)) process.exit(1); + ' "$state_file" 2>/dev/null; then + echo "invalid"; return 2 + fi + + echo "valid"; return 0 +} + +# --------------------------------------------------------------------------- +# screen-state-predicate: extract the pinned predicate from a persisted state +# file and re-run it through screen-cmd. Persisted commands are never trusted — +# a poisoned state file must not re-enter the loop with an unscreened command. +# Delegates the verdict (ok/refuse + exit) to screen-cmd; "invalid" exit 2 when +# the state has no pinned predicate. +# --------------------------------------------------------------------------- +screen-state-predicate() { + local state_file="${1:?usage: screen-state-predicate }" + if [[ ! -f "$state_file" ]]; then + echo "invalid"; return 2 + fi + + local pred + if ! pred=$(node -e ' + const fs = require("fs"); + const state = JSON.parse(fs.readFileSync(process.argv[1], "utf8")); + if (typeof state.predicate !== "string" || state.predicate.length === 0 || state.predicate.includes("\0")) process.exit(1); + process.stdout.write(state.predicate); + ' "$state_file" 2>/dev/null); then + echo "invalid"; return 2 + fi + + screen-cmd "$pred" +} + +case "${1:-}" in + classify) shift; classify "$@" ;; + next-hop) shift; next-hop "$@" ;; + units) shift; units "$@" ;; + plateau) shift; plateau "$@" ;; + screen-cmd) shift; screen-cmd "$@" ;; + verdict) shift; verdict "$@" ;; + validate-state) shift; validate-state "$@" ;; + screen-state-predicate) shift; screen-state-predicate "$@" ;; + *) echo "usage: $0 {classify|next-hop|units|plateau|screen-cmd|verdict|validate-state|screen-state-predicate}" >&2; exit 64 ;; +esac diff --git a/pi-extension/skills/autoresearch/scripts/score-regression.sh b/pi-extension/skills/autoresearch/scripts/score-regression.sh new file mode 100755 index 00000000..f6753052 --- /dev/null +++ b/pi-extension/skills/autoresearch/scripts/score-regression.sh @@ -0,0 +1,205 @@ +#!/usr/bin/env bash +# score-regression.sh — scoring backend for autoresearch:regression +# +# rubric [file] → grep-rubric quality score of regression.md → "SCORE: N" +# verdict → tiered stability verdict from a results TSV +# +# verdict logic: +# - any HARD row that is a green→red regression (classification=regression-eligible, regressed=true) → UNSTABLE +# - else weighted SCORE: per-dim worst subscore, weights renormalized over dims that ran, +# STABLE iff stability_score >= threshold (default 95) +# - classification in {pre-existing,new-coverage,baseline-unavailable,flaky} never gates +# - exit 0 STABLE / 1 UNSTABLE / 2 ERROR (CI-usable) · score math → stderr +# +# Overridable env: REG_THRESHOLD, REG_W_FLAKINESS, REG_W_PERFORMANCE, REG_W_RESOURCE, REG_W_VISUAL +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)" + +resolve_default_spec() { + local candidate + for candidate in \ + "$SCRIPT_DIR/../regression.md" \ + "$SCRIPT_DIR/../../../commands/autoresearch/regression.md" \ + "$SCRIPT_DIR/../../../commands/autoresearch_regression.md" \ + "$REPO_ROOT/.claude/commands/autoresearch/regression.md" \ + "$REPO_ROOT/claude-plugin/commands/autoresearch/regression.md"; do + [[ -f "$candidate" ]] && { printf '%s\n' "$candidate"; return; } + done + printf '%s\n' "$SCRIPT_DIR/../regression.md" +} + +SPEC_DEFAULT="$(resolve_default_spec)" + +REG_THRESHOLD="${REG_THRESHOLD:-95}" +REG_W_FLAKINESS="${REG_W_FLAKINESS:-0.30}" +REG_W_PERFORMANCE="${REG_W_PERFORMANCE:-0.30}" +REG_W_RESOURCE="${REG_W_RESOURCE:-0.20}" +REG_W_VISUAL="${REG_W_VISUAL:-0.20}" + +# --------------------------------------------------------------------------- +# rubric: grep the protocol spec for required invariants/sections/flags. +# Each pattern matched = +1. Prints "SCORE: N" only. +# --------------------------------------------------------------------------- +rubric() { + local file="${1:-$SPEC_DEFAULT}" + if [[ ! -f "$file" ]]; then echo "SCORE: 0"; return 0; fi + + local checks=( + "classification" + "green.{0,3}red" + "regression.eligible|[^a-z]eligible" + "pre-existing" + "new-coverage" + "baseline.unavailable|BASELINE_UNAVAILABLE" + "functional" + "api-contract" + "data-migration" + "integration-e2e" + "flakiness" + "performance" + "resource" + "visual" + "HARD" + "SCORE" + "worktree" + "--detach|detach" + "submodule" + "baseline.cache|--baseline-cache" + "--select" + "findRelatedTests|nx affected|affected" + "Mann.?Whitney" + "independent.process" + "effect.size|median delta" + "SSIM|maxDiffPixelRatio|pixel.ratio" + "samples" + "noise-band" + "forward-only" + "allowlist" + "fix-cycle|--fix-cycles" + "probe" + "auto-skip" + "--max-runs" + "handoff" + "verdict" + "STABLE" + "UNSTABLE" + ) + + local score=0 pat + for pat in "${checks[@]}"; do + if grep -qiE -- "$pat" "$file"; then score=$((score + 1)); fi + done + echo "SCORE: $score" +} + +# --------------------------------------------------------------------------- +# verdict: reduce a results TSV to STABLE|UNSTABLE + stability score. +# --------------------------------------------------------------------------- +verdict() { + local tsv="${1:?usage: verdict }" + if [[ ! -f "$tsv" ]]; then + echo "VERDICT: ERROR"; echo "score=0.0"; echo "blocking=missing-tsv" + return 2 + fi + + awk -v FS='\t' \ + -v wF="$REG_W_FLAKINESS" -v wP="$REG_W_PERFORMANCE" \ + -v wR="$REG_W_RESOURCE" -v wV="$REG_W_VISUAL" \ + -v thr="$REG_THRESHOLD" ' + /^#/ { next } # comment (e.g. metric_direction) + $1=="iteration" { next } # header + NF != 15 { invalid_schema=1; next } + { + dim=$3; tier=$5; cls=$6; regressed=$10; subv=$11+0; + valid_dim = (dim=="functional" || dim=="api-contract" || dim=="data-migration" || dim=="integration-e2e" || dim=="flakiness" || dim=="performance" || dim=="resource" || dim=="visual-ui"); + expected_tier = (dim=="functional" || dim=="api-contract" || dim=="data-migration" || dim=="integration-e2e") ? "HARD" : "SCORE"; + if (!valid_dim || tier!=expected_tier || (regressed!="true" && regressed!="false") || $11 !~ /^([0-9]+([.][0-9]+)?|[.][0-9]+)$/ || subv<0 || subv>100) { + invalid_schema=1; + next; + } + if (cls!="regression-eligible" && cls!="pre-existing" && cls!="new-coverage" && cls!="baseline-unavailable" && cls!="flaky") { + invalid_classification=cls; + next; + } + if (cls!="baseline-unavailable") { + nrows++; + present[dim]=1; + } + if (tier=="HARD" && regressed=="true" && cls=="regression-eligible") hardset[dim]=1; + if (tier=="SCORE" && cls!="baseline-unavailable") { + if (!(dim in dmin) || subv < dmin[dim]) dmin[dim]=subv; # per-dim worst case + } + } + END { + if (invalid_schema) { + printf "VERDICT: ERROR\n"; + printf "score=0.00\n"; + printf "blocking=invalid-schema\n"; + exit 2; + } + if (invalid_classification!="") { + printf "VERDICT: ERROR\n"; + printf "score=0.00\n"; + printf "blocking=invalid-classification\n"; + exit 2; + } + # No measurable data rows ran at all → nothing to gate on. Must NOT read as a + # green ship signal: an empty/header-only TSV or all-dims-unavailable run is + # advisory, not STABLE. Emit BASELINE_UNAVAILABLE + non-zero exit. + if (nrows==0) { + printf "VERDICT: BASELINE_UNAVAILABLE\n"; + printf "score=0.00\n"; + printf "blocking=no-dims-ran\n"; + printf "dims_ran=none\n"; + printf "dims_unavailable=functional,api-contract,data-migration,integration-e2e,flakiness,performance,resource,visual-ui\n"; + exit 2; + } + + hb=""; + for (d in hardset) hb = hb (hb==""?"":",") d; + + wt["flakiness"]=wF; wt["performance"]=wP; wt["resource"]=wR; wt["visual-ui"]=wV; + num=0; den=0; + for (d in dmin) { w=(d in wt)?wt[d]:0; if (w>0){ num+=w*dmin[d]; den+=w; } } + score = (den>0) ? num/den : 100; + + split("functional,api-contract,data-migration,integration-e2e,flakiness,performance,resource,visual-ui", reg, ","); + ran=""; unavail=""; + for (i=1;i<=8;i++){ + if (reg[i] in present) ran = ran (ran==""?"":",") reg[i]; + else unavail = unavail (unavail==""?"":",") reg[i]; + } + + unstable = (hb!="") || (score < thr); + verdict = unstable ? "UNSTABLE" : "STABLE"; + if (hb!="" && score= threshold while UNSTABLE. + # The gate itself (above) compares full precision. + disp = int(score*100)/100; + + printf "VERDICT: %s\n", verdict; + printf "score=%.2f\n", disp; + printf "blocking=%s\n", blocking; + printf "dims_ran=%s\n", (ran==""?"none":ran); + printf "dims_unavailable=%s\n", (unavail==""?"none":unavail); + + for (d in dmin) printf " %-12s subscore=%.2f weight=%s\n", d, int(dmin[d]*100)/100, ((d in wt)?wt[d]:"0") > "/dev/stderr"; + printf " stability_score=%.2f threshold=%s\n", disp, thr > "/dev/stderr"; + + exit (unstable ? 1 : 0); + } + ' "$tsv" +} + +case "${1:-}" in + rubric) shift; rubric "$@" ;; + verdict) shift; verdict "$@" ;; + *) echo "usage: $0 {rubric [file] | verdict }" >&2; exit 64 ;; +esac diff --git a/pi-extension/src/guardrails.ts b/pi-extension/src/guardrails.ts new file mode 100644 index 00000000..b58b501d --- /dev/null +++ b/pi-extension/src/guardrails.ts @@ -0,0 +1,219 @@ +// guardrails — wires the ported Claude hooks to pi's extension event system. +// +// Claude event → pi event → hooks +// PreToolUse (Bash) → tool_call "bash" → scout-block, privacy-block, dangerous-cmd-block +// PreToolUse (R/W/E) → tool_call read/write → scout-block, privacy-block +// UserPromptSubmit → before_agent_start → iteration-context, dev-rules-reminder (inject message) +// UserPromptSubmit → input → simplify-gate (block/warn shipping verbs) +// SessionStart → session_start → session-init (persist state) +// SessionEnd → session_shutdown → stop-notify (notify + cleanup) +// +// All hooks fail open: a guardrail malfunction never blocks legitimate work. + +import type { ExtensionAPI, ExtensionContext } from "@earendil-works/pi-coding-agent"; +import { isToolCallEventType } from "@earendil-works/pi-coding-agent"; + +import { checkDangerousCommand } from "./hooks/dangerous-cmd-block.js"; +import { checkBashCommand, checkStructuredPath } from "./hooks/scout-block.js"; +import { + checkBashSensitive, + checkStructuredSensitive, + resolveConfirm, +} from "./hooks/privacy-block.js"; +import { buildIterationContext } from "./hooks/iteration-context.js"; +import { buildDevRules } from "./hooks/dev-rules-reminder.js"; +import { checkSimplifyGate, surfaceWarning } from "./hooks/simplify-gate.js"; +import { runStopNotify } from "./hooks/stop-notify.js"; +import { + isHookEnabled, + loadSessionState, + log, + pruneStaleSessionFiles, + saveSessionState, +} from "./lib/session-state.js"; +import { execSync } from "node:child_process"; +import { join } from "node:path"; + +const SESSION_INIT = "session-init"; + +function resolveGitRoot(cwd: string): string { + try { + return execSync("git rev-parse --show-toplevel", { + encoding: "utf8", + timeout: 5000, + cwd, + }).trim(); + } catch { + return cwd; + } +} + +function resolveGitBranch(cwd: string): string { + try { + return execSync("git rev-parse --abbrev-ref HEAD", { + encoding: "utf8", + timeout: 5000, + cwd, + }).trim(); + } catch { + return ""; + } +} + +function getSessionId(ctx: ExtensionContext): string { + return ctx.sessionManager.getSessionId() || "unknown"; +} + +// session_start → session-init +function onSessionStart(pi: ExtensionAPI): void { + pi.on("session_start", async (_event, ctx) => { + if (!isHookEnabled(SESSION_INIT)) return; + try { + const cwd = ctx.cwd; + const sessionId = getSessionId(ctx); + const projectRoot = resolveGitRoot(cwd); + const gitBranch = resolveGitBranch(cwd); + + const state = { + projectRoot, + plansPath: join(projectRoot, "plans"), + reportsPath: join(projectRoot, "plans", "reports"), + gitBranch, + sessionId, + iterationCount: 0, + startedAt: new Date().toISOString(), + }; + saveSessionState(cwd, sessionId, state); + pruneStaleSessionFiles(); + log(SESSION_INIT, { projectRoot, gitBranch }); + } catch { + // fail-open + } + }); +} + +// session_shutdown → stop-notify +function onSessionShutdown(pi: ExtensionAPI): void { + pi.on("session_shutdown", async (_event, ctx) => { + const cwd = ctx.cwd; + const sessionId = getSessionId(ctx); + await runStopNotify(cwd, sessionId); + }); +} + +// tool_call → scout-block + privacy-block + dangerous-cmd-block +function onToolCall(pi: ExtensionAPI): void { + pi.on("tool_call", async (event, ctx) => { + const cwd = ctx.cwd; + + // --- bash tool --- + if (isToolCallEventType("bash", event)) { + const command = event.input.command as string; + + // dangerous-cmd-block (highest priority — hard block) + const dangerous = checkDangerousCommand(command); + if (dangerous.block) { + if (ctx.hasUI) ctx.ui.notify(dangerous.reason || "Blocked", "error"); + return { block: true, reason: dangerous.reason }; + } + + // scout-block (.ckignore) + const scout = checkBashCommand(command, cwd); + if (scout.block) { + if (ctx.hasUI) ctx.ui.notify(scout.reason || "Blocked", "error"); + return { block: true, reason: scout.reason }; + } + + // privacy-block (sensitive file → confirm; ambiguous → warn) + const privacy = checkBashSensitive(command); + if (privacy.warning && ctx.hasUI) { + ctx.ui.notify(privacy.warning, "warning"); + } + const resolved = await resolveConfirm(privacy, ctx); + if (resolved.block) { + return { block: true, reason: resolved.reason }; + } + return; + } + + // --- structured path tools: read / write / edit --- + const pathTools = ["read", "write", "edit"] as const; + for (const name of pathTools) { + if (isToolCallEventType(name, event)) { + const filePath = (event.input.path as string) || ""; + + // scout-block (.ckignore) + const scout = checkStructuredPath(filePath, cwd); + if (scout.block) { + if (ctx.hasUI) ctx.ui.notify(scout.reason || "Blocked", "error"); + return { block: true, reason: scout.reason }; + } + + // privacy-block (sensitive file → confirm) + const privacy = checkStructuredSensitive(filePath); + const resolved = await resolveConfirm(privacy, ctx); + if (resolved.block) { + return { block: true, reason: resolved.reason }; + } + return; + } + } + + return; + }); +} + +// before_agent_start → iteration-context + dev-rules-reminder (inject message) +function onBeforeAgentStart(pi: ExtensionAPI): void { + pi.on("before_agent_start", async (event, ctx) => { + const cwd = ctx.cwd; + const sessionId = getSessionId(ctx); + const prompt = event.prompt || ""; + + const parts: string[] = []; + + const iter = buildIterationContext(cwd, sessionId, prompt); + if (iter.text) parts.push(iter.text); + + const dev = buildDevRules(cwd, sessionId); + if (dev.text) parts.push(dev.text); + + if (parts.length === 0) return; + + return { + message: { + customType: "autoresearch-context", + content: parts.join("\n\n"), + display: false, + }, + }; + }); +} + +// input → simplify-gate (block/warn shipping verbs) +function onInput(pi: ExtensionAPI): void { + pi.on("input", async (event, ctx) => { + if (event.source === "extension") return { action: "continue" }; + + const result = checkSimplifyGate(event.text, ctx.cwd); + surfaceWarning(result, ctx); + + if (result.block) { + if (ctx.hasUI) ctx.ui.notify(result.reason || "Blocked", "error"); + return { action: "handled" }; + } + + return { action: "continue" }; + }); +} + +export function registerGuardrails(pi: ExtensionAPI): void { + onSessionStart(pi); + onSessionShutdown(pi); + onToolCall(pi); + onBeforeAgentStart(pi); + onInput(pi); +} + +// Re-export for tests / direct use. +export { loadSessionState }; diff --git a/pi-extension/src/hooks/dangerous-cmd-block.ts b/pi-extension/src/hooks/dangerous-cmd-block.ts new file mode 100644 index 00000000..310a114d --- /dev/null +++ b/pi-extension/src/hooks/dangerous-cmd-block.ts @@ -0,0 +1,102 @@ +// dangerous-cmd-block — ported from claude-plugin/hooks/dangerous-cmd-block.cjs. +// Blocks destructive bash commands: forced push, recursive forced rm, hard git +// reset, forced git clean, forced branch deletion, checkout/restore of the +// working tree. Regular `git push` is allowed. Fails open on any error. + +import { basename } from "node:path"; +import { shellSegments } from "../lib/shell.js"; +import { isHookEnabled, log } from "../lib/session-state.js"; + +const HOOK_NAME = "dangerous-cmd-block"; + +function gitSubcommandIndex(words: string[]): number { + const optionsWithValue = new Set([ + "-C", + "-c", + "--exec-path", + "--git-dir", + "--work-tree", + "--namespace", + "--super-prefix", + "--config-env", + ]); + let index = 1; + while (index < words.length) { + const option = words[index]; + if (option === "--") return index + 1; + if (!option.startsWith("-")) return index; + if (optionsWithValue.has(option)) { + index += 2; + } else if ( + option.startsWith("-C") || + option.startsWith("-c") || + [...optionsWithValue].some( + (name) => name.startsWith("--") && option.startsWith(name + "="), + ) + ) { + index += 1; + } else { + index += 1; + } + } + return -1; +} + +export function commandLabel(command: string): string | null { + for (const words of shellSegments(command)) { + const executable = basename(words[0] || ""); + if ( + executable === "push" && + words.slice(1).some((arg) => arg === "-f" || arg.startsWith("--force")) + ) + return "forced push"; + if (executable === "rm") { + const flags = words.slice(1).filter((word) => word.startsWith("-")).join(""); + if ( + (/[rR]/.test(flags) || flags.includes("--recursive")) && + (flags.includes("f") || flags.includes("--force")) + ) + return "recursive forced removal"; + } + if (executable !== "git") continue; + const subcommandIndex = gitSubcommandIndex(words); + const subcommand = words[subcommandIndex]; + const args = words.slice(subcommandIndex + 1); + if (subcommand === "push" && args.some((arg) => arg === "-f" || arg.startsWith("--force"))) + return "forced git push"; + if (subcommand === "reset" && args.includes("--hard")) return "hard git reset"; + if (subcommand === "clean" && args.some((arg) => /^-[^-]*f/.test(arg) || arg === "--force")) + return "forced git clean"; + if (subcommand === "branch") { + const flags = args.filter((arg) => arg.startsWith("-")); + const hasDelete = flags.some((arg) => arg === "--delete" || /^-[^-]*[dD]/.test(arg)); + const hasForce = flags.some((arg) => arg === "--force" || /^-[^-]*[fD]/.test(arg)); + if (hasDelete && hasForce) return "forced branch deletion"; + } + if ((subcommand === "checkout" || subcommand === "restore") && args.includes(".")) + return `git ${subcommand} of working tree`; + } + return null; +} + +export interface DangerousCmdResult { + block?: boolean; + reason?: string; +} + +export function checkDangerousCommand(command: string): DangerousCmdResult { + if (!isHookEnabled(HOOK_NAME)) return {}; + try { + const label = commandLabel(command); + if (label) { + log(HOOK_NAME, { action: "block", matched: label }); + return { + block: true, + reason: `BLOCKED: Destructive command detected (${label}). This command is blocked during autoresearch sessions.`, + }; + } + } catch { + // fail-open + } + return {}; +} diff --git a/pi-extension/src/hooks/dev-rules-reminder.ts b/pi-extension/src/hooks/dev-rules-reminder.ts new file mode 100644 index 00000000..cc7d1eca --- /dev/null +++ b/pi-extension/src/hooks/dev-rules-reminder.ts @@ -0,0 +1,44 @@ +// dev-rules-reminder — ported from claude-plugin/hooks/dev-rules-reminder.cjs. +// Injects dev-context (plan path, code-standards) every 5th iteration, but +// skips the turn when iteration-context already injected. Fails open. + +import { join } from "node:path"; +import { isHookEnabled, loadSessionState, log } from "../lib/session-state.js"; + +const HOOK_NAME = "dev-rules-reminder"; + +export interface DevRulesResult { + text: string | null; +} + +export function buildDevRules(cwd: string, sessionId: string): DevRulesResult { + if (!isHookEnabled(HOOK_NAME)) return { text: null }; + try { + const state = loadSessionState(cwd, sessionId); + + // Skip if iteration-context already injected this same turn (within 2s) + if (state.lastContextInjection && Date.now() - state.lastContextInjection < 2000) { + log(HOOK_NAME, { action: "skip", reason: "iteration-context-fired" }); + return { text: null }; + } + + // Only inject on every 5th iteration, same cadence as iteration-context + if ((state.iterationCount || 0) % 5 !== 0) { + return { text: null }; + } + + const plansPath = state.plansPath || join(cwd, "plans"); + + const text = [ + "## Dev context", + `- Plan: ${plansPath} (check for active plan.md)`, + "- Standards: docs/code-standards.md", + ].join("\n"); + + log(HOOK_NAME, { action: "inject", iterations: state.iterationCount }); + return { text }; + } catch { + // fail-open + return { text: null }; + } +} diff --git a/pi-extension/src/hooks/iteration-context.ts b/pi-extension/src/hooks/iteration-context.ts new file mode 100644 index 00000000..4304b533 --- /dev/null +++ b/pi-extension/src/hooks/iteration-context.ts @@ -0,0 +1,109 @@ +// iteration-context — ported from claude-plugin/hooks/iteration-context.cjs. +// On every 5th prompt, injects the active iteration TSV tail + iteration count +// as additional context so the agent remembers loop state across compaction. +// In Claude this ran on UserPromptSubmit and returned additionalContext; in pi +// we run on before_agent_start and return a message. Fails open. + +import { relative } from "node:path"; +import { + incrementCounter, + isHookEnabled, + loadSessionState, + log, + saveSessionState, +} from "../lib/session-state.js"; +import { findRecentTsv, readTsvTail } from "../lib/tsv.js"; + +const HOOK_NAME = "iteration-context"; + +const AR_COMMANDS = [ + "autoresearch", + "/autoresearch:", + "loop", + "debug", + "fix", + "scenario", + "predict", + "learn", + "reason", + "probe", + "security", + "ship", +]; + +function hasArCommand(prompt: string): boolean { + if (!prompt || typeof prompt !== "string") return false; + const lower = prompt.toLowerCase(); + return AR_COMMANDS.some((cmd) => lower.includes(cmd)); +} + +function formatRows(header: string, rows: string[]): string { + const lines: string[] = []; + if (header) lines.push(header); + for (const r of rows) lines.push(r); + return lines.join("\n"); +} + +function relativePath(cwd: string, absPath: string): string { + try { + return relative(cwd, absPath); + } catch { + return absPath; + } +} + +export interface IterationContextResult { + /** Additional context text to inject, or null to inject nothing. */ + text: string | null; +} + +export function buildIterationContext( + cwd: string, + sessionId: string, + prompt: string, +): IterationContextResult { + if (!isHookEnabled(HOOK_NAME)) return { text: null }; + try { + incrementCounter(cwd, sessionId, "iterationCount"); + const state = loadSessionState(cwd, sessionId); + const iterationCount = state.iterationCount; + + // Throttle: only inject every 5th prompt + if (iterationCount % 5 !== 0) { + log(HOOK_NAME, { action: "skip", iterations: iterationCount }); + return { text: null }; + } + + // Mark injection time so dev-rules-reminder can skip this turn + state.lastContextInjection = Date.now(); + saveSessionState(cwd, sessionId, state); + + const tsvPath = findRecentTsv(cwd, 30); + if (!tsvPath) { + log(HOOK_NAME, { action: "skip", iterations: iterationCount, reason: "no-tsv" }); + return { text: null }; + } + + const tsv = readTsvTail(tsvPath, 3); + if (!tsv) { + log(HOOK_NAME, { action: "skip", iterations: iterationCount, reason: "tsv-unreadable" }); + return { text: null }; + } + + const relTsv = relativePath(cwd, tsvPath); + const rowBlock = formatRows(tsv.header, tsv.rows); + + let text = + `## Active iteration state\n**TSV:** ${relTsv}\n**Iteration:** ${iterationCount} | **Rows:** ${tsv.total}\n\n${rowBlock}`; + + if (hasArCommand(prompt)) { + text += `\n\n**Loop state:** active — ${tsv.total} iterations recorded, last 3 rows above`; + } + + log(HOOK_NAME, { action: "inject", iterations: iterationCount, tsvRows: tsv.total }); + return { text }; + } catch { + // fail-open + return { text: null }; + } +} diff --git a/pi-extension/src/hooks/privacy-block.ts b/pi-extension/src/hooks/privacy-block.ts new file mode 100644 index 00000000..cf206818 --- /dev/null +++ b/pi-extension/src/hooks/privacy-block.ts @@ -0,0 +1,81 @@ +// privacy-block — ported from claude-plugin/hooks/privacy-block.cjs. +// Escalates clear sensitive-file access to a user confirmation. In Claude this +// returned permissionDecision: 'ask'; in pi we use ctx.ui.confirm() in +// interactive mode and fail open (allow) otherwise. Ambiguous commands get a +// warning injected. Fails open on any error. + +import { isHookEnabled, log } from "../lib/session-state.js"; +import { bashSensitivity, isSensitive } from "../lib/paths.js"; +import type { ExtensionContext } from "@earendil-works/pi-coding-agent"; + +const HOOK_NAME = "privacy-block"; + +export interface PrivacyResult { + block?: boolean; + reason?: string; + /** Text to inject as additional context (ambiguous-command warning). */ + warning?: string; + /** When true, the caller should run ctx.ui.confirm before proceeding. */ + needsConfirm?: boolean; + confirmReason?: string; +} + +export function checkStructuredSensitive(filePath: string): PrivacyResult { + if (!isHookEnabled(HOOK_NAME)) return {}; + try { + if (isSensitive(filePath)) { + log(HOOK_NAME, { action: "ask", tool: "structured", category: "sensitive-file" }); + return { + needsConfirm: true, + confirmReason: + "This operation targets a potentially sensitive file. Confirm access?", + }; + } + } catch { + // fail-open + } + return {}; +} + +export function checkBashSensitive(command: string): PrivacyResult { + if (!isHookEnabled(HOOK_NAME)) return {}; + try { + const sensitivity = bashSensitivity(command); + if (sensitivity === "clear") { + log(HOOK_NAME, { action: "ask", tool: "bash", category: "sensitive-command" }); + return { + needsConfirm: true, + confirmReason: + "This command clearly accesses a potentially sensitive file. Confirm access?", + }; + } + if (sensitivity === "ambiguous") { + log(HOOK_NAME, { action: "warn", tool: "bash", category: "ambiguous-sensitive-text" }); + return { + warning: + "WARNING: The command contains sensitive-looking text. Confirm that it does not expose credentials.", + }; + } + } catch { + // fail-open + } + return {}; +} + +// Resolve a needsConfirm result against the UI. Returns a block decision if +// the user declines; returns {} if allowed or no UI available (fail open). +export async function resolveConfirm( + result: PrivacyResult, + ctx: ExtensionContext, +): Promise { + if (!result.needsConfirm) return result; + if (!ctx.hasUI) { + // Non-interactive: fail open (allow). + return {}; + } + const ok = await ctx.ui.confirm("Sensitive file", result.confirmReason || "Allow?"); + if (!ok) { + return { block: true, reason: "Blocked: sensitive-file access declined by user." }; + } + return {}; +} diff --git a/pi-extension/src/hooks/scout-block.ts b/pi-extension/src/hooks/scout-block.ts new file mode 100644 index 00000000..eaf3ae2f --- /dev/null +++ b/pi-extension/src/hooks/scout-block.ts @@ -0,0 +1,61 @@ +// scout-block — ported from claude-plugin/hooks/scout-block.cjs. +// Blocks file access (via read/write/edit/bash tools) to paths matching +// .ckignore + baseline ignore patterns (node_modules, .git, etc.). Fails open. + +import { isHookEnabled, log } from "../lib/session-state.js"; +import { checkIgnore, extractPathTokens, findProjectRoot, loadCkIgnore } from "../lib/paths.js"; + +const HOOK_NAME = "scout-block"; + +export interface ScoutBlockResult { + block?: boolean; + reason?: string; +} + +// Check a structured-tool path (read/write/edit) against .ckignore. +export function checkStructuredPath( + filePath: string, + cwd: string, +): ScoutBlockResult { + if (!isHookEnabled(HOOK_NAME)) return {}; + try { + const projectRoot = findProjectRoot(cwd); + const ig = loadCkIgnore(projectRoot); + const matched = checkIgnore(filePath, ig, projectRoot, cwd); + if (matched) { + log(HOOK_NAME, { action: "block", tool: "structured", path: filePath, matched }); + return { + block: true, + reason: `BLOCKED: Access to '${matched}' denied by .ckignore\n\nTo allow, add to .ckignore: !${matched}`, + }; + } + } catch { + // fail-open + } + return {}; +} + +// Check path-like tokens in a bash command against .ckignore. +export function checkBashCommand( + command: string, + cwd: string, +): ScoutBlockResult { + if (!isHookEnabled(HOOK_NAME)) return {}; + try { + const projectRoot = findProjectRoot(cwd); + const ig = loadCkIgnore(projectRoot); + for (const token of extractPathTokens(command)) { + const matched = checkIgnore(token, ig, projectRoot, cwd); + if (matched) { + log(HOOK_NAME, { action: "block", tool: "bash", path: token, matched }); + return { + block: true, + reason: `BLOCKED: Access to '${matched}' denied by .ckignore\n\nTo allow, add to .ckignore: !${matched}`, + }; + } + } + } catch { + // fail-open + } + return {}; +} diff --git a/pi-extension/src/hooks/simplify-gate.ts b/pi-extension/src/hooks/simplify-gate.ts new file mode 100644 index 00000000..37d4784e --- /dev/null +++ b/pi-extension/src/hooks/simplify-gate.ts @@ -0,0 +1,118 @@ +// simplify-gate — ported from claude-plugin/hooks/simplify-gate.cjs. +// Warns or blocks shipping verbs (ship/merge/deploy/pr/publish/release) when +// too many lines have changed. In Claude this ran on UserPromptSubmit; in pi +// we run on the `input` event and can block or transform. Fails open. + +import { readFileSync } from "node:fs"; +import { execSync } from "node:child_process"; +import { isHookEnabled, log } from "../lib/session-state.js"; +import type { ExtensionContext } from "@earendil-works/pi-coding-agent"; + +const HOOK_NAME = "simplify-gate"; + +const SHIPPING_VERBS = ["ship", "merge", "deploy", "pr", "publish", "release"]; + +const NEGATION_PHRASES = [ + "don't ship", + "never deploy", + "not ready to merge", + "don't merge", + "don't deploy", + "don't publish", + "don't release", + "no ship", + "no merge", + "no deploy", +]; + +const WARN_THRESHOLD = 400; +const BLOCK_THRESHOLD = 800; + +function hasShippingVerb(prompt: string): boolean { + const lower = prompt.toLowerCase(); + for (const phrase of NEGATION_PHRASES) { + if (lower.includes(phrase)) return false; + } + for (const verb of SHIPPING_VERBS) { + const regex = new RegExp("\\b" + verb + "\\b", "i"); + if (regex.test(prompt)) return true; + } + return false; +} + +function pendingLoc(cwd: string): number { + const diff = execSync("git diff HEAD --numstat", { + encoding: "utf8", + timeout: 5000, + cwd, + }); + let loc = diff + .trim() + .split("\n") + .filter(Boolean) + .reduce((total, line) => { + const [added, removed] = line.split("\t"); + return ( + total + + (/^\d+$/.test(added) ? Number(added) : 0) + + (/^\d+$/.test(removed) ? Number(removed) : 0) + ); + }, 0); + const untracked = execSync("git ls-files --others --exclude-standard -z", { + encoding: "utf8", + timeout: 5000, + cwd, + }); + for (const file of untracked.split("\0").filter(Boolean)) { + try { + loc += readFileSync(file, "utf8").split("\n").length - 1; + } catch { + /* skip unreadable untracked file */ + } + } + return loc; +} + +export interface SimplifyGateResult { + /** Block the prompt entirely. */ + block?: boolean; + reason?: string; + /** Warning text to surface to the user (non-blocking). */ + warning?: string; +} + +export function checkSimplifyGate(prompt: string, cwd: string): SimplifyGateResult { + if (!isHookEnabled(HOOK_NAME)) return {}; + if (!prompt || typeof prompt !== "string") return {}; + + if (!hasShippingVerb(prompt)) return {}; + + let loc = 0; + try { + loc = pendingLoc(cwd); + } catch { + // fail-open on git errors + return {}; + } + + if (loc < WARN_THRESHOLD) return {}; + + log(HOOK_NAME, { loc, action: loc > BLOCK_THRESHOLD ? "block" : "warn" }); + + if (loc > BLOCK_THRESHOLD) { + return { + block: true, + reason: `BLOCKED: ${loc} lines changed exceeds ${BLOCK_THRESHOLD} LOC shipping threshold. Simplify before shipping. Use AR_DISABLE_SIMPLIFY_GATE=1 to override.`, + }; + } + + // 400–800 range: warn but allow + return { warning: `WARNING: ${loc} lines changed. Consider simplifying before shipping.` }; +} + +// Surface a non-blocking warning to the user without altering the prompt. +export function surfaceWarning(result: SimplifyGateResult, ctx: ExtensionContext): void { + if (result.warning && ctx.hasUI) { + ctx.ui.notify(result.warning, "warning"); + } +} diff --git a/pi-extension/src/hooks/stop-notify.ts b/pi-extension/src/hooks/stop-notify.ts new file mode 100644 index 00000000..a4fb25c7 --- /dev/null +++ b/pi-extension/src/hooks/stop-notify.ts @@ -0,0 +1,145 @@ +// stop-notify — ported from claude-plugin/hooks/stop-notify.cjs, with the +// multi-protocol terminal notification from pi's notify.ts example (OSC 777, +// Kitty OSC 99, Windows toast). Fires on session_shutdown: sends a terminal +// notification + optional webhook, then cleans up the session state file. + +import http from "node:http"; +import https from "node:https"; +import { basename } from "node:path"; +import { + cleanupSessionFile, + isHookEnabled, + loadSessionState, + log, +} from "../lib/session-state.js"; +import { findRecentTsv, readTsvTail } from "../lib/tsv.js"; + +const HOOK_NAME = "stop-notify"; + +function formatDuration(startedAt?: string): string { + if (!startedAt) return "unknown"; + const ms = Date.now() - new Date(startedAt).getTime(); + if (ms < 0) return "unknown"; + const totalSeconds = Math.floor(ms / 1000); + const hours = Math.floor(totalSeconds / 3600); + const minutes = Math.floor((totalSeconds % 3600) / 60); + const seconds = totalSeconds % 60; + if (hours > 0) return `${hours}h ${minutes}m`; + return `${minutes}m ${seconds}s`; +} + +function buildTsvSummary(projectRoot: string): { text: string; iterations: number } { + const tsvPath = findRecentTsv(projectRoot, 120); // look back 2 hours on session end + if (!tsvPath) return { text: "no iterations recorded", iterations: 0 }; + + const tsv = readTsvTail(tsvPath, 1); + if (!tsv) return { text: "no iterations recorded", iterations: 0 }; + + const lastRow = tsv.rows[0] || ""; + const metricMatch = lastRow.match(/[\t|]([0-9.-]+)(?:[\t|]|$)/); + const metric = metricMatch ? metricMatch[1] : "n/a"; + + return { text: `${tsv.total} iterations, metric: ${metric}`, iterations: tsv.total }; +} + +function windowsToastScript(title: string, body: string): string { + const type = "Windows.UI.Notifications"; + const mgr = `[${type}.ToastNotificationManager, ${type}, ContentType = WindowsRuntime]`; + const template = `[${type}.ToastTemplateType]::ToastText01`; + const toast = `[${type}.ToastNotification]::new($xml)`; + return [ + `${mgr} > $null`, + `$xml = [${type}.ToastNotificationManager]::GetTemplateContent(${template})`, + `$xml.GetElementsByTagName('text')[0].AppendChild($xml.CreateTextNode('${body}')) > $null`, + `[${type}.ToastNotificationManager]::CreateToastNotifier('${title}').Show(${toast})`, + ].join("; "); +} + +function notifyOSC777(title: string, body: string): void { + process.stdout.write(`\x1b]777;notify;${title};${body}\x07`); +} + +function notifyOSC99(title: string, body: string): void { + process.stdout.write(`\x1b]99;i=1:d=0;${title}\x1b\\`); + process.stdout.write(`\x1b]99;i=1:p=body;${body}\x1b\\`); +} + +function notifyWindows(title: string, body: string): void { + const { execFile } = require("node:child_process"); + execFile("powershell.exe", ["-NoProfile", "-Command", windowsToastScript(title, body)]); +} + +function notify(title: string, body: string): void { + if (process.env.WT_SESSION) { + notifyWindows(title, body); + } else if (process.env.KITTY_WINDOW_ID) { + notifyOSC99(title, body); + } else { + notifyOSC777(title, body); + } +} + +function postWebhook(webhookUrl: string, payload: Record): Promise { + return new Promise((resolve) => { + try { + const parsed = new URL(webhookUrl); + const client = parsed.protocol === "http:" ? http : https; + const body = JSON.stringify(payload); + const req = client.request( + { + hostname: parsed.hostname, + port: parsed.port || (parsed.protocol === "https:" ? 443 : 80), + path: parsed.pathname + parsed.search, + method: "POST", + headers: { + "Content-Type": "application/json", + "Content-Length": Buffer.byteLength(body), + }, + }, + (response) => { + response.resume(); + response.on("end", resolve); + }, + ); + req.setTimeout(2000, () => req.destroy()); + req.on("error", resolve); + req.write(body); + req.end(); + } catch { + resolve(); + } + }); +} + +export async function runStopNotify(cwd: string, sessionId: string): Promise { + if (!isHookEnabled(HOOK_NAME)) return; + try { + const state = loadSessionState(cwd, sessionId); + const duration = formatDuration(state.startedAt); + const tsvSummary = buildTsvSummary(state.projectRoot || cwd); + const projectName = basename(state.projectRoot || cwd); + + notify("autoresearch", `Session completed — ${projectName} (${duration})`); + + const webhookUrl = process.env.AR_NOTIFY_WEBHOOK; + if (webhookUrl) { + await postWebhook(webhookUrl, { + text: "autoresearch session completed", + project: projectName, + branch: state.gitBranch || "", + duration, + tsv_summary: tsvSummary.text, + }); + } + + log(HOOK_NAME, { + projectName, + duration, + iterations: tsvSummary.iterations, + }); + + cleanupSessionFile(cwd, sessionId); + } catch { + // fail-open + } +} diff --git a/pi-extension/src/index.ts b/pi-extension/src/index.ts new file mode 100644 index 00000000..daf753ce --- /dev/null +++ b/pi-extension/src/index.ts @@ -0,0 +1,43 @@ +// Autoresearch for pi — extension entry point. +// +// Ports the Claude Code autoresearch hooks (safety guardrails + iteration +// context injection) to pi's extension event system, and contributes the +// autoresearch skill + 14 command prompt templates via resources_discover. +// +// Install: pi install git:github.com/uditgoenka/autoresearch +// (then add "pi-extension" as an extension package, or drop this dir into +// ~/.pi/agent/extensions/autoresearch-pi/) +// Try: pi -e ./pi-extension +// +// Hooks → events: +// PreToolUse → tool_call (scout/privacy/dangerous-cmd block) +// UserPrompt → before_agent_start (iteration-context + dev-rules inject) +// UserPrompt → input (simplify-gate) +// SessionStart → session_start (session-init) +// SessionEnd → session_shutdown (stop-notify) +// +// Every hook fails open — a guardrail malfunction never blocks work. + +import { dirname, join } from "node:path"; +import { fileURLToPath } from "node:url"; +import type { ExtensionAPI } from "@earendil-works/pi-coding-agent"; + +import { registerGuardrails } from "./guardrails.js"; + +const baseDir = dirname(fileURLToPath(import.meta.url)); + +export default function autoresearchPi(pi: ExtensionAPI): void { + // 1. Safety guardrails + context injection (the "hooks"). + registerGuardrails(pi); + + // 2. Contribute the autoresearch skill + command prompt templates. + // skills/autoresearch/SKILL.md → /skill:autoresearch (dispatcher) + // prompts/autoresearch.md → /autoresearch (classic loop) + // prompts/autoresearch/*.md → /autoresearch_debug, /autoresearch_fix, ... + pi.on("resources_discover", () => { + return { + skillPaths: [join(baseDir, "..", "skills", "autoresearch", "SKILL.md")], + promptPaths: [join(baseDir, "..", "prompts")], + }; + }); +} diff --git a/pi-extension/src/lib/ignore.ts b/pi-extension/src/lib/ignore.ts new file mode 100644 index 00000000..c90a8766 --- /dev/null +++ b/pi-extension/src/lib/ignore.ts @@ -0,0 +1,83 @@ +// Minimal gitignore-spec pattern matcher — ported from +// claude-plugin/hooks/lib/ignore.cjs. Zero deps. Supports directory +// patterns (dir/), globs (*.ext), negation (!pattern), double-star (**/) +// for arbitrary depth, and comments (#). Used by scout-block to enforce +// .ckignore + baseline ignore patterns. + +interface Rule { + pattern: string; + negated: boolean; + re: RegExp; +} + +export class Ignore { + private _rules: Rule[] = []; + + add(patterns: string | string[]): this { + const lines = Array.isArray(patterns) ? patterns : patterns.split("\n"); + for (const raw of lines) { + const line = raw.trim(); + if (!line || line.startsWith("#")) continue; + const negated = line.startsWith("!"); + const pattern = negated ? line.slice(1) : line; + this._rules.push({ pattern, negated, re: this._compile(pattern) }); + } + return this; + } + + ignores(inputPath: string): boolean { + const p = inputPath.startsWith("/") ? inputPath.slice(1) : inputPath; + let ignored = false; + for (const rule of this._rules) { + if (rule.re.test(p)) { + ignored = !rule.negated; + } + } + return ignored; + } + + private _compile(pattern: string): RegExp { + let p = pattern; + if (p.endsWith("/")) p = p + "**"; + let re = ""; + let i = 0; + while (i < p.length) { + const c = p[i]; + if (c === "*") { + if (p[i + 1] === "*") { + if (p[i + 2] === "/") { + re += "(?:.+/)?"; + i += 3; + continue; + } + re += ".*"; + i += 2; + continue; + } + re += "[^/]*"; + i++; + } else if (c === "?") { + re += "[^/]"; + i++; + } else if (c === ".") { + re += "\\."; + i++; + } else if (c === "/") { + re += "/"; + i++; + } else { + re += c; + i++; + } + } + const anchored = pattern.includes("/") && !pattern.startsWith("**/"); + if (anchored) { + return new RegExp("^" + re + "(?:$|/)"); + } + return new RegExp("(?:^|/)" + re + "(?:$|/)"); + } +} + +export function ignore(): Ignore { + return new Ignore(); +} diff --git a/pi-extension/src/lib/paths.ts b/pi-extension/src/lib/paths.ts new file mode 100644 index 00000000..61f96118 --- /dev/null +++ b/pi-extension/src/lib/paths.ts @@ -0,0 +1,245 @@ +// Path + sensitive-file helpers — ported from claude-plugin/hooks/scout-block.cjs +// and privacy-block.cjs. Used by the scout-block and privacy-block guardrails. + +import { existsSync, readFileSync, statSync } from "node:fs"; +import { basename, dirname, isAbsolute, join, relative, resolve } from "node:path"; +import { Ignore } from "./ignore.js"; +import { shellSegments } from "./shell.js"; + +export const BASELINE_PATTERNS = [ + "node_modules/", + "__pycache__/", + ".git/", + "dist/", + "build/", + "out/", + "coverage/", + ".next/", + ".nuxt/", + "venv/", + ".venv/", + "env/", + ".terraform/", + ".aws/", + ".ssh/", + "*.log", +]; + +export function findProjectRoot(startDir: string): string { + let dir = startDir; + for (let i = 0; i < 20; i++) { + if (existsSync(join(dir, ".git"))) return dir; + const parent = dirname(dir); + if (parent === dir) break; + dir = parent; + } + return startDir; +} + +export function loadCkIgnore(projectRoot: string): Ignore { + const ig = new Ignore(); + ig.add(BASELINE_PATTERNS); + try { + const ckPath = join(projectRoot, ".ckignore"); + const content = readFileSync(ckPath, "utf8"); + ig.add(content); + } catch { + /* no .ckignore — baseline only */ + } + return ig; +} + +export function relativeToRoot(filePath: string, projectRoot: string, cwd: string): string { + const abs = isAbsolute(filePath) ? filePath : resolve(cwd, filePath); + const rel = relative(projectRoot, abs); + // If the path escapes the project root, use the absolute path for matching. + return (rel.startsWith("..") ? abs : rel).replace(/\\/g, "/"); +} + +export function checkIgnore( + filePath: string, + ig: Ignore, + projectRoot: string, + cwd: string, +): string | null { + if (!filePath || typeof filePath !== "string") return null; + const rel = relativeToRoot(filePath, projectRoot, cwd); + return ig.ignores(rel) ? rel : null; +} + +// --- Sensitive-file detection (privacy-block) --- + +export const SENSITIVE_PATTERNS = [ + ".env", + ".env.local", + ".env.production", + ".env.development", + ".pem", + ".key", + ".p12", + ".pfx", + "id_rsa", + "id_ed25519", + ".ssh/", + "credentials.json", + "credentials.yaml", + "secret", + "api_key", + "apikey", + ".aws/credentials", +]; + +export const ALLOWED_EXCEPTIONS = [ + ".env.example", + ".env.sample", + ".env.template", + ".env.test", +]; + +export function isSensitive(filePath: string): boolean { + if (!filePath || typeof filePath !== "string") return false; + const normalized = filePath.replace(/\\/g, "/").toLowerCase(); + const base = basename(filePath).toLowerCase(); + + for (const exc of ALLOWED_EXCEPTIONS) { + if (base === exc || normalized.endsWith("/" + exc)) return false; + } + + for (const pattern of SENSITIVE_PATTERNS) { + const lp = pattern.toLowerCase(); + if ( + base === lp || + normalized.endsWith("/" + lp) || + normalized.endsWith(lp) || + normalized.includes("/" + lp + "/") || + base.includes(lp) + ) { + return true; + } + } + + return false; +} + +export function isRemoteOperand(token: string): boolean { + if (/^[a-zA-Z]:[\\/]/.test(token)) return false; + return /^[^/\s:]+(?:@[^/\s:]+)?:/.test(token); +} + +// Extract path-like tokens from a bash command string for scout-block. +export function extractPathTokens(command: string): string[] { + const tokens: string[] = []; + for (const segmentTokens of shellSegments(command)) { + const executable = basename(segmentTokens[0] || ""); + const remoteShell = executable === "ssh" || executable === "tsh"; + if (remoteShell) { + const localPathOptions = new Set(["-i", "-F", "-E", "-S"]); + const optionsWithValue = new Set([ + "-B", + "-b", + "-c", + "-D", + "-e", + "-I", + "-J", + "-L", + "-l", + "-m", + "-O", + "-p", + "-Q", + "-R", + "-W", + "-w", + ]); + for (let i = 1; i < segmentTokens.length; i++) { + const token = segmentTokens[i]; + if (localPathOptions.has(token) && segmentTokens[i + 1]) { + tokens.push(segmentTokens[++i]); + } else if (/^-[iFES].+/.test(token)) { + tokens.push(token.slice(2)); + } else if (token === "-o" && segmentTokens[i + 1]) { + const value = segmentTokens[++i]; + const match = value.match( + /^(?:IdentityFile|UserKnownHostsFile|GlobalKnownHostsFile|CertificateFile)(?:=|\s+)(.+)$/i, + ); + if (match) tokens.push(match[1]); + } else if (/^-o(?:IdentityFile|UserKnownHostsFile|GlobalKnownHostsFile|CertificateFile)=/i.test(token)) { + tokens.push(token.slice(token.indexOf("=") + 1)); + } else if (optionsWithValue.has(token)) { + i += 1; + } else if (!token.startsWith("-")) { + break; + } + } + continue; + } + if (["echo", "printf"].includes(executable)) continue; + const candidates = + executable === "grep" || executable === "sed" || executable === "awk" + ? segmentTokens.slice(2) + : segmentTokens; + for (const token of candidates) { + if (!isRemoteOperand(token)) tokens.push(token); + } + } + return tokens.filter((t) => { + if (!t || t.startsWith("-")) return false; + return t.includes("/") || t.startsWith(".") || /\.[a-z]{1,6}$/.test(t); + }); +} + +// Bash sensitivity classification for privacy-block. +export function bashSensitivity(command: string): "clear" | "ambiguous" | "none" { + const clearOperations = new Set([ + "cat", + "head", + "tail", + "less", + "more", + "grep", + "cp", + "mv", + "scp", + "rsync", + "curl", + "wget", + "tee", + "sed", + "awk", + "chmod", + "chown", + "rm", + "truncate", + "source", + ".", + ]); + + for (const tokens of shellSegments(command)) { + const executable = basename(tokens[0] || "").toLowerCase(); + if (clearOperations.has(executable)) { + const operands = + executable === "grep" || executable === "sed" || executable === "awk" + ? tokens.slice(2) + : tokens.slice(1); + for (const token of operands) { + const operand = token.replace( + /^(?:--upload-file|-T|--data-binary|--data|--data-raw|--output|-o)=?/, + "", + ); + if (operand && !operand.startsWith("-") && !isRemoteOperand(operand) && isSensitive(operand)) + return "clear"; + } + } + for (let i = 0; i < tokens.length - 1; i++) { + if (/^(?:>|>>|<|<<)$/.test(tokens[i]) && isSensitive(tokens[i + 1])) return "clear"; + } + } + + const lower = command.toLowerCase(); + return SENSITIVE_PATTERNS.some((pattern) => lower.includes(pattern.toLowerCase())) + ? "ambiguous" + : "none"; +} + +export { statSync }; diff --git a/pi-extension/src/lib/session-state.ts b/pi-extension/src/lib/session-state.ts new file mode 100644 index 00000000..b26c7185 --- /dev/null +++ b/pi-extension/src/lib/session-state.ts @@ -0,0 +1,129 @@ +// Session state persistence — ported from claude-plugin/hooks/lib/ar-hook-utils.cjs. +// In Claude, hooks received a session_id on stdin. In pi, we derive a stable +// hash from cwd + session id via ExtensionContext. State lives in the OS temp +// dir keyed by that hash, exactly like the original. + +import { existsSync, mkdirSync, readFileSync, writeFileSync, appendFileSync, statSync, readdirSync, unlinkSync } from "node:fs"; +import { homedir, tmpdir } from "node:os"; +import { basename, dirname, join } from "node:path"; +import { createHash } from "node:crypto"; + +export interface SessionState { + projectRoot: string; + plansPath: string; + reportsPath: string; + gitBranch: string; + sessionId: string; + iterationCount: number; + startedAt?: string; + lastContextInjection?: number; +} + +const DEFAULT_STATE = (cwd: string, sessionId: string): SessionState => ({ + projectRoot: cwd, + plansPath: join(cwd, "plans"), + reportsPath: join(cwd, "plans", "reports"), + gitBranch: "", + sessionId, + iterationCount: 0, +}); + +export function sessionHash(cwd: string, sessionId: string): string { + return createHash("md5").update(`${cwd}:${sessionId}`).digest("hex").slice(0, 12); +} + +export function sessionStatePath(cwd: string, sessionId: string): string { + return join(tmpdir(), `ar-session-${sessionHash(cwd, sessionId)}.json`); +} + +export function loadSessionState(cwd: string, sessionId: string): SessionState { + try { + const raw = readFileSync(sessionStatePath(cwd, sessionId), "utf8"); + return { ...DEFAULT_STATE(cwd, sessionId), ...JSON.parse(raw) }; + } catch (error) { + if ((error as NodeJS.ErrnoException).code !== "ENOENT") throw error; + return DEFAULT_STATE(cwd, sessionId); + } +} + +export function saveSessionState(cwd: string, sessionId: string, state: SessionState): void { + writeFileSync(sessionStatePath(cwd, sessionId), JSON.stringify(state, null, 2)); +} + +export function incrementCounter( + cwd: string, + sessionId: string, + field: keyof SessionState, +): number { + const state = loadSessionState(cwd, sessionId); + const next = ((state[field] as number) || 0) + 1; + (state[field] as number) = next; + saveSessionState(cwd, sessionId, state); + return next; +} + +export function cleanupSessionFile(cwd: string, sessionId: string): void { + try { + unlinkSync(sessionStatePath(cwd, sessionId)); + } catch { + /* already gone */ + } +} + +const SESSION_MAX_AGE_MS = 24 * 60 * 60 * 1000; // 24 hours + +export function pruneStaleSessionFiles(): void { + try { + const now = Date.now(); + const tmp = tmpdir(); + for (const entry of readdirSync(tmp)) { + if (!entry.startsWith("ar-session-") || !entry.endsWith(".json")) continue; + const fp = join(tmp, entry); + try { + const stat = statSync(fp); + if (now - stat.mtimeMs > SESSION_MAX_AGE_MS) unlinkSync(fp); + } catch { + /* skip */ + } + } + } catch { + /* tmp dir unreadable */ + } +} + +// Bounded metadata-only runtime log, mirroring the original. Lives under the +// pi config dir (~/.pi/agent/hooks/.logs//hook-log.jsonl). Raw +// paths, commands, tool inputs, and secrets are intentionally excluded. +export function log(hookName: string, entry: Record): void { + try { + const cwd = process.cwd(); + const projectKey = createHash("md5").update(cwd).digest("hex").slice(0, 12); + const logDir = join(homedir(), ".pi", "agent", "hooks", ".logs", projectKey); + mkdirSync(logDir, { recursive: true }); + const logPath = join(logDir, "hook-log.jsonl"); + const safeEntry: Record = {}; + for (const key of [ + "action", + "tool", + "loc", + "duration", + "iterations", + "category", + "remediation", + "matched", + ]) { + if (entry && typeof entry[key] !== "undefined") safeEntry[key] = entry[key]; + } + const record = JSON.stringify({ ts: new Date().toISOString(), hook: hookName, ...safeEntry }); + appendFileSync(logPath, record + "\n"); + } catch { + /* fail-open */ + } +} + +export function isHookEnabled(hookName: string): boolean { + const envKey = "AR_DISABLE_" + hookName.replace(/-/g, "_").toUpperCase(); + return !process.env[envKey]; +} + +export { homedir, basename, dirname, existsSync, readFileSync }; diff --git a/pi-extension/src/lib/shell.ts b/pi-extension/src/lib/shell.ts new file mode 100644 index 00000000..b0106c6c --- /dev/null +++ b/pi-extension/src/lib/shell.ts @@ -0,0 +1,289 @@ +// Shell tokenizer — ported from claude-plugin/hooks/lib/ar-hook-utils.cjs. +// Splits a shell command into segments of tokens, unwrapping wrappers +// (env/sudo/nohup/exec/command/builtin), sh -c, xargs, find -exec, eval, +// and $()/`` substitutions. Used by the guardrail hooks to inspect what a +// bash command actually does without executing it. + +import { basename } from "node:path"; + +export interface ShellSegment extends Array {} + +export interface TokenizeResult { + segments: string[][]; + substitutions: string[]; +} + +function tokenizeShell(command: string): TokenizeResult { + const segments: string[][] = []; + const substitutions: string[] = []; + let tokens: string[] = []; + let token = ""; + let quote = ""; + let lineStart = true; + let heredocDelimiter = ""; + + const pushToken = () => { + if (token) tokens.push(token); + token = ""; + }; + const pushSegment = () => { + pushToken(); + if (tokens.length) segments.push(tokens); + tokens = []; + }; + + for (let i = 0; i < command.length; i++) { + const char = command[i]; + if (heredocDelimiter && lineStart) { + const lineEnd = command.indexOf("\n", i); + const end = lineEnd === -1 ? command.length : lineEnd; + if (command.slice(i, end).trim() === heredocDelimiter) heredocDelimiter = ""; + i = lineEnd === -1 ? command.length : lineEnd; + lineStart = true; + continue; + } + if (char === "\\" && quote !== "'") { + if (i + 1 < command.length) token += command[++i]; + continue; + } + if (char === "'" || char === '"') { + if (!quote) quote = char; + else if (quote === char) quote = ""; + else token += char; + continue; + } + if (char === "$" && command[i + 1] === "(" && quote !== "'") { + let depth = 1; + let inner = ""; + let innerQuote = ""; + i += 2; + for (; i < command.length && depth > 0; i++) { + const current = command[i]; + if (current === "\\" && innerQuote !== "'" && i + 1 < command.length) { + inner += current + command[++i]; + } else if (current === "'" || current === '"') { + if (!innerQuote) innerQuote = current; + else if (innerQuote === current) innerQuote = ""; + inner += current; + } else if (!innerQuote && current === "(") { + depth += 1; + inner += current; + } else if (!innerQuote && current === ")" && --depth === 0) { + break; + } else { + inner += current; + } + } + i -= 1; + substitutions.push(inner); + token += "$()"; + continue; + } + if (char === "`" && quote !== "'") { + let inner = ""; + for (i += 1; i < command.length && command[i] !== "`"; i++) { + if (command[i] === "\\" && i + 1 < command.length) + inner += command[i++] + command[i]; + else inner += command[i]; + } + substitutions.push(inner); + token += "``"; + continue; + } + if (!quote && /[\t ]/.test(char)) { + pushToken(); + continue; + } + if (!quote && char === "#" && !token) { + pushSegment(); + const lineEnd = command.indexOf("\n", i); + if (lineEnd === -1) break; + i = lineEnd - 1; + lineStart = true; + continue; + } + if ( + !quote && + (char === "\n" || + char === ";" || + char === "|" || + char === "&" || + char === "(" || + char === ")") + ) { + pushSegment(); + if (char === "|" && command[i + 1] === "|") i += 1; + if (char === "&" && command[i + 1] === "&") i += 1; + lineStart = char === "\n"; + continue; + } + if (!quote && (char === "<" || char === ">")) { + pushToken(); + let redirect = char; + while (command[i + 1] === char) redirect += command[++i]; + tokens.push(redirect); + if (redirect === "<<") { + let next = i + 1; + while (/[\t ]/.test(command[next] || "")) next += 1; + const match = command + .slice(next) + .match(/^(?:['"]([^'"]+)['"]|([^\s;|&]+))/); + if (match) heredocDelimiter = match[1] || match[2]; + } + continue; + } + token += char; + lineStart = false; + } + pushSegment(); + return { segments, substitutions }; +} + +function unwrapCommand(tokens: string[]): string[] { + let index = 0; + while (index < tokens.length) { + if (/^[A-Za-z_][A-Za-z0-9_]*=/.test(tokens[index])) { + index += 1; + continue; + } + const executable = basename(tokens[index]).toLowerCase(); + if ( + executable === "command" || + executable === "builtin" || + executable === "nohup" || + executable === "exec" + ) { + index += 1; + while (tokens[index] && tokens[index].startsWith("-")) index += 1; + continue; + } + if (executable === "env") { + index += 1; + while (tokens[index]) { + const option = tokens[index]; + if (/^[A-Za-z_][A-Za-z0-9_]*=/.test(option)) index += 1; + else if (option === "-S" || option === "--split-string") return []; + else if (["-u", "--unset", "-C", "--chdir"].includes(option)) index += 2; + else if (option.startsWith("-")) index += 1; + else break; + } + continue; + } + if (executable === "sudo") { + index += 1; + const optionsWithValue = new Set([ + "-C", + "-D", + "-g", + "-h", + "-p", + "-R", + "-r", + "-T", + "-t", + "-U", + "-u", + "--chdir", + "--close-from", + "--group", + "--host", + "--prompt", + "--role", + "--type", + "--user", + ]); + while (tokens[index] && tokens[index].startsWith("-")) { + const option = tokens[index++]; + if (optionsWithValue.has(option)) index += 1; + } + continue; + } + break; + } + return tokens.slice(index); +} + +export function shellSegments(command: string, depth = 0): string[][] { + if (depth > 8 || typeof command !== "string") return []; + const parsed = tokenizeShell(command); + const segments: string[][] = []; + for (const original of parsed.segments) { + if (basename(original[0] || "").toLowerCase() === "env") { + const splitIndex = original.findIndex( + (item) => item === "-S" || item === "--split-string", + ); + if (splitIndex >= 0 && original[splitIndex + 1]) { + segments.push(...shellSegments(original[splitIndex + 1], depth + 1)); + continue; + } + } + let tokens = unwrapCommand(original); + if (!tokens.length) continue; + while ( + tokens.length && + /^(?:if|then|elif|else|fi|while|until|do|done|for|case|esac|select|time|!|\{|\})$/.test( + tokens[0], + ) + ) + tokens = tokens.slice(1); + if (!tokens.length) continue; + segments.push(tokens); + const executable = basename(tokens[0]).toLowerCase(); + if (["sh", "bash", "dash", "zsh", "ksh"].includes(executable)) { + const commandIndex = tokens.findIndex( + (item, index) => + index > 0 && (/^-[^-]*c/.test(item) || item === "--command"), + ); + if (commandIndex >= 0 && tokens[commandIndex + 1]) { + segments.push(...shellSegments(tokens[commandIndex + 1], depth + 1)); + } + } + if (executable === "xargs") { + let commandIndex = 1; + while (tokens[commandIndex] && tokens[commandIndex].startsWith("-")) { + const option = tokens[commandIndex++]; + if ( + [ + "-a", + "--arg-file", + "-d", + "--delimiter", + "-E", + "-I", + "--replace", + "-L", + "--max-lines", + "-n", + "--max-args", + "-P", + "--max-procs", + "-s", + "--max-chars", + ].includes(option) + ) + commandIndex += 1; + } + if (tokens[commandIndex]) segments.push(tokens.slice(commandIndex)); + } + if (executable === "find") { + for (let i = 1; i < tokens.length; i++) { + if ( + (tokens[i] === "-exec" || tokens[i] === "-execdir") && + tokens[i + 1] + ) { + const end = tokens.findIndex( + (item, index) => index > i && (item === ";" || item === "+"), + ); + segments.push(tokens.slice(i + 1, end === -1 ? undefined : end)); + } + } + } + if (executable === "eval" && tokens[1]) { + segments.push(...shellSegments(tokens.slice(1).join(" "), depth + 1)); + } + } + for (const substitution of parsed.substitutions) { + segments.push(...shellSegments(substitution, depth + 1)); + } + return segments; +} diff --git a/pi-extension/src/lib/tsv.ts b/pi-extension/src/lib/tsv.ts new file mode 100644 index 00000000..233ea88e --- /dev/null +++ b/pi-extension/src/lib/tsv.ts @@ -0,0 +1,77 @@ +// TSV helpers — ported from claude-plugin/hooks/lib/ar-hook-utils.cjs. +// readTsvTail returns the header + last N data rows of an iteration results +// TSV. findRecentTsv walks autoresearch/-/ for the most recently +// modified *.tsv within an age window. + +import { readdirSync, readFileSync, statSync } from "node:fs"; +import { join } from "node:path"; + +export interface TsvTail { + header: string; + rows: string[]; + total: number; +} + +export function readTsvTail(filePath: string, n: number): TsvTail | null { + try { + const content = readFileSync(filePath, "utf8"); + const lines = content.split("\n").filter((l) => l.trim()); + const headerLines = lines.filter( + (l) => + l.startsWith("#") || + l.startsWith("iteration\t") || + l.startsWith("iteration|"), + ); + const dataLines = lines.filter( + (l) => + !l.startsWith("#") && + l.trim() && + !l.startsWith("iteration\t") && + !l.startsWith("iteration|"), + ); + const header = headerLines.length > 0 ? headerLines[headerLines.length - 1] : ""; + const tail = dataLines.slice(-n); + return { header, rows: tail, total: dataLines.length }; + } catch { + return null; + } +} + +export function findRecentTsv(cwd: string, maxAgeMinutes: number): string | null { + const maxAge = maxAgeMinutes * 60 * 1000; + const now = Date.now(); + const arDir = join(cwd, "autoresearch"); + let best: string | null = null; + let bestMtime = 0; + + try { + for (const dir of readdirSync(arDir)) { + const subdir = join(arDir, dir); + let stat; + try { + stat = statSync(subdir); + } catch { + continue; + } + if (!stat.isDirectory()) continue; + try { + for (const f of readdirSync(subdir)) { + if (!f.endsWith(".tsv")) continue; + const fp = join(subdir, f); + const fstat = statSync(fp); + const age = now - fstat.mtimeMs; + if (age < maxAge && fstat.mtimeMs > bestMtime) { + best = fp; + bestMtime = fstat.mtimeMs; + } + } + } catch { + continue; + } + } + } catch { + /* no autoresearch dir */ + } + + return best; +} From 329dcfc850cc287e284fdcb002b4027e0ab72836 Mon Sep 17 00:00:00 2001 From: Jean-Baptiste Alleaume Date: Wed, 9 Sep 2026 12:30:55 +0000 Subject: [PATCH 2/6] fix(pi): contribute skill + prompts only via resources_discover Declaring skills and prompts in both the package.json `pi` manifest AND the extension's `resources_discover` handler caused pi to register each resource twice, emitting `collision` diagnostics for every prompt (each colliding with itself) and the skill. Drop `pi.skills`/`pi.prompts` from the manifest; keep only `pi.extensions` (the entry point). The skill + 14 prompt templates are contributed solely via resources_discover at runtime, which works for `pi install`, `pi -e ./pi-extension` (dir), and `pi -e ./src/index.ts` (bare file). Verified /autoresearch_plan and /skill:autoresearch still load correctly with no manifest declaration. README documents the single-contribution mechanism and the skill-name dedup behavior (extension copy wins; stale global/project copies can be removed). --- pi-extension/README.md | 8 +++++++- pi-extension/package.json | 4 +--- 2 files changed, 8 insertions(+), 4 deletions(-) diff --git a/pi-extension/README.md b/pi-extension/README.md index 83678db7..e6137200 100644 --- a/pi-extension/README.md +++ b/pi-extension/README.md @@ -14,7 +14,7 @@ This package keeps the same spirit as the [Claude Code](../claude-plugin) and [O ``` pi-extension/ -├── package.json # pi package manifest (extension + skill + prompts) +├── package.json # pi package manifest (declares the extension entry only) ├── src/ │ ├── index.ts # entry: registers guardrails + resources_discover │ ├── guardrails.ts # wires hooks → pi events @@ -51,6 +51,12 @@ Every hook **fails open** — a guardrail malfunction never blocks legitimate wo > **Note on subagent context:** Claude's `SubagentStart` hook injected iteration state into subagents. pi subagents inherit the loaded autoresearch skill, so the active-iteration context already flows via the skill + `iteration-context` on the parent. There is no direct subagent-spawn event in pi's extension API, so the `subagent-context` hook is not ported. +### Resource contribution + +The skill and prompt templates are contributed **only** via the extension's `resources_discover` event (not duplicated in the `package.json` `pi` manifest). Declaring them in both places causes pi to register each resource twice and emit `collision` diagnostics. The manifest declares just the extension entry (`pi.extensions`); `resources_discover` supplies the skill + prompts at runtime. This works for `pi install`, `pi -e ./pi-extension` (dir), and `pi -e ./src/index.ts` (bare file). + +> **Skill-name collisions:** if you already have the autoresearch skill installed elsewhere (e.g. a global `~/.pi/agent/skills/autoresearch/` copy, or the repo's own `.agents/skills/autoresearch/`), pi dedupes by name — the extension's copy wins and the others are skipped. The skipped entries show as `collision` diagnostics in the startup banner. To silence them, remove the stale copies you no longer need (the extension bundles its own). + ## Commands The 14 autoresearch commands are exposed as pi **prompt templates** (markdown protocols, invoked via `/name`) and the dispatcher is a pi **skill** (`/skill:autoresearch`): diff --git a/pi-extension/package.json b/pi-extension/package.json index 6f526de1..822de6f4 100644 --- a/pi-extension/package.json +++ b/pi-extension/package.json @@ -21,8 +21,6 @@ "agent" ], "pi": { - "extensions": ["./src/index.ts"], - "skills": ["./skills/autoresearch/SKILL.md"], - "prompts": ["./prompts"] + "extensions": ["./src/index.ts"] } } From 4929a77b7e6cebcda6bf168f3adfffdff74a78eb Mon Sep 17 00:00:00 2001 From: Jean-Baptiste Alleaume Date: Wed, 9 Sep 2026 12:44:46 +0000 Subject: [PATCH 3/6] fix(pi): move hook log dir out of deprecated ~/.pi/agent/hooks/ MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The extension wrote its bounded runtime log to ~/.pi/agent/hooks/.logs/, which created a legacy hooks/ directory and triggered pi's migration warning ("Global hooks/ directory found. Hooks have been renamed to extensions."). Move the log to ~/.pi/agent/autoresearch/.logs//hook-log.jsonl — a namespaced dir that does not trip the hooks→extensions deprecation. Verified the hooks/ directory is no longer created and guardrails still fire + log. --- pi-extension/README.md | 2 +- pi-extension/src/lib/session-state.ts | 8 +++++--- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/pi-extension/README.md b/pi-extension/README.md index e6137200..6d88365f 100644 --- a/pi-extension/README.md +++ b/pi-extension/README.md @@ -142,7 +142,7 @@ Per-hook disable flags (same as the Claude plugin): | `AR_DISABLE_STOP_NOTIFY` | session-end notification | | `AR_NOTIFY_WEBHOOK` | optional webhook URL for session-end notifications | -Runtime logs (bounded metadata only — no paths, commands, or secrets) are written to `~/.pi/agent/hooks/.logs//hook-log.jsonl`. +Runtime logs (bounded metadata only — no paths, commands, or secrets) are written to `~/.pi/agent/autoresearch/.logs//hook-log.jsonl`. ## License diff --git a/pi-extension/src/lib/session-state.ts b/pi-extension/src/lib/session-state.ts index b26c7185..e4e6bbe2 100644 --- a/pi-extension/src/lib/session-state.ts +++ b/pi-extension/src/lib/session-state.ts @@ -92,13 +92,15 @@ export function pruneStaleSessionFiles(): void { } // Bounded metadata-only runtime log, mirroring the original. Lives under the -// pi config dir (~/.pi/agent/hooks/.logs//hook-log.jsonl). Raw -// paths, commands, tool inputs, and secrets are intentionally excluded. +// pi config dir (~/.pi/agent/autoresearch/.logs//hook-log.jsonl). +// NOTE: do NOT use ~/.pi/agent/hooks/ — pi renamed hooks→extensions and warns +// on a legacy hooks/ directory, so we use a namespaced autoresearch/ dir instead. +// Raw paths, commands, tool inputs, and secrets are intentionally excluded. export function log(hookName: string, entry: Record): void { try { const cwd = process.cwd(); const projectKey = createHash("md5").update(cwd).digest("hex").slice(0, 12); - const logDir = join(homedir(), ".pi", "agent", "hooks", ".logs", projectKey); + const logDir = join(homedir(), ".pi", "agent", "autoresearch", ".logs", projectKey); mkdirSync(logDir, { recursive: true }); const logPath = join(logDir, "hook-log.jsonl"); const safeEntry: Record = {}; From f64b4b8316b833c53640cbb2a094ad9f15625455 Mon Sep 17 00:00:00 2001 From: Jean-Baptiste Alleaume Date: Wed, 9 Sep 2026 13:16:48 +0000 Subject: [PATCH 4/6] =?UTF-8?q?feat(pi):=20complete=20integration=20?= =?UTF-8?q?=E2=80=94=20installer,=20tests,=20CI,=20docs?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Wire the pi extension into the same distribution, test, and release surface as Claude/OpenCode/Codex: - scripts/install.sh: add `--pi` mode (global/local). Copies the pi-extension/ bundle to /extensions/autoresearch-pi so pi auto-discovers it. Honors PI_AGENT_DIR for the config dir; pi-aware overwrite confirmation. - tests/test-pi.sh: 53 cases covering the ported guardrails (shell tokenizer, ignore matcher, dangerous-cmd-block, scout-block, privacy-block, simplify-gate, iteration-context, env-flag disables) + a tsc --strict type-check against pi-coding-agent types. Gracefully skips the type-check when pi types aren't installed. All 53 pass; cleans up its own artifacts. - .github/workflows/release-readiness.yml: install pi-coding-agent and run tests/test-pi.sh on every OS in the conformance matrix. - scripts/release.sh + scripts/release.md: add test-pi.sh to the release gate. - README.md: pi badge, intro, hook-parity note, and a pi Quick Start section (guided install / pi install / try-without-installing). All five test suites green: test-pi 53, test-hooks 228, test-orchestrator 195, test-regression 65, test-maintenance 50. --- .github/workflows/release-readiness.yml | 4 + README.md | 29 +- scripts/install.sh | 36 +- scripts/release.md | 3 +- scripts/release.sh | 2 +- tests/test-pi.sh | 480 ++++++++++++++++++++++++ 6 files changed, 544 insertions(+), 10 deletions(-) create mode 100755 tests/test-pi.sh diff --git a/.github/workflows/release-readiness.yml b/.github/workflows/release-readiness.yml index f7736260..c890bbb6 100644 --- a/.github/workflows/release-readiness.yml +++ b/.github/workflows/release-readiness.yml @@ -37,6 +37,10 @@ jobs: test -z "$(git status --porcelain)" - name: Verify Claude guardrails run: bash tests/test-hooks.sh + - name: Verify pi extension guardrails + run: | + npm install -g @earendil-works/pi-coding-agent + bash tests/test-pi.sh - name: Verify orchestration and clean installs run: bash tests/test-orchestrator.sh - name: Verify regression scoring diff --git a/README.md b/README.md index 5b672c2e..fd073ae5 100644 --- a/README.md +++ b/README.md @@ -2,13 +2,14 @@ # Autoresearch -**Turn [Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), or [OpenAI Codex](https://developers.openai.com/codex) into a relentless improvement engine.** +**Turn [Claude Code](https://docs.anthropic.com/en/docs/claude-code), [OpenCode](https://opencode.ai), [OpenAI Codex](https://developers.openai.com/codex), or [pi](https://github.com/earendil-works/pi-coding-agent) into a relentless improvement engine.** Based on [Karpathy's autoresearch](https://github.com/karpathy/autoresearch) — constraint + mechanical metric + autonomous iteration = compounding gains. [![Claude Code Skill](https://img.shields.io/badge/Claude_Code-Skill-blue?logo=anthropic&logoColor=white)](https://docs.anthropic.com/en/docs/claude-code) [![OpenCode](https://img.shields.io/badge/OpenCode-Skill-purple)](https://opencode.ai) [![Codex](https://img.shields.io/badge/Codex-Skill-green?logo=openai&logoColor=white)](https://developers.openai.com/codex) +[![pi](https://img.shields.io/badge/pi-Extension-teal)](https://github.com/earendil-works/pi-coding-agent) [![Version](https://img.shields.io/badge/version-2.2.2-blue.svg)](https://github.com/uditgoenka/autoresearch/releases) [![License: MIT](https://img.shields.io/badge/License-MIT-green.svg)](LICENSE) @@ -22,7 +23,7 @@ Based on [Karpathy's autoresearch](https://github.com/karpathy/autoresearch) — *You don't need AGI. You need a goal, a metric, and a loop that never quits.* -**Supports Claude Code, OpenCode, and OpenAI Codex for the core skill, bundled runtime, installation, and verification surface. Hook guardrails are Claude Code-only.** +**Supports Claude Code, OpenCode, OpenAI Codex, and pi for the core skill, bundled runtime, installation, and verification surface. Hook guardrails ship for Claude Code (native hooks) and pi (extension events).** > **v2.2.0 — Autonomous Orchestrator:** Type a plain-language goal to `/autoresearch` and it classifies your goal, derives a Success predicate, confirms it once, then loops across subcommands until done. No manual chaining required. `Metric:`/`Verify:` invocations run the classic loop unchanged. See [guide/autoresearch-orchestrator.md](guide/autoresearch-orchestrator.md). @@ -116,7 +117,7 @@ Before looping, Claude performs a one-time setup: ## Hooks & Safety -Hooks are defense-in-depth guardrails, not a security sandbox. Claude Code ships the hook surface; OpenCode and Codex ship the core skill/runtime/install surface without hook parity. +Hooks are defense-in-depth guardrails, not a security sandbox. Claude Code ships the hook surface natively; pi ships the same guardrails as extension events (`tool_call`, `before_agent_start`, `input`, `session_start`, `session_shutdown`). OpenCode and Codex ship the core skill/runtime/install surface without hook parity. ### What's Protected @@ -314,6 +315,28 @@ cp -r autoresearch/.agents/skills/autoresearch ~/.codex/skills/autoresearch > Invoke via `$autoresearch` mention syntax. Subcommands are keywords: `$autoresearch plan`, `$autoresearch debug`, `$autoresearch evals`, etc. > The installed Codex package includes the bundled orchestrator and regression helpers under `plugins/autoresearch/skills/autoresearch/` and `.agents/skills/autoresearch/`. +### pi Quick Start + +**Option A — Guided installer (recommended):** +```bash +git clone https://github.com/uditgoenka/autoresearch.git +cd autoresearch +./scripts/install.sh --pi --global +``` + +**Option B — pi package install:** +```bash +pi install git:github.com/uditgoenka/autoresearch +``` + +**Option C — Try without installing:** +```bash +pi -e ./pi-extension +``` + +> Invoke commands as pi prompt templates: `/autoresearch`, `/autoresearch_debug`, `/autoresearch_fix`, … (underscore names). The dispatcher is a skill: `/skill:autoresearch`. +> The pi extension ports the Claude hook guardrails (scout/privacy/dangerous-cmd block, iteration context, simplify gate, stop-notify) to pi extension events. See [`pi-extension/README.md`](pi-extension/README.md) for the full hook→event mapping and `AR_DISABLE_*` flags. + ### Run It ``` diff --git a/scripts/install.sh b/scripts/install.sh index 8b69ed16..d2193b88 100755 --- a/scripts/install.sh +++ b/scripts/install.sh @@ -22,6 +22,7 @@ Options: --claude Install for Claude Code --opencode Install for OpenCode --codex Install for OpenAI Codex + --pi Install for pi (coding agent) -g, --global Install globally -l, --local Install in the current project -c, --config-dir Override the global config directory @@ -33,6 +34,7 @@ Examples: ./scripts/install.sh --claude --global ./scripts/install.sh --opencode --local ./scripts/install.sh --codex --global + ./scripts/install.sh --pi --global EOF } @@ -61,6 +63,9 @@ parse_args() { --codex) if [[ -n "$TOOL" && "$TOOL" != "codex" ]]; then die "choose only one tool"; fi TOOL="codex" ;; + --pi) + if [[ -n "$TOOL" && "$TOOL" != "pi" ]]; then die "choose only one tool"; fi + TOOL="pi" ;; -g|--global) if [[ -n "$LOCATION" && "$LOCATION" != "global" ]]; then die "choose --global or --local"; fi LOCATION="global" ;; @@ -103,6 +108,9 @@ get_global_dir() { codex) if [[ -n "${CODEX_HOME:-}" ]]; then expand_path "$CODEX_HOME" else printf '%s\n' "$HOME/.codex"; fi ;; + pi) + if [[ -n "${PI_AGENT_DIR:-}" ]]; then expand_path "$PI_AGENT_DIR" + else printf '%s\n' "$HOME/.pi/agent"; fi ;; esac } @@ -113,6 +121,7 @@ get_target_dir() { claude) printf '%s\n' "$PWD/.claude" ;; opencode) printf '%s\n' "$PWD/.opencode" ;; codex) printf '%s\n' "$PWD/.codex" ;; + pi) printf '%s\n' "$PWD/.pi" ;; esac return fi @@ -121,12 +130,13 @@ get_target_dir() { prompt_tool() { local answer - printf 'Select the tool to install:\n 1) Claude Code\n 2) OpenCode\n 3) OpenAI Codex\nChoice [1]: ' + printf 'Select the tool to install:\n 1) Claude Code\n 2) OpenCode\n 3) OpenAI Codex\n 4) pi (coding agent)\nChoice [1]: ' read -r answer || cancelled case "${answer:-1}" in 1) TOOL="claude" ;; 2) TOOL="opencode" ;; 3) TOOL="codex" ;; + 4) TOOL="pi" ;; *) die "invalid selection: $answer" ;; esac } @@ -134,7 +144,7 @@ prompt_tool() { prompt_location() { local global_dir answer local_dir global_dir="$(get_global_dir "$TOOL")" - case "$TOOL" in claude) local_dir="$PWD/.claude" ;; opencode) local_dir="$PWD/.opencode" ;; codex) local_dir="$PWD/.codex" ;; esac + case "$TOOL" in claude) local_dir="$PWD/.claude" ;; opencode) local_dir="$PWD/.opencode" ;; codex) local_dir="$PWD/.codex" ;; pi) local_dir="$PWD/.pi" ;; esac printf 'Install location:\n 1) Global (%s)\n 2) Local (%s)\nChoice [1]: ' "$global_dir" "$local_dir" read -r answer || cancelled case "${answer:-1}" in @@ -162,9 +172,14 @@ sync_dir() { sync_file() { mkdir -p "$(dirname "$2")"; cp "$1" "$2"; } confirm_overwrite() { - local target_root="$1" + local target_root="$1" existing + if [[ "$TOOL" == "pi" ]]; then + existing="$target_root/extensions/autoresearch-pi" + else + existing="$target_root/skills/autoresearch" + fi if [[ $FORCE -eq 1 ]]; then return 0; fi - if [[ ! -d "$target_root/skills/autoresearch" ]]; then return 0; fi + if [[ ! -d "$existing" ]]; then return 0; fi if ! is_interactive; then return 0; fi local answer printf 'Existing autoresearch files found in %s. Replace? [Y/n]: ' "$target_root" @@ -285,6 +300,15 @@ install_codex() { sync_dir "$REPO_ROOT/.agents/skills/autoresearch" "$t/skills/autoresearch" } +install_pi() { + # pi auto-discovers extensions under /extensions//index.ts. + # We copy the whole pi-extension/ bundle (src + skills + prompts) so the + # extension is self-contained: guardrails + skill + 14 prompt templates. + local t="$1" + mkdir -p "$t/extensions" + sync_dir "$REPO_ROOT/pi-extension" "$t/extensions/autoresearch-pi" +} + main() { parse_args "$@" ensure_context @@ -299,17 +323,19 @@ main() { confirm_overwrite "$target_root" local label - case "$TOOL" in claude) label="Claude Code" ;; opencode) label="OpenCode" ;; codex) label="OpenAI Codex" ;; esac + case "$TOOL" in claude) label="Claude Code" ;; opencode) label="OpenCode" ;; codex) label="OpenAI Codex" ;; pi) label="pi (coding agent)" ;; esac printf 'Installing Autoresearch for %s (%s)\nTarget: %s\n' "$label" "$LOCATION" "$target_root" case "$TOOL" in claude) install_claude "$target_root" ;; opencode) install_opencode "$target_root" ;; codex) install_codex "$target_root" ;; + pi) install_pi "$target_root" ;; esac case "$TOOL" in codex) printf 'Done. Use $autoresearch in Codex to start.\n' ;; + pi) printf 'Done. Run /autoresearch in pi to start. (Restart pi or run /reload if already open.)\n' ;; *) printf 'Done. Run /autoresearch to start.\n' ;; esac } diff --git a/scripts/release.md b/scripts/release.md index 96f83593..ae191231 100644 --- a/scripts/release.md +++ b/scripts/release.md @@ -24,7 +24,7 @@ The preparation script stops after the PR is opened. It does not merge, tag, or | Identity and workspace | clean tree, `master`, `gh`, `uditgoenka` Git author, `uditgoenka` GitHub login | | Transform cleanliness | `bash scripts/transform.sh`, then `git diff --exit-code` and `git status --porcelain` | | Version alignment | `claude-plugin/.claude-plugin/plugin.json`, `.claude-plugin/marketplace.json`, `.claude/skills/autoresearch/SKILL.md`, `README.md`, `guide/README.md`, and the generated `SKILL.md` mirrors | -| Release suites | `bash tests/test-hooks.sh`, `bash tests/test-orchestrator.sh`, `bash tests/test-regression.sh`, `bash tests/test-maintenance.sh` | +| Release suites | `bash tests/test-hooks.sh`, `bash tests/test-pi.sh`, `bash tests/test-orchestrator.sh`, `bash tests/test-regression.sh`, `bash tests/test-maintenance.sh` | | Clean-install smoke | disposable installs for Claude, OpenCode, and Codex run bundled `scripts/orchestrate.sh classify`, `scripts/score-regression.sh verdict`, and `scripts/score-regression.sh rubric` from outside the source checkout | | Publication boundary | PR creation only; merge, tag creation, and GitHub release creation are separate explicit owner actions | @@ -37,6 +37,7 @@ Before running the script, verify: - [ ] The working tree is clean and on `master` - [ ] `scripts/orchestrate.sh` and `scripts/score-regression.sh` are executable in every installed bundle - [ ] `bash tests/test-hooks.sh` +- [ ] `bash tests/test-pi.sh` - [ ] `bash tests/test-orchestrator.sh` - [ ] `bash tests/test-regression.sh` - [ ] `bash tests/test-maintenance.sh` diff --git a/scripts/release.sh b/scripts/release.sh index 75929916..3cd5dac8 100755 --- a/scripts/release.sh +++ b/scripts/release.sh @@ -102,7 +102,7 @@ const checks = [ if (checks.some(([actual, expected]) => actual !== expected)) process.exit(1); NODE -for suite in test-hooks.sh test-orchestrator.sh test-regression.sh test-maintenance.sh; do +for suite in test-hooks.sh test-pi.sh test-orchestrator.sh test-regression.sh test-maintenance.sh; do bash "tests/$suite" done diff --git a/tests/test-pi.sh b/tests/test-pi.sh new file mode 100755 index 00000000..369fec21 --- /dev/null +++ b/tests/test-pi.sh @@ -0,0 +1,480 @@ +#!/usr/bin/env bash +# Tests for the pi extension guardrails — the TypeScript ports of the Claude hooks. +# +# These exercise the pure-logic functions exported by pi-extension/src/ (shell +# tokenizer, ignore matcher, dangerous-cmd-block, scout-block, privacy-block, +# simplify-gate, iteration-context) via tsx, mirroring the behaviors asserted +# in test-hooks.sh for the Claude .cjs hooks. They do NOT spin up a pi session. +set -euo pipefail + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)" +PI_EXT="$REPO_ROOT/pi-extension" + +PASS=0 +FAIL=0 +TOTAL=0 + +# Resolve a tsx binary (bundled with pi-coding-agent's node_modules if available, +# else npx). tsx transpiles TS on the fly like pi's jiti loader. +TSX="" +if [[ -x "$REPO_ROOT/node_modules/.bin/tsx" ]]; then + TSX="$REPO_ROOT/node_modules/.bin/tsx" +elif command -v tsx >/dev/null 2>&1; then + TSX="tsx" +else + # Fall back to npx (cached). Pinned to avoid network on every run. + TSX="npx -y tsx@4" +fi + +# Run a TS test snippet that imports from the extension and prints a single +# line: "PASS" or "FAIL: ". The snippet must call process.exit with 0/1. +run_case() { + local label="$1" + local code="$2" + TOTAL=$((TOTAL + 1)) + local result + result=$(printf '%s' "$code" | (cd "$PI_EXT" && $TSX -e "$(cat)" 2>&1) ) || true + if [[ "$result" == "PASS" ]]; then + printf ' PASS: %s\n' "$label" + PASS=$((PASS + 1)) + else + printf ' FAIL: %s — %s\n' "$label" "$result" + FAIL=$((FAIL + 1)) + fi +} + +# Convenience: assert a JS expression is truthy. The expression runs in a module +# that imports the named symbols. +ok() { + local label="$1" + local imports="$2" + local expr="$3" + run_case "$label" " +import { $imports } from \"./src/lib/shell.ts\"; // placeholder, replaced below +" + # The above placeholder approach is awkward; use a direct inline module instead. +} + +# We use a cleaner pattern: each case is a self-contained inline TS program. +case_ok() { + local label="$1" + local program="$2" + TOTAL=$((TOTAL + 1)) + local result + result=$(printf '%s' "$program" | (cd "$PI_EXT" && $TSX --eval "$program" 2>&1)) || true + if [[ "$result" == "PASS" ]]; then + printf ' PASS: %s\n' "$label" + PASS=$((PASS + 1)) + else + printf ' FAIL: %s — %s\n' "$label" "$result" + FAIL=$((FAIL + 1)) + fi +} + +# ============================================================================ +printf '\n--- Testing shell tokenizer (shell.ts) ---\n' +# ============================================================================ + +case_ok "shellSegments: git push -f" ' +import { shellSegments } from "./src/lib/shell.ts"; +const s = shellSegments("git push -f origin master")[0].join(" "); +process.stdout.write(s === "git push -f origin master" ? "PASS" : "FAIL: " + s); +' + +case_ok "shellSegments: sudo rm unwraps" ' +import { shellSegments } from "./src/lib/shell.ts"; +const s = shellSegments("sudo rm -rf /tmp/x")[0].join(" "); +process.stdout.write(s === "rm -rf /tmp/x" ? "PASS" : "FAIL: " + s); +' + +case_ok "shellSegments: env VAR=1 unwraps" ' +import { shellSegments } from "./src/lib/shell.ts"; +const s = shellSegments("env FOO=bar node x.js")[0].join(" "); +process.stdout.write(s === "node x.js" ? "PASS" : "FAIL: " + s); +' + +case_ok "shellSegments: sh -c nested" ' +import { shellSegments } from "./src/lib/shell.ts"; +const segs = shellSegments("bash -c \x27git reset --hard\x27"); +const s = segs[1] && segs[1].join(" "); +process.stdout.write(s === "git reset --hard" ? "PASS" : "FAIL: " + s); +' + +case_ok "shellSegments: command substitution inspected" ' +import { shellSegments } from "./src/lib/shell.ts"; +const segs = shellSegments("echo $(git push -f)"); +process.stdout.write(segs.length >= 1 ? "PASS" : "FAIL: no segments"); +' + +# ============================================================================ +printf '\n--- Testing dangerous-cmd-block ---\n' +# ============================================================================ + +case_ok "dangerous: blocks git push --force" ' +import { commandLabel } from "./src/hooks/dangerous-cmd-block.ts"; +const r = commandLabel("git push --force"); +process.stdout.write(r === "forced git push" ? "PASS" : "FAIL: " + r); +' + +case_ok "dangerous: blocks git push -f" ' +import { commandLabel } from "./src/hooks/dangerous-cmd-block.ts"; +const r = commandLabel("git push -f origin main"); +process.stdout.write(r === "forced git push" ? "PASS" : "FAIL: " + r); +' + +case_ok "dangerous: allows regular git push" ' +import { commandLabel } from "./src/hooks/dangerous-cmd-block.ts"; +const r = commandLabel("git push origin main"); +process.stdout.write(r === null ? "PASS" : "FAIL: " + r); +' + +case_ok "dangerous: blocks git reset --hard" ' +import { commandLabel } from "./src/hooks/dangerous-cmd-block.ts"; +const r = commandLabel("git reset --hard HEAD~1"); +process.stdout.write(r === "hard git reset" ? "PASS" : "FAIL: " + r); +' + +case_ok "dangerous: blocks rm -rf" ' +import { commandLabel } from "./src/hooks/dangerous-cmd-block.ts"; +const r = commandLabel("rm -rf /"); +process.stdout.write(r === "recursive forced removal" ? "PASS" : "FAIL: " + r); +' + +case_ok "dangerous: blocks git clean -f" ' +import { commandLabel } from "./src/hooks/dangerous-cmd-block.ts"; +const r = commandLabel("git clean -f"); +process.stdout.write(r === "forced git clean" ? "PASS" : "FAIL: " + r); +' + +case_ok "dangerous: blocks git branch -D" ' +import { commandLabel } from "./src/hooks/dangerous-cmd-block.ts"; +const r = commandLabel("git branch -D feature"); +process.stdout.write(r === "forced branch deletion" ? "PASS" : "FAIL: " + r); +' + +case_ok "dangerous: blocks git checkout ." ' +import { commandLabel } from "./src/hooks/dangerous-cmd-block.ts"; +const r = commandLabel("git checkout . "); +process.stdout.write(r === "git checkout of working tree" ? "PASS" : "FAIL: " + r); +' + +case_ok "dangerous: allows git status" ' +import { commandLabel } from "./src/hooks/dangerous-cmd-block.ts"; +const r = commandLabel("git status"); +process.stdout.write(r === null ? "PASS" : "FAIL: " + r); +' + +case_ok "dangerous: blocks force-with-lease" ' +import { commandLabel } from "./src/hooks/dangerous-cmd-block.ts"; +const r = commandLabel("git push origin main --force-with-lease"); +process.stdout.write(r === "forced git push" ? "PASS" : "FAIL: " + r); +' + +case_ok "dangerous: harmless echoed text allowed" ' +import { commandLabel } from "./src/hooks/dangerous-cmd-block.ts"; +const r = commandLabel("echo rm -rf /"); +process.stdout.write(r === null ? "PASS" : "FAIL: " + r); +' + +case_ok "dangerous: sudo wrapper cannot hide hard reset" ' +import { commandLabel } from "./src/hooks/dangerous-cmd-block.ts"; +const r = commandLabel("sudo git reset --hard HEAD"); +process.stdout.write(r === "hard git reset" ? "PASS" : "FAIL: " + r); +' + +case_ok "dangerous: comment text not executed" ' +import { commandLabel } from "./src/hooks/dangerous-cmd-block.ts"; +const r = commandLabel("true # ; git reset --hard HEAD"); +process.stdout.write(r === null ? "PASS" : "FAIL: " + r); +' + +case_ok "dangerous: checkDangerousCommand block result" ' +import { checkDangerousCommand } from "./src/hooks/dangerous-cmd-block.ts"; +const r = checkDangerousCommand("git push -f origin master"); +process.stdout.write(r.block === true ? "PASS" : "FAIL: not blocked"); +' + +case_ok "dangerous: disabled via env var" ' +delete process.env.AR_DISABLE_DANGEROUS_CMD_BLOCK; +process.env.AR_DISABLE_DANGEROUS_CMD_BLOCK = "1"; +import { checkDangerousCommand } from "./src/hooks/dangerous-cmd-block.ts"; +const r = checkDangerousCommand("git push -f origin master"); +process.stdout.write(r.block ? "FAIL: still blocked when disabled" : "PASS"); +' + +# ============================================================================ +printf '\n--- Testing ignore matcher (ignore.ts) ---\n' +# ============================================================================ + +case_ok "ignore: node_modules ignored" ' +import { ignore } from "./src/lib/ignore.ts"; +const ig = ignore().add(["node_modules/", "*.log"]); +process.stdout.write(ig.ignores("node_modules/foo.js") ? "PASS" : "FAIL"); +' + +case_ok "ignore: *.log ignored" ' +import { ignore } from "./src/lib/ignore.ts"; +const ig = ignore().add(["*.log"]); +process.stdout.write(ig.ignores("debug.log") ? "PASS" : "FAIL"); +' + +case_ok "ignore: negation un-ignores" ' +import { ignore } from "./src/lib/ignore.ts"; +const ig = ignore().add(["node_modules/", "!node_modules/keep/"]); +process.stdout.write(!ig.ignores("node_modules/keep/x.js") ? "PASS" : "FAIL"); +' + +case_ok "ignore: src allowed" ' +import { ignore } from "./src/lib/ignore.ts"; +const ig = ignore().add(["node_modules/"]); +process.stdout.write(!ig.ignores("src/index.ts") ? "PASS" : "FAIL"); +' + +# ============================================================================ +printf '\n--- Testing scout-block (.ckignore) ---\n' +# ============================================================================ + +case_ok "scout: blocks node_modules Read" ' +import { checkStructuredPath } from "./src/hooks/scout-block.ts"; +const r = checkStructuredPath("node_modules/express/index.js", "/tmp"); +process.stdout.write(r.block === true ? "PASS" : "FAIL"); +' + +case_ok "scout: allows normal file Read" ' +import { checkStructuredPath } from "./src/hooks/scout-block.ts"; +const r = checkStructuredPath("src/main.ts", "/tmp"); +process.stdout.write(!r.block ? "PASS" : "FAIL: blocked"); +' + +case_ok "scout: blocks .git access" ' +import { checkStructuredPath } from "./src/hooks/scout-block.ts"; +const r = checkStructuredPath(".git/config", "/tmp"); +process.stdout.write(r.block === true ? "PASS" : "FAIL"); +' + +case_ok "scout: blocks Bash with node_modules path" ' +import { checkBashCommand } from "./src/hooks/scout-block.ts"; +const r = checkBashCommand("cat node_modules/foo/bar.js", "/tmp"); +process.stdout.write(r.block === true ? "PASS" : "FAIL"); +' + +case_ok "scout: allows build tool (npm)" ' +import { checkBashCommand } from "./src/hooks/scout-block.ts"; +const r = checkBashCommand("npm test", "/tmp"); +process.stdout.write(!r.block ? "PASS" : "FAIL: blocked"); +' + +case_ok "scout: bash string literal false-positive prevention" ' +import { checkBashCommand } from "./src/hooks/scout-block.ts"; +const r = checkBashCommand("echo \x27testing node_modules string\x27", "/tmp"); +process.stdout.write(!r.block ? "PASS" : "FAIL: blocked"); +' + +case_ok "scout: remote ssh path not local" ' +import { checkBashCommand } from "./src/hooks/scout-block.ts"; +const r = checkBashCommand("ssh deploy@prod cat node_modules/server.js", "/tmp"); +process.stdout.write(!r.block ? "PASS" : "FAIL: blocked remote"); +' + +case_ok "scout: disabled via env var" ' +process.env.AR_DISABLE_SCOUT_BLOCK = "1"; +import { checkBashCommand } from "./src/hooks/scout-block.ts"; +const r = checkBashCommand("cat node_modules/foo.js", "/tmp"); +process.stdout.write(!r.block ? "PASS" : "FAIL: still blocked"); +' + +# ============================================================================ +printf '\n--- Testing privacy-block (sensitive files) ---\n' +# ============================================================================ + +case_ok "privacy: .env sensitive" ' +import { isSensitive } from "./src/lib/paths.ts"; +process.stdout.write(isSensitive(".env") ? "PASS" : "FAIL"); +' + +case_ok "privacy: .env.example allowed" ' +import { isSensitive } from "./src/lib/paths.ts"; +process.stdout.write(!isSensitive(".env.example") ? "PASS" : "FAIL"); +' + +case_ok "privacy: id_rsa sensitive" ' +import { isSensitive } from "./src/lib/paths.ts"; +process.stdout.write(isSensitive("~/.ssh/id_rsa") ? "PASS" : "FAIL"); +' + +case_ok "privacy: credentials.json sensitive" ' +import { isSensitive } from "./src/lib/paths.ts"; +process.stdout.write(isSensitive("credentials.json") ? "PASS" : "FAIL"); +' + +case_ok "privacy: src normal" ' +import { isSensitive } from "./src/lib/paths.ts"; +process.stdout.write(!isSensitive("src/index.ts") ? "PASS" : "FAIL"); +' + +case_ok "privacy: cat .env is clear" ' +import { bashSensitivity } from "./src/lib/paths.ts"; +const r = bashSensitivity("cat .env"); +process.stdout.write(r === "clear" ? "PASS" : "FAIL: " + r); +' + +case_ok "privacy: echo src none" ' +import { bashSensitivity } from "./src/lib/paths.ts"; +const r = bashSensitivity("echo src/index.ts"); +process.stdout.write(r === "none" ? "PASS" : "FAIL: " + r); +' + +case_ok "privacy: checkStructuredSensitive needsConfirm" ' +import { checkStructuredSensitive } from "./src/hooks/privacy-block.ts"; +const r = checkStructuredSensitive(".env"); +process.stdout.write(r.needsConfirm === true ? "PASS" : "FAIL"); +' + +case_ok "privacy: disabled via env var" ' +process.env.AR_DISABLE_PRIVACY_BLOCK = "1"; +import { checkStructuredSensitive } from "./src/hooks/privacy-block.ts"; +const r = checkStructuredSensitive(".env"); +process.stdout.write(!r.needsConfirm ? "PASS" : "FAIL: still asking"); +' + +# ============================================================================ +printf '\n--- Testing simplify-gate ---\n' +# ============================================================================ + +# simplify-gate needs a git repo to count LOC. Build a throwaway repo. +GATE_REPO="$(mktemp -d)" +trap 'rm -rf "$GATE_REPO"' EXIT +( cd "$GATE_REPO" && git init -q && git config user.name t && git config user.email t@x && printf "x\n" > a.txt && git add a.txt && git commit -qm init ) >/dev/null 2>&1 + +case_ok "simplify-gate: no shipping verb allows" " +import { checkSimplifyGate } from \"./src/hooks/simplify-gate.ts\"; +const r = checkSimplifyGate('fix the bug', '$GATE_REPO'); +process.stdout.write(!r.block ? 'PASS' : 'FAIL: blocked'); +" + +case_ok "simplify-gate: small diff allows shipping" " +import { checkSimplifyGate } from \"./src/hooks/simplify-gate.ts\"; +const r = checkSimplifyGate('ship this', '$GATE_REPO'); +process.stdout.write(!r.block ? 'PASS' : 'FAIL: blocked'); +" + +case_ok "simplify-gate: negation phrase allows" " +import { checkSimplifyGate } from \"./src/hooks/simplify-gate.ts\"; +const r = checkSimplifyGate(\"don't ship yet\", '$GATE_REPO'); +process.stdout.write(!r.block ? 'PASS' : 'FAIL: blocked'); +" + +case_ok "simplify-gate: large diff blocks" " +import { checkSimplifyGate } from \"./src/hooks/simplify-gate.ts\"; +const fs = require('fs'); +// Append 811 lines to a TRACKED file so git diff HEAD --numstat (cwd-aware) counts them. +fs.appendFileSync('$GATE_REPO/a.txt', '\\n' + Array(811).fill('b').join('\\n')); +const r = checkSimplifyGate('release now', '$GATE_REPO'); +process.stdout.write(r.block === true ? 'PASS' : 'FAIL: not blocked'); +" + +case_ok "simplify-gate: disabled via env var" " +process.env.AR_DISABLE_SIMPLIFY_GATE = '1'; +import { checkSimplifyGate } from \"./src/hooks/simplify-gate.ts\"; +const r = checkSimplifyGate('ship this', '$GATE_REPO'); +process.stdout.write(!r.block ? 'PASS' : 'FAIL: still blocked'); +" + +# ============================================================================ +printf '\n--- Testing iteration-context ---\n' +# ============================================================================ + +ITER_REPO="$(mktemp -d)" +( cd "$ITER_REPO" && mkdir -p autoresearch/run001 && printf 'iteration\tstatus\tmetric\n1\tpass\t0.85\n2\tpass\t0.87\n3\tpass\t0.88\n' > autoresearch/run001/results.tsv ) >/dev/null 2>&1 + +case_ok "iteration-context: skips before 5th" " +import { buildIterationContext } from \"./src/hooks/iteration-context.ts\"; +const r = buildIterationContext('$ITER_REPO', 'sess-skip', 'fix the bug'); +process.stdout.write(r.text === null ? 'PASS' : 'FAIL: injected early'); +" + +case_ok "iteration-context: injects on 5th" " +import { buildIterationContext } from \"./src/hooks/iteration-context.ts\"; +// advance the counter to 4 first +for (let i=0;i<4;i++) buildIterationContext('$ITER_REPO', 'sess-inject', 'x'); +const r = buildIterationContext('$ITER_REPO', 'sess-inject', 'autoresearch loop'); +process.stdout.write(r.text && r.text.includes('Active iteration state') ? 'PASS' : 'FAIL: ' + (r.text||'null')); +" + +case_ok "iteration-context: loop state for AR commands" " +import { buildIterationContext } from \"./src/hooks/iteration-context.ts\"; +for (let i=0;i<9;i++) buildIterationContext('$ITER_REPO', 'sess-loop', 'x'); +const r = buildIterationContext('$ITER_REPO', 'sess-loop', 'autoresearch: loop over scenarios'); +process.stdout.write(r.text && r.text.includes('Loop state') ? 'PASS' : 'FAIL: ' + (r.text||'null')); +" + +case_ok "iteration-context: disabled via env var" " +process.env.AR_DISABLE_ITERATION_CONTEXT = '1'; +import { buildIterationContext } from \"./src/hooks/iteration-context.ts\"; +const r = buildIterationContext('$ITER_REPO', 'sess-disabled', 'x'); +process.stdout.write(r.text === null ? 'PASS' : 'FAIL: still injecting'); +" + +rm -rf "$ITER_REPO" + +# ============================================================================ +printf '\n--- Testing isHookEnabled env flags ---\n' +# ============================================================================ + +case_ok "isHookEnabled: enabled by default" ' +import { isHookEnabled } from "./src/lib/session-state.ts"; +delete process.env.AR_DISABLE_SCOUT_BLOCK; +process.stdout.write(isHookEnabled("scout-block") ? "PASS" : "FAIL"); +' + +case_ok "isHookEnabled: disabled by env" ' +import { isHookEnabled } from "./src/lib/session-state.ts"; +process.env.AR_DISABLE_SCOUT_BLOCK = "1"; +process.stdout.write(!isHookEnabled("scout-block") ? "PASS" : "FAIL"); +' + +# ============================================================================ +# Type-check: the extension must compile clean against pi types. +# ============================================================================ +printf '\n--- Testing TypeScript type-check ---\n' +TOTAL=$((TOTAL + 1)) +PI_NODE="$(cd "$PI_EXT" && npm root -g 2>/dev/null)/@earendil-works/pi-coding-agent" +if [[ ! -d "$PI_NODE" ]]; then + # Try the nvm global used in dev + PI_NODE="$(dirname "$(dirname "$(readlink -f "$(command -v pi)")")")/lib/node_modules/@earendil-works/pi-coding-agent" +fi +if [[ ! -d "$PI_NODE" ]]; then + printf ' SKIP: pi-coding-agent types not found (non-pi environment)\n' + PASS=$((PASS + 1)) +else + # Symlink pi's types into the extension's own node_modules so tsc's + # upward module resolution (src -> pi-extension/node_modules) finds them. + NM="$PI_EXT/node_modules" + mkdir -p "$NM/@earendil-works" + ln -sf "$PI_NODE" "$NM/@earendil-works/pi-coding-agent" + [[ -d "$PI_NODE/../typebox" ]] && ln -sf "$PI_NODE/../typebox" "$NM/typebox" + [[ -d "$PI_NODE/../pi-ai" ]] && ln -sf "$PI_NODE/../pi-ai" "$NM/@earendil-works/pi-ai" + printf '{"compilerOptions":{"target":"ES2022","module":"ESNext","moduleResolution":"Bundler","strict":true,"noEmit":true,"skipLibCheck":true,"esModuleInterop":true,"resolveJsonModule":true,"lib":["ES2022"],"types":[]},"include":["src/**/*.ts"]}' > "$PI_EXT/tsconfig.test.json" + if npx -y -p typescript@5.6 tsc --noEmit -p "$PI_EXT/tsconfig.test.json" >/tmp/pi-tsc.log 2>&1; then + printf ' PASS: tsc --strict clean\n' + PASS=$((PASS + 1)) + else + printf ' FAIL: tsc --strict\n' + head -20 /tmp/pi-tsc.log >&2 + FAIL=$((FAIL + 1)) + fi + rm -rf "$NM" "$PI_EXT/tsconfig.test.json" +fi + +# ============================================================================ +# Summary +# ============================================================================ +printf '\n=== Results: %d/%d passed ===' "$PASS" "$TOTAL" +if [[ "$FAIL" -gt 0 ]]; then + printf ' (%d FAILED)\n' "$FAIL" + exit 1 +else + printf ' (all passed)\n' + exit 0 +fi From b58bd4abc42865fad610f68d735861f0cabcabd9 Mon Sep 17 00:00:00 2001 From: Jean-Baptiste Alleaume Date: Wed, 9 Sep 2026 13:27:15 +0000 Subject: [PATCH 5/6] feat(pi): generate pi bundle from canonical .claude/ via transform.sh MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Bring the pi extension fully into the canonical-source model documented in CONTRIBUTING.md: .claude/ is the source of truth, transform.sh regenerates every platform port. Previously the pi-extension skill + prompts were hand-copied from the Codex .agents/ bundle, which used $autoresearch syntax and request_user_input — wrong conventions for pi (pi uses /autoresearch slash commands and ctx.ui). - scripts/transform.sh: add transform_pi() + --pi flag. Regenerates pi-extension/skills/autoresearch/ (SKILL.md + references) and pi-extension/prompts/ (14 flat underscore templates) from .claude/ with pi adaptations (colon→underscore, /autoresearch:X→/autoresearch_X, AskUserQuestion→ctx.ui). Adds the pi skills dir to sync_runtime_helpers so orchestrator/regression scripts stay in sync. Idempotent. - pi-extension/: regenerated bundle now uses /autoresearch slash commands and ctx.ui (no more $autoresearch drift from Codex). - CONTRIBUTING.md: document pi in the repo structure, file table, quick start, multi-platform sync, testing checks, and add a "pi Hook Development" section mirroring the Claude hook guide. All five suites green (591 tests). transform.sh idempotent. /autoresearch_fix prompt loads and expands correctly with the regenerated bundle. --- CONTRIBUTING.md | 51 ++++++++++-- pi-extension/prompts/autoresearch.md | 2 +- pi-extension/prompts/autoresearch_debug.md | 2 +- pi-extension/prompts/autoresearch_evals.md | 4 +- pi-extension/prompts/autoresearch_fix.md | 2 +- pi-extension/prompts/autoresearch_improve.md | 6 +- pi-extension/prompts/autoresearch_learn.md | 2 +- pi-extension/prompts/autoresearch_plan.md | 6 +- pi-extension/prompts/autoresearch_predict.md | 2 +- pi-extension/prompts/autoresearch_probe.md | 6 +- pi-extension/prompts/autoresearch_reason.md | 2 +- .../prompts/autoresearch_regression.md | 2 +- pi-extension/prompts/autoresearch_scenario.md | 2 +- pi-extension/prompts/autoresearch_security.md | 2 +- pi-extension/prompts/autoresearch_ship.md | 2 +- pi-extension/skills/autoresearch/SKILL.md | 36 ++++----- scripts/transform.sh | 81 ++++++++++++++++++- 17 files changed, 162 insertions(+), 48 deletions(-) diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index d4326a46..b1963948 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -4,7 +4,7 @@ Whether you're fixing a typo, adding examples, creating a new sub-command, or im ## Quick Start -Autoresearch is Markdown files that Claude Code, OpenCode, and Codex discover from `skills/` and `commands/` directories. No build step, no compilation — edit a `.md` file, invoke the skill, see your changes. +Autoresearch is Markdown files that Claude Code, OpenCode, Codex, and pi discover from `skills/` and `commands/` directories. No build step, no compilation — edit a `.md` file, invoke the skill, see your changes. ```bash # 1. Clone the repo @@ -15,6 +15,7 @@ cd autoresearch ./scripts/install.sh --claude --global # Claude Code ./scripts/install.sh --opencode --global # OpenCode ./scripts/install.sh --codex --global # Codex +./scripts/install.sh --pi --global # pi (coding agent) # 3. Or symlink for live editing (recommended for development) ln -s $(pwd)/.claude/skills/autoresearch ~/.claude/skills/autoresearch @@ -27,9 +28,10 @@ ln -s $(pwd)/.claude/commands/autoresearch.md ~/.claude/commands/autoresearch.md The canonical source is `.claude/`. After making changes, run the transform to sync all platforms: ```bash -./scripts/transform.sh # sync to OpenCode + Codex +./scripts/transform.sh # sync to OpenCode + Codex + pi ./scripts/transform.sh --opencode # OpenCode only ./scripts/transform.sh --codex # Codex only +./scripts/transform.sh --pi # pi only ``` ## Repository Structure (v2.2.2) @@ -45,10 +47,14 @@ autoresearch/ │ └── autoresearch/ ← 13 subcommand files (self-contained) ├── .opencode/ ← OpenCode port (generated via transform.sh) ├── .agents/ + plugins/ ← Codex port (generated via transform.sh) +├── pi-extension/ ← pi (coding agent) extension (generated via transform.sh) +│ ├── src/ ← Guardrails ported from Claude hooks → pi events +│ ├── skills/autoresearch/ ← Skill + references + scripts +│ └── prompts/ ← 14 command prompt templates ├── claude-plugin/ ← Distribution package (Claude Code plugin install) ├── scripts/ -│ ├── install.sh ← Guided installer (3 platforms) -│ ├── transform.sh ← .claude/ → .opencode/ + .agents/ sync +│ ├── install.sh ← Guided installer (4 platforms) +│ ├── transform.sh ← .claude/ → .opencode/ + .agents/ + pi-extension/ sync │ ├── release.sh ← Release automation │ └── release.md ← Release checklist ├── guide/ ← Guides — one per command + advanced patterns @@ -67,8 +73,9 @@ autoresearch/ | `references/security-checklist.md` | STRIDE + OWASP checklist (loaded by security command) | Adding security checks | | `references/predict-personas.md` | 5 expert personas (loaded by predict command) | Adding/modifying personas | | `references/reason-judge-protocol.md` | Adversarial refinement protocol (loaded by reason command) | Changing judge/critic behavior | -| `scripts/transform.sh` | Canonical transform (.claude/ → .opencode/ + .agents/ + claude-plugin/) | Adding new commands, reference files, or generated helper updates | +| `scripts/transform.sh` | Canonical transform (.claude/ → .opencode/ + .agents/ + claude-plugin/ + pi-extension/) | Adding new commands, reference files, or generated helper updates | | `claude-plugin/` | Distribution package — synced from .claude/ during release | Don't edit directly — edit .claude/ | +| `pi-extension/` | pi extension — skills + prompts synced from .claude/ via transform.sh; guardrails in `src/` ported from Claude hooks | Edit `src/` for guardrail logic; skills/prompts are generated — edit .claude/ | ## What to Contribute @@ -114,7 +121,7 @@ Only create a reference in `references/` if shared by multiple commands. Single- ### 4. Run transform + update docs ```bash -./scripts/transform.sh # sync to OpenCode + Codex +./scripts/transform.sh # sync to OpenCode + Codex + pi ``` Update: README.md (commands table), guide/ (new guide file), COMPARISON.md (subcommand count). @@ -154,6 +161,7 @@ For maintainer workflows, the canonical checks are: - `bash scripts/transform.sh` — regenerate platform distributions and bundled runtime helpers - `bash tests/test-maintenance.sh` — transform idempotence and release-prep guards - `bash tests/test-hooks.sh` — Claude hook contracts and fail-open behavior +- `bash tests/test-pi.sh` — pi extension guardrail contracts (ported hooks) and TypeScript type-check ## Release Process @@ -218,3 +226,34 @@ echo "Exit code: $?" # Full test suite bash tests/test-hooks.sh ``` + +## pi Hook Development + +The pi extension ports the Claude hooks to pi's extension event system. Guardrail logic lives in `pi-extension/src/hooks/*.ts` as pure functions (no stdin/stdout — pi passes events in-process); `pi-extension/src/guardrails.ts` wires them to pi events (`tool_call`, `before_agent_start`, `input`, `session_start`, `session_shutdown`). Shared helpers are in `pi-extension/src/lib/`. + +### Adding a New pi Hook + +1. Create `pi-extension/src/hooks/{name}.ts` exporting a pure function that returns a `{ block?, reason?, warning?, needsConfirm?, text? }` result. +2. Wire it to the matching pi event in `pi-extension/src/guardrails.ts`. +3. Guard with `isHookEnabled("{name}")` (honors `AR_DISABLE_{NAME}`). +4. Fail open — wrap logic in try/catch, never throw to the event handler. +5. Add cases to `tests/test-pi.sh`. +6. Run `bash tests/test-pi.sh` (and `tsc --strict` is covered by the suite's type-check step). + +### pi Hook Rules + +- **Fail-open:** wrap in try/catch; a guardrail malfunction never blocks work +- **No external deps:** Node.js builtins only (vendored `lib/ignore.ts`) +- **No stdin/stdout:** pi passes events in-process; return an object, don't `process.exit` +- **State:** use `loadSessionState()` / `saveSessionState()` (OS temp dir, keyed by cwd + session id) +- **Logs:** `~/.pi/agent/autoresearch/.logs//hook-log.jsonl` (bounded metadata only — no paths, commands, or secrets). Do NOT write under `~/.pi/agent/hooks/` — pi renamed hooks→extensions and warns on a legacy `hooks/` dir. + +### Testing pi Hooks + +```bash +# Full test suite (53 cases: guardrail logic + tsc --strict type-check) +bash tests/test-pi.sh + +# Type-check only (needs pi-coding-agent types resolvable) +cd pi-extension && npx -y -p typescript@5.6 tsc --noEmit --strict --moduleResolution Bundler --module ESNext --target ES2022 --skipLibCheck --esModuleInterop --lib ES2022 --types "[]" src/index.ts +``` diff --git a/pi-extension/prompts/autoresearch.md b/pi-extension/prompts/autoresearch.md index 7eb1cf12..b46dbecf 100644 --- a/pi-extension/prompts/autoresearch.md +++ b/pi-extension/prompts/autoresearch.md @@ -22,7 +22,7 @@ Extract from $ARGUMENTS: ## Setup (if required context missing) -If Goal, Scope, Metric, or Verify missing → use request_user_input (single batched call): +If Goal, Scope, Metric, or Verify missing → use ctx.ui (single batched call): Q1 (Goal): "What do you want to improve?" Q2 (Scope): "Which files?" — suggest globs from project Q3 (Metric+Verify): "How to measure? Provide a shell command that outputs a number" diff --git a/pi-extension/prompts/autoresearch_debug.md b/pi-extension/prompts/autoresearch_debug.md index 7c481e7a..c41d161b 100644 --- a/pi-extension/prompts/autoresearch_debug.md +++ b/pi-extension/prompts/autoresearch_debug.md @@ -21,7 +21,7 @@ Extract from $ARGUMENTS: If Scope and Symptom both missing: 1. Auto-scan: run tests, lint, typecheck to detect existing failures -2. request_user_input (single batch): +2. ctx.ui (single batch): Q1 (Issue): "What's the problem?" — hunt all bugs, specific error, failing tests, CI failure, performance Q2 (Scope): "Which files?" — suggested globs + entire codebase Q3 (Depth): "How deep?" — quick (5), standard (15), deep (30+), unlimited diff --git a/pi-extension/prompts/autoresearch_evals.md b/pi-extension/prompts/autoresearch_evals.md index c0f940ca..053eb5c1 100644 --- a/pi-extension/prompts/autoresearch_evals.md +++ b/pi-extension/prompts/autoresearch_evals.md @@ -17,8 +17,8 @@ Extract from $ARGUMENTS: 1. If path provided → use that TSV directly 2. If no path → scan current directory + `autoresearch/*/` for `*-results.tsv` files -3. If multiple found → request_user_input: "Which results to analyze?" — list found files -4. If none found → request_user_input: "Provide path to results TSV" +3. If multiple found → ctx.ui: "Which results to analyze?" — list found files +4. If none found → ctx.ui: "Provide path to results TSV" 5. Also scan project root for v2.0.03 legacy TSV files (backward compat) ## Parse TSV diff --git a/pi-extension/prompts/autoresearch_fix.md b/pi-extension/prompts/autoresearch_fix.md index 4ffb7c1e..53329af8 100644 --- a/pi-extension/prompts/autoresearch_fix.md +++ b/pi-extension/prompts/autoresearch_fix.md @@ -21,7 +21,7 @@ Extract from $ARGUMENTS: If Target and Scope both missing: 1. Auto-detect failures: run test suite, type checker, linter, build -2. Present results via request_user_input (single batched call): +2. Present results via ctx.ui (single batched call): Q1 (Fix What): "Found [N] test failures, [M] type errors, [K] lint errors. Fix what?" — everything, only tests, only types, only lint Q2 (Guard): "Safety command that must always pass?" — npm test, tsc, npm run build, skip Q3 (Scope): "Which files can I modify?" — suggested globs from error locations + all diff --git a/pi-extension/prompts/autoresearch_improve.md b/pi-extension/prompts/autoresearch_improve.md index 7b3860c6..dc815880 100644 --- a/pi-extension/prompts/autoresearch_improve.md +++ b/pi-extension/prompts/autoresearch_improve.md @@ -27,7 +27,7 @@ If upstream `handoff.json` exists in CWD → read it. Map source findings to def ## Setup (if Goal or ICP missing) -request_user_input (single batch): +ctx.ui (single batch): Q1 (Goal): "What product area to improve?" — open text Q2 (ICP): "Who is your ideal customer?" — open text describing target buyer/user Q3 (Pain points): "Top 3 pain points your customers face?" — open text @@ -43,7 +43,7 @@ Resolve product context (priority chain): 3. `package.json` / `pyproject.toml` / `Cargo.toml` description (≥10 chars) → use it 4. If ALL above absent AND NOT `--no-discover` → auto-discover: scan 10 key files (manifest, routes, models, config), cap 1500 tokens 5. If `--discover` → force scan regardless of above -6. If nothing found → warn: "No product context. Run `$autoresearch learn --mode summarize` for better results." +6. If nothing found → warn: "No product context. Run `/autoresearch_learn --mode summarize` for better results." ## Phase 2: Research Loop @@ -82,7 +82,7 @@ Print: `--- Eval Checkpoint (iterations {X}-{Y}) ---\nInsights: {total} (+{new}) Write `improvement-plan.md` with full tiered ranking. -request_user_input (multi-select): present tiered list, user selects which features become PRDs. +ctx.ui (multi-select): present tiered list, user selects which features become PRDs. If `--features` provided → pre-select matching items, still show for confirmation. ## Phase 4: PRD Generation diff --git a/pi-extension/prompts/autoresearch_learn.md b/pi-extension/prompts/autoresearch_learn.md index 96e319c3..dc3404ca 100644 --- a/pi-extension/prompts/autoresearch_learn.md +++ b/pi-extension/prompts/autoresearch_learn.md @@ -24,7 +24,7 @@ Extract from $ARGUMENTS: ## Setup (if Mode or Scope missing) -request_user_input (single batch): +ctx.ui (single batch): Q1 (Mode): "What to do?" — init (generate docs), update (refresh), check (validate), summarize (overview), wiki (knowledge base) Q2 (Scope): "Which files?" — suggested globs + entire codebase Q3 (Depth): "How detailed?" — overview only, standard, comprehensive diff --git a/pi-extension/prompts/autoresearch_plan.md b/pi-extension/prompts/autoresearch_plan.md index a7cfe510..cf8e7fa1 100644 --- a/pi-extension/prompts/autoresearch_plan.md +++ b/pi-extension/prompts/autoresearch_plan.md @@ -17,7 +17,7 @@ Remaining text = goal description. ## Setup (if Goal missing) -request_user_input (single batch): +ctx.ui (single batch): Q1 (Goal): "What do you want to achieve?" — open text Q2 (Type): "What kind of goal?" — improve a metric, fix errors, audit security, explore edge cases, document code, ship something If Goal provided → skip. @@ -45,7 +45,7 @@ For metric-driven goals: For subjective goals: - Suggest proxy metrics where possible -- Or recommend $autoresearch reason for non-measurable goals +- Or recommend /autoresearch_reason for non-measurable goals ## Phase 4: Derive Verify Command @@ -76,7 +76,7 @@ Based on goal complexity: Output a ready-to-run autoresearch config block: ``` -$autoresearch +/autoresearch Goal: {derived goal} Scope: {derived globs} Metric: {derived metric} diff --git a/pi-extension/prompts/autoresearch_predict.md b/pi-extension/prompts/autoresearch_predict.md index 8ed34e54..85563481 100644 --- a/pi-extension/prompts/autoresearch_predict.md +++ b/pi-extension/prompts/autoresearch_predict.md @@ -24,7 +24,7 @@ Remaining text not matching flags = goal description. ## Setup (if Scope or Goal missing) -request_user_input (single batch): +ctx.ui (single batch): Q1 (Scope): "Which files to analyze?" — suggested globs + entire codebase Q2 (Goal): "What should personas focus on?" — code quality, security, performance, architecture, all Q3 (Depth): "How deep?" — shallow (3 personas, 1 round), standard (5, 2 — recommended), deep (8, 3) diff --git a/pi-extension/prompts/autoresearch_probe.md b/pi-extension/prompts/autoresearch_probe.md index 1cbfce6b..86350c41 100644 --- a/pi-extension/prompts/autoresearch_probe.md +++ b/pi-extension/prompts/autoresearch_probe.md @@ -14,14 +14,14 @@ Extract from $ARGUMENTS: - `Depth:` or `--depth` — shallow (5 rounds), standard (15), deep (30) - `--personas N` or `Personas:` — active persona count (3-8, default 6) - `--saturation-threshold N` — net-new constraints/round below which counts toward saturation (default 2) -- `--mode` or `Mode:` — interactive (default, uses request_user_input) or autonomous (self-answers from codebase) +- `--mode` or `Mode:` — interactive (default, uses ctx.ui) or autonomous (self-answers from codebase) - `--adversarial` — rotate hostile personas to front - `Iterations:` or `--iterations` — default 15 rounds. "unlimited" for unbounded. - `--evals`, `--evals-interval N`, `--chain`, `--` ## Setup (if Topic missing) -request_user_input (single batch): +ctx.ui (single batch): Q1 (Topic): "What to probe?" — open text describing feature, requirement, or design Q2 (Scope): "Which files for context?" — suggested globs + entire codebase Q3 (Depth): "How deep?" — shallow (5 rounds), standard (15), deep (30), unlimited @@ -60,7 +60,7 @@ If --adversarial: rotate Skeptic + Contradiction Finder + Edge-Case Hunter to fr - Annotate questions with: relevant file:line, existing behavior, gaps ### Phase 4: Answer Capture -- **Interactive mode:** present questions via request_user_input, collect answers +- **Interactive mode:** present questions via ctx.ui, collect answers - **Autonomous mode:** infer answers from codebase context, label confidence (high/medium/low) ### Phase 5: Constraint Extraction diff --git a/pi-extension/prompts/autoresearch_reason.md b/pi-extension/prompts/autoresearch_reason.md index d17dc90f..51042d48 100644 --- a/pi-extension/prompts/autoresearch_reason.md +++ b/pi-extension/prompts/autoresearch_reason.md @@ -24,7 +24,7 @@ Remaining text not matching flags = task description. ## Setup (if Task or Domain missing) -request_user_input (single batch): +ctx.ui (single batch): Q1 (Task): "What should be reasoned about?" — open text Q2 (Domain): "What domain?" — software architecture, product strategy, business decision, security, research, content Q3 (Mode): "Refinement mode?" — convergent (stop when winner repeats), creative (never auto-stop), debate (no synthesis) diff --git a/pi-extension/prompts/autoresearch_regression.md b/pi-extension/prompts/autoresearch_regression.md index 1a97ea7d..18fd4928 100644 --- a/pi-extension/prompts/autoresearch_regression.md +++ b/pi-extension/prompts/autoresearch_regression.md @@ -24,7 +24,7 @@ Extract from $ARGUMENTS: ## Setup / Probe-on-launch 1. Auto-detect per-dimension verify commands: `package.json` scripts, `Makefile`, `nx`, migrate config, bench/snapshot/size scripts. -2. request_user_input (single batch) to confirm detected commands + base ref + which dimensions to run. +2. ctx.ui (single batch) to confirm detected commands + base ref + which dimensions to run. 3. **Auto-skip probe** when CI / no-TTY / `--mode autonomous` / complete-config / chained-handoff — log the inferred config instead of asking. ## Classification Phase (first-class, before any differential) diff --git a/pi-extension/prompts/autoresearch_scenario.md b/pi-extension/prompts/autoresearch_scenario.md index 6569bbe7..40724c23 100644 --- a/pi-extension/prompts/autoresearch_scenario.md +++ b/pi-extension/prompts/autoresearch_scenario.md @@ -20,7 +20,7 @@ Extract from $ARGUMENTS: ## Setup (if Scenario or Domain missing) -request_user_input (single batch): +ctx.ui (single batch): Q1 (Scenario): "Describe the feature/flow to explore" Q2 (Domain): "What domain?" — web app, mobile app, API, CLI, data pipeline, infrastructure Q3 (Scope): "Which files for context?" — suggested globs + entire codebase diff --git a/pi-extension/prompts/autoresearch_security.md b/pi-extension/prompts/autoresearch_security.md index 625be5c0..5ed6b134 100644 --- a/pi-extension/prompts/autoresearch_security.md +++ b/pi-extension/prompts/autoresearch_security.md @@ -22,7 +22,7 @@ Extract from $ARGUMENTS: If Scope missing and no --diff: 1. Scan codebase for tech stack, frameworks, API routes -2. request_user_input (single batch): +2. ctx.ui (single batch): Q1 (Scope): "What to audit?" — entire codebase, API + middleware, auth, external-facing Q2 (Depth): "How thorough?" — quick (5), standard (15), deep (30+), unlimited Q3 (Action): "What to do with findings?" — report only, report + auto-fix, report + CI gate diff --git a/pi-extension/prompts/autoresearch_ship.md b/pi-extension/prompts/autoresearch_ship.md index 78d35594..165fb415 100644 --- a/pi-extension/prompts/autoresearch_ship.md +++ b/pi-extension/prompts/autoresearch_ship.md @@ -29,7 +29,7 @@ Remaining text = description of what to ship. - Has Dockerfile / deploy config → deployment - Has markdown / content files → content - Has package.json version change → package -2. If still unclear → request_user_input (single batch): +2. If still unclear → ctx.ui (single batch): Q1 (What): "What are you shipping?" — code PR, release, deployment, content, docs, package Q2 (Target): "Specific target?" — current branch, specific PR, specific path Q3 (Mode): "How to ship?" — full workflow, dry-run only, checklist only diff --git a/pi-extension/skills/autoresearch/SKILL.md b/pi-extension/skills/autoresearch/SKILL.md index 22f2056a..ee9c557c 100644 --- a/pi-extension/skills/autoresearch/SKILL.md +++ b/pi-extension/skills/autoresearch/SKILL.md @@ -12,7 +12,7 @@ version: 2.2.2 - All results logged to `autoresearch/{subcommand}-{YYMMDD}-{HHMM}/` directory. - Chain handoff via `handoff.json`. Evals reads `*-results.tsv`. -## Dispatch (bare `$autoresearch`) +## Dispatch (bare `/autoresearch`) Parse the invocation in this order: @@ -30,20 +30,20 @@ Print a banner on every invocation: `[autoresearch] mode: classic | orchestrator | Command | Does | Default Iterations | |---|---|---| -| `$autoresearch` | Iterate against a metric: modify → verify → keep/discard | 25 | -| `$autoresearch plan` | Convert a goal into validated Scope, Metric, Verify config | N/A | -| `$autoresearch debug` | Hunt bugs: hypothesize → test → falsify → repeat | 15 | -| `$autoresearch fix` | Crush errors one-by-one until zero remain | 20 | -| `$autoresearch security` | STRIDE + OWASP audit with red-team personas | 15 | -| `$autoresearch ship` | Ship through 8 phases: checklist → dry-run → deploy → verify | N/A | -| `$autoresearch scenario` | Generate edge cases across 12 dimensions | 20 | -| `$autoresearch predict` | 5 expert personas debate before implementation | N/A | -| `$autoresearch learn` | Scout codebase → generate docs or wiki → validate → fix loop | 10 | -| `$autoresearch reason` | Adversarial debate with blind judges until convergence | 8 | -| `$autoresearch probe` | 8 personas interrogate requirements until saturation | 15 | -| `$autoresearch improve` | Research ICP challenges, discover improvements, generate PRDs | 15 | -| `$autoresearch evals` | Analyze iteration results: trends, plateaus, regressions | N/A | -| `$autoresearch regression` | Regression stability gate: baseline vs candidate, verdict STABLE/UNSTABLE | N/A | +| `/autoresearch` | Iterate against a metric: modify → verify → keep/discard | 25 | +| `/autoresearch_plan` | Convert a goal into validated Scope, Metric, Verify config | N/A | +| `/autoresearch_debug` | Hunt bugs: hypothesize → test → falsify → repeat | 15 | +| `/autoresearch_fix` | Crush errors one-by-one until zero remain | 20 | +| `/autoresearch_security` | STRIDE + OWASP audit with red-team personas | 15 | +| `/autoresearch_ship` | Ship through 8 phases: checklist → dry-run → deploy → verify | N/A | +| `/autoresearch_scenario` | Generate edge cases across 12 dimensions | 20 | +| `/autoresearch_predict` | 5 expert personas debate before implementation | N/A | +| `/autoresearch_learn` | Scout codebase → generate docs or wiki → validate → fix loop | 10 | +| `/autoresearch_reason` | Adversarial debate with blind judges until convergence | 8 | +| `/autoresearch_probe` | 8 personas interrogate requirements until saturation | 15 | +| `/autoresearch_improve` | Research ICP challenges, discover improvements, generate PRDs | 15 | +| `/autoresearch_evals` | Analyze iteration results: trends, plateaus, regressions | N/A | +| `/autoresearch_regression` | Regression stability gate: baseline vs candidate, verdict STABLE/UNSTABLE | N/A | ## Universal Flags @@ -57,8 +57,8 @@ Print a banner on every invocation: `[autoresearch] mode: classic | orchestrator | `--` | All | Shorthand for `--chain ` | | `--dry-run` | Orchestrator | Print derived config + planned pipeline; no execution | | `--max-cycles N` | Orchestrator | Hard ceiling on orchestration cycles (default 50) | -| `--classic` | Bare `$autoresearch` | Force Classic metric-loop mode | -| `--auto` | Bare `$autoresearch` | Force Orchestrator mode | +| `--classic` | Bare `/autoresearch` | Force Classic metric-loop mode | +| `--auto` | Bare `/autoresearch` | Force Orchestrator mode | ## Orchestrator @@ -76,7 +76,7 @@ Backed by `scripts/orchestrate.sh` (deterministic seam — all routing logic liv 1. **Classify** — `scripts/orchestrate.sh classify ""` → archetype label + mode. 2. **Derive predicate** — reuse `plan` logic to produce a concrete Success predicate: exact shell command + expected output. For `optimize-metric`, run the full plan/wizard derivation internally. -3. **Confirm** — ONE `request_user_input` showing: archetype, mode, concrete predicate (command + expected output), terminal choice (stop-at-verified vs proceed-to-ship). Misclassifications are caught here, not mid-run. +3. **Confirm** — ONE `ctx.ui` showing: archetype, mode, concrete predicate (command + expected output), terminal choice (stop-at-verified vs proceed-to-ship). Misclassifications are caught here, not mid-run. 4. **Round-0 dry-run** — prove the predicate command runs and returns a value; safety-screen every derived command via `screen-cmd`; print projected cycle budget. Stop here if `--dry-run`. 5. **Loop** until predicate satisfied: a. Assess state via cheap signals (last `handoff.json`, regression verdict, error count) + affected-test verify. diff --git a/scripts/transform.sh b/scripts/transform.sh index 79f17375..18bcd6c5 100755 --- a/scripts/transform.sh +++ b/scripts/transform.sh @@ -18,12 +18,14 @@ CLAUDE_COMMANDS="$REPO_ROOT/.claude/commands" DO_OPENCODE=1 DO_CODEX=1 DO_CLAUDE=1 +DO_PI=1 while [[ $# -gt 0 ]]; do case "$1" in - --opencode) DO_OPENCODE=1; DO_CODEX=0; DO_CLAUDE=0 ;; - --codex) DO_OPENCODE=0; DO_CODEX=1; DO_CLAUDE=0 ;; - -h|--help) printf 'Usage: %s [--opencode|--codex]\n' "$0"; exit 0 ;; + --opencode) DO_OPENCODE=1; DO_CODEX=0; DO_CLAUDE=0; DO_PI=0 ;; + --codex) DO_OPENCODE=0; DO_CODEX=1; DO_CLAUDE=0; DO_PI=0 ;; + --pi) DO_OPENCODE=0; DO_CODEX=0; DO_CLAUDE=0; DO_PI=1 ;; + -h|--help) printf 'Usage: %s [--opencode|--codex|--pi]\n' "$0"; exit 0 ;; *) printf 'Unknown flag: %s\n' "$1" >&2; exit 1 ;; esac shift @@ -48,6 +50,9 @@ sync_runtime_helpers() { if [[ $DO_CODEX -eq 1 ]]; then destinations+=("$REPO_ROOT/.agents/skills/autoresearch" "$REPO_ROOT/plugins/autoresearch/skills/autoresearch") fi + if [[ $DO_PI -eq 1 ]]; then + destinations+=("$REPO_ROOT/pi-extension/skills/autoresearch") + fi for dst in "${destinations[@]}"; do rm -rf "$dst/scripts" @@ -249,11 +254,81 @@ transform_hooks() { printf 'Hooks: transformed %s → claude-plugin/hooks/\n' ".claude/hooks/autoresearch/" } +# --- pi Transform --- +# Differences from canonical .claude/: colon → underscore in command names, +# /autoresearch:X → /autoresearch_X, AskUserQuestion → ctx.ui. The 14 commands +# become flat prompt templates under pi-extension/prompts/; the skill + +# references live under pi-extension/skills/autoresearch/. + +transform_pi() { + local dst_skills="$REPO_ROOT/pi-extension/skills/autoresearch" + local dst_prompts="$REPO_ROOT/pi-extension/prompts" + + rm -rf "$dst_skills" "$dst_prompts"/autoresearch*.md + mkdir -p "$dst_skills/references" "$dst_prompts" + + adapt_pi() { + sed \ + -e 's/`AskUserQuestion`/`ctx.ui`/g' \ + -e 's/AskUserQuestion/ctx.ui/g' \ + -e 's|/autoresearch:plan|/autoresearch_plan|g' \ + -e 's|/autoresearch:debug|/autoresearch_debug|g' \ + -e 's|/autoresearch:fix|/autoresearch_fix|g' \ + -e 's|/autoresearch:security|/autoresearch_security|g' \ + -e 's|/autoresearch:ship|/autoresearch_ship|g' \ + -e 's|/autoresearch:scenario|/autoresearch_scenario|g' \ + -e 's|/autoresearch:predict|/autoresearch_predict|g' \ + -e 's|/autoresearch:learn|/autoresearch_learn|g' \ + -e 's|/autoresearch:reason|/autoresearch_reason|g' \ + -e 's|/autoresearch:probe|/autoresearch_probe|g' \ + -e 's|/autoresearch:evals|/autoresearch_evals|g' \ + -e 's|/autoresearch:improve|/autoresearch_improve|g' \ + -e 's|/autoresearch:regression|/autoresearch_regression|g' \ + -e 's|name: autoresearch:plan|name: autoresearch_plan|g' \ + -e 's|name: autoresearch:debug|name: autoresearch_debug|g' \ + -e 's|name: autoresearch:fix|name: autoresearch_fix|g' \ + -e 's|name: autoresearch:security|name: autoresearch_security|g' \ + -e 's|name: autoresearch:ship|name: autoresearch_ship|g' \ + -e 's|name: autoresearch:scenario|name: autoresearch_scenario|g' \ + -e 's|name: autoresearch:predict|name: autoresearch_predict|g' \ + -e 's|name: autoresearch:learn|name: autoresearch_learn|g' \ + -e 's|name: autoresearch:reason|name: autoresearch_reason|g' \ + -e 's|name: autoresearch:probe|name: autoresearch_probe|g' \ + -e 's|name: autoresearch:evals|name: autoresearch_evals|g' \ + -e 's|name: autoresearch:improve|name: autoresearch_improve|g' \ + -e 's|name: autoresearch:regression|name: autoresearch_regression|g' \ + -e 's|\.claude/skills/|pi-extension/skills/|g' \ + -e 's|\.claude/commands/|pi-extension/prompts/|g' \ + "$1" + } + + # SKILL.md + references + adapt_pi "$CLAUDE_SKILLS/SKILL.md" > "$dst_skills/SKILL.md" + for ref in "$CLAUDE_SKILLS"/references/*.md; do + [[ -f "$ref" ]] || continue + adapt_pi "$ref" > "$dst_skills/references/$(basename "$ref")" + done + + # Core command → flat prompt template + adapt_pi "$CLAUDE_COMMANDS/autoresearch.md" > "$dst_prompts/autoresearch.md" + + # Subcommand files (colon → underscore in filename) + for cmd in "$CLAUDE_COMMANDS"/autoresearch/*.md; do + [[ -f "$cmd" ]] || continue + local base + base="$(basename "$cmd")" + adapt_pi "$cmd" > "$dst_prompts/autoresearch_${base}" + done + + printf 'pi: transformed %s → %s\n' ".claude/" "pi-extension/" +} + # --- Main --- if [[ $DO_CLAUDE -eq 1 ]]; then transform_claude; fi if [[ $DO_OPENCODE -eq 1 ]]; then transform_opencode; fi if [[ $DO_CODEX -eq 1 ]]; then transform_codex; fi +if [[ $DO_PI -eq 1 ]]; then transform_pi; fi sync_runtime_helpers if [[ $DO_CLAUDE -eq 1 ]]; then transform_hooks; fi From d82342d9b622e0ce3a722c7b76e8145bfff846bb Mon Sep 17 00:00:00 2001 From: Jean-Baptiste Alleaume Date: Wed, 9 Sep 2026 13:39:36 +0000 Subject: [PATCH 6/6] feat(pi): port subagent-context hook, integrating with pi-subagents MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The subagent-context hook was the one Claude hook not yet ported. pi has no native subagent-spawn event in the extension API, but the widely-used pi-subagents extension provides two clean connection points: - Child side: pi subagents are full pi sessions that load this same extension and fire before_agent_start. When PI_SUBAGENT_CHILD=1, the hook injects the project/branch/active-TSV/iteration block — the direct port of Claude's SubagentStart injection. iteration-context already flows into children via the same handler; subagent-context adds the structured header. - Parent side: listen for pi-subagents' `prompt-template:subagent:request` event on the shared pi.events bus to log delegation launches (bounded metadata only). Notification only — the request isn't mutated. Both paths fail open if pi-subagents isn't installed (event never fires / env not set). AR_DISABLE_SUBAGENT_CONTEXT gates both. - pi-extension/src/hooks/subagent-context.ts: buildSubagentContext + isSubagentChild - pi-extension/src/guardrails.ts: wire child-side injection at before_agent_start + parent-side pi.events listener - tests/test-pi.sh: 5 new cases (58 total, all green) - README/index.ts: document the pi-subagents integration All five suites green (596 tests). transform.sh idempotent. --- pi-extension/README.md | 3 +- pi-extension/src/guardrails.ts | 41 ++++++++ pi-extension/src/hooks/subagent-context.ts | 109 +++++++++++++++++++++ pi-extension/src/index.ts | 1 + tests/test-pi.sh | 40 ++++++++ 5 files changed, 193 insertions(+), 1 deletion(-) create mode 100644 pi-extension/src/hooks/subagent-context.ts diff --git a/pi-extension/README.md b/pi-extension/README.md index 6d88365f..3600a73a 100644 --- a/pi-extension/README.md +++ b/pi-extension/README.md @@ -46,10 +46,11 @@ pi-extension/ | `UserPromptSubmit` | `input` | simplify-gate (block/warn shipping verbs) | | `SessionStart` | `session_start` | session-init (persist project/branch state) | | `SessionEnd` | `session_shutdown` | stop-notify (notify + webhook + cleanup) | +| `SubagentStart` | `before_agent_start` (child) + `pi.events` (parent) | subagent-context (port of Claude SubagentStart; integrates with pi-subagents) | Every hook **fails open** — a guardrail malfunction never blocks legitimate work. -> **Note on subagent context:** Claude's `SubagentStart` hook injected iteration state into subagents. pi subagents inherit the loaded autoresearch skill, so the active-iteration context already flows via the skill + `iteration-context` on the parent. There is no direct subagent-spawn event in pi's extension API, so the `subagent-context` hook is not ported. +> **Note on subagent context:** Claude's `SubagentStart` hook injected iteration state into subagents. pi has no native subagent-spawn event in the extension API, but the widely-used [pi-subagents](https://github.com/earendil-works/pi-subagents) extension emits delegation events on the shared `pi.events` bus, and pi subagents are full pi sessions that load this same extension and fire `before_agent_start`. The `subagent-context` hook uses both: on the **parent** side it listens for `prompt-template:subagent:request` to log launches; on the **child** side (`PI_SUBAGENT_CHILD=1`) it injects the project/branch/active-TSV/iteration block at `before_agent_start` — the direct port of the Claude `SubagentStart` injection. If pi-subagents isn't installed, both paths fail open (no-op). ### Resource contribution diff --git a/pi-extension/src/guardrails.ts b/pi-extension/src/guardrails.ts index b58b501d..abbbad09 100644 --- a/pi-extension/src/guardrails.ts +++ b/pi-extension/src/guardrails.ts @@ -7,6 +7,8 @@ // UserPromptSubmit → input → simplify-gate (block/warn shipping verbs) // SessionStart → session_start → session-init (persist state) // SessionEnd → session_shutdown → stop-notify (notify + cleanup) +// SubagentStart → before_agent_start → subagent-context (child-side, PI_SUBAGENT_CHILD=1) +// + pi.events → subagent-context (parent-side delegation logging) // // All hooks fail open: a guardrail malfunction never blocks legitimate work. @@ -22,6 +24,7 @@ import { } from "./hooks/privacy-block.js"; import { buildIterationContext } from "./hooks/iteration-context.js"; import { buildDevRules } from "./hooks/dev-rules-reminder.js"; +import { buildSubagentContext, isSubagentChild } from "./hooks/subagent-context.js"; import { checkSimplifyGate, surfaceWarning } from "./hooks/simplify-gate.js"; import { runStopNotify } from "./hooks/stop-notify.js"; import { @@ -172,6 +175,16 @@ function onBeforeAgentStart(pi: ExtensionAPI): void { const parts: string[] = []; + // When running inside a pi-subagents child, inject the subagent context + // block (project/branch/active TSV/iteration) — the port of Claude's + // SubagentStart hook. iteration-context already flows into children via + // this same handler, but the subagent block adds the structured header + // and is the canonical entry point for child-side autoresearch state. + if (isSubagentChild()) { + const sub = buildSubagentContext(cwd, sessionId); + if (sub.text) parts.push(sub.text); + } + const iter = buildIterationContext(cwd, sessionId, prompt); if (iter.text) parts.push(iter.text); @@ -207,12 +220,40 @@ function onInput(pi: ExtensionAPI): void { }); } +// Parent-side observability for pi-subagents: log delegation requests so +// subagent launches show up in the bounded hook log. This is notification +// only — the request cannot be mutated here; child-side context injection +// happens via onBeforeAgentStart (PI_SUBAGENT_CHILD=1). Fails open if +// pi-subagents is not installed (the event simply never fires). +const SUBAGENT_REQUEST_EVENT = "prompt-template:subagent:request"; + +function onSubagentDelegation(pi: ExtensionAPI): void { + try { + pi.events.on(SUBAGENT_REQUEST_EVENT, (data: unknown) => { + if (!isHookEnabled("subagent-context")) return; + try { + const req = data as { agent?: string; context?: string } | null; + log("subagent-context", { + action: "delegation", + tool: req?.agent || "unknown", + category: req?.context || "unknown", + }); + } catch { + /* fail-open */ + } + }); + } catch { + /* pi.events unavailable — fail open */ + } +} + export function registerGuardrails(pi: ExtensionAPI): void { onSessionStart(pi); onSessionShutdown(pi); onToolCall(pi); onBeforeAgentStart(pi); onInput(pi); + onSubagentDelegation(pi); } // Re-export for tests / direct use. diff --git a/pi-extension/src/hooks/subagent-context.ts b/pi-extension/src/hooks/subagent-context.ts new file mode 100644 index 00000000..1fd59f90 --- /dev/null +++ b/pi-extension/src/hooks/subagent-context.ts @@ -0,0 +1,109 @@ +// subagent-context — ported from claude-plugin/hooks/subagent-context.cjs. +// +// In Claude this ran on SubagentStart and injected project/branch/TSV state +// into the subagent. pi has no native subagent-spawn event in the extension +// API, but the widely-used pi-subagents extension emits delegation events on +// the shared `pi.events` bus, and — crucially — pi subagents are full pi +// sessions that load this same extension and fire `before_agent_start`. So +// iteration context already flows into children via iteration-context.ts. +// +// This module does two things: +// 1. (child side) When this extension is running inside a subagent +// (PI_SUBAGENT_CHILD=1), build the same context block the Claude hook +// injected: project, branch, plans/reports path, active TSV, iteration +// count, latest row summary. Injected via before_agent_start. +// 2. (parent side) Listen for pi-subagents delegation events on pi.events to +// log subagent launches for observability (bounded metadata only). +// +// Fails open on any error. + +import { join, relative } from "node:path"; +import { + isHookEnabled, + loadSessionState, + log, +} from "../lib/session-state.js"; +import { findRecentTsv, readTsvTail } from "../lib/tsv.js"; + +const HOOK_NAME = "subagent-context"; + +/** True when this extension instance is running inside a pi-subagents child. */ +export function isSubagentChild(): boolean { + return process.env.PI_SUBAGENT_CHILD === "1"; +} + +function relativePath(cwd: string, absPath: string): string { + try { + return relative(cwd, absPath); + } catch { + return absPath; + } +} + +function summarizeLastRow(header: string, row: string | undefined): string { + if (!row) return "none"; + const headerCols = header ? header.split(/\t|\|/) : []; + const rowCols = row.split(/\t|\|/); + const parts: string[] = []; + rowCols.forEach((val, i) => { + const col = (headerCols[i] || "").toLowerCase(); + if ( + col.includes("status") || + col.includes("result") || + col.includes("pass") || + col.includes("fail") + ) { + parts.push(val.trim()); + } else if (/^-?\d+(\.\d+)?$/.test(val.trim()) && parts.length < 3) { + const label = headerCols[i] ? headerCols[i].trim() + "=" : ""; + parts.push(label + val.trim()); + } + }); + return parts.length > 0 ? parts.join(", ") : rowCols.slice(0, 3).join(", "); +} + +export interface SubagentContextResult { + /** Context block to inject into the subagent, or null. */ + text: string | null; +} + +// Build the subagent context block. Called from before_agent_start when +// PI_SUBAGENT_CHILD=1. Mirrors subagent-context.cjs output exactly. +export function buildSubagentContext( + cwd: string, + sessionId: string, +): SubagentContextResult { + if (!isHookEnabled(HOOK_NAME)) return { text: null }; + try { + const state = loadSessionState(cwd, sessionId); + const tsvPath = findRecentTsv(cwd, 30); + + if (!tsvPath) { + log(HOOK_NAME, { action: "skip", reason: "no-active-tsv" }); + return { text: null }; + } + + const tsv = readTsvTail(tsvPath, 1); + const relTsv = relativePath(cwd, tsvPath); + const latestSummary = tsv + ? summarizeLastRow(tsv.header, tsv.rows[0]) + : "none"; + + const text = [ + "## Autoresearch context (for subagent)", + `- Project: ${state.projectRoot || cwd}`, + `- Branch: ${state.gitBranch || "unknown"}`, + `- Plans: ${state.plansPath || join(cwd, "plans")}`, + `- Reports: ${state.reportsPath || join(cwd, "plans", "reports")}`, + `- Active TSV: ${relTsv}`, + `- Iteration: ${state.iterationCount || 0}`, + `- Latest: ${latestSummary}`, + ].join("\n"); + + log(HOOK_NAME, { action: "inject", subagent: "child" }); + return { text }; + } catch { + // fail-open + return { text: null }; + } +} diff --git a/pi-extension/src/index.ts b/pi-extension/src/index.ts index daf753ce..81d9164b 100644 --- a/pi-extension/src/index.ts +++ b/pi-extension/src/index.ts @@ -15,6 +15,7 @@ // UserPrompt → input (simplify-gate) // SessionStart → session_start (session-init) // SessionEnd → session_shutdown (stop-notify) +// SubagentStart → before_agent_start (subagent-context, child-side) + pi.events (parent) // // Every hook fails open — a guardrail malfunction never blocks work. diff --git a/tests/test-pi.sh b/tests/test-pi.sh index 369fec21..fe4cad97 100755 --- a/tests/test-pi.sh +++ b/tests/test-pi.sh @@ -434,6 +434,46 @@ process.env.AR_DISABLE_SCOUT_BLOCK = "1"; process.stdout.write(!isHookEnabled("scout-block") ? "PASS" : "FAIL"); ' +# ============================================================================ +printf '\n--- Testing subagent-context ---\n' +# ============================================================================ + +SUB_REPO="$(mktemp -d)" +( cd "$SUB_REPO" && mkdir -p autoresearch/run001 && printf 'iteration\tstatus\tmetric\n1\tpass\t0.85\n2\tpass\t0.87\n' > autoresearch/run001/results.tsv ) >/dev/null 2>&1 + +case_ok "subagent-context: isSubagentChild detects PI_SUBAGENT_CHILD" ' +import { isSubagentChild } from "./src/hooks/subagent-context.ts"; +process.env.PI_SUBAGENT_CHILD = "1"; +process.stdout.write(isSubagentChild() ? "PASS" : "FAIL"); +' + +case_ok "subagent-context: no TSV returns null" ' +import { buildSubagentContext } from "./src/hooks/subagent-context.ts"; +const r = buildSubagentContext("/nonexistent-cwd-12345", "sess-no-tsv"); +process.stdout.write(r.text === null ? "PASS" : "FAIL: " + r.text); +' + +case_ok "subagent-context: with active TSV injects header" " +import { buildSubagentContext } from \"./src/hooks/subagent-context.ts\"; +const r = buildSubagentContext('$SUB_REPO', 'sess-sub'); +process.stdout.write(r.text && r.text.includes('Autoresearch context (for subagent)') ? 'PASS' : 'FAIL: ' + (r.text||'null')); +" + +case_ok "subagent-context: contains TSV path + iteration" " +import { buildSubagentContext } from \"./src/hooks/subagent-context.ts\"; +const r = buildSubagentContext('$SUB_REPO', 'sess-sub2'); +process.stdout.write(r.text && r.text.includes('Active TSV:') && r.text.includes('Iteration:') ? 'PASS' : 'FAIL: ' + (r.text||'null')); +" + +case_ok "subagent-context: disabled via env var" " +process.env.AR_DISABLE_SUBAGENT_CONTEXT = '1'; +import { buildSubagentContext } from \"./src/hooks/subagent-context.ts\"; +const r = buildSubagentContext('$SUB_REPO', 'sess-disabled'); +process.stdout.write(r.text === null ? 'PASS' : 'FAIL: still injecting'); +" + +rm -rf "$SUB_REPO" + # ============================================================================ # Type-check: the extension must compile clean against pi types. # ============================================================================