From 2a74533463aa6f533c4592a5d48b422a3d6027ec Mon Sep 17 00:00:00 2001 From: BGamboa13 Date: Fri, 2 Oct 2026 11:37:44 -0400 Subject: [PATCH 1/2] feat(agents): ordered per-role model fallbacks on provider quota exhaustion A role in subagents.json model_profiles can list ordered `fallbacks`. When the role's model reports credit or quota exhaustion, the same task continues on the next fallback in the same child session and keeps the role's thinking level. Only real exhaustion triggers it: the usage limits Pi itself refuses to retry, plus explicit exhaustion wording (402, credit balance). Rate limits, concurrency caps, overloads and ordinary errors never do; Pi's own retry still handles them. The thread, subagent_status and subagent_result show the fallback taken. Refs #964 --- docs/gentle-shell.md | 24 +++++++ extensions/gentle-agents.ts | 17 ++++- extensions/gentle-ai.ts | 17 ++++- lib/agents-config.ts | 50 ++++++++++++-- lib/agents-protocol.ts | 17 ++++- lib/agents-quota.ts | 33 +++++++++ lib/agents-runner.ts | 67 ++++++++++++++++++- tests/agents-config.test.ts | 52 +++++++++++++++ tests/agents-quota.test.ts | 82 +++++++++++++++++++++++ tests/agents-runner.test.ts | 129 +++++++++++++++++++++++++++++++++++- tests/gentle-ai.test.ts | 21 ++++++ 11 files changed, 493 insertions(+), 16 deletions(-) create mode 100644 lib/agents-quota.ts create mode 100755 tests/agents-quota.test.ts diff --git a/docs/gentle-shell.md b/docs/gentle-shell.md index fda812207..7c9481e6c 100644 --- a/docs/gentle-shell.md +++ b/docs/gentle-shell.md @@ -202,6 +202,30 @@ The `subagent_*` tools and the agents card replace the third-party subagents pac Agent paths follow `GENTLE_PI_AGENT_HOME`, then `PI_CODING_AGENT_DIR`, then `~/.pi/agent` for definitions, config, history, child sessions, and transcripts. These overrides select the agent profile; they do not sandbox project or shared global resources. +#### Role model fallbacks + +A `model_profiles` entry in `subagents.json` (global or project) may list ordered `fallbacks` for the role. When the role's model reports explicit credit or quota exhaustion, the same task continues on the next fallback with the role's thinking level unchanged: + +```json +{ + "model_profiles": { + "gentle-ai-worker": { + "model": "provider-a/primary-model", + "effort": "high", + "fallbacks": ["provider-b/fallback-model", "provider-c/primary-model"] + } + } +} +``` + +- Only explicit exhaustion triggers a fallback: the usage limits Pi itself refuses to retry (`insufficient_quota`, `quota exceeded`, `billing`, `available balance`, `out of budget`, monthly, free and subscription usage limits), plus HTTP 402 and an exhausted quota or credit balance. Rate limits never do, even when worded as a quota (`Quota exceeded … per minute`) or reported only as `RESOURCE_EXHAUSTED`; neither do concurrency caps, overloads or ordinary errors. Pi's own retries handle the transient ones before the child settles, so a quota error Pi does retry falls back only once that retry budget is spent. +- A fallback may reuse the primary's model id on another provider (account rotation). An entry equal to the primary itself is ignored, as are non-string or empty entries. +- A project `fallbacks` list replaces the global one for that role; a project entry without `fallbacks` inherits it, and `"fallbacks": []` clears it. Fallbacks apply whichever source supplied the primary (profile, agent frontmatter or `default_model`). +- The task continues in the failed child's own session when its file exists, so finished work and context carry over; otherwise it restarts with the original prompt and context. Each attempt is one new child, bounded by the list. +- The agents card and `subagent_status` show the model currently running; `subagent_status` and the result name every model tried, in order, and the reason (`fallback: provider-a/primary-model -> provider-b/fallback-model (provider quota exhausted)`), and the task thread records a note. When every model is exhausted the task fails with the models tried and a hint to add credits or another fallback. +- Applying a profile from `/gentle:profiles` or `/gentle:models`, or a repository's pinned profile, keeps a role's `fallbacks` while its primary model is unchanged (including a role that names no primary on either side) and drops them when the primary changes. +- Fallbacks cover subagent roles. The primary orchestrator (#882) and in-process review lenses keep their own routing, and continuing a finished task later starts again from the role's primary. + ```text ╭─ ❀ Agents · 1 active · 1 done ─────────────────────────────── 1m24s ╮ │ ✓ gentle-ai-explore map footer sources gpt-5.6-terra · 34k · $0.27 · 25s │ diff --git a/extensions/gentle-agents.ts b/extensions/gentle-agents.ts index 45b145256..0cb84b65f 100644 --- a/extensions/gentle-agents.ts +++ b/extensions/gentle-agents.ts @@ -254,12 +254,22 @@ function taskDetails(task: TaskRecord): Record { export function describeTask(task: TaskRecord): string { const head = `${task.id} · ${task.agent} · ${task.status} · ${task.mode}`; const detail = task.error ? `\n${task.error}` : ""; - return `${head} · cwd: ${task.cwd} · ${task.turns} turns · ${task.toolCalls} tool calls · last: ${task.lastStep}${detail}`; + return `${head} · cwd: ${task.cwd} · ${task.turns} turns · ${task.toolCalls} tool calls · last: ${task.lastStep}${fallbackSuffix(task)}${detail}`; +} + +// Which model really did the work after a role fallback, and why it moved. +function fallbackNote(task: TaskRecord): string { + return task.fallback ? `fallback: ${task.fallback.models.join(" -> ")} (${task.fallback.reason})` : ""; +} + +function fallbackSuffix(task: TaskRecord): string { + return task.fallback ? ` · ${fallbackNote(task)}` : ""; } function finishedText(task: TaskRecord): string { - if (task.status === "completed") return task.result ?? "(the subagent returned no text)"; - return `Subagent ${task.agent} ${task.status}${task.error ? `: ${task.error}` : ""}${task.result ? `\n\nLast answer:\n${task.result}` : ""}`; + const note = task.fallback ? `\n\n[${fallbackNote(task)}]` : ""; + if (task.status === "completed") return `${task.result ?? "(the subagent returned no text)"}${note}`; + return `Subagent ${task.agent} ${task.status}${task.error ? `: ${task.error}` : ""}${task.result ? `\n\nLast answer:\n${task.result}` : ""}${note}`; } // pi's keybinding hint needs a live theme; outside one (tests, headless) the @@ -1299,6 +1309,7 @@ export default function gentleAgents(pi: ExtensionAPI, env: NodeJS.ProcessEnv = ...(target === undefined || foreign ? {} : { onLaunch: () => { registry.register(target, "subagent:spawn"); } }), model: profile.model, thinking: profile.thinking, + ...(profile.fallbacks === undefined ? {} : { fallbacks: profile.fallbacks }), sessionDir, resumeSessionPath: resume, ...(deps.childExtensionPaths && deps.childExtensionPaths.length > 0 ? { extensionPaths: [...deps.childExtensionPaths] } : {}), diff --git a/extensions/gentle-ai.ts b/extensions/gentle-ai.ts index 99c1015c5..53d66198d 100644 --- a/extensions/gentle-ai.ts +++ b/extensions/gentle-ai.ts @@ -2410,6 +2410,19 @@ function modelProfileForRoutingEntry( return Object.keys(profile).length > 0 ? profile : undefined; } +// The profile store knows nothing about role fallbacks, so rewriting a role +// from it must not erase the list a user keeps in subagents.json. The list +// belongs to the primary it was written for: it survives only while the +// materialized model is unchanged. +function withPreservedFallbacks( + profile: Record | undefined, + existing: unknown, +): Record | undefined { + if (!profile || !isRecord(existing) || !Array.isArray(existing.fallbacks)) return profile; + if (existing.model !== profile.model) return profile; + return { ...profile, fallbacks: existing.fallbacks }; +} + function updateSubagentModelProfileAtPath( path: string, name: string, @@ -2428,7 +2441,7 @@ function updateSubagentModelProfileAtPath( const modelProfiles = isRecord(config.model_profiles) ? { ...config.model_profiles } : {}; - const profile = modelProfileForRoutingEntry(entry); + const profile = withPreservedFallbacks(modelProfileForRoutingEntry(entry), modelProfiles[name]); // A write that would leave the profile as it is (including removing a // profile that was never there) is not an update and touches no file. if (JSON.stringify(modelProfiles[name]) === JSON.stringify(profile)) return false; @@ -2461,7 +2474,7 @@ async function updateSubagentModelProfileAtPathAsync( const modelProfiles = isRecord(config.model_profiles) ? { ...config.model_profiles } : {}; - const profile = modelProfileForRoutingEntry(entry); + const profile = withPreservedFallbacks(modelProfileForRoutingEntry(entry), modelProfiles[name]); // A write that would leave the profile as it is (including removing a // profile that was never there) is not an update and touches no file. if (JSON.stringify(modelProfiles[name]) === JSON.stringify(profile)) return false; diff --git a/lib/agents-config.ts b/lib/agents-config.ts index 258818d1a..166becb4c 100644 --- a/lib/agents-config.ts +++ b/lib/agents-config.ts @@ -62,6 +62,9 @@ export interface AgentDefinitionError { export interface ModelProfile { model: ModelRef | undefined; thinking: ThinkingLevel | undefined; + // Ordered models tried after the primary reports quota exhaustion. Absent + // means "not configured" (a lower scope may supply it); empty means "none". + fallbacks?: ModelRef[]; } export interface AgentsConfig { @@ -83,6 +86,8 @@ export interface ProfileSources { export interface ResolvedProfile { model: ModelRef | undefined; thinking: ThinkingLevel | undefined; + // Present only when the role has fallbacks distinct from the resolved primary. + fallbacks?: ModelRef[]; source: ProfileSources; } @@ -239,6 +244,21 @@ function positiveInteger(value: unknown, fallback: number): number { return typeof value === "number" && Number.isInteger(value) && value > 0 ? value : fallback; } +// A non-array value is "not configured"; unusable entries are dropped and +// repeats keep their first position. +function parseFallbacks(value: unknown): ModelRef[] | undefined { + if (!Array.isArray(value)) return undefined; + const seen = new Set(); + const refs: ModelRef[] = []; + for (const item of value) { + const ref = parseModelRef(item); + if (!ref || seen.has(formatModelRef(ref))) continue; + seen.add(formatModelRef(ref)); + refs.push(ref); + } + return refs; +} + function parseProfiles(value: unknown): Record { const profiles: Record = {}; if (!value || typeof value !== "object") return profiles; @@ -246,7 +266,12 @@ function parseProfiles(value: unknown): Record { if (!raw || typeof raw !== "object") continue; const entry = raw as Record; const thinking = parseThinking(entry.effort ?? entry.thinking); - profiles[name] = { model: parseModelRef(entry.model), thinking: thinking !== undefined && THINKING_LEVELS.includes(thinking) ? (thinking as ThinkingLevel) : undefined }; + const fallbacks = parseFallbacks(entry.fallbacks); + profiles[name] = { + model: parseModelRef(entry.model), + thinking: thinking !== undefined && THINKING_LEVELS.includes(thinking) ? (thinking as ThinkingLevel) : undefined, + ...(fallbacks === undefined ? {} : { fallbacks }), + }; } return profiles; } @@ -254,7 +279,8 @@ function parseProfiles(value: unknown): Record { function mergeProfiles(base: Record, override: Record): Record { const merged = { ...base }; for (const [name, profile] of Object.entries(override)) { - merged[name] = { model: profile.model ?? base[name]?.model, thinking: profile.thinking ?? base[name]?.thinking }; + const fallbacks = profile.fallbacks ?? base[name]?.fallbacks; + merged[name] = { model: profile.model ?? base[name]?.model, thinking: profile.thinking ?? base[name]?.thinking, ...(fallbacks === undefined ? {} : { fallbacks }) }; } return merged; } @@ -309,7 +335,20 @@ export function withPinnedModelProfiles( // repository that pinned a different one, which is the exact conflict a pin // exists to remove. Only `modelProfiles` moves: the orchestrator routing and // every operational default stay global. - return { ...config, modelProfiles: parseProfiles(pinned) }; + const modelProfiles = parseProfiles(pinned); + // The pin store carries no fallbacks. A role that keeps the very same primary + // keeps the fallbacks configured for it; a different primary starts clean. + for (const [name, profile] of Object.entries(modelProfiles)) { + const configured = config.modelProfiles[name]; + // Both unset means the role still inherits the same primary (definition or default_model). + const samePrimary = profile.model === undefined + ? configured?.model === undefined + : configured?.model !== undefined && formatModelRef(profile.model) === formatModelRef(configured.model); + if (profile.fallbacks === undefined && configured?.fallbacks !== undefined && samePrimary) { + profile.fallbacks = configured.fallbacks; + } + } + return { ...config, modelProfiles }; } function pick(candidates: Array<[T | undefined, ProfileSource]>): [T | undefined, ProfileSource] { @@ -328,7 +367,10 @@ export function resolveAgentProfile(agent: AgentDefinition, config: AgentsConfig [agent.thinking, PROFILE_SOURCE.DEFINITION], [config.defaultThinking, PROFILE_SOURCE.DEFAULT], ]); - return { model, thinking, source: { model: modelSource, thinking: thinkingSource } }; + // A fallback equal to the primary would only repeat the exhausted route. The + // same model id on another provider (account rotation) is a different ref. + const fallbacks = (profile?.fallbacks ?? []).filter((ref) => formatModelRef(ref) !== formatModelRef(model)); + return { model, thinking, ...(fallbacks.length > 0 ? { fallbacks } : {}), source: { model: modelSource, thinking: thinkingSource } }; } export function formatModelRef(model: ModelRef | undefined): string { diff --git a/lib/agents-protocol.ts b/lib/agents-protocol.ts index e8e3e9065..07a00a569 100644 --- a/lib/agents-protocol.ts +++ b/lib/agents-protocol.ts @@ -1,4 +1,5 @@ import { createHash } from "node:crypto"; +import { isQuotaExhaustion } from "./agents-quota.ts"; import { sanitizeTerminalText } from "./terminal-theme.ts"; // Gentle Agents protocol. A child pi process streams RPC events; the host @@ -64,6 +65,8 @@ export interface AgentEndEvent { text: string; outcome: "success" | "error" | "aborted" | "empty"; diagnostic?: string; + /** The provider explicitly reported credit/quota exhaustion (see agents-quota.ts). */ + quotaExhausted?: true; } export interface AgentSettledEvent { type: typeof TASK_EVENT.AGENT_SETTLED } export interface ErrorEvent { type: typeof TASK_EVENT.ERROR; message: string } @@ -137,6 +140,8 @@ export interface TaskRecord { toolCalls: number; tokens: number; cost: number; + /** Set once a role fallback took over: why, and every model tried in order (the last one is current). */ + fallback?: { reason: string; models: string[] }; } export interface TaskSummary { @@ -175,12 +180,18 @@ function keepTail(text: string, max: number): string { function terminalAssistant(messages: unknown): Omit { if (!Array.isArray(messages)) return { text: "", outcome: "empty", diagnostic: "assistant returned no final report" }; for (let index = messages.length - 1; index >= 0; index -= 1) { - const message = messages[index] as { role?: string; content?: unknown; stopReason?: unknown }; + const message = messages[index] as { role?: string; content?: unknown; stopReason?: unknown; errorMessage?: unknown }; if (message?.role !== "assistant") continue; const stopReason = clean(message.stopReason).toLowerCase(); // Do not preserve unbounded provider error payloads. The terminal reason is - // enough for an operator to distinguish failure from an empty report. - if (stopReason === "error") return { text: "", outcome: "error", diagnostic: "assistant reported an error" }; + // enough for an operator to distinguish failure from an empty report. The + // one thing the payload may say that routing needs is quota exhaustion, so + // it is classified here and only a fixed phrase and a flag leave this scope. + if (stopReason === "error") { + return isQuotaExhaustion(message.errorMessage) + ? { text: "", outcome: "error", diagnostic: "assistant reported an error: provider quota exhausted", quotaExhausted: true } + : { text: "", outcome: "error", diagnostic: "assistant reported an error" }; + } if (stopReason === "aborted") return { text: "", outcome: "aborted", diagnostic: "assistant aborted" }; const text = contentText(message.content); return text.length > 0 diff --git a/lib/agents-quota.ts b/lib/agents-quota.ts new file mode 100644 index 000000000..a17194045 --- /dev/null +++ b/lib/agents-quota.ts @@ -0,0 +1,33 @@ +// Explicit provider credit/quota exhaustion, the only failure that moves a role +// to its next fallback model. A rate limit, a concurrency cap or an overload is +// transient and stays with Pi's own retry policy; routing must never change for +// those, so this classifier is deliberately conservative. + +const MAX_CLASSIFIED_CHARS = 2_000; + +// Pi's own non-retryable provider-limit vocabulary (pi-ai `isRetryableAssistantError`). +// Not retrying is not proof of exhaustion: a per-minute "quota exceeded" is a rate +// limit, so the transient guard below applies to these too. +const PI_PROVIDER_LIMIT = /GoUsageLimitError|FreeUsageLimitError|Monthly usage limit reached|available balance|insufficient_quota|out of budget|quota exceeded|billing|subscription_sharing_usage_limit_exceeded/i; + +// Further explicit exhaustion wording from providers outside Pi's list. +const QUOTA_EXHAUSTED = [ + /^\s*402\b/, + /\bpayment required\b/i, + /\binsufficient[_\s-]+(quota|credits?|funds|balance)\b/i, + /\b(exceeded|exhausted|out of)\b.{0,40}\b(quota|credits?)\b/i, + /\bquota\b.{0,40}\b(exhausted|reached)\b/i, + /\bcredit balance\b/i, + /usage[_\s-]?limit/i, +]; + +// Transient limiters always win. RESOURCE_EXHAUSTED alone is not evidence either: +// Google also uses it for RPM/TPM throttling, so it needs exhaustion wording above. +const TRANSIENT_LIMIT = /\b(concurren\w*|simultaneous|per[\s-]+(minute|second)|rpm|tpm|rate[_\s-]?limit\w*|too many requests|overloaded)\b/i; + +export function isQuotaExhaustion(message: unknown): boolean { + if (typeof message !== "string") return false; + const text = message.slice(0, MAX_CLASSIFIED_CHARS); + if (TRANSIENT_LIMIT.test(text)) return false; + return PI_PROVIDER_LIMIT.test(text) || QUOTA_EXHAUSTED.some((pattern) => pattern.test(text)); +} diff --git a/lib/agents-runner.ts b/lib/agents-runner.ts index d48c260ce..ed3d6d0d8 100644 --- a/lib/agents-runner.ts +++ b/lib/agents-runner.ts @@ -125,6 +125,12 @@ export interface TaskRequest { thinking: string | undefined; sessionDir: string; resumeSessionPath: string | undefined; + // Untried models for this role, in order. When the child settles failed with + // explicit provider quota exhaustion, the runner continues the same task on the + // next one, keeping the role's thinking level. Omitted means no fallback. + fallbacks?: ModelRef[]; + // Runner-owned: every model this task has run on, in order. Set by a fallback. + attemptedModels?: string[]; env: NodeJS.ProcessEnv; // Untrusted narrowing intent; paths come only from matching host provenance. extensionPaths?: string[]; @@ -205,8 +211,14 @@ interface LiveTask { // STDERR_TAIL_MAX characters. Only surfaced on the stall and pre-settle exit // terminal paths, never on completed, cancelled, or other failure reasons. stderrTail: string; + // The latest agent_end reported explicit quota exhaustion (a later successful + // retry inside the child clears it), and the request that replaces this child + // once it has exited. Set only at settlement. + quotaExhausted?: boolean; + fallback?: TaskRequest; } +const FALLBACK_REASON = "provider quota exhausted"; const STDERR_TAIL_MAX = 512; const DIR_MODE = 0o700; const FILE_MODE = 0o600; @@ -419,7 +431,13 @@ export class AgentRunner { this.finish(id, TASK_STATUS.CANCELLED, `${reason} before start`); return true; } - if (!this.live.has(id)) return false; + const live = this.live.get(id); + if (!live) return false; + // A child already stopping for a fallback must not be relaunched after this. + if (live.fallback) { + live.fallback = undefined; + if (live.terminal) live.terminal = { status: TASK_STATUS.CANCELLED, error: reason }; + } this.requestStop(id, TASK_STATUS.CANCELLED, reason, true); return true; } @@ -804,10 +822,18 @@ export class AgentRunner { } } if (event.type === TASK_EVENT.ASK) void this.answer(id, request, live, event.request, raw); + if (event.type === TASK_EVENT.AGENT_END) live.quotaExhausted = event.quotaExhausted === true; if (event.type === TASK_EVENT.AGENT_SETTLED) { if (live.observations) live.observations.agentSettled = true; const terminal = this.store.get(id); - if (terminal?.error) this.requestStop(id, TASK_STATUS.FAILED, terminal.error); + if (terminal?.error) { + let error = terminal.error; + if (live.quotaExhausted) { + live.fallback = this.fallbackRequest(id, request); + if (!live.fallback && request.attemptedModels) error = `${error}; every configured model is out of quota (tried ${request.attemptedModels.join(", ")}). Top up credits or add another fallback in model_profiles`; + } + this.requestStop(id, TASK_STATUS.FAILED, error); + } else if (terminal?.result) this.requestStop(id, TASK_STATUS.COMPLETED, null); else this.requestStop(id, TASK_STATUS.FAILED, "assistant settled without a final report"); } @@ -973,10 +999,47 @@ export class AgentRunner { return; } const terminal = live.terminal; + if (live.fallback && terminal?.status === TASK_STATUS.FAILED) { + this.relaunchOnFallback(id, live.fallback); + return; + } const exitDesc = formatChildExit(live.childExit, live.childExitSignal); this.finish(id, terminal ? terminal.status : TASK_STATUS.FAILED, terminal ? terminal.error : `pi exited with ${exitDesc} before agent_settled${this.stderrSuffix(live)}`, live); } + // The next route for a role whose model just exhausted its quota, or undefined + // when no fallback is left. Continuing the failed child's own session keeps the + // work it already did (files edited, tool results read) and the role's context, + // so the new model only gets a short nudge. A session file that never reached + // disk cannot be resumed; then the same prompt and context start over, which is + // safe because nothing was recorded to continue from. Thinking is never touched. + private fallbackRequest(id: string, request: TaskRequest): TaskRequest | undefined { + const [model, ...rest] = request.fallbacks ?? []; + if (!model) return undefined; + const session = this.store.get(id)?.sessionPath; + const resumed = session && existsSync(session) ? session : undefined; + return { + ...request, + model, + fallbacks: rest, + attemptedModels: [...(request.attemptedModels ?? [formatModelRef(request.model)]), formatModelRef(model)], + ...(resumed ? { resumeSessionPath: resumed, prompt: `The previous model stopped because its provider reported ${FALLBACK_REASON}. You are now ${formatModelRef(model)}. Continue the task from where the conversation stopped and do not redo finished work.`, context: undefined } : {}), + // The first launch already registered the worktree and opened telemetry. + onLaunch: undefined, + prepareResponseObservations: undefined, + canCollectResponseObservations: undefined, + collectResponseObservations: false, + }; + } + + private relaunchOnFallback(id: string, next: TaskRequest): void { + const models = next.attemptedModels ?? []; + const mode = next.resumeSessionPath ? "continuing the same session" : "restarting the task"; + this.store.update(id, { model: formatModelRef(next.model), result: null, error: null, endedAt: null, fallback: { reason: FALLBACK_REASON, models: [...models] } }); + this.store.apply(id, { type: TASK_EVENT.NOTE, text: `fallback: ${models[models.length - 2]} reported ${FALLBACK_REASON}; ${mode} on ${models[models.length - 1]}` }, this.deps.now()); + this.launch(id, next); + } + private finish(id: string, status: TaskRecord["status"], error: string | null, live?: LiveTask): void { const current = this.store.get(id); if (!current || isFinished(current.status)) return; diff --git a/tests/agents-config.test.ts b/tests/agents-config.test.ts index 9c0293383..50484fec4 100644 --- a/tests/agents-config.test.ts +++ b/tests/agents-config.test.ts @@ -247,3 +247,55 @@ test("a pinned profile replaces subagent routing and leaves every other default // repository can pin "everything inherits" without touching the global store. assert.deepEqual(withPinnedModelProfiles(global, {}).modelProfiles, {}); }); + +test("model profiles parse ordered fallbacks and ignore unusable entries", () => { + const config = parseAgentsConfig({ + model_profiles: { + "sdd-apply": { model: "provider-a/primary", thinking: "high", fallbacks: ["provider-b/fallback", " ", 7, null, "provider-b/fallback", "provider-c/other"] }, + plain: { model: "provider-a/primary" }, + notAList: { model: "provider-a/primary", fallbacks: "provider-b/fallback" }, + }, + }, undefined); + assert.deepEqual(config.modelProfiles["sdd-apply"].fallbacks, [{ provider: "provider-b", id: "fallback" }, { provider: "provider-c", id: "other" }]); + assert.equal("fallbacks" in config.modelProfiles.plain, false, "a profile without the key keeps today's shape"); + assert.equal("fallbacks" in config.modelProfiles.notAList, false, "a non-array value counts as not configured"); +}); + +test("a project fallback list replaces the global one and an absent list inherits it", () => { + const config = parseAgentsConfig( + { model_profiles: { worker: { model: "a/primary", fallbacks: ["b/one", "c/two"] }, reviewer: { model: "a/primary", fallbacks: ["b/one"] }, other: { fallbacks: ["b/one"] } } }, + { model_profiles: { worker: { effort: "low" }, reviewer: { fallbacks: ["d/three"] }, other: { fallbacks: [] } } }, + ); + assert.deepEqual(config.modelProfiles.worker.fallbacks?.map((ref) => ref.id), ["one", "two"], "a project that sets no list inherits the global one"); + assert.deepEqual(config.modelProfiles.reviewer.fallbacks, [{ provider: "d", id: "three" }], "a project list overrides the global list"); + assert.deepEqual(config.modelProfiles.other.fallbacks, [], "an explicit empty project list clears the fallbacks"); +}); + +test("resolveAgentProfile exposes fallbacks other than the resolved primary", () => { + const agent = parseAgentDefinition("---\nname: worker\n---\nbody", "/worker.md", "global") as Parameters[0]; + const config = parseAgentsConfig({ model_profiles: { worker: { model: "a/primary", effort: "high", fallbacks: ["a/primary", "b/primary", "c/other"] } } }, undefined); + const profile = resolveAgentProfile(agent, config); + assert.deepEqual(profile.fallbacks?.map((ref) => `${ref.provider}/${ref.id}`), ["b/primary", "c/other"], "the same id on another provider is kept, the exhausted route itself is not"); + assert.equal(profile.thinking, "high"); + assert.equal("fallbacks" in resolveAgentProfile(agent, parseAgentsConfig(undefined, undefined)), false); +}); + +test("a pin keeps configured fallbacks only for a role that keeps its primary", () => { + const global = parseAgentsConfig({ model_profiles: { + kept: { model: "a/primary", fallbacks: ["b/one"] }, + changed: { model: "a/primary", fallbacks: ["b/one"] }, + inherited: { effort: "high", fallbacks: ["b/one"] }, + newlySet: { fallbacks: ["b/one"] }, + } }, undefined); + const pinned = withPinnedModelProfiles(global, { + kept: { model: "a/primary", thinking: "low" }, + changed: { model: "z/other" }, + inherited: { thinking: "low" }, + newlySet: { model: "z/other" }, + }); + assert.deepEqual(pinned.modelProfiles.kept.fallbacks, [{ provider: "b", id: "one" }]); + assert.equal("fallbacks" in pinned.modelProfiles.changed, false); + // A role that names no primary on either side still inherits the same one. + assert.deepEqual(pinned.modelProfiles.inherited.fallbacks, [{ provider: "b", id: "one" }]); + assert.equal("fallbacks" in pinned.modelProfiles.newlySet, false); +}); diff --git a/tests/agents-quota.test.ts b/tests/agents-quota.test.ts new file mode 100755 index 000000000..991fded94 --- /dev/null +++ b/tests/agents-quota.test.ts @@ -0,0 +1,82 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import { isRetryableAssistantError, type AssistantMessage } from "@earendil-works/pi-ai"; +import { isQuotaExhaustion } from "../lib/agents-quota.ts"; +import { TASK_EVENT, normalizeRpcEvent } from "../lib/agents-protocol.ts"; + +// Only explicit credit/quota exhaustion may change a role's routing. + +// Pi's own non-retryable provider-limit messages that state real exhaustion: Pi +// fails these fast, so only a fallback keeps the work going. +const PI_LIMIT_MESSAGES = [ + "GoUsageLimitError: weekly limit reached", + "FreeUsageLimitError: free tier used up", + "Monthly usage limit reached", + "Your available balance is insufficient", + '429: {"error":{"type":"insufficient_quota"}}', + "The key is out of budget", + "Error: billing hard limit has been reached", + "subscription_sharing_usage_limit_exceeded", +]; + +// Pi does not retry this one either, but a per-minute quota is a rate limit, not +// exhaustion, so it must not change routing. +const PI_NON_RETRYABLE_RATE_LIMIT = "Quota exceeded for metric requests per minute"; + +test("isQuotaExhaustion accepts explicit credit and quota exhaustion", () => { + for (const message of [ + ...PI_LIMIT_MESSAGES, + '402: {"error":{"message":"Payment Required"}}', + '429: {"error":{"type":"insufficient_quota","message":"You exceeded your current quota, please check your plan and billing details."}}', + "Your credit balance is too low to access the API", + "You have hit your usage limit. Upgrade your plan.", + "Quota exceeded for this project", + "usage_limit_exceeded", + "The account is out of credits", + "insufficient credits", + "Error: billing hard limit has been reached", + '{"error":{"code":429,"status":"RESOURCE_EXHAUSTED","message":"You exceeded your current quota"}}', + ]) assert.equal(isQuotaExhaustion(message), true, message); +}); + +test("isQuotaExhaustion rejects rate limits, concurrency caps and unrelated failures", () => { + for (const message of [ + '429: {"message":"qwen3.6 concurrency limit: max 5 simultaneous requests.","type":"rate_limit_error"}', + "429 Too Many Requests", + "Rate limit reached for requests, please retry in 20s", + "overloaded_error: Overloaded", + PI_NON_RETRYABLE_RATE_LIMIT, + '429: {"error":{"code":429,"status":"RESOURCE_EXHAUSTED"}}', + "429 RESOURCE_EXHAUSTED", + "WebSocket error: connection reset", + "500 internal server error", + "context length exceeded", + "", + undefined, + null, + 42, + ]) assert.equal(isQuotaExhaustion(message), false, String(message)); +}); + +test("the classifier agrees with Pi's retry policy at the boundary", () => { + const failed = (errorMessage: string) => ({ role: "assistant", content: [], stopReason: "error", errorMessage }) as unknown as AssistantMessage; + // Exhaustion Pi will not retry: falling back is the only way forward. + for (const message of [...PI_LIMIT_MESSAGES, PI_NON_RETRYABLE_RATE_LIMIT]) assert.equal(isRetryableAssistantError(failed(message)), false, message); + // Transient limits Pi retries itself: routing must stay on the role's model. + for (const message of ['429: {"message":"qwen3.6 concurrency limit: max 5 simultaneous requests.","type":"rate_limit_error"}', "429 Too Many Requests", "Rate limit reached for requests, please retry in 20s", "overloaded_error: Overloaded"]) { + assert.equal(isRetryableAssistantError(failed(message)), true, message); + assert.equal(isQuotaExhaustion(message), false, message); + } +}); + +test("isQuotaExhaustion only inspects a bounded prefix", () => { + assert.equal(isQuotaExhaustion(`${"x".repeat(5_000)} insufficient_quota`), false); +}); + +test("an errored assistant message surfaces a quota flag without copying the provider text", () => { + const quota = normalizeRpcEvent({ type: "agent_end", messages: [{ role: "assistant", content: [], stopReason: "error", errorMessage: "insufficient_quota secret=never-copy" }] }); + assert.deepEqual(quota, [{ type: TASK_EVENT.AGENT_END, text: "", outcome: "error", diagnostic: "assistant reported an error: provider quota exhausted", quotaExhausted: true }]); + assert.doesNotMatch(JSON.stringify(quota), /never-copy/); + const rate = normalizeRpcEvent({ type: "agent_end", messages: [{ role: "assistant", content: [], stopReason: "error", errorMessage: "429 concurrency limit" }] }); + assert.deepEqual(rate, [{ type: TASK_EVENT.AGENT_END, text: "", outcome: "error", diagnostic: "assistant reported an error" }]); +}); diff --git a/tests/agents-runner.test.ts b/tests/agents-runner.test.ts index b9dbdd96a..5b39adea1 100644 --- a/tests/agents-runner.test.ts +++ b/tests/agents-runner.test.ts @@ -1,8 +1,9 @@ import assert from "node:assert/strict"; import test from "node:test"; -import fs, { existsSync, readFileSync, statSync } from "node:fs"; +import fs, { existsSync, mkdtempSync, readFileSync, rmSync, statSync, writeFileSync } from "node:fs"; import { syncBuiltinESMExports } from "node:module"; -import { basename, dirname } from "node:path"; +import { basename, dirname, join } from "node:path"; +import { tmpdir } from "node:os"; import { PassThrough } from "node:stream"; import { AGENT_MODE, parseAgentsConfig, resolveAgentProfile, type AgentDefinition } from "../lib/agents-config.ts"; import { TASK_STATUS, TaskStore, type TaskRecord } from "../lib/agents-protocol.ts"; @@ -1693,3 +1694,127 @@ test("temporary instructions transport file is cleaned up if child emits an earl assert.ok(!existsSync(capturedPromptPath), "temporary transport file must be cleaned up on early child error"); assert.ok(!existsSync(dirname(capturedPromptPath)), "temporary transport directory must be cleaned up on early child error"); }); + +// Role fallbacks (#964): explicit provider quota exhaustion moves the same task to +// the role's next model. Anything else, and an empty list, fails as it always did. + +const QUOTA_ERROR = { role: "assistant", content: [], stopReason: "error", errorMessage: '402: {"error":{"type":"insufficient_quota","message":"You exceeded your current quota"}}' }; +const CONCURRENCY_ERROR = { role: "assistant", content: [], stopReason: "error", errorMessage: '429: {"message":"qwen3.6 concurrency limit: max 5 simultaneous requests.","type":"rate_limit_error"}' }; +const FINAL_REPORT = { role: "assistant", content: [{ type: "text", text: "fallback report" }], stopReason: "stop" }; +const PRIMARY = { provider: "openai-codex", id: "gpt-5.6-terra" }; +const FALLBACK_A = { provider: "provider-b", id: "fallback-a" }; +const FALLBACK_B = { provider: "provider-c", id: "fallback-b" }; + +function settle(child: FakeChild, message: Record): void { + child.emit({ type: "agent_end", messages: [message] }); + child.emit({ type: "agent_settled" }); +} + +function modelArgument(args: string[]): string | undefined { + return args[args.indexOf("--model") + 1]; +} + +test("AgentRunner relaunches the same task on the next fallback after quota exhaustion", async () => { + const { store, runner, children, spawnOptions } = harness(); + const task = runner.run(request({ fallbacks: [FALLBACK_A, FALLBACK_B] })); + await tick(); + assert.equal(modelArgument(spawnOptions[0].args), "openai-codex/gpt-5.6-terra:high"); + + settle(children[0], QUOTA_ERROR); + await tick(); + await tick(); + + assert.equal(children.length, 2, "one replacement child, not one per remaining fallback"); + assert.equal(modelArgument(spawnOptions[1].args), "provider-b/fallback-a:high", "the role's thinking level is preserved"); + const running = store.get(task.id)!; + assert.equal(running.status, TASK_STATUS.RUNNING, "the task is not finished between attempts"); + assert.equal(running.model, "provider-b/fallback-a"); + assert.deepEqual(running.fallback, { reason: "provider quota exhausted", models: ["openai-codex/gpt-5.6-terra", "provider-b/fallback-a"] }); + assert.equal(children[1].written.find((command) => command.type === "prompt")?.message, "Map the repo", "no session file to continue, so the same prompt is sent again"); + assert.match(JSON.stringify(store.thread(task.id)), /fallback: openai-codex\/gpt-5.6-terra reported provider quota exhausted; restarting the task on provider-b\/fallback-a/); + + settle(children[1], FINAL_REPORT); + const finished = await runner.waitFor(task.id); + assert.equal(finished.status, TASK_STATUS.COMPLETED); + assert.equal(finished.result, "fallback report"); + assert.equal(finished.error, null); + assert.equal(finished.fallback?.models.at(-1), "provider-b/fallback-a", "the record names the model that completed the work"); +}); + +test("AgentRunner continues the failed child's session on a fallback when the session file exists", async () => { + const dir = mkdtempSync(join(tmpdir(), "gentle-fallback-")); + try { + const sessionFile = join(dir, "child.jsonl"); + writeFileSync(sessionFile, "{}\n"); + const { runner, children, spawnOptions } = harness({ state: { sessionFile } }); + const task = runner.run(request({ fallbacks: [FALLBACK_A], context: "extra context" })); + await tick(); + settle(children[0], QUOTA_ERROR); + await tick(); + await tick(); + + const args = spawnOptions[1].args; + assert.equal(args[args.indexOf("--session") + 1], sessionFile); + assert.equal(modelArgument(args), "provider-b/fallback-a:high"); + const prompt = String(children[1].written.find((command) => command.type === "prompt")?.message); + assert.match(prompt, /provider reported provider quota exhausted/); + assert.match(prompt, /Continue the task/); + assert.doesNotMatch(prompt, /extra context/, "the session already holds the task context"); + assert.equal(runner.cancel(task.id), true); + } finally { + rmSync(dir, { recursive: true, force: true }); + } +}); + +test("AgentRunner walks the fallback list in order and fails with the models tried when all are exhausted", async () => { + const { store, runner, children, spawnOptions } = harness(); + const task = runner.run(request({ fallbacks: [FALLBACK_A, FALLBACK_B] })); + await tick(); + for (let attempt = 0; attempt < 3; attempt += 1) { + settle(children[attempt], QUOTA_ERROR); + await tick(); + await tick(); + } + + assert.equal(children.length, 3, "attempts are bounded by the configured list"); + assert.deepEqual(spawnOptions.map((spawn) => modelArgument(spawn.args)), ["openai-codex/gpt-5.6-terra:high", "provider-b/fallback-a:high", "provider-c/fallback-b:high"]); + const failed = await runner.waitFor(task.id); + assert.equal(failed.status, TASK_STATUS.FAILED); + assert.match(failed.error ?? "", /every configured model is out of quota \(tried openai-codex\/gpt-5\.6-terra, provider-b\/fallback-a, provider-c\/fallback-b\)/); + assert.match(failed.error ?? "", /model_profiles/); + assert.equal(store.get(task.id)?.fallback?.models.length, 3); +}); + +test("AgentRunner does not fall back for rate limits, concurrency caps, generic errors or an empty list", async () => { + const scenarios = [ + { name: "concurrency limit", message: CONCURRENCY_ERROR, fallbacks: [FALLBACK_A] }, + { name: "generic error", message: { role: "assistant", content: [], stopReason: "error", errorMessage: "WebSocket error" }, fallbacks: [FALLBACK_A] }, + { name: "error without a message", message: { role: "assistant", content: [], stopReason: "error" }, fallbacks: [FALLBACK_A] }, + { name: "quota without fallbacks", message: QUOTA_ERROR, fallbacks: [] }, + { name: "quota with no list at all", message: QUOTA_ERROR, fallbacks: undefined }, + ]; + for (const scenario of scenarios) { + const { runner, children } = harness(); + const task = runner.run(request({ fallbacks: scenario.fallbacks })); + await tick(); + settle(children[0], scenario.message); + const failed = await runner.waitFor(task.id); + assert.equal(failed.status, TASK_STATUS.FAILED, scenario.name); + assert.equal(children.length, 1, `${scenario.name} must not relaunch`); + assert.equal(failed.fallback, undefined, scenario.name); + assert.doesNotMatch(failed.error ?? "", /out of quota/, scenario.name); + } +}); + +test("AgentRunner never relaunches a fallback after the task was cancelled while its child stopped", async () => { + const { runner, children } = harness({ exitOnKill: false }); + const task = runner.run(request({ fallbacks: [FALLBACK_A] })); + await tick(); + settle(children[0], QUOTA_ERROR); + await tick(); + assert.equal(runner.cancel(task.id, "user cancelled"), true); + children[0].exit(0); + const finished = await runner.waitFor(task.id); + assert.equal(finished.status, TASK_STATUS.CANCELLED); + assert.equal(children.length, 1); +}); diff --git a/tests/gentle-ai.test.ts b/tests/gentle-ai.test.ts index 371d13caf..904d9ecb4 100644 --- a/tests/gentle-ai.test.ts +++ b/tests/gentle-ai.test.ts @@ -2995,3 +2995,24 @@ test("switchLiveOrchestrator returns note when setModel fails", async () => { const result = await __testing.switchLiveOrchestrator(ctx, live, entry); assert.equal(result, "\nno authentication is configured for openai; this session keeps its current model."); }); + +test("applying a profile keeps role fallbacks only for a role whose primary model is unchanged", async (t) => { + const { fixture, writeStore, writeSettings } = profilesStoreFixture(t); + writeSettings(); + const helperPath = join(fixture.root, ".pi", "agents", "helper.md"); + writeMarkdown(helperPath, "---\nname: helper\ndescription: Helper\nmodel: openai/beta\n---\nbody\n"); + const subagentsPath = join(fixture.root, ".pi", "subagents.json"); + writeFileSync(subagentsPath, `${JSON.stringify({ model_profiles: { + worker: { model: "openai/alpha", effort: "low", fallbacks: ["other/alpha"] }, + helper: { model: "openai/beta", fallbacks: ["other/beta"] }, + } }, null, 2)}\n`); + writeStore({ team: { worker: { model: "openai/alpha", thinking: "high" }, helper: { model: "openai/gamma" } } }); + applyOnce(fixture); + await fixture.run("gentle:profiles"); + + const profiles = JSON.parse(readFileSync(subagentsPath, "utf8")); + assert.deepEqual(profiles.model_profiles, { + worker: { model: "openai/alpha", effort: "high", fallbacks: ["other/alpha"] }, + helper: { model: "openai/gamma" }, + }, "fallbacks follow the primary they were written for"); +}); From 8d14dc62a54f586c3dd3380213eba66fb70cc078 Mon Sep 17 00:00:00 2001 From: BGamboa13 Date: Fri, 2 Oct 2026 15:58:02 -0400 Subject: [PATCH 2/2] fix(agents): keep a fallback-only profile when its role's routing is cleared Clearing a role's routing entry produced no profile, and both profile writers then deleted the role's model_profiles entry, including a fallback-only one. Fallbacks survive while the primary is unchanged, and a cleared entry over a fallback-only profile changes no primary, so its fallbacks now stay. A role whose primary is cleared still drops them. --- extensions/gentle-ai.ts | 7 ++++--- tests/gentle-ai.test.ts | 20 ++++++++++++++++++++ 2 files changed, 24 insertions(+), 3 deletions(-) diff --git a/extensions/gentle-ai.ts b/extensions/gentle-ai.ts index 53d66198d..aac1b3e55 100644 --- a/extensions/gentle-ai.ts +++ b/extensions/gentle-ai.ts @@ -2413,13 +2413,14 @@ function modelProfileForRoutingEntry( // The profile store knows nothing about role fallbacks, so rewriting a role // from it must not erase the list a user keeps in subagents.json. The list // belongs to the primary it was written for: it survives only while the -// materialized model is unchanged. +// materialized model is unchanged. That includes a cleared entry over a +// fallback-only profile, where neither side names a primary. function withPreservedFallbacks( profile: Record | undefined, existing: unknown, ): Record | undefined { - if (!profile || !isRecord(existing) || !Array.isArray(existing.fallbacks)) return profile; - if (existing.model !== profile.model) return profile; + if (!isRecord(existing) || !Array.isArray(existing.fallbacks)) return profile; + if (existing.model !== profile?.model) return profile; return { ...profile, fallbacks: existing.fallbacks }; } diff --git a/tests/gentle-ai.test.ts b/tests/gentle-ai.test.ts index 904d9ecb4..f8023aec2 100644 --- a/tests/gentle-ai.test.ts +++ b/tests/gentle-ai.test.ts @@ -3016,3 +3016,23 @@ test("applying a profile keeps role fallbacks only for a role whose primary mode helper: { model: "openai/gamma" }, }, "fallbacks follow the primary they were written for"); }); + +test("clearing a role keeps a fallback-only profile and drops fallbacks whose primary is cleared", async (t) => { + const { fixture, writeStore, writeSettings } = profilesStoreFixture(t); + writeSettings(); + const helperPath = join(fixture.root, ".pi", "agents", "helper.md"); + writeMarkdown(helperPath, "---\nname: helper\ndescription: Helper\nmodel: openai/beta\n---\nbody\n"); + const subagentsPath = join(fixture.root, ".pi", "subagents.json"); + writeFileSync(subagentsPath, `${JSON.stringify({ model_profiles: { + worker: { fallbacks: ["other/alpha"] }, + helper: { model: "openai/beta", fallbacks: ["other/beta"] }, + } }, null, 2)}\n`); + writeStore({ team: { worker: {}, helper: {} } }); + applyOnce(fixture); + await fixture.run("gentle:profiles"); + + const profiles = JSON.parse(readFileSync(subagentsPath, "utf8")); + assert.deepEqual(profiles.model_profiles, { + worker: { fallbacks: ["other/alpha"] }, + }, "a cleared entry keeps fallbacks only where no primary changed"); +});