From fb39138adca36c33640637c141988db94d310990 Mon Sep 17 00:00:00 2001 From: Aaron Elijah Mars <61592645+aaronjmars@users.noreply.github.com> Date: Wed, 7 Oct 2026 16:02:41 -0400 Subject: [PATCH] chore(models): move the haiku tier to Claude Haiku 5.5 --- .github/workflows/aeon.yml | 5 +++-- CHANGELOG.md | 7 +++++++ aeon.yml | 4 ++-- apps/dashboard/lib/config.test.ts | 4 ++-- apps/dashboard/lib/constants.test.ts | 2 +- apps/dashboard/lib/constants.ts | 6 ++++-- apps/dashboard/lib/dispatch.test.ts | 2 +- docs/CONFIGURATION.md | 4 ++-- scripts/fleet-scorecard.mjs | 3 ++- scripts/llm-gateway.sh | 8 ++++++-- scripts/notify-jsonrender.sh | 2 +- scripts/tests/test_fleet_scorecard.mjs | 5 +++-- scripts/tests/test_llm_gateway.sh | 10 ++++++++-- skills/create-skill/SKILL.md | 4 ++-- skills/skill-repair/SKILL.md | 2 +- 15 files changed, 45 insertions(+), 23 deletions(-) diff --git a/.github/workflows/aeon.yml b/.github/workflows/aeon.yml index 64426a89856..16cca822cad 100644 --- a/.github/workflows/aeon.yml +++ b/.github/workflows/aeon.yml @@ -34,9 +34,10 @@ on: - '(config default)' - claude-sonnet-5-5 - claude-opus-5-5 - - claude-haiku-4-5-20251001 + - claude-haiku-5-5 - claude-sonnet-5 - claude-opus-4-8 + - claude-haiku-4-5-20251001 - grok-4.7 - grok-4.6 - grok-4.5 @@ -1750,7 +1751,7 @@ jobs: JSON format: {\"score\": N, \"assessment\": \"one line\", \"flags\": []}" - SCORE_MODEL="${ANTHROPIC_DEFAULT_HAIKU_MODEL:-claude-haiku-4-5-20251001}" + SCORE_MODEL="${ANTHROPIC_DEFAULT_HAIKU_MODEL:-claude-haiku-5-5}" # Score through the SAME harness the skill ran on, ALWAYS via the # harness-adapter — no native `claude -p` / `run-grok.sh` path remains. # A repo authed for only one harness has no Claude creds, so a direct diff --git a/CHANGELOG.md b/CHANGELOG.md index 788e919cba4..8c865d32b1b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -176,6 +176,13 @@ from or pin to; the template keeps serving the latest `main` to new forks. ### Changed +- **Haiku tier moves to Claude Haiku 5.5.** `claude-haiku-5-5` replaces `claude-haiku-4-5-20251001` + in the dashboard picker, the workflow_dispatch choices, the quality scorer's default, the + json-render notifier and the docs; `claude-haiku-4-5-20251001` stays a valid dispatch choice and + shows as "Haiku 4.5 (configured)". The OpenRouter haiku slot defaults to + `anthropic/claude-haiku-5.5`; Surplus does not serve it yet, so its haiku tier stays on + `claude-haiku-4.5`. Fleet scorecard prices Haiku 5.5 at $0.10 in / $0.50 out per 1M tokens. + - **Operator-console plugin fills its Claude directory listing.** `plugin/.claude-plugin/plugin.json` now sets `displayName`, author email, `icon` and the docs, support, privacy and terms links that Anthropic's plugin directory reads for the listing, and a shorter `description` that fits the diff --git a/aeon.yml b/aeon.yml index 2c85fe36879..ef2a6d6d835 100644 --- a/aeon.yml +++ b/aeon.yml @@ -2,7 +2,7 @@ # # Enable skills by setting enabled: true. Multiple skills at the same time run in parallel. # Set var: "value" to pass a default parameter to the skill. -# Set model: "claude-haiku-4-5-20251001" per skill to override the default model (cost optimization). +# Set model: "claude-haiku-5-5" per skill to override the default model (cost optimization). # Set harness: "" per skill (claude, grok, codex, pi, vibe, kimi, fx, cursor, # hermes) to run it on a different coding-agent CLI than the default Claude Code # harness (see the top-level `harness:` key below). @@ -229,7 +229,7 @@ chains: # - { skill: digest, consume: [token-movers, github-trending] } # Default model for all skills. Override per-run via workflow_dispatch. -# Options: claude-sonnet-5-5, claude-opus-5-5, claude-haiku-4-5-20251001 +# Options: claude-sonnet-5-5, claude-opus-5-5, claude-haiku-5-5 # (older claude-sonnet-5 / claude-opus-4-8 still accepted) # (Grok harness model: grok-4.7, the flagship reasoning model and grok's current # default, run with --no-subagents in CI; grok-4.6 is also offered, and an existing diff --git a/apps/dashboard/lib/config.test.ts b/apps/dashboard/lib/config.test.ts index bf29ab017ca..58d6b3e1673 100644 --- a/apps/dashboard/lib/config.test.ts +++ b/apps/dashboard/lib/config.test.ts @@ -190,9 +190,9 @@ describe("updateModelInConfig", () => { }); it("replaces an existing model", () => { - const updated = updateModelInConfig(FULL_YAML, "claude-haiku-4-5-20251001"); + const updated = updateModelInConfig(FULL_YAML, "claude-haiku-5-5"); const config = parseConfig(updated); - assert.equal(config.model, "claude-haiku-4-5-20251001"); + assert.equal(config.model, "claude-haiku-5-5"); }); }); diff --git a/apps/dashboard/lib/constants.test.ts b/apps/dashboard/lib/constants.test.ts index 19e25c2c7bf..7ec94a5b76d 100644 --- a/apps/dashboard/lib/constants.test.ts +++ b/apps/dashboard/lib/constants.test.ts @@ -73,7 +73,7 @@ describe("keyProvidedByHarness", () => { describe("MODELS", () => { it("offers exactly the current three Claude models", () => { - assert.deepEqual(MODELS.map(m => m.id), ["claude-sonnet-5-5", "claude-opus-5-5", "claude-haiku-4-5-20251001"]); + assert.deepEqual(MODELS.map(m => m.id), ["claude-sonnet-5-5", "claude-opus-5-5", "claude-haiku-5-5"]); }); }); diff --git a/apps/dashboard/lib/constants.ts b/apps/dashboard/lib/constants.ts index 5e27585e58c..9e0a89e01da 100644 --- a/apps/dashboard/lib/constants.ts +++ b/apps/dashboard/lib/constants.ts @@ -6,11 +6,12 @@ import type { Harness } from './types' // harness-switch snap uses (modelsForHarness(...)[0] in app/page.tsx). Keep it in // sync with the config default in lib/config.ts and aeon.yml `model:`. // Just the current three. A repo or per-skill pin that still names an older id -// (claude-sonnet-5, claude-opus-4-8) keeps a visible entry via pickerOptions. +// (claude-sonnet-5, claude-opus-4-8, claude-haiku-4-5-20251001) keeps a visible +// entry via pickerOptions. export const MODELS = [ { id: 'claude-sonnet-5-5', label: 'Sonnet 5.5' }, { id: 'claude-opus-5-5', label: 'Opus 5.5' }, - { id: 'claude-haiku-4-5-20251001', label: 'Haiku 4.5' }, + { id: 'claude-haiku-5-5', label: 'Haiku 5.5' }, ] // Models offered when the Grok (`grok`) harness is selected: the ids the @@ -137,6 +138,7 @@ export const HARNESSES = [ const RETIRED_MODEL_LABELS: Record = { 'claude-sonnet-5': 'Sonnet 5', 'claude-opus-4-8': 'Opus 4.8', + 'claude-haiku-4-5-20251001': 'Haiku 4.5', 'openai/gpt-5.1-codex-mini': 'GPT-5.1 Codex Mini', 'openai/gpt-5-mini': 'GPT-5 Mini', 'openai/gpt-5.3-codex': 'GPT-5.3 Codex', diff --git a/apps/dashboard/lib/dispatch.test.ts b/apps/dashboard/lib/dispatch.test.ts index ded6c6c2592..38f6ac65ebc 100644 --- a/apps/dashboard/lib/dispatch.test.ts +++ b/apps/dashboard/lib/dispatch.test.ts @@ -31,7 +31,7 @@ describe("sanitizeModel", () => { it("keeps dotted versions and colon variant suffixes", () => { assert.equal(sanitizeModel("grok-4.5"), "grok-4.5"); assert.equal(sanitizeModel("vendor/model:free"), "vendor/model:free"); - assert.equal(sanitizeModel("claude-haiku-4-5-20251001"), "claude-haiku-4-5-20251001"); + assert.equal(sanitizeModel("claude-haiku-5-5"), "claude-haiku-5-5"); }); it("strips characters outside the id charset", () => { diff --git a/docs/CONFIGURATION.md b/docs/CONFIGURATION.md index 92d79d1acf7..992842e99ee 100644 --- a/docs/CONFIGURATION.md +++ b/docs/CONFIGURATION.md @@ -297,11 +297,11 @@ The default model for all skills is set in `aeon.yml` (or from the dashboard hea model: claude-sonnet-5-5 ``` -Options: `claude-sonnet-5-5` (default), `claude-opus-5-5`, `claude-haiku-4-5-20251001` (the older `claude-sonnet-5` and `claude-opus-4-8` are still accepted). Per-run overrides are available via workflow dispatch, and individual skills can override to optimize cost: +Options: `claude-sonnet-5-5` (default), `claude-opus-5-5`, `claude-haiku-5-5` (the older `claude-sonnet-5`, `claude-opus-4-8` and `claude-haiku-4-5-20251001` are still accepted). Per-run overrides are available via workflow dispatch, and individual skills can override to optimize cost: ```yaml skills: - token-movers: { enabled: true, schedule: "30 12 * * *", model: "claude-haiku-4-5-20251001" } + token-movers: { enabled: true, schedule: "30 12 * * *", model: "claude-haiku-5-5" } ``` > Model ids are for the **claude** harness; each other [harness](harnesses.md) carries its own list, which the dashboard picker swaps in when you select it. diff --git a/scripts/fleet-scorecard.mjs b/scripts/fleet-scorecard.mjs index 6da979e50f7..4f6dac9122d 100755 --- a/scripts/fleet-scorecard.mjs +++ b/scripts/fleet-scorecard.mjs @@ -118,13 +118,14 @@ const CLAUDE_PRICES = { 'claude-sonnet-4-6': { in: 3, out: 15, cr: 0.30 }, 'claude-sonnet-4-5': { in: 3, out: 15, cr: 0.30 }, 'claude-sonnet-4': { in: 3, out: 15, cr: 0.30 }, + 'claude-haiku-5-5': { in: 0.10, out: 0.50, cr: 0.01 }, 'claude-haiku-4-5': { in: 1, out: 5, cr: 0.10 }, 'claude-3-5-haiku': { in: 0.80, out: 4, cr: 0.08 }, }; // A Claude id with a version this table does not know yet (a new release) is // priced at its family's current flagship row, so it still shows up in the // cost columns instead of vanishing. -const CLAUDE_FAMILY_FALLBACK = { opus: 'claude-opus-5-5', sonnet: 'claude-sonnet-5-5', haiku: 'claude-haiku-4-5' }; +const CLAUDE_FAMILY_FALLBACK = { opus: 'claude-opus-5-5', sonnet: 'claude-sonnet-5-5', haiku: 'claude-haiku-5-5' }; const normalizeModel = (model) => String(model || '').trim().toLowerCase() .replace(/^anthropic\//, '').replace(/-\d{8}$/, '').replace(/\./g, '-'); // Rates in $ per token for a model, or null when it is not a Claude model. Rows diff --git a/scripts/llm-gateway.sh b/scripts/llm-gateway.sh index cf5e115312e..716ef1d33c0 100644 --- a/scripts/llm-gateway.sh +++ b/scripts/llm-gateway.sh @@ -262,7 +262,7 @@ case "${GATEWAY:-direct}" in # Map EVERY model slot Claude Code uses to OpenRouter slugs (opus/sonnet/haiku). export ANTHROPIC_DEFAULT_OPUS_MODEL="${OPENROUTER_MODEL:-anthropic/claude-opus-5.5}" export ANTHROPIC_DEFAULT_SONNET_MODEL="${OPENROUTER_MODEL_SONNET:-anthropic/claude-sonnet-5.5}" - export ANTHROPIC_DEFAULT_HAIKU_MODEL="${OPENROUTER_MODEL_HAIKU:-anthropic/claude-haiku-4.5}" + export ANTHROPIC_DEFAULT_HAIKU_MODEL="${OPENROUTER_MODEL_HAIKU:-anthropic/claude-haiku-5.5}" # Tiered mapping, same as the glm arm: the run's resolved model id picks the # slot, so sonnet-tier skills (and the scorer) stay on sonnet instead of every # run being billed as Opus. @@ -360,8 +360,12 @@ X-Title: ${OPENROUTER_APP_TITLE:-Aeon}" # - to . (claude-opus-5-5 -> claude-opus-5.5). # SURPLUS_MODEL overrides; opus-5.5 is the fallback when $MODEL is unset. # Surplus served claude-opus-5.5 and claude-sonnet-5.5 on 2026-10-01 - # (/api/inference/v1/models). + # (/api/inference/v1/models). It did not serve claude-haiku-5.5 yet on + # 2026-10-07, so the haiku tier stays on claude-haiku-4.5 there. surplus_model="${SURPLUS_MODEL:-$(printf '%s' "${MODEL:-claude-opus-5-5}" | sed -E 's/-[0-9]{8}$//; s/([0-9])-([0-9])/\1.\2/g')}" + if [ -z "${SURPLUS_MODEL:-}" ] && [ "$surplus_model" = "claude-haiku-5.5" ]; then + surplus_model="claude-haiku-4.5" + fi start_ccr_sidecar surplus \ "https://www.surplusintelligence.ai/api/inference/v1/chat/completions" \ "$SURPLUS_API_KEY" "$surplus_model" diff --git a/scripts/notify-jsonrender.sh b/scripts/notify-jsonrender.sh index 9a28bd8af68..2b0650c9f7f 100755 --- a/scripts/notify-jsonrender.sh +++ b/scripts/notify-jsonrender.sh @@ -60,7 +60,7 @@ CRITICAL RULES: # Use claude CLI — works with both API key and OAuth token SPEC=$(echo "$CONTENT" | claude -p "Convert this skill output into a json-render spec. Skill: ${SKILL}" \ - --model claude-haiku-4-5-20251001 \ + --model claude-haiku-5-5 \ --system-prompt "$SYSTEM" \ --max-turns 1 \ --output-format text 2>/dev/null) diff --git a/scripts/tests/test_fleet_scorecard.mjs b/scripts/tests/test_fleet_scorecard.mjs index 42fdb104417..711b7039869 100644 --- a/scripts/tests/test_fleet_scorecard.mjs +++ b/scripts/tests/test_fleet_scorecard.mjs @@ -56,6 +56,7 @@ test('prices Claude rows by model version and leaves non-Claude rows unpriced', '2026-10-01,a,claude-sonnet-5,1000000,1000000,0,0', // 2 + 10 = 12 '2026-10-01,a,claude-sonnet-4-6,1000000,1000000,0,0', // 3 + 15 = 18 '2026-10-01,a,claude-haiku-4-5-20251001,1000000,1000000,1000000,1000000', // 1 + 5 + read 0.1 + write 1.25 = 7.35 + '2026-10-01,a,claude-haiku-5-5,10000000,10000000,10000000,10000000', // 10M each at a tenth of 4.5's rates = 7.35 '2026-10-01,b,openai/gpt-5.1-codex-mini,1000000,1000000,0,0', // unpriced '2026-10-01,b,codex-default,1000000,1000000,0,0', // unpriced '2026-10-01,b,grok-4.5,1000000,1000000,0,0', // unpriced @@ -74,8 +75,8 @@ test('prices Claude rows by model version and leaves non-Claude rows unpriced', try { await import(new URL(`../fleet-scorecard.mjs?test=pricing-${Date.now()}`, import.meta.url)) const metrics = JSON.parse(readFileSync('/tmp/fleet-scorecard/metrics.json', 'utf8')) - assert.equal(metrics.generations, 8) - assert.equal(metrics.est_cost_usd, 24 + 30 + 12 + 18 + 7.35) + assert.equal(metrics.generations, 9) + assert.equal(metrics.est_cost_usd, Math.round((24 + 30 + 12 + 18 + 7.35 + 7.35) * 100) / 100) assert.equal(metrics.unpriced_generations, 3) assert.equal(metrics.unpriced_tokens, 6_000_000) const body = readFileSync('/tmp/fleet-scorecard/scorecard-body.md', 'utf8') diff --git a/scripts/tests/test_llm_gateway.sh b/scripts/tests/test_llm_gateway.sh index 8048eefb7d3..2140e9eb958 100755 --- a/scripts/tests/test_llm_gateway.sh +++ b/scripts/tests/test_llm_gateway.sh @@ -99,6 +99,9 @@ sidecar_model() { # $1 = gateway, $2 = run model id ("" = unset); prints model= [ "$(sidecar_model surplus claude-haiku-4-5-20251001)" = "claude-haiku-4.5" ] \ && pass "surplus: date suffix stripped, then dot-form" \ || bad "surplus: date suffix stripped, then dot-form (got $(sidecar_model surplus claude-haiku-4-5-20251001))" +[ "$(sidecar_model surplus claude-haiku-5-5)" = "claude-haiku-4.5" ] \ + && pass "surplus: haiku-5-5 (not carried yet) stays on claude-haiku-4.5" \ + || bad "surplus: haiku-5-5 (not carried yet) stays on claude-haiku-4.5 (got $(sidecar_model surplus claude-haiku-5-5))" [ "$(sidecar_model surplus "")" = "claude-opus-5.5" ] \ && pass "surplus: unset MODEL falls back to opus-5.5" \ || bad "surplus: unset MODEL falls back to opus-5.5 (got $(sidecar_model surplus ""))" @@ -127,9 +130,12 @@ or_model() { # $1 = run model id; prints the MODEL the arm resolves [ "$(or_model claude-opus-5-5)" = "anthropic/claude-opus-5.5" ] \ && pass "openrouter: opus-pinned run gets the opus slug" \ || bad "openrouter: opus-pinned run gets the opus slug (got $(or_model claude-opus-5-5))" -[ "$(or_model claude-haiku-4-5-20251001)" = "anthropic/claude-haiku-4.5" ] \ +[ "$(or_model claude-haiku-5-5)" = "anthropic/claude-haiku-5.5" ] \ && pass "openrouter: haiku-tier run gets the haiku slug" \ - || bad "openrouter: haiku-tier run gets the haiku slug" + || bad "openrouter: haiku-tier run gets the haiku slug (got $(or_model claude-haiku-5-5))" +[ "$(or_model claude-haiku-4-5-20251001)" = "anthropic/claude-haiku-5.5" ] \ + && pass "openrouter: an older haiku id still lands on the haiku slot" \ + || bad "openrouter: an older haiku id still lands on the haiku slot (got $(or_model claude-haiku-4-5-20251001))" ( export GATEWAY=openrouter OPENROUTER_API_KEY=test-key MODEL=claude-sonnet-5 \ OPENROUTER_MODEL=x/opus OPENROUTER_MODEL_SONNET=x/sonnet # shellcheck disable=SC1090 diff --git a/skills/create-skill/SKILL.md b/skills/create-skill/SKILL.md index 6b0ad22b21b..2e8ad330f3f 100644 --- a/skills/create-skill/SKILL.md +++ b/skills/create-skill/SKILL.md @@ -70,7 +70,7 @@ Today is ${today}. Your task is to generate a complete, production-ready skill f - **Variable behavior** — what `${var}` controls; what happens when empty (sane default OR clean abort with notify). - **Steps** — 4-8 numbered, following the standard pattern: read context → fetch/search → process/analyze → write output → log → notify. - **Schedule suggestion** — choose a cron slot. Read existing schedules in `aeon.yml`; avoid co-scheduling at the same minute as heavy skills (article, repo-scanner, deep-research, telegram-digest) unless the new skill is lightweight (<30s expected). Prefer a `:30` minute offset if the natural hour is already crowded. - - **Model** - default `claude-sonnet-5-5`. Pick `claude-haiku-4-5-20251001` if the skill is high-frequency aggregation/digestion (cost optimization), or `claude-opus-5-5` if it needs the strongest reasoning. Document the choice in the PR body. + - **Model** - default `claude-sonnet-5-5`. Pick `claude-haiku-5-5` if the skill is high-frequency aggregation/digestion (cost optimization), or `claude-opus-5-5` if it needs the strongest reasoning. Document the choice in the PR body. - **Category** - the pack the skill joins. Pick exactly one of `core` `evolution` `basics` `dev` `crypto` `productivity` (the set `scripts/check-skill-categories.sh` enforces; anything else, or a missing category, fails CI). For a new user-facing skill that is usually `basics`, `dev`, `crypto`, or `productivity`. See `docs/skill-packs.md`. 6. **Write the SKILL.md draft** at `skills/{skill-name}/SKILL.md` with this exact structure: @@ -145,7 +145,7 @@ Today is ${today}. Your task is to generate a complete, production-ready skill f 9. **Register in `aeon.yml`.** Insert the new skill in the appropriate time-slot section: - Format: ` {skill-name}: { enabled: false, schedule: "{suggested_cron}" }` - - Add `model: "claude-haiku-4-5-20251001"` (or `"claude-opus-5-5"`) if chosen in step 5. + - Add `model: "claude-haiku-5-5"` (or `"claude-opus-5-5"`) if chosen in step 5. - Add `var: ""` if the skill takes a default var. - Add a brief trailing comment if the name doesn't make purpose obvious. - Place near related skills (crypto with crypto, content with content, etc.). diff --git a/skills/skill-repair/SKILL.md b/skills/skill-repair/SKILL.md index db991e72262..bdd3bc576b4 100644 --- a/skills/skill-repair/SKILL.md +++ b/skills/skill-repair/SKILL.md @@ -122,7 +122,7 @@ Categories follow `CLAUDE.md`. Pick the **most specific** category that fits the |---|---| | **`api-change`** | WebFetch the live API spec / status page / release notes. Update endpoints, payload shape, headers, error codes in the skill. Cite the spec URL in the PR body. Never guess — if WebFetch fails, drop to `REPAIR_DIAGNOSED_NO_FIX`. | | **`rate-limit`** | Add backoff (`sleep`), reduce request count, or add a fallback endpoint. Never raise the limit from the skill side. If the skill's `schedule` is too aggressive, propose a less-frequent cron in the PR body but **don't edit `aeon.yml`** unless the issue file already authorizes it. | -| **`timeout`** | Split work into stages, add early-return on partial success, downgrade `model:` to `claude-haiku-4-5-20251001` for the skill that doesn't need Sonnet or Opus. | +| **`timeout`** | Split work into stages, add early-return on partial success, downgrade `model:` to `claude-haiku-5-5` for the skill that doesn't need Sonnet or Opus. | | **`sandbox-limitation`** | Usually the "sandbox blocks the network" myth — there is **no** network sandbox. The real cause is a bare `$SECRET` on the command line (refused by the Bash permission layer) or a non-allowlisted command. Fix: route auth-required calls through `./secretcurl` with a `{ENV_NAME}` placeholder, or `gh api` for GitHub (auth handled internally). **Irreversible side-effects** (email / spend / on-chain / deploy) run **in-run** via `./secretcurl` as the skill's final, fail-closed action — never for reads. Add/refresh a "Network note" section. (There are **no** `scripts/prefetch-*.sh` or `scripts/postprocess-*.sh` scripts — both patterns were retired; auth'd reads and irreversible sends alike happen in-run.) | | **`prompt-bug`** | Minimum-edit specificity insertion. Don't rewrite — add the missing constraint, a forbidden phrase, a required output structure, or a clarifying example. Diff should be < 30 added/removed lines. | | **`output-format`** / **`quality-regression`** | Re-read the target skill's own output spec in its `SKILL.md`. Edit the skill so the next run satisfies that spec. Cite the exact requirement (section / line) in the PR body. |