From 0139e0d35144ea7d4dd1499a60640863cedf8ce5 Mon Sep 17 00:00:00 2001 From: Alan-TheGentleman Date: Thu, 1 Oct 2026 21:56:02 +0200 Subject: [PATCH] fix(provider): map NaN reasoning levels per model Align Pi thinking levels with the efforts each NaN model accepts: GLM maps minimal/xhigh onto low/max and cannot be disabled; Qwen 3.6 and Gemma 4 map off to none so reasoning can actually be turned off; fixed-depth models (DeepSeek V4 Flash, Qwen 3.8 Flash, MiMo) expose a single medium level. Advertise GLM 5.3 image input and raise output caps so reasoning that shares max_tokens does not starve the answer. --- lib/nan-provider.ts | 60 ++++++++++++++++++++++------- tests/nan-provider.test.ts | 78 ++++++++++++++++++++++++++++++++------ 2 files changed, 113 insertions(+), 25 deletions(-) diff --git a/lib/nan-provider.ts b/lib/nan-provider.ts index 4720efc0b..3e7697a16 100644 --- a/lib/nan-provider.ts +++ b/lib/nan-provider.ts @@ -18,23 +18,50 @@ const ZERO_COST = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }; // Maintained chat subset from https://nan.builders/docs/models. // Decimal bounds conservatively interpret the documented 1M/262K/131K labels. -// Pi models text/image inputs only; MiMo's documented audio input is not advertised. -// Where no output maximum is published, 8,192 is our conservative configured cap for coding with reasoning, not NaN's limit. -const CHAT_MODELS: NanChatModelConfig[] = [ - { id: "glm5.3", name: "GLM 5.3", input: ["text"], contextWindow: 1_000_000 }, - { id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", input: ["text", "image"], contextWindow: 1_000_000 }, - { id: "glm5.3-flash", name: "GLM 5.3 Flash", input: ["text", "image"], contextWindow: 1_000_000 }, - { id: "qwen3.8-flash", name: "Qwen 3.8 Flash", input: ["text", "image"], contextWindow: 1_048_576, maxTokens: 131_000 }, - { id: "mimo-v2.6-flash", name: "MiMo V2.6 Flash", input: ["text", "image"], contextWindow: 1_000_000 }, - { id: "gemma4", name: "Gemma 4", input: ["text", "image"], contextWindow: 262_000 }, - { id: "qwen3.6", name: "Qwen 3.6", input: ["text", "image"], contextWindow: 262_000 }, -].map((model) => ({ +// Pi models text/image inputs only; documented audio/video inputs are not advertised. +// Reasoning shares max_tokens with the answer: 65,536 leaves answer room after a +// 32,768 reasoning budget; DeepSeek uses NaN's 16,384 floor. Qwen 3.8 retains its +// published 131K output maximum. The remaining 32,768 caps are configured, not NaN-published limits. +// Pi requires explicit max mappings to offer that level; missing ordinary levels pass through. +const FIXED_THINKING_LEVEL_MAP: NanChatModelConfig["thinkingLevelMap"] = { + off: null, minimal: null, low: null, high: null, xhigh: null, max: null, +}; // Only medium remains usable: depth is fixed and reasoning cannot be disabled. + +const CHAT_MODELS: NanChatModelConfig[] = ([ + { + id: "glm5.3", name: "GLM 5.3", input: ["text", "image"], contextWindow: 1_000_000, + thinkingLevelMap: { off: null, minimal: "low", xhigh: "max", max: "max" }, + }, + { + id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", input: ["text", "image"], contextWindow: 1_000_000, + thinkingLevelMap: FIXED_THINKING_LEVEL_MAP, maxTokens: 16_384, + }, + { + id: "glm5.3-flash", name: "GLM 5.3 Flash", input: ["text", "image"], contextWindow: 1_000_000, + thinkingLevelMap: { off: null, minimal: "low", xhigh: "max", max: "max" }, + }, + { + id: "qwen3.8-flash", name: "Qwen 3.8 Flash", input: ["text", "image"], contextWindow: 1_048_576, + thinkingLevelMap: FIXED_THINKING_LEVEL_MAP, maxTokens: 131_000, + }, + { + id: "mimo-v2.6-flash", name: "MiMo V2.6 Flash", input: ["text", "image"], contextWindow: 1_000_000, + thinkingLevelMap: FIXED_THINKING_LEVEL_MAP, + }, + { + id: "gemma4", name: "Gemma 4", input: ["text", "image"], contextWindow: 262_000, + thinkingLevelMap: { off: "none", xhigh: "max", max: "max" }, maxTokens: 65_536, + }, + { + id: "qwen3.6", name: "Qwen 3.6", input: ["text", "image"], contextWindow: 262_000, + thinkingLevelMap: { off: "none", xhigh: "max", max: "max" }, maxTokens: 65_536, + }, +] satisfies Partial[]).map((model) => ({ ...model, - input: model.input as NanChatModelConfig["input"], api: "openai-completions", reasoning: true, cost: ZERO_COST, - maxTokens: model.maxTokens ?? 8_192, + maxTokens: model.maxTokens ?? 32_768, })); // The cold/offline baseline declares documented chat support, not key entitlement. @@ -42,7 +69,12 @@ const CHAT_MODELS: NanChatModelConfig[] = [ const OFFLINE_MODELS = CHAT_MODELS; function cloneModel(model: NanChatModelConfig): NanChatModelConfig { - return { ...model, input: [...model.input], cost: { ...model.cost } }; + return { + ...model, + input: [...model.input], + cost: { ...model.cost }, + thinkingLevelMap: model.thinkingLevelMap ? { ...model.thinkingLevelMap } : undefined, + }; } function knownChatModels(ids: readonly string[]): NanChatModelConfig[] { diff --git a/tests/nan-provider.test.ts b/tests/nan-provider.test.ts index daa58f1d3..1046116df 100644 --- a/tests/nan-provider.test.ts +++ b/tests/nan-provider.test.ts @@ -1,7 +1,7 @@ import assert from "node:assert/strict"; import test from "node:test"; import type { RefreshModelsContext } from "@earendil-works/pi-ai"; -import { createModels, InMemoryCredentialStore } from "@earendil-works/pi-ai"; +import { clampThinkingLevel, createModels, getSupportedThinkingLevels, InMemoryCredentialStore } from "@earendil-works/pi-ai"; import type { Provider } from "@earendil-works/pi-ai"; import nanProviderExtension from "../extensions/nan-provider.ts"; import { createNanProviderConfig as createNativeProvider, NAN_PROVIDER_BASE_URL, NAN_PROVIDER_ID } from "../lib/nan-provider.ts"; @@ -191,10 +191,66 @@ test("initial catalog contains all seven documented chat models with configured assert.deepEqual(model?.input, ["text", "image"]); assert.equal(model?.contextWindow, 1_000_000); assert.equal(models.length, 7); - assert.equal(model?.maxTokens, 8_192); + assert.equal(model?.maxTokens, 16_384); assert.deepEqual(model?.cost, { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }); }); +test("adjustable models map Pi thinking levels to accepted NaN efforts", () => { + const models = createNativeProvider().getModels(); + for (const id of ["glm5.3", "glm5.3-flash"]) { + const model = models.find((model) => model.id === id); + assert.ok(model); + assert.deepEqual(model.thinkingLevelMap, { off: null, minimal: "low", xhigh: "max", max: "max" }); + assert.deepEqual(getSupportedThinkingLevels(model), ["minimal", "low", "medium", "high", "xhigh", "max"]); + assert.equal(clampThinkingLevel(model, "off"), "minimal"); + } + for (const id of ["qwen3.6", "gemma4"]) { + const model = models.find((model) => model.id === id); + assert.ok(model); + assert.deepEqual(model.thinkingLevelMap, { off: "none", xhigh: "max", max: "max" }); + assert.deepEqual(getSupportedThinkingLevels(model), ["off", "minimal", "low", "medium", "high", "xhigh", "max"]); + assert.equal(clampThinkingLevel(model, "off"), "off"); + } +}); + +test("fixed-depth models expose only medium and clamp unsupported thinking levels", () => { + const models = createNativeProvider().getModels(); + for (const id of ["deepseek-v4-flash", "qwen3.8-flash", "mimo-v2.6-flash"]) { + const model = models.find((model) => model.id === id); + assert.ok(model); + assert.deepEqual(model.thinkingLevelMap, { + off: null, minimal: null, low: null, high: null, xhigh: null, max: null, + }); + assert.deepEqual(getSupportedThinkingLevels(model), ["medium"]); + for (const level of ["off", "minimal", "low", "medium", "high", "xhigh", "max"] as const) { + assert.equal(clampThinkingLevel(model, level), "medium"); + } + } +}); + +test("thinking maps are isolated across snapshots, providers, and live catalog refreshes", async () => { + const provider = createNativeProvider({ + fetchImpl: async () => jsonResponse({ data: DOCUMENTED_CHAT_IDS.map((id) => ({ id })) }), + }); + const baseline = provider.getModels(); + for (const model of provider.getModels()) { + assert.ok(model.thinkingLevelMap); + model.thinkingLevelMap.off = "mutated"; + model.thinkingLevelMap.medium = null; + } + assert.deepEqual(provider.getModels(), baseline); + assert.deepEqual(createNativeProvider().getModels(), baseline); + await provider.refreshModels!(refreshContext({ type: "api_key", key: "map-key" })); + assert.deepEqual(provider.getModels(), baseline); + for (const model of provider.getModels()) { + assert.ok(model.thinkingLevelMap); + model.thinkingLevelMap.xhigh = "mutated"; + } + assert.deepEqual(provider.getModels(), baseline); + await provider.refreshModels!(refreshContext({ type: "api_key", key: "map-key" })); + assert.deepEqual(provider.getModels(), baseline); +}); + test("catalog snapshots cannot mutate the offline baseline or another provider", async () => { const provider = createNativeProvider(); const snapshot = [...provider.getModels()]; @@ -204,7 +260,7 @@ test("catalog snapshots cannot mutate the offline baseline or another provider", snapshot.pop(); const fresh = provider.getModels(); assert.deepEqual(fresh.map((model) => model.id), DOCUMENTED_CHAT_IDS); - assert.deepEqual(fresh[0].input, ["text"]); + assert.deepEqual(fresh[0].input, ["text", "image"]); assert.equal(fresh[0].cost.input, 0); assert.deepEqual(createNativeProvider().getModels(), fresh); await provider.refreshModels!({ @@ -236,21 +292,21 @@ test("live discovery uses the key-scoped endpoint and replaces the fallback with const known = models?.[0]; assert.equal(known?.api, "openai-completions"); assert.equal(known?.reasoning, true); - assert.deepEqual(known?.input, ["text"]); + assert.deepEqual(known?.input, ["text", "image"]); assert.equal(known?.contextWindow, 1_000_000); - assert.equal(known?.maxTokens, 8_192); + assert.equal(known?.maxTokens, 32_768); assert.deepEqual(known?.cost, { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 }); }); test("known chat models retain documented capabilities without advertising audio", async () => { const expected = [ - ["glm5.3", 1_000_000, ["text"], 8_192], - ["deepseek-v4-flash", 1_000_000, ["text", "image"], 8_192], - ["glm5.3-flash", 1_000_000, ["text", "image"], 8_192], + ["glm5.3", 1_000_000, ["text", "image"], 32_768], + ["deepseek-v4-flash", 1_000_000, ["text", "image"], 16_384], + ["glm5.3-flash", 1_000_000, ["text", "image"], 32_768], ["qwen3.8-flash", 1_048_576, ["text", "image"], 131_000], - ["mimo-v2.6-flash", 1_000_000, ["text", "image"], 8_192], - ["gemma4", 262_000, ["text", "image"], 8_192], - ["qwen3.6", 262_000, ["text", "image"], 8_192], + ["mimo-v2.6-flash", 1_000_000, ["text", "image"], 32_768], + ["gemma4", 262_000, ["text", "image"], 65_536], + ["qwen3.6", 262_000, ["text", "image"], 65_536], ] as const; const config = createNanProviderConfig({ fetchImpl: async () => jsonResponse({ data: expected.map(([id]) => ({ id })) }),