Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
60 changes: 46 additions & 14 deletions lib/nan-provider.ts
Original file line number Diff line number Diff line change
Expand Up @@ -18,31 +18,63 @@ const ZERO_COST = { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 };

// Maintained chat subset from https://nan.builders/docs/models.
// Decimal bounds conservatively interpret the documented 1M/262K/131K labels.
// Pi models text/image inputs only; MiMo's documented audio input is not advertised.
// Where no output maximum is published, 8,192 is our conservative configured cap for coding with reasoning, not NaN's limit.
const CHAT_MODELS: NanChatModelConfig[] = [
{ id: "glm5.3", name: "GLM 5.3", input: ["text"], contextWindow: 1_000_000 },
{ id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", input: ["text", "image"], contextWindow: 1_000_000 },
{ id: "glm5.3-flash", name: "GLM 5.3 Flash", input: ["text", "image"], contextWindow: 1_000_000 },
{ id: "qwen3.8-flash", name: "Qwen 3.8 Flash", input: ["text", "image"], contextWindow: 1_048_576, maxTokens: 131_000 },
{ id: "mimo-v2.6-flash", name: "MiMo V2.6 Flash", input: ["text", "image"], contextWindow: 1_000_000 },
{ id: "gemma4", name: "Gemma 4", input: ["text", "image"], contextWindow: 262_000 },
{ id: "qwen3.6", name: "Qwen 3.6", input: ["text", "image"], contextWindow: 262_000 },
].map((model) => ({
// Pi models text/image inputs only; documented audio/video inputs are not advertised.
// Reasoning shares max_tokens with the answer: 65,536 leaves answer room after a
// 32,768 reasoning budget; DeepSeek uses NaN's 16,384 floor. Qwen 3.8 retains its
// published 131K output maximum. The remaining 32,768 caps are configured, not NaN-published limits.
// Pi requires explicit max mappings to offer that level; missing ordinary levels pass through.
const FIXED_THINKING_LEVEL_MAP: NanChatModelConfig["thinkingLevelMap"] = {
off: null, minimal: null, low: null, high: null, xhigh: null, max: null,
}; // Only medium remains usable: depth is fixed and reasoning cannot be disabled.

const CHAT_MODELS: NanChatModelConfig[] = ([
{
id: "glm5.3", name: "GLM 5.3", input: ["text", "image"], contextWindow: 1_000_000,
thinkingLevelMap: { off: null, minimal: "low", xhigh: "max", max: "max" },
},
{
id: "deepseek-v4-flash", name: "DeepSeek V4 Flash", input: ["text", "image"], contextWindow: 1_000_000,
thinkingLevelMap: FIXED_THINKING_LEVEL_MAP, maxTokens: 16_384,
},
{
id: "glm5.3-flash", name: "GLM 5.3 Flash", input: ["text", "image"], contextWindow: 1_000_000,
thinkingLevelMap: { off: null, minimal: "low", xhigh: "max", max: "max" },
},
{
id: "qwen3.8-flash", name: "Qwen 3.8 Flash", input: ["text", "image"], contextWindow: 1_048_576,
thinkingLevelMap: FIXED_THINKING_LEVEL_MAP, maxTokens: 131_000,
},
{
id: "mimo-v2.6-flash", name: "MiMo V2.6 Flash", input: ["text", "image"], contextWindow: 1_000_000,
thinkingLevelMap: FIXED_THINKING_LEVEL_MAP,
},
{
id: "gemma4", name: "Gemma 4", input: ["text", "image"], contextWindow: 262_000,
thinkingLevelMap: { off: "none", xhigh: "max", max: "max" }, maxTokens: 65_536,
},
{
id: "qwen3.6", name: "Qwen 3.6", input: ["text", "image"], contextWindow: 262_000,
thinkingLevelMap: { off: "none", xhigh: "max", max: "max" }, maxTokens: 65_536,
},
] satisfies Partial<NanChatModelConfig>[]).map((model) => ({
...model,
input: model.input as NanChatModelConfig["input"],
api: "openai-completions",
reasoning: true,
cost: ZERO_COST,
maxTokens: model.maxTokens ?? 8_192,
maxTokens: model.maxTokens ?? 32_768,
}));

// The cold/offline baseline declares documented chat support, not key entitlement.
// A successful live catalog remains authoritative for the credential that fetched it.
const OFFLINE_MODELS = CHAT_MODELS;

function cloneModel(model: NanChatModelConfig): NanChatModelConfig {
return { ...model, input: [...model.input], cost: { ...model.cost } };
return {
...model,
input: [...model.input],
cost: { ...model.cost },
thinkingLevelMap: model.thinkingLevelMap ? { ...model.thinkingLevelMap } : undefined,
};
}

function knownChatModels(ids: readonly string[]): NanChatModelConfig[] {
Expand Down
78 changes: 67 additions & 11 deletions tests/nan-provider.test.ts
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
import assert from "node:assert/strict";
import test from "node:test";
import type { RefreshModelsContext } from "@earendil-works/pi-ai";
import { createModels, InMemoryCredentialStore } from "@earendil-works/pi-ai";
import { clampThinkingLevel, createModels, getSupportedThinkingLevels, InMemoryCredentialStore } from "@earendil-works/pi-ai";
import type { Provider } from "@earendil-works/pi-ai";
import nanProviderExtension from "../extensions/nan-provider.ts";
import { createNanProviderConfig as createNativeProvider, NAN_PROVIDER_BASE_URL, NAN_PROVIDER_ID } from "../lib/nan-provider.ts";
Expand Down Expand Up @@ -191,10 +191,66 @@ test("initial catalog contains all seven documented chat models with configured
assert.deepEqual(model?.input, ["text", "image"]);
assert.equal(model?.contextWindow, 1_000_000);
assert.equal(models.length, 7);
assert.equal(model?.maxTokens, 8_192);
assert.equal(model?.maxTokens, 16_384);
assert.deepEqual(model?.cost, { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 });
});

test("adjustable models map Pi thinking levels to accepted NaN efforts", () => {
const models = createNativeProvider().getModels();
for (const id of ["glm5.3", "glm5.3-flash"]) {
const model = models.find((model) => model.id === id);
assert.ok(model);
assert.deepEqual(model.thinkingLevelMap, { off: null, minimal: "low", xhigh: "max", max: "max" });
assert.deepEqual(getSupportedThinkingLevels(model), ["minimal", "low", "medium", "high", "xhigh", "max"]);
assert.equal(clampThinkingLevel(model, "off"), "minimal");
}
for (const id of ["qwen3.6", "gemma4"]) {
const model = models.find((model) => model.id === id);
assert.ok(model);
assert.deepEqual(model.thinkingLevelMap, { off: "none", xhigh: "max", max: "max" });
assert.deepEqual(getSupportedThinkingLevels(model), ["off", "minimal", "low", "medium", "high", "xhigh", "max"]);
assert.equal(clampThinkingLevel(model, "off"), "off");
}
});

test("fixed-depth models expose only medium and clamp unsupported thinking levels", () => {
const models = createNativeProvider().getModels();
for (const id of ["deepseek-v4-flash", "qwen3.8-flash", "mimo-v2.6-flash"]) {
const model = models.find((model) => model.id === id);
assert.ok(model);
assert.deepEqual(model.thinkingLevelMap, {
off: null, minimal: null, low: null, high: null, xhigh: null, max: null,
});
assert.deepEqual(getSupportedThinkingLevels(model), ["medium"]);
for (const level of ["off", "minimal", "low", "medium", "high", "xhigh", "max"] as const) {
assert.equal(clampThinkingLevel(model, level), "medium");
}
}
});

test("thinking maps are isolated across snapshots, providers, and live catalog refreshes", async () => {
const provider = createNativeProvider({
fetchImpl: async () => jsonResponse({ data: DOCUMENTED_CHAT_IDS.map((id) => ({ id })) }),
});
const baseline = provider.getModels();
for (const model of provider.getModels()) {
assert.ok(model.thinkingLevelMap);
model.thinkingLevelMap.off = "mutated";
model.thinkingLevelMap.medium = null;
}
assert.deepEqual(provider.getModels(), baseline);
assert.deepEqual(createNativeProvider().getModels(), baseline);
await provider.refreshModels!(refreshContext({ type: "api_key", key: "map-key" }));
assert.deepEqual(provider.getModels(), baseline);
for (const model of provider.getModels()) {
assert.ok(model.thinkingLevelMap);
model.thinkingLevelMap.xhigh = "mutated";
}
assert.deepEqual(provider.getModels(), baseline);
await provider.refreshModels!(refreshContext({ type: "api_key", key: "map-key" }));
assert.deepEqual(provider.getModels(), baseline);
});

test("catalog snapshots cannot mutate the offline baseline or another provider", async () => {
const provider = createNativeProvider();
const snapshot = [...provider.getModels()];
Expand All @@ -204,7 +260,7 @@ test("catalog snapshots cannot mutate the offline baseline or another provider",
snapshot.pop();
const fresh = provider.getModels();
assert.deepEqual(fresh.map((model) => model.id), DOCUMENTED_CHAT_IDS);
assert.deepEqual(fresh[0].input, ["text"]);
assert.deepEqual(fresh[0].input, ["text", "image"]);
assert.equal(fresh[0].cost.input, 0);
assert.deepEqual(createNativeProvider().getModels(), fresh);
await provider.refreshModels!({
Expand Down Expand Up @@ -236,21 +292,21 @@ test("live discovery uses the key-scoped endpoint and replaces the fallback with
const known = models?.[0];
assert.equal(known?.api, "openai-completions");
assert.equal(known?.reasoning, true);
assert.deepEqual(known?.input, ["text"]);
assert.deepEqual(known?.input, ["text", "image"]);
assert.equal(known?.contextWindow, 1_000_000);
assert.equal(known?.maxTokens, 8_192);
assert.equal(known?.maxTokens, 32_768);
assert.deepEqual(known?.cost, { input: 0, output: 0, cacheRead: 0, cacheWrite: 0 });
});

test("known chat models retain documented capabilities without advertising audio", async () => {
const expected = [
["glm5.3", 1_000_000, ["text"], 8_192],
["deepseek-v4-flash", 1_000_000, ["text", "image"], 8_192],
["glm5.3-flash", 1_000_000, ["text", "image"], 8_192],
["glm5.3", 1_000_000, ["text", "image"], 32_768],
["deepseek-v4-flash", 1_000_000, ["text", "image"], 16_384],
["glm5.3-flash", 1_000_000, ["text", "image"], 32_768],
["qwen3.8-flash", 1_048_576, ["text", "image"], 131_000],
["mimo-v2.6-flash", 1_000_000, ["text", "image"], 8_192],
["gemma4", 262_000, ["text", "image"], 8_192],
["qwen3.6", 262_000, ["text", "image"], 8_192],
["mimo-v2.6-flash", 1_000_000, ["text", "image"], 32_768],
["gemma4", 262_000, ["text", "image"], 65_536],
["qwen3.6", 262_000, ["text", "image"], 65_536],
] as const;
const config = createNanProviderConfig({
fetchImpl: async () => jsonResponse({ data: expected.map(([id]) => ({ id })) }),
Expand Down
Loading