diff --git a/docs/en/configuration/env-vars.md b/docs/en/configuration/env-vars.md index 4e9d1b6b278..d16631f6499 100644 --- a/docs/en/configuration/env-vars.md +++ b/docs/en/configuration/env-vars.md @@ -181,6 +181,7 @@ Switches that control the behavior of subsystems such as telemetry, background t | `KIMI_MODEL_TOP_P` | Nucleus-sampling `top_p` for every request; `kimi` provider only (global) | Number, e.g. `0.95` | | `KIMI_MODEL_THINKING_EFFORT` | Force a thinking effort (`thinking.effort`), bypassing the model's declared `support_efforts`; `kimi` provider only | An effort value, e.g. `max` | | `KIMI_MODEL_THINKING_KEEP` | Preserved-thinking passthrough: `thinking.keep` on `kimi`, a `clear_thinking_20251015` edit on `anthropic`; overrides `[thinking] keep` | A value the API accepts, e.g. `all`; an off-value (`false`/`0`/`no`/`off`/`none`/`null`) disables it | +| `KIMI_CODE_LLM_HEADERS_TIMEOUT_MS` | Max time (ms) an LLM request on the `openai` / `openai-responses` / `anthropic` protocols may wait for the response headers (first byte), replacing the HTTP client's default 300 s headers timeout — raise it for long non-streaming thinking; unset keeps the default; the SDK's own total request timeout (default 10 minutes) still caps each attempt; not supported for `google-genai` (its SDK exposes no dispatcher option) or behind a SOCKS proxy (requests still go through the proxy) | Positive integer no greater than 2147483647; invalid values fail the request | | `KIMI_CODE_NO_AUTO_UPDATE` | Fully disable the update preflight: no check, background install, or prompt. Legacy alias `KIMI_CLI_NO_AUTO_UPDATE` also honored | Truthy: `1`/`true`/`yes`/`on` | | `KIMI_DISABLE_CRON` | Disable the scheduled-task tool (`CronCreate` rejects new schedules; existing tasks do not fire) | `1` to disable | diff --git a/docs/zh/configuration/env-vars.md b/docs/zh/configuration/env-vars.md index ed087d23f62..4f033953853 100644 --- a/docs/zh/configuration/env-vars.md +++ b/docs/zh/configuration/env-vars.md @@ -181,6 +181,7 @@ kimi | `KIMI_MODEL_TOP_P` | 每次请求的核采样 `top_p`,仅对 `kimi` 供应商生效(全局生效) | 数字,如 `0.95` | | `KIMI_MODEL_THINKING_EFFORT` | 在线上强制使用指定的思考强度,绕过模型声明的 `support_efforts`;仅 `kimi` 供应商生效 | 思考强度值,如 `max` | | `KIMI_MODEL_THINKING_KEEP` | 保留思考透传;`kimi` 以 `thinking.keep` 发送,`anthropic` 以 `clear_thinking_20251015` 编辑发送;覆盖 `[thinking] keep` | API 接受的值,如 `all`;传入关值(`false`/`0`/`no`/`off`/`none`/`null`)可禁用 | +| `KIMI_CODE_LLM_HEADERS_TIMEOUT_MS` | `openai` / `openai-responses` / `anthropic` 协议的 LLM 请求等待响应头(首字节)的最长时间(毫秒),替代 HTTP 客户端默认的 300 秒响应头超时——长时间非流式 thinking 可调大;未设置保持默认;SDK 自身的总请求超时(默认 10 分钟)仍对单次尝试生效;`google-genai` 不支持(其 SDK 没有 dispatcher 配置项);SOCKS 代理下不生效(请求仍走代理) | 正整数,不超过 2147483647;非法值会使请求失败 | | `KIMI_CODE_NO_AUTO_UPDATE` | 完全禁用更新预检:不检查、不后台安装、不提示。同时兼容旧名 `KIMI_CLI_NO_AUTO_UPDATE` | 真值:`1`/`true`/`yes`/`on` | | `KIMI_DISABLE_CRON` | 禁用定时任务工具(`CronCreate` 拒绝新计划,已有任务不触发) | `1` 表示禁用 | diff --git a/packages/agent-core-v2/src/human/llm/requester/bases/anthropic/requester.ts b/packages/agent-core-v2/src/human/llm/requester/bases/anthropic/requester.ts index 99dc22e22a5..f555246edde 100644 --- a/packages/agent-core-v2/src/human/llm/requester/bases/anthropic/requester.ts +++ b/packages/agent-core-v2/src/human/llm/requester/bases/anthropic/requester.ts @@ -20,6 +20,10 @@ import { type LlmRequestEvent, type ToolCallIdPolicy, } from '#/llm/requester/requester'; +import { + getLlmHeadersTimeoutDispatcher, + resolveLlmHeadersTimeoutMs, +} from '#/llm/requester/timeout'; import { normalizeToolCallIdsForProvider, @@ -83,12 +87,16 @@ function buildDefaultHeaders( } function createClient(model: LlmModel, headers: Record | undefined): Anthropic { + const headersTimeoutMs = resolveLlmHeadersTimeoutMs(); + const dispatcher = + headersTimeoutMs === undefined ? undefined : getLlmHeadersTimeoutDispatcher(headersTimeoutMs); return new Anthropic({ apiKey: model.apiKey ?? 'unused', authToken: null, baseURL: model.baseUrl ?? null, defaultHeaders: buildDefaultHeaders(headers), maxRetries: 0, + fetchOptions: dispatcher === undefined ? undefined : { dispatcher }, }); } diff --git a/packages/agent-core-v2/src/human/llm/requester/bases/openai-responses/requester.ts b/packages/agent-core-v2/src/human/llm/requester/bases/openai-responses/requester.ts index 357e54035c7..35909c2fce2 100644 --- a/packages/agent-core-v2/src/human/llm/requester/bases/openai-responses/requester.ts +++ b/packages/agent-core-v2/src/human/llm/requester/bases/openai-responses/requester.ts @@ -20,6 +20,10 @@ import { type LlmRequestEvent, type ToolCallIdPolicy, } from '#/llm/requester/requester'; +import { + getLlmHeadersTimeoutDispatcher, + resolveLlmHeadersTimeoutMs, +} from '#/llm/requester/timeout'; import { normalizeToolCallIdsForProvider, @@ -49,11 +53,15 @@ const OPENAI_RESPONSES_TOOL_CALL_ID_POLICY: ToolCallIdPolicy = { }; function createClient(model: LlmModel, headers: Record | undefined): OpenAI { + const headersTimeoutMs = resolveLlmHeadersTimeoutMs(); + const dispatcher = + headersTimeoutMs === undefined ? undefined : getLlmHeadersTimeoutDispatcher(headersTimeoutMs); return new OpenAI({ apiKey: model.apiKey ?? 'unused', baseURL: model.baseUrl, defaultHeaders: headers, maxRetries: 0, + fetchOptions: dispatcher === undefined ? undefined : { dispatcher }, }); } diff --git a/packages/agent-core-v2/src/human/llm/requester/bases/openai/requester.ts b/packages/agent-core-v2/src/human/llm/requester/bases/openai/requester.ts index a19bfc1c66b..d78f1b4fc49 100644 --- a/packages/agent-core-v2/src/human/llm/requester/bases/openai/requester.ts +++ b/packages/agent-core-v2/src/human/llm/requester/bases/openai/requester.ts @@ -20,6 +20,10 @@ import { type LlmRequestEvent, type ToolCallIdPolicy, } from '#/llm/requester/requester'; +import { + getLlmHeadersTimeoutDispatcher, + resolveLlmHeadersTimeoutMs, +} from '#/llm/requester/timeout'; import { normalizeToolCallIdsForProvider, @@ -50,11 +54,15 @@ const OPENAI_CHAT_TOOL_CALL_ID_POLICY: ToolCallIdPolicy = { }; function createClient(model: LlmModel, headers: Record | undefined): OpenAI { + const headersTimeoutMs = resolveLlmHeadersTimeoutMs(); + const dispatcher = + headersTimeoutMs === undefined ? undefined : getLlmHeadersTimeoutDispatcher(headersTimeoutMs); return new OpenAI({ apiKey: model.apiKey ?? 'unused', baseURL: model.baseUrl, defaultHeaders: headers, maxRetries: 0, + fetchOptions: dispatcher === undefined ? undefined : { dispatcher }, }); } diff --git a/packages/agent-core-v2/src/human/llm/requester/timeout.ts b/packages/agent-core-v2/src/human/llm/requester/timeout.ts new file mode 100644 index 00000000000..8fb99b8b44e --- /dev/null +++ b/packages/agent-core-v2/src/human/llm/requester/timeout.ts @@ -0,0 +1,123 @@ +import { Agent, EnvHttpProxyAgent, type Dispatcher } from 'undici'; + +export const LLM_HEADERS_TIMEOUT_ENV = 'KIMI_CODE_LLM_HEADERS_TIMEOUT_MS'; + +const MAX_HEADERS_TIMEOUT_MS = 2 ** 31 - 1; + +type Env = Readonly>; + +export function resolveLlmHeadersTimeoutMs(env: Env = process.env): number | undefined { + const raw = env[LLM_HEADERS_TIMEOUT_ENV]; + if (raw === undefined || raw.trim() === '') return undefined; + const value = Number(raw); + if (!Number.isInteger(value) || value <= 0 || value > MAX_HEADERS_TIMEOUT_MS) { + throw new Error( + `${LLM_HEADERS_TIMEOUT_ENV} must be a positive integer no greater than ${MAX_HEADERS_TIMEOUT_MS}, got ${JSON.stringify(raw)}.`, + ); + } + return value; +} + +const SOCKS_SCHEMES = new Set(['socks', 'socks4', 'socks4a', 'socks5', 'socks5h']); +const LOOPBACK_NO_PROXY = ['localhost', '127.0.0.1', '::1', '[::1]'] as const; + +function schemeOf(value: string): string | undefined { + return /^([a-z][a-z0-9+.-]*):/i.exec(value)?.[1]?.toLowerCase(); +} + +function firstNonBlank(env: Env, keys: readonly string[]): string | undefined { + for (const key of keys) { + const value = env[key]?.trim(); + if (value !== undefined && value.length > 0) return value; + } + return undefined; +} + +function httpSchemeValue(value: string | undefined): string | undefined { + return value !== undefined && !SOCKS_SCHEMES.has(schemeOf(value) ?? '') ? value : undefined; +} + +function resolveHttpProxyUrls(env: Env): { httpProxy?: string; httpsProxy?: string } | undefined { + const allProxy = httpSchemeValue(firstNonBlank(env, ['all_proxy', 'ALL_PROXY'])); + const httpProxy = httpSchemeValue(firstNonBlank(env, ['http_proxy', 'HTTP_PROXY'])) ?? allProxy; + const httpsProxy = httpSchemeValue(firstNonBlank(env, ['https_proxy', 'HTTPS_PROXY'])) ?? allProxy; + if (httpProxy === undefined && httpsProxy === undefined) return undefined; + return { httpProxy, httpsProxy }; +} + +function hasSocksProxy(env: Env): boolean { + return [ + firstNonBlank(env, ['all_proxy', 'ALL_PROXY']), + firstNonBlank(env, ['https_proxy', 'HTTPS_PROXY']), + firstNonBlank(env, ['http_proxy', 'HTTP_PROXY']), + ].some((value) => value !== undefined && SOCKS_SCHEMES.has(schemeOf(value) ?? '')); +} + +function resolveNoProxy(env: Env): string { + const raw = + [env['no_proxy'], env['NO_PROXY']].find((value) => (value?.trim() ?? '').length > 0) ?? ''; + const hosts = raw + .split(',') + .map((host) => host.trim()) + .filter((host) => host.length > 0); + if (hosts.includes('*')) return '*'; + for (const loopback of LOOPBACK_NO_PROXY) { + if (!hosts.includes(loopback)) hosts.push(loopback); + } + return hosts.join(','); +} + +let warnedSocksProxy = false; +let warnedInvalidProxy = false; +let cached: { readonly key: string; readonly dispatcher: Dispatcher | undefined } | undefined; + +export function getLlmHeadersTimeoutDispatcher( + timeoutMs: number, + env: Env = process.env, +): Dispatcher | undefined { + const key = JSON.stringify([ + timeoutMs, + env['http_proxy'], + env['HTTP_PROXY'], + env['https_proxy'], + env['HTTPS_PROXY'], + env['all_proxy'], + env['ALL_PROXY'], + env['no_proxy'], + env['NO_PROXY'], + ]); + if (cached?.key === key) return cached.dispatcher; + let dispatcher: Dispatcher | undefined; + const httpProxyUrls = resolveHttpProxyUrls(env); + if (httpProxyUrls !== undefined) { + try { + dispatcher = new EnvHttpProxyAgent({ + httpProxy: httpProxyUrls.httpProxy ?? '', + httpsProxy: httpProxyUrls.httpsProxy ?? '', + noProxy: resolveNoProxy(env), + headersTimeout: timeoutMs, + }); + } catch (error) { + if (!warnedInvalidProxy) { + warnedInvalidProxy = true; + const reason = error instanceof Error ? error.message : String(error); + process.stderr.write( + `kimi: ${LLM_HEADERS_TIMEOUT_ENV}: ignoring invalid proxy configuration (${reason}); requests keep the default headers timeout\n`, + ); + } + dispatcher = undefined; + } + } else if (hasSocksProxy(env)) { + if (!warnedSocksProxy) { + warnedSocksProxy = true; + process.stderr.write( + `kimi: ${LLM_HEADERS_TIMEOUT_ENV} is not supported with SOCKS proxies; requests keep the default headers timeout\n`, + ); + } + dispatcher = undefined; + } else { + dispatcher = new Agent({ headersTimeout: timeoutMs }); + } + cached = { key, dispatcher }; + return dispatcher; +} diff --git a/packages/agent-core-v2/src/human/test/llm/errors.test.ts b/packages/agent-core-v2/src/human/test/llm/errors.test.ts index d7632b8fc16..2b9a96864b8 100644 --- a/packages/agent-core-v2/src/human/test/llm/errors.test.ts +++ b/packages/agent-core-v2/src/human/test/llm/errors.test.ts @@ -87,17 +87,17 @@ describe('convertOpenAIError', () => { expect(convertOpenAIError(raw)).toMatchObject({ kind: 'context_overflow', statusCode: 400 }); }); - it('maps 413 too-large messages to request_too_large', () => { - const raw = new RawOpenAISDKAPIError( + it('maps remaining status errors to their kinds', () => { + const tooLarge = new RawOpenAISDKAPIError( 413, { message: 'request entity too large' }, undefined, new Headers(), ); - expect(convertOpenAIError(raw)).toMatchObject({ kind: 'request_too_large', statusCode: 413 }); - }); - - it('maps remaining status errors to their kinds', () => { + expect(convertOpenAIError(tooLarge)).toMatchObject({ + kind: 'request_too_large', + statusCode: 413, + }); const overloaded = new RawOpenAISDKAPIError(529, {}, 'overloaded', new Headers()); expect(convertOpenAIError(overloaded)).toMatchObject({ kind: 'overloaded', statusCode: 529 }); const generic = new RawOpenAISDKAPIError(500, {}, 'server error', new Headers()); diff --git a/packages/agent-core-v2/src/human/test/llm/headers-timeout.test.ts b/packages/agent-core-v2/src/human/test/llm/headers-timeout.test.ts new file mode 100644 index 00000000000..ab0ab7fd5d6 --- /dev/null +++ b/packages/agent-core-v2/src/human/test/llm/headers-timeout.test.ts @@ -0,0 +1,162 @@ +import { describe, expect, it, vi } from 'vitest'; + +import { UNKNOWN_CAPABILITY } from '#/llm/capability'; +import { createUserMessage, type Message } from '#/llm/message'; +import type { LlmModel } from '#/llm/model'; +import type { LlmRequester, LlmRequestEvent } from '#/llm/requester/requester'; +import { LLM_HEADERS_TIMEOUT_ENV } from '#/llm/requester/timeout'; +import { createAnthropicRequester } from '#/llm/requester/bases/anthropic/requester'; +import { createOpenAIRequester } from '#/llm/requester/bases/openai/requester'; +import { createOpenAIResponsesRequester } from '#/llm/requester/bases/openai-responses/requester'; + +const model: LlmModel = { + provider: 'test', + model: 'test-model', + capability: UNKNOWN_CAPABILITY, + baseUrl: 'https://example.test/v1', +}; +const messages: readonly Message[] = [createUserMessage('hi')]; + +const openAISse = [ + 'data: {"id":"chatcmpl-1","object":"chat.completion.chunk","created":0,"model":"test-model","choices":[{"index":0,"delta":{"role":"assistant","content":"hi"},"finish_reason":"stop"}]}', + '', + 'data: [DONE]', + '', + '', +].join('\n'); + +const responsesSse = [ + 'event: response.completed', + 'data: {"type":"response.completed","response":{"id":"resp_1","status":"completed","usage":{"input_tokens":1,"output_tokens":1,"total_tokens":2}}}', + '', + '', +].join('\n'); + +const anthropicSse = [ + 'event: message_start', + 'data: {"type":"message_start","message":{"id":"msg_1","type":"message","role":"assistant","content":[],"model":"test-model","stop_reason":null,"stop_sequence":null,"usage":{"input_tokens":10,"output_tokens":1}}}', + '', + 'event: content_block_start', + 'data: {"type":"content_block_start","index":0,"content_block":{"type":"text","text":""}}', + '', + 'event: content_block_delta', + 'data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"hi"}}', + '', + 'event: content_block_stop', + 'data: {"type":"content_block_stop","index":0}', + '', + 'event: message_delta', + 'data: {"type":"message_delta","delta":{"stop_reason":"end_turn","stop_sequence":null},"usage":{"output_tokens":2}}', + '', + 'event: message_stop', + 'data: {"type":"message_stop"}', + '', + '', +].join('\n'); + +async function generate(requester: LlmRequester): Promise { + const events: LlmRequestEvent[] = []; + await requester.generate( + { model }, + { messages }, + { signal: new AbortController().signal, onEvent: (event) => events.push(event) }, + ); + return events; +} + +describe('default client headers timeout', () => { + it('passes the configured headers-timeout dispatcher to fetch and fails the request on invalid values', async () => { + const proxyEnvKeys = [ + 'http_proxy', + 'HTTP_PROXY', + 'https_proxy', + 'HTTPS_PROXY', + 'all_proxy', + 'ALL_PROXY', + 'no_proxy', + 'NO_PROXY', + ]; + const savedEnv = Object.fromEntries( + [LLM_HEADERS_TIMEOUT_ENV, ...proxyEnvKeys].map((key) => [key, process.env[key]]), + ); + for (const key of [LLM_HEADERS_TIMEOUT_ENV, ...proxyEnvKeys]) delete process.env[key]; + let currentSse = openAISse; + const fetchStub = vi.fn( + async () => + new Response(currentSse, { + status: 200, + headers: { 'content-type': 'text/event-stream' }, + }), + ); + vi.stubGlobal('fetch', fetchStub); + try { + const protocols = [ + { create: () => createOpenAIRequester(), sse: openAISse }, + { create: () => createOpenAIResponsesRequester(), sse: responsesSse }, + { create: () => createAnthropicRequester(), sse: anthropicSse }, + ]; + let calls = 0; + for (const protocol of protocols) { + currentSse = protocol.sse; + let events = await generate(protocol.create()); + calls += 1; + expect(events.at(-1)).toMatchObject({ type: 'llm.done' }); + expect(fetchStub).toHaveBeenCalledTimes(calls); + expect(fetchStub.mock.calls[calls - 1]?.[1]).not.toHaveProperty('dispatcher'); + + process.env[LLM_HEADERS_TIMEOUT_ENV] = '45000'; + events = await generate(protocol.create()); + calls += 1; + expect(events.at(-1)).toMatchObject({ type: 'llm.done' }); + expect(fetchStub).toHaveBeenCalledTimes(calls); + expect(fetchStub.mock.calls[calls - 1]?.[1]).toHaveProperty('dispatcher'); + delete process.env[LLM_HEADERS_TIMEOUT_ENV]; + } + + currentSse = openAISse; + process.env[LLM_HEADERS_TIMEOUT_ENV] = '45000'; + process.env['HTTP_PROXY'] = 'http://127.0.0.1:3128'; + await generate(createOpenAIRequester()); + calls += 1; + expect(fetchStub).toHaveBeenCalledTimes(calls); + expect(fetchStub.mock.calls[calls - 1]?.[1]).toHaveProperty('dispatcher'); + delete process.env['HTTP_PROXY']; + + process.env['ALL_PROXY'] = 'socks5://127.0.0.1:1080'; + await generate(createOpenAIRequester()); + calls += 1; + expect(fetchStub).toHaveBeenCalledTimes(calls); + expect(fetchStub.mock.calls[calls - 1]?.[1]).not.toHaveProperty('dispatcher'); + delete process.env['ALL_PROXY']; + + process.env[LLM_HEADERS_TIMEOUT_ENV] = 'abc'; + let events = await generate(createOpenAIRequester()); + expect(events.at(-1)).toMatchObject({ + type: 'llm.failed.remote', + error: { message: expect.stringContaining(LLM_HEADERS_TIMEOUT_ENV) }, + }); + + process.env[LLM_HEADERS_TIMEOUT_ENV] = '3000000000'; + events = await generate(createOpenAIRequester()); + expect(events.at(-1)).toMatchObject({ + type: 'llm.failed.remote', + error: { message: expect.stringContaining(LLM_HEADERS_TIMEOUT_ENV) }, + }); + + process.env[LLM_HEADERS_TIMEOUT_ENV] = '45000'; + process.env['HTTP_PROXY'] = 'not-a-url'; + events = await generate(createOpenAIRequester()); + calls += 1; + expect(events.at(-1)).toMatchObject({ type: 'llm.done' }); + expect(fetchStub).toHaveBeenCalledTimes(calls); + expect(fetchStub.mock.calls[calls - 1]?.[1]).not.toHaveProperty('dispatcher'); + delete process.env['HTTP_PROXY']; + } finally { + vi.unstubAllGlobals(); + for (const [key, value] of Object.entries(savedEnv)) { + if (value === undefined) delete process.env[key]; + else process.env[key] = value; + } + } + }); +});