-
Notifications
You must be signed in to change notification settings - Fork 1.7k
Expand file tree
/
Copy pathmodel_registry.yaml
More file actions
67 lines (55 loc) · 2.03 KB
/
Copy pathmodel_registry.yaml
File metadata and controls
67 lines (55 loc) · 2.03 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
# Model registry — context window and output token limits.
#
# Point SKILLSPECTOR_MODEL_REGISTRY at this file (or your own) so the tool
# knows each model's token budget. This is the fallback when the dynamic
# metadata API is unavailable (e.g. open-source deployments).
#
# Format:
# models:
# "<model-label>":
# context_length: <int> # total context window in tokens (required)
# max_output_tokens: <int> # model's max output cap (optional)
models:
# Stock OpenAI model IDs (for direct api.openai.com or compatible endpoints).
"gpt-5.2":
context_length: 400000
max_output_tokens: 128000
"gpt-5.3-chat":
context_length: 128000
max_output_tokens: 16384
# Pricing (USD per 1M tokens) is informational only; SkillSpector does not
# compute cost. Prompts over 272K input tokens bill at 2x input / 1.5x output.
# Source: https://developers.openai.com/api/docs/models/<model> (as of 2026-09-21)
# $4.00 input / $0.40 cached input / $20.00 output (promotional through
# 2026-11-21). Input is capped at 922K tokens; the 75% input budget
# (~787K) stays under it.
"gpt-5.6-sol":
context_length: 1050000
max_output_tokens: 128000
# $2.00 input / $0.20 cached input / $12.00 output
"gpt-5.6-terra":
context_length: 1050000
max_output_tokens: 128000
# $0.20 input / $0.02 cached input / $1.20 output
"gpt-5.6-luna":
context_length: 1050000
max_output_tokens: 128000
# Provider-prefixed IDs for inference gateways that accept them.
"azure/anthropic/claude-opus-4-5":
context_length: 200000
max_output_tokens: 64000
"azure/anthropic/claude-sonnet-4-6":
context_length: 1000000
max_output_tokens: 128000
"azure/anthropic/claude-opus-4-6":
context_length: 1000000
max_output_tokens: 128000
"openai/openai/gpt-5.2":
context_length: 400000
max_output_tokens: 128000
"openai/openai/gpt-5.3-chat":
context_length: 128000
max_output_tokens: 16384
"gemini-3.5-flash":
context_length: 1048576
max_output_tokens: 65536