Skip to content

Commit d37d555

Browse files
authored
Merge pull request #181 from SharpAI/fix/turbokv-ctx-size-warning
fix(turbokv): warn when --turbo-kv can't apply (e.g. with --ctx-size)
2 parents 323bd6b + f5f0649 commit d37d555

1 file changed

Lines changed: 15 additions & 0 deletions

File tree

‎Sources/SwiftLM/Server.swift‎

Lines changed: 15 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -227,6 +227,11 @@ func isVLMCheckpointMismatch(_ error: any Error) -> Bool {
227227
}
228228
}
229229

230+
/// One-shot flags for TurboKV configuration notices.
231+
enum TurboKVNotice {
232+
nonisolated(unsafe) static var warnedNoLayers = false
233+
}
234+
230235
func sanitizeForJinja(_ value: any Sendable) -> (any Sendable)? {
231236
if value is NSNull { return nil }
232237
let mirror = Mirror(reflecting: value)
@@ -1287,6 +1292,9 @@ struct MLXServer: AsyncParsableCommand {
12871292
let turboKVStr = config.turboKV ? "enabled" : "disabled"
12881293
let mtpStr = config.mtp ? "enabled (\(config.numMtpTokens) tokens/round)" : "disabled"
12891294
print("[SwiftLM] Config: ctx_size=\(ctxSizeStr), temp=\(config.temp), top_p=\(config.topP), top_k=\(topKStr), min_p=\(minPStr), repeat_penalty=\(penaltyStr), parallel=\(parallelSlots), cors=\(corsStr), mem_limit=\(memLimitStr), auth=\(authStr), thinking=\(thinkingStr), ssd_stream=\(ssdStr), turbo_kv=\(turboKVStr), mtp=\(mtpStr)")
1295+
if config.turboKV, let ctx = config.ctxSize {
1296+
print("[SwiftLM] ⚠️ --turbo-kv has no effect with --ctx-size \(ctx): a bounded context gives the attention layers a RotatingKVCache, and TurboKV only compresses KVCacheSimple. Drop --ctx-size to use --turbo-kv.")
1297+
}
12901298

12911299
// ── Build Hummingbird router ──
12921300
let router = Router()
@@ -2038,11 +2046,18 @@ func handleChatCompletion(
20382046
// This compresses cache history older than 8192 tokens into 3.5-bit Polar+QJL
20392047
// form, halving KV RAM for long-context (100k+) requests.
20402048
if config.turboKV {
2049+
var enabledLayers = 0
20412050
for layer in cache {
20422051
if let simple = layer as? KVCacheSimple {
20432052
simple.turboQuantEnabled = true
2053+
enabledLayers += 1
20442054
}
20452055
}
2056+
if enabledLayers == 0 && !TurboKVNotice.warnedNoLayers {
2057+
// Warn once. A benign race here only means a duplicate log line.
2058+
TurboKVNotice.warnedNoLayers = true
2059+
print("[SwiftLM] ⚠️ --turbo-kv is enabled but this model's cache has no KVCacheSimple layers (\(cache.count) layers, e.g. RotatingKVCache from --ctx-size), so no KV compression is applied.")
2060+
}
20462061
}
20472062

20482063
// ── Prompt cache: bypass for multimodal inputs ──

0 commit comments

Comments
 (0)