Skip to content
Merged
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
15 changes: 15 additions & 0 deletions Sources/SwiftLM/Server.swift
Original file line number Diff line number Diff line change
Expand Up @@ -227,6 +227,11 @@ func isVLMCheckpointMismatch(_ error: any Error) -> Bool {
}
}

/// One-shot flags for TurboKV configuration notices.
enum TurboKVNotice {
nonisolated(unsafe) static var warnedNoLayers = false
}

func sanitizeForJinja(_ value: any Sendable) -> (any Sendable)? {
if value is NSNull { return nil }
let mirror = Mirror(reflecting: value)
Expand Down Expand Up @@ -1287,6 +1292,9 @@ struct MLXServer: AsyncParsableCommand {
let turboKVStr = config.turboKV ? "enabled" : "disabled"
let mtpStr = config.mtp ? "enabled (\(config.numMtpTokens) tokens/round)" : "disabled"
print("[SwiftLM] Config: ctx_size=\(ctxSizeStr), temp=\(config.temp), top_p=\(config.topP), top_k=\(topKStr), min_p=\(minPStr), repeat_penalty=\(penaltyStr), parallel=\(parallelSlots), cors=\(corsStr), mem_limit=\(memLimitStr), auth=\(authStr), thinking=\(thinkingStr), ssd_stream=\(ssdStr), turbo_kv=\(turboKVStr), mtp=\(mtpStr)")
if config.turboKV, let ctx = config.ctxSize {
print("[SwiftLM] ⚠️ --turbo-kv has no effect with --ctx-size \(ctx): a bounded context gives the attention layers a RotatingKVCache, and TurboKV only compresses KVCacheSimple. Drop --ctx-size to use --turbo-kv.")
}

// ── Build Hummingbird router ──
let router = Router()
Expand Down Expand Up @@ -2038,11 +2046,18 @@ func handleChatCompletion(
// This compresses cache history older than 8192 tokens into 3.5-bit Polar+QJL
// form, halving KV RAM for long-context (100k+) requests.
if config.turboKV {
var enabledLayers = 0
for layer in cache {
if let simple = layer as? KVCacheSimple {
simple.turboQuantEnabled = true
enabledLayers += 1
}
}
if enabledLayers == 0 && !TurboKVNotice.warnedNoLayers {
// Warn once. A benign race here only means a duplicate log line.
TurboKVNotice.warnedNoLayers = true
print("[SwiftLM] ⚠️ --turbo-kv is enabled but this model's cache has no KVCacheSimple layers (\(cache.count) layers, e.g. RotatingKVCache from --ctx-size), so no KV compression is applied.")
}
}

// ── Prompt cache: bypass for multimodal inputs ──
Expand Down
Loading