Skip to content

Commit a256c25

Browse files
committed
fix: replace SWAP-ASSISTED warning with SSD STREAMING label when streamExperts active
Expert weights are mmap'd via OS page cache - no swap used. Suppress misleading swap slowdown warnings when SSD streaming is on.
1 parent 48e49dc commit a256c25

1 file changed

Lines changed: 6 additions & 7 deletions

File tree

‎Sources/mlx-server/Server.swift‎

Lines changed: 6 additions & 7 deletions
Original file line numberDiff line numberDiff line change
@@ -255,29 +255,28 @@ struct MLXServer: AsyncParsableCommand {
255255
print("[mlx-server] \(plan.strategy.emoji) Memory strategy: FULL GPU (\(String(format: "%.1f", plan.weightMemoryGB))GB model, \(String(format: "%.1f", system.availableRAMGB))GB available)")
256256
case .swapAssisted:
257257
if self.streamExperts {
258-
// SSD Streaming: bypass Metal's capped working set and use physical RAM budget
259-
// (85% of total RAM minus 4GB OS reservation)
258+
// SSD Streaming: expert weights are mmap'd from SSD via the OS page cache.
259+
// No swap involved — the page cache evicts stale expert pages cleanly.
260260
let physicalBudget = Int(Double(system.totalRAMBytes) * 0.85) - (4 * 1024 * 1024 * 1024)
261261
Memory.cacheLimit = physicalBudget
262262
Memory.memoryLimit = 200 * 1024 * 1024 * 1024 // 200GB sentinel to bypass MLX eval_impl spin loop
263-
print("[mlx-server] \(plan.strategy.emoji) Memory strategy: SWAP-ASSISTED + SSD Streaming. Cache limit: \(physicalBudget / (1024*1024*1024))GB (physical RAM budget).")
263+
print("[mlx-server] 💾 Memory strategy: SSD STREAMING (page-cache managed, \(physicalBudget / (1024*1024*1024))GB RAM budget, no swap)")
264264
} else {
265265
Memory.cacheLimit = plan.recommendedCacheLimit
266266
print("[mlx-server] \(plan.strategy.emoji) Memory strategy: SWAP-ASSISTED (\(String(format: "%.1f", plan.overcommitRatio))× overcommit, cache limited to \(plan.recommendedCacheLimit / (1024*1024))MB)")
267+
for w in plan.warnings { print("[mlx-server] \(w)") }
267268
}
268-
for w in plan.warnings { print("[mlx-server] \(w)") }
269269
case .layerPartitioned:
270270
if self.streamExperts {
271-
// SSD Streaming: bypass Metal's capped working set and use physical RAM budget
272271
let physicalBudget = Int(Double(system.totalRAMBytes) * 0.85) - (4 * 1024 * 1024 * 1024)
273272
Memory.cacheLimit = physicalBudget
274273
Memory.memoryLimit = 200 * 1024 * 1024 * 1024 // 200GB sentinel to bypass MLX eval_impl spin loop
275-
print("[mlx-server] \(plan.strategy.emoji) Memory strategy: LAYER PARTITIONED + SSD Streaming. Cache limit: \(physicalBudget / (1024*1024*1024))GB (physical RAM budget).")
274+
print("[mlx-server] 💾 Memory strategy: SSD STREAMING (page-cache managed, \(physicalBudget / (1024*1024*1024))GB RAM budget, no swap)")
276275
} else {
277276
Memory.cacheLimit = plan.recommendedCacheLimit
278277
print("[mlx-server] \(plan.strategy.emoji) Memory strategy: LAYER PARTITIONED (\(plan.recommendedGPULayers)/\(plan.totalLayers) GPU layers, cache limited to \(plan.recommendedCacheLimit / (1024*1024))MB)")
278+
for w in plan.warnings { print("[mlx-server] \(w)") }
279279
}
280-
for w in plan.warnings { print("[mlx-server] \(w)") }
281280
case .tooLarge:
282281
Memory.cacheLimit = plan.recommendedCacheLimit
283282
print("[mlx-server] \(plan.strategy.emoji) WARNING: Model is \(String(format: "%.1f", plan.overcommitRatio))× system RAM. Loading will be extremely slow.")

0 commit comments

Comments
 (0)