From 966140c3efac42429c1c48fa1774cd1dace1ceb2 Mon Sep 17 00:00:00 2001 From: Brian Hetherman Date: Sat, 5 Sep 2026 13:40:50 -0400 Subject: [PATCH] Give the MTP llama-server 4 parallel slots instead of 1 With -np 1, open-webui's browser chat and the voice assistant (and anything else hitting this OpenAI-compatible endpoint) contended for a single slot -- whichever touched it last evicted the other's live KV cache, forcing a multi-second full prompt reprocess on the next request from the loser. n_ctx_slot = n_ctx / n_parallel, so ctx is quadrupled to 131072 (this model's native n_ctx_train) to keep each of the 4 slots at the same 32768 budget as before. Measured KV cache cost is only 271 MiB per slot, so this only costs ~810 MiB more VRAM than the old config. Co-Authored-By: Claude Sonnet 5 --- docker-compose.mtp.yaml | 17 +++++++++++++++-- 1 file changed, 15 insertions(+), 2 deletions(-) diff --git a/docker-compose.mtp.yaml b/docker-compose.mtp.yaml index a74f116857..880307c089 100644 --- a/docker-compose.mtp.yaml +++ b/docker-compose.mtp.yaml @@ -25,8 +25,21 @@ services: # was raised to match the README since Gemma-4's hybrid SWA # architecture (most layers use a small fixed attention window, only # every 5th layer is full-context) keeps KV cache growth with n_ctx - # much cheaper than a plain transformer's. - - 'MTP_CTX_SIZE=32768' + # much cheaper than a plain transformer's -- measured at only 271 MiB + # of KV cache (target + draft) per 32768-token slot. + # + # -np 4 gives open-webui's browser chat and the voice assistant (and + # anything else hitting this OpenAI-compatible endpoint) their own + # slot each instead of one contending for a single shared slot -- + # with -np 1, whichever of them touched the slot last would evict the + # other's live KV cache, forcing a full prompt reprocess (multi-second + # stall) on the next request from the loser. n_ctx_slot = n_ctx / + # n_parallel, so total ctx is quadrupled to keep each slot at the same + # 32768 budget as before -- 131072 matches this model's native + # n_ctx_train exactly, and only costs ~810 MiB more KV cache VRAM + # than the old -np 1 config (still well under the ~7GB that's free). + - 'MTP_CTX_SIZE=131072' + - 'MTP_N_PARALLEL=4' - 'MTP_BATCH_SIZE=512' # Server-side default -- only applies when the client doesn't send its # own temperature (Open WebUI won't unless you set it in the model's