diff --git a/docker-compose.mtp.yaml b/docker-compose.mtp.yaml index a74f116857..880307c089 100644 --- a/docker-compose.mtp.yaml +++ b/docker-compose.mtp.yaml @@ -25,8 +25,21 @@ services: # was raised to match the README since Gemma-4's hybrid SWA # architecture (most layers use a small fixed attention window, only # every 5th layer is full-context) keeps KV cache growth with n_ctx - # much cheaper than a plain transformer's. - - 'MTP_CTX_SIZE=32768' + # much cheaper than a plain transformer's -- measured at only 271 MiB + # of KV cache (target + draft) per 32768-token slot. + # + # -np 4 gives open-webui's browser chat and the voice assistant (and + # anything else hitting this OpenAI-compatible endpoint) their own + # slot each instead of one contending for a single shared slot -- + # with -np 1, whichever of them touched the slot last would evict the + # other's live KV cache, forcing a full prompt reprocess (multi-second + # stall) on the next request from the loser. n_ctx_slot = n_ctx / + # n_parallel, so total ctx is quadrupled to keep each slot at the same + # 32768 budget as before -- 131072 matches this model's native + # n_ctx_train exactly, and only costs ~810 MiB more KV cache VRAM + # than the old -np 1 config (still well under the ~7GB that's free). + - 'MTP_CTX_SIZE=131072' + - 'MTP_N_PARALLEL=4' - 'MTP_BATCH_SIZE=512' # Server-side default -- only applies when the client doesn't send its # own temperature (Open WebUI won't unless you set it in the model's