services: llama-mtp: build: context: ../mtp-relaxed-decoding container_name: llama-mtp restart: unless-stopped ports: - '8030:8030' volumes: # Same host dir ollama's /gguf-import mount points at -- target + # MTP drafter already live there. - /home/brian-llm/models/gguf:/models:ro environment: - 'MTP_AUTH_TOKEN=${MTP_AUTH_TOKEN}' - 'MTP_TARGET_GGUF=/models/gemma-4-E2B-it-Q4_K_M.gguf' - 'MTP_DRAFT_GGUF=/models/mtp-gemma-4-E2B-it.gguf' # n_max beyond ~2 wasn't worth it for this drafter, and relaxed_top_n # 2-20 all performed similarly -- see ../mtp-relaxed-decoding/README.md # ยง6. Benchmarked on an RTX 5090 there; re-check on this Titan Xp # before trusting these as tuned rather than just reasonable defaults. - 'MTP_N_MAX=2' - 'MTP_RELAXED_TOP_N=5' # Titan Xp has 12GB VRAM shared with asr/tts/ollama -- batch size kept # below the README's RTX-5090 sizing (1024) for headroom, but context # was raised to match the README since Gemma-4's hybrid SWA # architecture (most layers use a small fixed attention window, only # every 5th layer is full-context) keeps KV cache growth with n_ctx # much cheaper than a plain transformer's -- measured at only 271 MiB # of KV cache (target + draft) per 32768-token slot. # # -np 4 gives open-webui's browser chat and the voice assistant (and # anything else hitting this OpenAI-compatible endpoint) their own # slot each instead of one contending for a single shared slot -- # with -np 1, whichever of them touched the slot last would evict the # other's live KV cache, forcing a full prompt reprocess (multi-second # stall) on the next request from the loser. n_ctx_slot = n_ctx / # n_parallel, so total ctx is quadrupled to keep each slot at the same # 32768 budget as before -- 131072 matches this model's native # n_ctx_train exactly, and only costs ~810 MiB more KV cache VRAM # than the old -np 1 config (still well under the ~7GB that's free). - 'MTP_CTX_SIZE=131072' - 'MTP_N_PARALLEL=4' - 'MTP_BATCH_SIZE=512' # Server-side default -- only applies when the client doesn't send its # own temperature (Open WebUI won't unless you set it in the model's # Advanced Params). - 'MTP_TEMP=0.6' deploy: resources: reservations: devices: - driver: nvidia count: 1 capabilities: [gpu] open-webui: depends_on: - llama-mtp environment: # Registers this alongside (not instead of) the Ollama connection -- # Open WebUI lists both in the model picker. - 'OPENAI_API_BASE_URLS=http://llama-mtp:8030/v1' - 'OPENAI_API_KEYS=${MTP_AUTH_TOKEN}' ollama-auth: depends_on: - llama-mtp environment: # ollama-auth/default.conf.template now proxies to llama-mtp (Ollama # itself serves no models anymore) and swaps the client's # OLLAMA_AUTH_TOKEN for this before forwarding upstream. - 'MTP_AUTH_TOKEN=${MTP_AUTH_TOKEN}'