From 6f583deff99138a19ef77cdd5ebc0dd4a4cd15cd Mon Sep 17 00:00:00 2001 From: Brian Hetherman Date: Fri, 4 Sep 2026 11:33:12 -0400 Subject: [PATCH] Add relaxed-MTP llama-server as a new connection, surface generation speed Wires the custom relaxed-acceptance llama-server (see ../mtp-relaxed-decoding) into the stack as a new llama-mtp service, registered as an additional OpenAI-compatible connection alongside Ollama. ollama-auth now proxies to llama-mtp instead of the now-empty Ollama, so LAN clients (e.g. Home Assistant's voice pipeline) keep working against the same URL/token with no reconfiguration. Also merges llama.cpp's `timings` extension (dropped by the generic OpenAI response schema) into message.usage and surfaces it as a labeled tok/s badge, so relaxed-MTP responses get the same visible generation-speed info Ollama responses already had. Co-Authored-By: Claude Sonnet 5 --- docker-compose.mtp.yaml | 59 +++++++++++++++++++ docker-compose.yaml | 7 ++- docker-stack.sh | 2 +- ollama-auth/default.conf.template | 22 ++++--- src/lib/components/chat/Chat.svelte | 11 +++- .../chat/Messages/ResponseMessage.svelte | 25 ++++++++ 6 files changed, 113 insertions(+), 13 deletions(-) create mode 100644 docker-compose.mtp.yaml diff --git a/docker-compose.mtp.yaml b/docker-compose.mtp.yaml new file mode 100644 index 0000000000..a74f116857 --- /dev/null +++ b/docker-compose.mtp.yaml @@ -0,0 +1,59 @@ +services: + llama-mtp: + build: + context: ../mtp-relaxed-decoding + container_name: llama-mtp + restart: unless-stopped + ports: + - '8030:8030' + volumes: + # Same host dir ollama's /gguf-import mount points at -- target + + # MTP drafter already live there. + - /home/brian-llm/models/gguf:/models:ro + environment: + - 'MTP_AUTH_TOKEN=${MTP_AUTH_TOKEN}' + - 'MTP_TARGET_GGUF=/models/gemma-4-E2B-it-Q4_K_M.gguf' + - 'MTP_DRAFT_GGUF=/models/mtp-gemma-4-E2B-it.gguf' + # n_max beyond ~2 wasn't worth it for this drafter, and relaxed_top_n + # 2-20 all performed similarly -- see ../mtp-relaxed-decoding/README.md + # ยง6. Benchmarked on an RTX 5090 there; re-check on this Titan Xp + # before trusting these as tuned rather than just reasonable defaults. + - 'MTP_N_MAX=2' + - 'MTP_RELAXED_TOP_N=5' + # Titan Xp has 12GB VRAM shared with asr/tts/ollama -- batch size kept + # below the README's RTX-5090 sizing (1024) for headroom, but context + # was raised to match the README since Gemma-4's hybrid SWA + # architecture (most layers use a small fixed attention window, only + # every 5th layer is full-context) keeps KV cache growth with n_ctx + # much cheaper than a plain transformer's. + - 'MTP_CTX_SIZE=32768' + - 'MTP_BATCH_SIZE=512' + # Server-side default -- only applies when the client doesn't send its + # own temperature (Open WebUI won't unless you set it in the model's + # Advanced Params). + - 'MTP_TEMP=0.6' + deploy: + resources: + reservations: + devices: + - driver: nvidia + count: 1 + capabilities: [gpu] + + open-webui: + depends_on: + - llama-mtp + environment: + # Registers this alongside (not instead of) the Ollama connection -- + # Open WebUI lists both in the model picker. + - 'OPENAI_API_BASE_URLS=http://llama-mtp:8030/v1' + - 'OPENAI_API_KEYS=${MTP_AUTH_TOKEN}' + + ollama-auth: + depends_on: + - llama-mtp + environment: + # ollama-auth/default.conf.template now proxies to llama-mtp (Ollama + # itself serves no models anymore) and swaps the client's + # OLLAMA_AUTH_TOKEN for this before forwarding upstream. + - 'MTP_AUTH_TOKEN=${MTP_AUTH_TOKEN}' diff --git a/docker-compose.yaml b/docker-compose.yaml index d68c76b535..aa82dd6a25 100644 --- a/docker-compose.yaml +++ b/docker-compose.yaml @@ -66,7 +66,12 @@ services: environment: - 'OLLAMA_BASE_URL=http://ollama:11434' - 'WEBUI_SECRET_KEY=' - - 'DEFAULT_MODELS=gemma-4-e2b-mtp' + # Only takes effect on a fresh DB -- once ui.default_models exists in + # the config table, that value wins (see open-webui's PersistentConfig + # seed_defaults behavior). Currently gemma-4-e2b-mtp was removed from + # Ollama in favor of the relaxed-MTP llama-mtp service; the live value + # is 'gemma-4-e2b,/models/gemma-4-E2B-it-Q4_K_M.gguf'. + - 'DEFAULT_MODELS=gemma-4-e2b,/models/gemma-4-E2B-it-Q4_K_M.gguf' - 'ENABLE_OAUTH_SIGNUP=true' # Links Authentik login to the existing local account with the same # email instead of rejecting it as a collision -- safe here since this diff --git a/docker-stack.sh b/docker-stack.sh index 8821f4c5b7..88ff3d4aef 100755 --- a/docker-stack.sh +++ b/docker-stack.sh @@ -5,7 +5,7 @@ cd "$(dirname "${BASH_SOURCE[0]}")" # Wraps docker compose with the full file set for this stack (base + GPU # passthrough + audio/ollama-auth services) so a plain "up"/"build" can't # accidentally drop the ollama GPU override again. -readonly COMPOSE_FILES=(-f docker-compose.yaml -f docker-compose.gpu.yaml -f docker-compose.audio.yaml) +readonly COMPOSE_FILES=(-f docker-compose.yaml -f docker-compose.gpu.yaml -f docker-compose.audio.yaml -f docker-compose.mtp.yaml) if [[ $# -eq 0 ]]; then exec docker compose "${COMPOSE_FILES[@]}" up -d --build diff --git a/ollama-auth/default.conf.template b/ollama-auth/default.conf.template index 08092d61f4..12bb7bf90b 100644 --- a/ollama-auth/default.conf.template +++ b/ollama-auth/default.conf.template @@ -13,18 +13,24 @@ server { return 401; } - # Docker's embedded DNS (127.0.0.11) can reassign the "ollama" - # hostname to a new container IP on restart. proxy_pass to a - # literal upstream caches that IP for the container's lifetime; - # routing through a variable forces nginx to re-resolve via this - # resolver on each request instead, so a restarted ollama doesn't - # need a matching ollama-auth restart to be reachable again. + # Ollama itself now serves no models -- this proxy targets + # llama-mtp (the relaxed-MTP server) instead, so LAN clients that + # already trust this URL/token (e.g. Home Assistant's voice + # pipeline) don't need reconfiguring. Docker's embedded DNS + # (127.0.0.11) can reassign a container hostname to a new IP on + # restart; proxy_pass to a literal upstream caches that IP for the + # container's lifetime, so route through a variable to force + # nginx to re-resolve via this resolver on each request instead. resolver 127.0.0.11 valid=10s; - set $ollama_upstream ollama; - proxy_pass http://$ollama_upstream:11434; + set $mtp_upstream llama-mtp; + proxy_pass http://$mtp_upstream:8030; proxy_http_version 1.1; proxy_set_header Connection ""; proxy_set_header Host $host; + # Client presents OLLAMA_AUTH_TOKEN (validated above); llama-server + # behind this proxy expects its own MTP_AUTH_TOKEN instead, so swap + # it here rather than requiring every client to be reconfigured. + proxy_set_header Authorization "Bearer ${MTP_AUTH_TOKEN}"; proxy_buffering off; proxy_read_timeout 600s; proxy_send_timeout 600s; diff --git a/src/lib/components/chat/Chat.svelte b/src/lib/components/chat/Chat.svelte index 9aa7ad720b..5a0120b77c 100644 --- a/src/lib/components/chat/Chat.svelte +++ b/src/lib/components/chat/Chat.svelte @@ -2744,7 +2744,8 @@ }; const chatCompletionEventHandler = async (data, message, chatId) => { - const { id, done, choices, content, output, sources, selected_model_id, error, usage } = data; + const { id, done, choices, content, output, sources, selected_model_id, error, usage, timings } = + data; // Store raw OR-aligned output items from backend if (output) { @@ -2797,8 +2798,12 @@ message.arena = true; } - if (usage) { - message.usage = usage; + if (usage || timings) { + // llama.cpp-compatible servers (e.g. our relaxed-MTP llama-server) send + // generation speed as a sibling `timings` object rather than inside + // `usage` -- merge it in so it surfaces in the same response-info + // tooltip Ollama's eval_count/eval_duration fields already populate. + message.usage = { ...(usage || {}), ...(timings || {}) }; } history.messages[message.id] = message; diff --git a/src/lib/components/chat/Messages/ResponseMessage.svelte b/src/lib/components/chat/Messages/ResponseMessage.svelte index 2826111865..f3c38d9465 100644 --- a/src/lib/components/chat/Messages/ResponseMessage.svelte +++ b/src/lib/components/chat/Messages/ResponseMessage.svelte @@ -144,6 +144,23 @@ } } + // Generation speed, normalized across backends: Ollama reports + // eval_count/eval_duration (nanoseconds), llama.cpp-compatible servers + // (e.g. our relaxed-MTP llama-server) report timings.predicted_per_second + // directly -- note that field name is llama.cpp's generic term for "the + // main generation loop's output," i.e. the actual response stream, NOT + // the MTP drafter in isolation (which is separately broken out as + // draft_n/draft_n_accepted). + const getTokensPerSecond = (usage: Record | undefined | null) => { + if (!usage) return null; + if (typeof usage.predicted_per_second === 'number') return usage.predicted_per_second; + if (typeof usage.eval_count === 'number' && typeof usage.eval_duration === 'number' && usage.eval_duration > 0) { + return (usage.eval_count / usage.eval_duration) * 1e9; + } + return null; + }; + $: tokensPerSecond = getTokensPerSecond(message.usage); + export let siblings; export let setInputText: Function = () => {}; @@ -1189,6 +1206,14 @@ {/if} + {#if tokensPerSecond} + + {tokensPerSecond.toFixed(1)} tok/s + + {/if} + {#if message.usage}