Add relaxed-MTP llama-server as a new connection, surface generation speed
Create and publish Docker images with specific build args / build (map[arch:linux/amd64 runner:ubuntu-latest], map[build_args: free_disk:false name:main suffix:]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/amd64 runner:ubuntu-latest], map[build_args:USE_CUDA=true
USE_CUDA_VER=cu126
free_disk:true name:cuda126 suffix:-cuda126]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/amd64 runner:ubuntu-latest], map[build_args:USE_CUDA=true free_disk:true name:cuda suffix:-cuda]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/amd64 runner:ubuntu-latest], map[build_args:USE_OLLAMA=true free_disk:false name:ollama suffix:-ollama]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/amd64 runner:ubuntu-latest], map[build_args:USE_SLIM=true free_disk:false name:slim suffix:-slim]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/arm64 runner:ubuntu-24.04-arm], map[build_args: free_disk:false name:main suffix:]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/arm64 runner:ubuntu-24.04-arm], map[build_args:USE_CUDA=true
USE_CUDA_VER=cu126
free_disk:true name:cuda126 suffix:-cuda126]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/arm64 runner:ubuntu-24.04-arm], map[build_args:USE_CUDA=true free_disk:true name:cuda suffix:-cuda]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/arm64 runner:ubuntu-24.04-arm], map[build_args:USE_OLLAMA=true free_disk:false name:ollama suffix:-ollama]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/arm64 runner:ubuntu-24.04-arm], map[build_args:USE_SLIM=true free_disk:false name:slim suffix:-slim]) (push) Canceled after 0s
Frontend Build / Format & Build (push) Canceled after 0s
Frontend Build / Unit Tests (push) Canceled after 0s
Release to PyPI / release (push) Canceled after 0s
Release / publish (push) Canceled after 0s
Create and publish Docker images with specific build args / merge (map[name:cuda suffix:-cuda]) (push) Canceled after 0s
Create and publish Docker images with specific build args / merge (map[name:cuda126 suffix:-cuda126]) (push) Canceled after 0s
Create and publish Docker images with specific build args / merge (map[name:main suffix:]) (push) Canceled after 0s
Create and publish Docker images with specific build args / merge (map[name:ollama suffix:-ollama]) (push) Canceled after 0s
Create and publish Docker images with specific build args / merge (map[name:slim suffix:-slim]) (push) Canceled after 0s
Create and publish Docker images with specific build args / notify-helm-charts (push) Canceled after 0s
Create and publish Docker images with specific build args / copy-to-dockerhub (, main) (push) Canceled after 0s
Create and publish Docker images with specific build args / copy-to-dockerhub (-cuda, cuda) (push) Canceled after 0s
Create and publish Docker images with specific build args / copy-to-dockerhub (-cuda126, cuda126) (push) Canceled after 0s
Create and publish Docker images with specific build args / copy-to-dockerhub (-ollama, ollama) (push) Canceled after 0s
Create and publish Docker images with specific build args / copy-to-dockerhub (-slim, slim) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/amd64 runner:ubuntu-latest], map[build_args: free_disk:false name:main suffix:]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/amd64 runner:ubuntu-latest], map[build_args:USE_CUDA=true
USE_CUDA_VER=cu126
free_disk:true name:cuda126 suffix:-cuda126]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/amd64 runner:ubuntu-latest], map[build_args:USE_CUDA=true free_disk:true name:cuda suffix:-cuda]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/amd64 runner:ubuntu-latest], map[build_args:USE_OLLAMA=true free_disk:false name:ollama suffix:-ollama]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/amd64 runner:ubuntu-latest], map[build_args:USE_SLIM=true free_disk:false name:slim suffix:-slim]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/arm64 runner:ubuntu-24.04-arm], map[build_args: free_disk:false name:main suffix:]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/arm64 runner:ubuntu-24.04-arm], map[build_args:USE_CUDA=true
USE_CUDA_VER=cu126
free_disk:true name:cuda126 suffix:-cuda126]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/arm64 runner:ubuntu-24.04-arm], map[build_args:USE_CUDA=true free_disk:true name:cuda suffix:-cuda]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/arm64 runner:ubuntu-24.04-arm], map[build_args:USE_OLLAMA=true free_disk:false name:ollama suffix:-ollama]) (push) Canceled after 0s
Create and publish Docker images with specific build args / build (map[arch:linux/arm64 runner:ubuntu-24.04-arm], map[build_args:USE_SLIM=true free_disk:false name:slim suffix:-slim]) (push) Canceled after 0s
Frontend Build / Format & Build (push) Canceled after 0s
Frontend Build / Unit Tests (push) Canceled after 0s
Release to PyPI / release (push) Canceled after 0s
Release / publish (push) Canceled after 0s
Create and publish Docker images with specific build args / merge (map[name:cuda suffix:-cuda]) (push) Canceled after 0s
Create and publish Docker images with specific build args / merge (map[name:cuda126 suffix:-cuda126]) (push) Canceled after 0s
Create and publish Docker images with specific build args / merge (map[name:main suffix:]) (push) Canceled after 0s
Create and publish Docker images with specific build args / merge (map[name:ollama suffix:-ollama]) (push) Canceled after 0s
Create and publish Docker images with specific build args / merge (map[name:slim suffix:-slim]) (push) Canceled after 0s
Create and publish Docker images with specific build args / notify-helm-charts (push) Canceled after 0s
Create and publish Docker images with specific build args / copy-to-dockerhub (, main) (push) Canceled after 0s
Create and publish Docker images with specific build args / copy-to-dockerhub (-cuda, cuda) (push) Canceled after 0s
Create and publish Docker images with specific build args / copy-to-dockerhub (-cuda126, cuda126) (push) Canceled after 0s
Create and publish Docker images with specific build args / copy-to-dockerhub (-ollama, ollama) (push) Canceled after 0s
Create and publish Docker images with specific build args / copy-to-dockerhub (-slim, slim) (push) Canceled after 0s
Wires the custom relaxed-acceptance llama-server (see ../mtp-relaxed-decoding) into the stack as a new llama-mtp service, registered as an additional OpenAI-compatible connection alongside Ollama. ollama-auth now proxies to llama-mtp instead of the now-empty Ollama, so LAN clients (e.g. Home Assistant's voice pipeline) keep working against the same URL/token with no reconfiguration. Also merges llama.cpp's `timings` extension (dropped by the generic OpenAI response schema) into message.usage and surfaces it as a labeled tok/s badge, so relaxed-MTP responses get the same visible generation-speed info Ollama responses already had. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,59 @@
|
|||||||
|
services:
|
||||||
|
llama-mtp:
|
||||||
|
build:
|
||||||
|
context: ../mtp-relaxed-decoding
|
||||||
|
container_name: llama-mtp
|
||||||
|
restart: unless-stopped
|
||||||
|
ports:
|
||||||
|
- '8030:8030'
|
||||||
|
volumes:
|
||||||
|
# Same host dir ollama's /gguf-import mount points at -- target +
|
||||||
|
# MTP drafter already live there.
|
||||||
|
- /home/brian-llm/models/gguf:/models:ro
|
||||||
|
environment:
|
||||||
|
- 'MTP_AUTH_TOKEN=${MTP_AUTH_TOKEN}'
|
||||||
|
- 'MTP_TARGET_GGUF=/models/gemma-4-E2B-it-Q4_K_M.gguf'
|
||||||
|
- 'MTP_DRAFT_GGUF=/models/mtp-gemma-4-E2B-it.gguf'
|
||||||
|
# n_max beyond ~2 wasn't worth it for this drafter, and relaxed_top_n
|
||||||
|
# 2-20 all performed similarly -- see ../mtp-relaxed-decoding/README.md
|
||||||
|
# §6. Benchmarked on an RTX 5090 there; re-check on this Titan Xp
|
||||||
|
# before trusting these as tuned rather than just reasonable defaults.
|
||||||
|
- 'MTP_N_MAX=2'
|
||||||
|
- 'MTP_RELAXED_TOP_N=5'
|
||||||
|
# Titan Xp has 12GB VRAM shared with asr/tts/ollama -- batch size kept
|
||||||
|
# below the README's RTX-5090 sizing (1024) for headroom, but context
|
||||||
|
# was raised to match the README since Gemma-4's hybrid SWA
|
||||||
|
# architecture (most layers use a small fixed attention window, only
|
||||||
|
# every 5th layer is full-context) keeps KV cache growth with n_ctx
|
||||||
|
# much cheaper than a plain transformer's.
|
||||||
|
- 'MTP_CTX_SIZE=32768'
|
||||||
|
- 'MTP_BATCH_SIZE=512'
|
||||||
|
# Server-side default -- only applies when the client doesn't send its
|
||||||
|
# own temperature (Open WebUI won't unless you set it in the model's
|
||||||
|
# Advanced Params).
|
||||||
|
- 'MTP_TEMP=0.6'
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
reservations:
|
||||||
|
devices:
|
||||||
|
- driver: nvidia
|
||||||
|
count: 1
|
||||||
|
capabilities: [gpu]
|
||||||
|
|
||||||
|
open-webui:
|
||||||
|
depends_on:
|
||||||
|
- llama-mtp
|
||||||
|
environment:
|
||||||
|
# Registers this alongside (not instead of) the Ollama connection --
|
||||||
|
# Open WebUI lists both in the model picker.
|
||||||
|
- 'OPENAI_API_BASE_URLS=http://llama-mtp:8030/v1'
|
||||||
|
- 'OPENAI_API_KEYS=${MTP_AUTH_TOKEN}'
|
||||||
|
|
||||||
|
ollama-auth:
|
||||||
|
depends_on:
|
||||||
|
- llama-mtp
|
||||||
|
environment:
|
||||||
|
# ollama-auth/default.conf.template now proxies to llama-mtp (Ollama
|
||||||
|
# itself serves no models anymore) and swaps the client's
|
||||||
|
# OLLAMA_AUTH_TOKEN for this before forwarding upstream.
|
||||||
|
- 'MTP_AUTH_TOKEN=${MTP_AUTH_TOKEN}'
|
||||||
+6
-1
@@ -66,7 +66,12 @@ services:
|
|||||||
environment:
|
environment:
|
||||||
- 'OLLAMA_BASE_URL=http://ollama:11434'
|
- 'OLLAMA_BASE_URL=http://ollama:11434'
|
||||||
- 'WEBUI_SECRET_KEY='
|
- 'WEBUI_SECRET_KEY='
|
||||||
- 'DEFAULT_MODELS=gemma-4-e2b-mtp'
|
# Only takes effect on a fresh DB -- once ui.default_models exists in
|
||||||
|
# the config table, that value wins (see open-webui's PersistentConfig
|
||||||
|
# seed_defaults behavior). Currently gemma-4-e2b-mtp was removed from
|
||||||
|
# Ollama in favor of the relaxed-MTP llama-mtp service; the live value
|
||||||
|
# is 'gemma-4-e2b,/models/gemma-4-E2B-it-Q4_K_M.gguf'.
|
||||||
|
- 'DEFAULT_MODELS=gemma-4-e2b,/models/gemma-4-E2B-it-Q4_K_M.gguf'
|
||||||
- 'ENABLE_OAUTH_SIGNUP=true'
|
- 'ENABLE_OAUTH_SIGNUP=true'
|
||||||
# Links Authentik login to the existing local account with the same
|
# Links Authentik login to the existing local account with the same
|
||||||
# email instead of rejecting it as a collision -- safe here since this
|
# email instead of rejecting it as a collision -- safe here since this
|
||||||
|
|||||||
+1
-1
@@ -5,7 +5,7 @@ cd "$(dirname "${BASH_SOURCE[0]}")"
|
|||||||
# Wraps docker compose with the full file set for this stack (base + GPU
|
# Wraps docker compose with the full file set for this stack (base + GPU
|
||||||
# passthrough + audio/ollama-auth services) so a plain "up"/"build" can't
|
# passthrough + audio/ollama-auth services) so a plain "up"/"build" can't
|
||||||
# accidentally drop the ollama GPU override again.
|
# accidentally drop the ollama GPU override again.
|
||||||
readonly COMPOSE_FILES=(-f docker-compose.yaml -f docker-compose.gpu.yaml -f docker-compose.audio.yaml)
|
readonly COMPOSE_FILES=(-f docker-compose.yaml -f docker-compose.gpu.yaml -f docker-compose.audio.yaml -f docker-compose.mtp.yaml)
|
||||||
|
|
||||||
if [[ $# -eq 0 ]]; then
|
if [[ $# -eq 0 ]]; then
|
||||||
exec docker compose "${COMPOSE_FILES[@]}" up -d --build
|
exec docker compose "${COMPOSE_FILES[@]}" up -d --build
|
||||||
|
|||||||
@@ -13,18 +13,24 @@ server {
|
|||||||
return 401;
|
return 401;
|
||||||
}
|
}
|
||||||
|
|
||||||
# Docker's embedded DNS (127.0.0.11) can reassign the "ollama"
|
# Ollama itself now serves no models -- this proxy targets
|
||||||
# hostname to a new container IP on restart. proxy_pass to a
|
# llama-mtp (the relaxed-MTP server) instead, so LAN clients that
|
||||||
# literal upstream caches that IP for the container's lifetime;
|
# already trust this URL/token (e.g. Home Assistant's voice
|
||||||
# routing through a variable forces nginx to re-resolve via this
|
# pipeline) don't need reconfiguring. Docker's embedded DNS
|
||||||
# resolver on each request instead, so a restarted ollama doesn't
|
# (127.0.0.11) can reassign a container hostname to a new IP on
|
||||||
# need a matching ollama-auth restart to be reachable again.
|
# restart; proxy_pass to a literal upstream caches that IP for the
|
||||||
|
# container's lifetime, so route through a variable to force
|
||||||
|
# nginx to re-resolve via this resolver on each request instead.
|
||||||
resolver 127.0.0.11 valid=10s;
|
resolver 127.0.0.11 valid=10s;
|
||||||
set $ollama_upstream ollama;
|
set $mtp_upstream llama-mtp;
|
||||||
proxy_pass http://$ollama_upstream:11434;
|
proxy_pass http://$mtp_upstream:8030;
|
||||||
proxy_http_version 1.1;
|
proxy_http_version 1.1;
|
||||||
proxy_set_header Connection "";
|
proxy_set_header Connection "";
|
||||||
proxy_set_header Host $host;
|
proxy_set_header Host $host;
|
||||||
|
# Client presents OLLAMA_AUTH_TOKEN (validated above); llama-server
|
||||||
|
# behind this proxy expects its own MTP_AUTH_TOKEN instead, so swap
|
||||||
|
# it here rather than requiring every client to be reconfigured.
|
||||||
|
proxy_set_header Authorization "Bearer ${MTP_AUTH_TOKEN}";
|
||||||
proxy_buffering off;
|
proxy_buffering off;
|
||||||
proxy_read_timeout 600s;
|
proxy_read_timeout 600s;
|
||||||
proxy_send_timeout 600s;
|
proxy_send_timeout 600s;
|
||||||
|
|||||||
@@ -2744,7 +2744,8 @@
|
|||||||
};
|
};
|
||||||
|
|
||||||
const chatCompletionEventHandler = async (data, message, chatId) => {
|
const chatCompletionEventHandler = async (data, message, chatId) => {
|
||||||
const { id, done, choices, content, output, sources, selected_model_id, error, usage } = data;
|
const { id, done, choices, content, output, sources, selected_model_id, error, usage, timings } =
|
||||||
|
data;
|
||||||
|
|
||||||
// Store raw OR-aligned output items from backend
|
// Store raw OR-aligned output items from backend
|
||||||
if (output) {
|
if (output) {
|
||||||
@@ -2797,8 +2798,12 @@
|
|||||||
message.arena = true;
|
message.arena = true;
|
||||||
}
|
}
|
||||||
|
|
||||||
if (usage) {
|
if (usage || timings) {
|
||||||
message.usage = usage;
|
// llama.cpp-compatible servers (e.g. our relaxed-MTP llama-server) send
|
||||||
|
// generation speed as a sibling `timings` object rather than inside
|
||||||
|
// `usage` -- merge it in so it surfaces in the same response-info
|
||||||
|
// tooltip Ollama's eval_count/eval_duration fields already populate.
|
||||||
|
message.usage = { ...(usage || {}), ...(timings || {}) };
|
||||||
}
|
}
|
||||||
|
|
||||||
history.messages[message.id] = message;
|
history.messages[message.id] = message;
|
||||||
|
|||||||
@@ -144,6 +144,23 @@
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
// Generation speed, normalized across backends: Ollama reports
|
||||||
|
// eval_count/eval_duration (nanoseconds), llama.cpp-compatible servers
|
||||||
|
// (e.g. our relaxed-MTP llama-server) report timings.predicted_per_second
|
||||||
|
// directly -- note that field name is llama.cpp's generic term for "the
|
||||||
|
// main generation loop's output," i.e. the actual response stream, NOT
|
||||||
|
// the MTP drafter in isolation (which is separately broken out as
|
||||||
|
// draft_n/draft_n_accepted).
|
||||||
|
const getTokensPerSecond = (usage: Record<string, unknown> | undefined | null) => {
|
||||||
|
if (!usage) return null;
|
||||||
|
if (typeof usage.predicted_per_second === 'number') return usage.predicted_per_second;
|
||||||
|
if (typeof usage.eval_count === 'number' && typeof usage.eval_duration === 'number' && usage.eval_duration > 0) {
|
||||||
|
return (usage.eval_count / usage.eval_duration) * 1e9;
|
||||||
|
}
|
||||||
|
return null;
|
||||||
|
};
|
||||||
|
$: tokensPerSecond = getTokensPerSecond(message.usage);
|
||||||
|
|
||||||
export let siblings;
|
export let siblings;
|
||||||
|
|
||||||
export let setInputText: Function = () => {};
|
export let setInputText: Function = () => {};
|
||||||
@@ -1189,6 +1206,14 @@
|
|||||||
</Tooltip>
|
</Tooltip>
|
||||||
{/if}
|
{/if}
|
||||||
|
|
||||||
|
{#if tokensPerSecond}
|
||||||
|
<span
|
||||||
|
class="text-xs text-gray-400 dark:text-gray-500 px-1 self-center whitespace-nowrap"
|
||||||
|
>
|
||||||
|
{tokensPerSecond.toFixed(1)} tok/s
|
||||||
|
</span>
|
||||||
|
{/if}
|
||||||
|
|
||||||
{#if message.usage}
|
{#if message.usage}
|
||||||
<Tooltip
|
<Tooltip
|
||||||
content={message.usage
|
content={message.usage
|
||||||
|
|||||||
Reference in New Issue
Block a user