Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
966140c3ef | ||
|
|
6f583deff9 | ||
|
|
7eff527d16 |
+1
-1
@@ -28,7 +28,7 @@ FROM --platform=$BUILDPLATFORM node:22-alpine3.20 AS build
|
||||
ARG BUILD_HASH
|
||||
|
||||
# Set Node.js options (heap limit Allocation failed - JavaScript heap out of memory)
|
||||
# ENV NODE_OPTIONS="--max-old-space-size=4096"
|
||||
ENV NODE_OPTIONS="--max-old-space-size=8192"
|
||||
|
||||
WORKDIR /app
|
||||
|
||||
|
||||
+22
-13
@@ -1,5 +1,6 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
cd "$(dirname "${BASH_SOURCE[0]}")"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Interactive docker compose launcher for Open WebUI.
|
||||
@@ -100,38 +101,46 @@ done
|
||||
# ── Drop mode ─────────────────────────────────────────────────────────────────
|
||||
|
||||
if [[ "$drop_project" == true ]]; then
|
||||
docker compose down --remove-orphans
|
||||
./docker-stack.sh down --remove-orphans
|
||||
echo -e "${GREEN}${BOLD}Compose project stopped and cleaned up.${RESET}"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# ── Build compose command ─────────────────────────────────────────────────────
|
||||
#
|
||||
# The base + GPU + audio file set is owned by docker-stack.sh (the single
|
||||
# source of truth for "what this stack always needs"), so it can't drift out
|
||||
# of sync between the two entry points the way docker-compose.audio.yaml once
|
||||
# did here. This script only ever layers its own optional extras (api/data/
|
||||
# playwright) on top and delegates the actual `docker compose` call.
|
||||
|
||||
compose_files=("-f" "docker-compose.yaml")
|
||||
|
||||
if [[ "$enable_gpu" == true ]]; then
|
||||
if ! [[ "$gpu_count" =~ ^([0-9]+|all)$ ]]; then
|
||||
echo >&2 "Error: Invalid GPU count '${gpu_count}'. Must be a number or 'all'."
|
||||
exit 1
|
||||
fi
|
||||
if [[ "$gpu_count" =~ ^([0-9]+|all)$ ]]; then
|
||||
export OLLAMA_GPU_DRIVER
|
||||
OLLAMA_GPU_DRIVER=$(detect_gpu_driver)
|
||||
export OLLAMA_GPU_COUNT="$gpu_count"
|
||||
compose_files+=("-f" "docker-compose.gpu.yaml")
|
||||
elif [[ "$enable_gpu" == true ]]; then
|
||||
echo >&2 "Error: Invalid GPU count '${gpu_count}'. Must be a number or 'all'."
|
||||
exit 1
|
||||
fi
|
||||
# docker-stack.sh always merges docker-compose.gpu.yaml (defaults: nvidia,
|
||||
# count 1, from that file's own ${VAR-default} fallbacks) -- GPU passthrough
|
||||
# is no longer optional here, matching docker-stack.sh's behavior. --enable-gpu
|
||||
# now only customizes the driver/count via the env vars set above.
|
||||
|
||||
extra_files=()
|
||||
|
||||
if [[ "$enable_api" == true ]]; then
|
||||
export OLLAMA_WEBAPI_PORT="$api_port"
|
||||
compose_files+=("-f" "docker-compose.api.yaml")
|
||||
extra_files+=("-f" "docker-compose.api.yaml")
|
||||
fi
|
||||
|
||||
if [[ -n "$data_dir" ]]; then
|
||||
export OLLAMA_DATA_DIR="$data_dir"
|
||||
compose_files+=("-f" "docker-compose.data.yaml")
|
||||
extra_files+=("-f" "docker-compose.data.yaml")
|
||||
fi
|
||||
|
||||
if [[ "$enable_playwright" == true ]]; then
|
||||
compose_files+=("-f" "docker-compose.playwright.yaml")
|
||||
extra_files+=("-f" "docker-compose.playwright.yaml")
|
||||
fi
|
||||
|
||||
export OPEN_WEBUI_PORT="$webui_port"
|
||||
@@ -163,7 +172,7 @@ fi
|
||||
|
||||
# ── Launch ────────────────────────────────────────────────────────────────────
|
||||
|
||||
if docker compose "${compose_files[@]}" "${up_args[@]}"; then
|
||||
if ./docker-stack.sh "${extra_files[@]}" "${up_args[@]}"; then
|
||||
echo
|
||||
echo -e "${GREEN}${BOLD}${CHECK_MARK} Compose project started successfully.${RESET}"
|
||||
else
|
||||
|
||||
@@ -0,0 +1,85 @@
|
||||
services:
|
||||
asr:
|
||||
build:
|
||||
context: ../asr-server
|
||||
container_name: asr
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- '8020:8020'
|
||||
environment:
|
||||
- 'ASR_AUTH_TOKEN=${ASR_AUTH_TOKEN}'
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: 1
|
||||
capabilities: [gpu]
|
||||
|
||||
tts:
|
||||
build:
|
||||
context: ../tts-server
|
||||
container_name: tts
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- '8010:8010'
|
||||
environment:
|
||||
- 'TTS_AUTH_TOKEN=${TTS_AUTH_TOKEN}'
|
||||
- 'HF_HOME=/app/data'
|
||||
# Fallback voice for direct /v1/audio/speech calls that use
|
||||
# AUDIO_TTS_MODEL=sopro-v2-turbo without a `voice` field. Set to the
|
||||
# basename (no .wav) of a clip dropped into services/tts-server/voices/.
|
||||
# - 'TTS_SOPRO_DEFAULT_VOICE=your_voice_name'
|
||||
volumes:
|
||||
# Caches the downloaded Kokoro + Sopro weights across rebuilds.
|
||||
- tts-data:/app/data
|
||||
# Sopro voice reference clips (<name>.wav); drop files in here on the
|
||||
# host and reference them by name via AUDIO_TTS_VOICE / `voice`.
|
||||
- ../tts-server/voices:/app/voices:ro
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: 1
|
||||
capabilities: [gpu]
|
||||
|
||||
open-webui:
|
||||
depends_on:
|
||||
- asr
|
||||
- tts
|
||||
environment:
|
||||
- 'AUDIO_STT_ENGINE=openai'
|
||||
- 'AUDIO_STT_OPENAI_API_BASE_URL=http://asr:8020/v1'
|
||||
- 'AUDIO_STT_OPENAI_API_KEY=${ASR_AUTH_TOKEN}'
|
||||
- 'AUDIO_STT_MODEL=nemotron-3.5-asr-streaming-0.6b'
|
||||
- 'AUDIO_TTS_ENGINE=openai'
|
||||
- 'AUDIO_TTS_OPENAI_API_BASE_URL=http://tts:8010/v1'
|
||||
- 'AUDIO_TTS_OPENAI_API_KEY=${TTS_AUTH_TOKEN}'
|
||||
# To switch back to Kokoro (fixed presets), set:
|
||||
# AUDIO_TTS_MODEL=kokoro-82m
|
||||
# AUDIO_TTS_VOICE=af_heart
|
||||
- 'AUDIO_TTS_MODEL=sopro-v2-turbo'
|
||||
- 'AUDIO_TTS_VOICE=jarvis-target'
|
||||
|
||||
wyoming-openai:
|
||||
image: ghcr.io/roryeckel/wyoming_openai:latest
|
||||
container_name: wyoming-openai
|
||||
restart: unless-stopped
|
||||
depends_on:
|
||||
- asr
|
||||
- tts
|
||||
ports:
|
||||
- '10300:10300'
|
||||
environment:
|
||||
- 'STT_OPENAI_URL=http://asr:8020/v1'
|
||||
- 'STT_OPENAI_KEY=${ASR_AUTH_TOKEN}'
|
||||
- 'STT_MODELS=nemotron-3.5-asr-streaming-0.6b'
|
||||
- 'TTS_OPENAI_URL=http://tts:8010/v1'
|
||||
- 'TTS_OPENAI_KEY=${TTS_AUTH_TOKEN}'
|
||||
- 'TTS_MODELS=sopro-v2-turbo'
|
||||
- 'TTS_VOICES=jarvis-target'
|
||||
- 'TTS_SPEED=1.2'
|
||||
|
||||
volumes:
|
||||
tts-data: {}
|
||||
@@ -0,0 +1,72 @@
|
||||
services:
|
||||
llama-mtp:
|
||||
build:
|
||||
context: ../mtp-relaxed-decoding
|
||||
container_name: llama-mtp
|
||||
restart: unless-stopped
|
||||
ports:
|
||||
- '8030:8030'
|
||||
volumes:
|
||||
# Same host dir ollama's /gguf-import mount points at -- target +
|
||||
# MTP drafter already live there.
|
||||
- /home/brian-llm/models/gguf:/models:ro
|
||||
environment:
|
||||
- 'MTP_AUTH_TOKEN=${MTP_AUTH_TOKEN}'
|
||||
- 'MTP_TARGET_GGUF=/models/gemma-4-E2B-it-Q4_K_M.gguf'
|
||||
- 'MTP_DRAFT_GGUF=/models/mtp-gemma-4-E2B-it.gguf'
|
||||
# n_max beyond ~2 wasn't worth it for this drafter, and relaxed_top_n
|
||||
# 2-20 all performed similarly -- see ../mtp-relaxed-decoding/README.md
|
||||
# §6. Benchmarked on an RTX 5090 there; re-check on this Titan Xp
|
||||
# before trusting these as tuned rather than just reasonable defaults.
|
||||
- 'MTP_N_MAX=2'
|
||||
- 'MTP_RELAXED_TOP_N=5'
|
||||
# Titan Xp has 12GB VRAM shared with asr/tts/ollama -- batch size kept
|
||||
# below the README's RTX-5090 sizing (1024) for headroom, but context
|
||||
# was raised to match the README since Gemma-4's hybrid SWA
|
||||
# architecture (most layers use a small fixed attention window, only
|
||||
# every 5th layer is full-context) keeps KV cache growth with n_ctx
|
||||
# much cheaper than a plain transformer's -- measured at only 271 MiB
|
||||
# of KV cache (target + draft) per 32768-token slot.
|
||||
#
|
||||
# -np 4 gives open-webui's browser chat and the voice assistant (and
|
||||
# anything else hitting this OpenAI-compatible endpoint) their own
|
||||
# slot each instead of one contending for a single shared slot --
|
||||
# with -np 1, whichever of them touched the slot last would evict the
|
||||
# other's live KV cache, forcing a full prompt reprocess (multi-second
|
||||
# stall) on the next request from the loser. n_ctx_slot = n_ctx /
|
||||
# n_parallel, so total ctx is quadrupled to keep each slot at the same
|
||||
# 32768 budget as before -- 131072 matches this model's native
|
||||
# n_ctx_train exactly, and only costs ~810 MiB more KV cache VRAM
|
||||
# than the old -np 1 config (still well under the ~7GB that's free).
|
||||
- 'MTP_CTX_SIZE=131072'
|
||||
- 'MTP_N_PARALLEL=4'
|
||||
- 'MTP_BATCH_SIZE=512'
|
||||
# Server-side default -- only applies when the client doesn't send its
|
||||
# own temperature (Open WebUI won't unless you set it in the model's
|
||||
# Advanced Params).
|
||||
- 'MTP_TEMP=0.6'
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: 1
|
||||
capabilities: [gpu]
|
||||
|
||||
open-webui:
|
||||
depends_on:
|
||||
- llama-mtp
|
||||
environment:
|
||||
# Registers this alongside (not instead of) the Ollama connection --
|
||||
# Open WebUI lists both in the model picker.
|
||||
- 'OPENAI_API_BASE_URLS=http://llama-mtp:8030/v1'
|
||||
- 'OPENAI_API_KEYS=${MTP_AUTH_TOKEN}'
|
||||
|
||||
ollama-auth:
|
||||
depends_on:
|
||||
- llama-mtp
|
||||
environment:
|
||||
# ollama-auth/default.conf.template now proxies to llama-mtp (Ollama
|
||||
# itself serves no models anymore) and swaps the client's
|
||||
# OLLAMA_AUTH_TOKEN for this before forwarding upstream.
|
||||
- 'MTP_AUTH_TOKEN=${MTP_AUTH_TOKEN}'
|
||||
+89
-1
@@ -2,11 +2,49 @@ services:
|
||||
ollama:
|
||||
volumes:
|
||||
- ollama:/root/.ollama
|
||||
# Read-only mount for importing custom GGUF files (e.g. via `ollama create`)
|
||||
- /home/brian-llm/models/gguf:/gguf-import:ro
|
||||
container_name: ollama
|
||||
pull_policy: always
|
||||
tty: true
|
||||
restart: unless-stopped
|
||||
image: ollama/ollama:${OLLAMA_DOCKER_TAG-latest}
|
||||
environment:
|
||||
# Never unload a loaded model on idle -- default is 5 minutes.
|
||||
- 'OLLAMA_KEEP_ALIVE=-1'
|
||||
- 'OLLAMA_NUM_PARALLEL=4'
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
count: 1
|
||||
capabilities: [gpu]
|
||||
# No ports published here -- Ollama has no auth of its own, so LAN access
|
||||
# goes through the ollama-auth proxy below instead. open-webui still
|
||||
# reaches it directly over the internal docker network (see
|
||||
# OLLAMA_BASE_URL), which doesn't need the token.
|
||||
|
||||
ollama-auth:
|
||||
image: nginx:alpine
|
||||
container_name: ollama-auth
|
||||
restart: unless-stopped
|
||||
depends_on:
|
||||
- ollama
|
||||
volumes:
|
||||
- ./ollama-auth/default.conf.template:/etc/nginx/templates/default.conf.template:ro
|
||||
environment:
|
||||
- 'OLLAMA_AUTH_TOKEN=${OLLAMA_AUTH_TOKEN}'
|
||||
ports:
|
||||
# Reachable from the LAN directly; restrict to the LAN subnet with the
|
||||
# host firewall (see ufw rules) as defense in depth, but the real
|
||||
# boundary is now the Bearer token check in ollama-auth/default.conf.template.
|
||||
# This is plain HTTP, so the token is readable by anything already on
|
||||
# the LAN segment that can sniff traffic -- accepted here since the
|
||||
# trust boundary is "the LAN subnet," not "the wire." Revisit with TLS
|
||||
# if that assumption ever stops holding (untrusted/guest/IoT devices
|
||||
# sharing this LAN, etc).
|
||||
- '11434:11434'
|
||||
|
||||
open-webui:
|
||||
build:
|
||||
@@ -19,14 +57,64 @@ services:
|
||||
depends_on:
|
||||
- ollama
|
||||
ports:
|
||||
- ${OPEN_WEBUI_PORT-3000}:8080
|
||||
# Published on all interfaces since nginx runs on a separate LAN host
|
||||
# (192.168.50.224) and must reach this port directly. Firewalling this
|
||||
# to that host is still good hygiene (see ufw/DOCKER-USER notes) but is
|
||||
# no longer the sole security boundary -- OIDC auth below is real
|
||||
# authentication against Authentik regardless of network path.
|
||||
- '${OPEN_WEBUI_PORT-3000}:8080'
|
||||
environment:
|
||||
- 'OLLAMA_BASE_URL=http://ollama:11434'
|
||||
- 'WEBUI_SECRET_KEY='
|
||||
# Only takes effect on a fresh DB -- once ui.default_models exists in
|
||||
# the config table, that value wins (see open-webui's PersistentConfig
|
||||
# seed_defaults behavior). Currently gemma-4-e2b-mtp was removed from
|
||||
# Ollama in favor of the relaxed-MTP llama-mtp service; the live value
|
||||
# is 'gemma-4-e2b,/models/gemma-4-E2B-it-Q4_K_M.gguf'.
|
||||
- 'DEFAULT_MODELS=gemma-4-e2b,/models/gemma-4-E2B-it-Q4_K_M.gguf'
|
||||
- 'ENABLE_OAUTH_SIGNUP=true'
|
||||
# Links Authentik login to the existing local account with the same
|
||||
# email instead of rejecting it as a collision -- safe here since this
|
||||
# is a single-admin personal instance, not an open-signup multi-tenant one.
|
||||
- 'OAUTH_MERGE_ACCOUNTS_BY_EMAIL=true'
|
||||
# Fill these in after creating the OAuth2/OpenID Provider + Application
|
||||
# in Authentik (see instructions) -- OAUTH_CLIENT_ID/SECRET come from
|
||||
# that provider.
|
||||
- 'OAUTH_CLIENT_ID=${OAUTH_CLIENT_ID}'
|
||||
- 'OAUTH_CLIENT_SECRET=${OAUTH_CLIENT_SECRET}'
|
||||
- 'OPENID_PROVIDER_URL=https://auth.hetherman.cloud/application/o/ai-open-webui/.well-known/openid-configuration'
|
||||
- 'OPENID_REDIRECT_URI=https://ai.hetherman.cloud/oauth/oidc/callback'
|
||||
- 'OAUTH_SCOPES=openid email profile'
|
||||
# SSO-only: Authentik becomes the sole way to authenticate, admin
|
||||
# included. If Authentik is ever unreachable, nobody can log in until
|
||||
# it's fixed -- no local password fallback.
|
||||
- 'ENABLE_PASSWORD_AUTH=false'
|
||||
- 'ENABLE_LOGIN_FORM=false'
|
||||
- 'ENABLE_SIGNUP=false'
|
||||
extra_hosts:
|
||||
- host.docker.internal:host-gateway
|
||||
restart: unless-stopped
|
||||
|
||||
# ollama-keepwarm:
|
||||
# image: curlimages/curl:latest
|
||||
# container_name: ollama-keepwarm
|
||||
# restart: unless-stopped
|
||||
# depends_on:
|
||||
# - ollama
|
||||
# # Pings Ollama every 60s to load the default model and refresh its
|
||||
# # keep_alive timer -- covers first boot and re-loads it any time the
|
||||
# # ollama container itself restarts (which flushes it from memory).
|
||||
# # Keep this model name in sync with DEFAULT_MODELS above.
|
||||
# entrypoint:
|
||||
# - sh
|
||||
# - -c
|
||||
# - |
|
||||
# while true; do
|
||||
# curl -s -o /dev/null -X POST http://ollama:11434/api/generate \
|
||||
# -d '{"model":"gemma-4-e2b","keep_alive":-1}'
|
||||
# sleep 60
|
||||
# done
|
||||
|
||||
volumes:
|
||||
ollama: {}
|
||||
open-webui: {}
|
||||
|
||||
Executable
+14
@@ -0,0 +1,14 @@
|
||||
#!/usr/bin/env bash
|
||||
set -euo pipefail
|
||||
cd "$(dirname "${BASH_SOURCE[0]}")"
|
||||
|
||||
# Wraps docker compose with the full file set for this stack (base + GPU
|
||||
# passthrough + audio/ollama-auth services) so a plain "up"/"build" can't
|
||||
# accidentally drop the ollama GPU override again.
|
||||
readonly COMPOSE_FILES=(-f docker-compose.yaml -f docker-compose.gpu.yaml -f docker-compose.audio.yaml -f docker-compose.mtp.yaml)
|
||||
|
||||
if [[ $# -eq 0 ]]; then
|
||||
exec docker compose "${COMPOSE_FILES[@]}" up -d --build
|
||||
fi
|
||||
|
||||
exec docker compose "${COMPOSE_FILES[@]}" "$@"
|
||||
@@ -0,0 +1,39 @@
|
||||
map_hash_bucket_size 128;
|
||||
|
||||
map $http_authorization $ollama_auth_ok {
|
||||
"Bearer ${OLLAMA_AUTH_TOKEN}" 1;
|
||||
default 0;
|
||||
}
|
||||
|
||||
server {
|
||||
listen 11434;
|
||||
|
||||
location / {
|
||||
if ($ollama_auth_ok = 0) {
|
||||
return 401;
|
||||
}
|
||||
|
||||
# Ollama itself now serves no models -- this proxy targets
|
||||
# llama-mtp (the relaxed-MTP server) instead, so LAN clients that
|
||||
# already trust this URL/token (e.g. Home Assistant's voice
|
||||
# pipeline) don't need reconfiguring. Docker's embedded DNS
|
||||
# (127.0.0.11) can reassign a container hostname to a new IP on
|
||||
# restart; proxy_pass to a literal upstream caches that IP for the
|
||||
# container's lifetime, so route through a variable to force
|
||||
# nginx to re-resolve via this resolver on each request instead.
|
||||
resolver 127.0.0.11 valid=10s;
|
||||
set $mtp_upstream llama-mtp;
|
||||
proxy_pass http://$mtp_upstream:8030;
|
||||
proxy_http_version 1.1;
|
||||
proxy_set_header Connection "";
|
||||
proxy_set_header Host $host;
|
||||
# Client presents OLLAMA_AUTH_TOKEN (validated above); llama-server
|
||||
# behind this proxy expects its own MTP_AUTH_TOKEN instead, so swap
|
||||
# it here rather than requiring every client to be reconfigured.
|
||||
proxy_set_header Authorization "Bearer ${MTP_AUTH_TOKEN}";
|
||||
proxy_buffering off;
|
||||
proxy_read_timeout 600s;
|
||||
proxy_send_timeout 600s;
|
||||
client_max_body_size 0;
|
||||
}
|
||||
}
|
||||
@@ -2744,7 +2744,8 @@
|
||||
};
|
||||
|
||||
const chatCompletionEventHandler = async (data, message, chatId) => {
|
||||
const { id, done, choices, content, output, sources, selected_model_id, error, usage } = data;
|
||||
const { id, done, choices, content, output, sources, selected_model_id, error, usage, timings } =
|
||||
data;
|
||||
|
||||
// Store raw OR-aligned output items from backend
|
||||
if (output) {
|
||||
@@ -2797,8 +2798,12 @@
|
||||
message.arena = true;
|
||||
}
|
||||
|
||||
if (usage) {
|
||||
message.usage = usage;
|
||||
if (usage || timings) {
|
||||
// llama.cpp-compatible servers (e.g. our relaxed-MTP llama-server) send
|
||||
// generation speed as a sibling `timings` object rather than inside
|
||||
// `usage` -- merge it in so it surfaces in the same response-info
|
||||
// tooltip Ollama's eval_count/eval_duration fields already populate.
|
||||
message.usage = { ...(usage || {}), ...(timings || {}) };
|
||||
}
|
||||
|
||||
history.messages[message.id] = message;
|
||||
|
||||
@@ -144,6 +144,23 @@
|
||||
}
|
||||
}
|
||||
|
||||
// Generation speed, normalized across backends: Ollama reports
|
||||
// eval_count/eval_duration (nanoseconds), llama.cpp-compatible servers
|
||||
// (e.g. our relaxed-MTP llama-server) report timings.predicted_per_second
|
||||
// directly -- note that field name is llama.cpp's generic term for "the
|
||||
// main generation loop's output," i.e. the actual response stream, NOT
|
||||
// the MTP drafter in isolation (which is separately broken out as
|
||||
// draft_n/draft_n_accepted).
|
||||
const getTokensPerSecond = (usage: Record<string, unknown> | undefined | null) => {
|
||||
if (!usage) return null;
|
||||
if (typeof usage.predicted_per_second === 'number') return usage.predicted_per_second;
|
||||
if (typeof usage.eval_count === 'number' && typeof usage.eval_duration === 'number' && usage.eval_duration > 0) {
|
||||
return (usage.eval_count / usage.eval_duration) * 1e9;
|
||||
}
|
||||
return null;
|
||||
};
|
||||
$: tokensPerSecond = getTokensPerSecond(message.usage);
|
||||
|
||||
export let siblings;
|
||||
|
||||
export let setInputText: Function = () => {};
|
||||
@@ -1189,6 +1206,14 @@
|
||||
</Tooltip>
|
||||
{/if}
|
||||
|
||||
{#if tokensPerSecond}
|
||||
<span
|
||||
class="text-xs text-gray-400 dark:text-gray-500 px-1 self-center whitespace-nowrap"
|
||||
>
|
||||
{tokensPerSecond.toFixed(1)} tok/s
|
||||
</span>
|
||||
{/if}
|
||||
|
||||
{#if message.usage}
|
||||
<Tooltip
|
||||
content={message.usage
|
||||
|
||||
Reference in New Issue
Block a user