diff --git a/Dockerfile b/Dockerfile index d70244583b..0c28709574 100644 --- a/Dockerfile +++ b/Dockerfile @@ -28,7 +28,7 @@ FROM --platform=$BUILDPLATFORM node:22-alpine3.20 AS build ARG BUILD_HASH # Set Node.js options (heap limit Allocation failed - JavaScript heap out of memory) -# ENV NODE_OPTIONS="--max-old-space-size=4096" +ENV NODE_OPTIONS="--max-old-space-size=8192" WORKDIR /app diff --git a/docker-compose-launcher.sh b/docker-compose-launcher.sh index 8305a10346..e7427e9190 100755 --- a/docker-compose-launcher.sh +++ b/docker-compose-launcher.sh @@ -1,5 +1,6 @@ #!/usr/bin/env bash set -euo pipefail +cd "$(dirname "${BASH_SOURCE[0]}")" # --------------------------------------------------------------------------- # Interactive docker compose launcher for Open WebUI. @@ -100,38 +101,46 @@ done # ── Drop mode ───────────────────────────────────────────────────────────────── if [[ "$drop_project" == true ]]; then - docker compose down --remove-orphans + ./docker-stack.sh down --remove-orphans echo -e "${GREEN}${BOLD}Compose project stopped and cleaned up.${RESET}" exit 0 fi # ── Build compose command ───────────────────────────────────────────────────── +# +# The base + GPU + audio file set is owned by docker-stack.sh (the single +# source of truth for "what this stack always needs"), so it can't drift out +# of sync between the two entry points the way docker-compose.audio.yaml once +# did here. This script only ever layers its own optional extras (api/data/ +# playwright) on top and delegates the actual `docker compose` call. -compose_files=("-f" "docker-compose.yaml") - -if [[ "$enable_gpu" == true ]]; then - if ! [[ "$gpu_count" =~ ^([0-9]+|all)$ ]]; then - echo >&2 "Error: Invalid GPU count '${gpu_count}'. Must be a number or 'all'." - exit 1 - fi +if [[ "$gpu_count" =~ ^([0-9]+|all)$ ]]; then export OLLAMA_GPU_DRIVER OLLAMA_GPU_DRIVER=$(detect_gpu_driver) export OLLAMA_GPU_COUNT="$gpu_count" - compose_files+=("-f" "docker-compose.gpu.yaml") +elif [[ "$enable_gpu" == true ]]; then + echo >&2 "Error: Invalid GPU count '${gpu_count}'. Must be a number or 'all'." + exit 1 fi +# docker-stack.sh always merges docker-compose.gpu.yaml (defaults: nvidia, +# count 1, from that file's own ${VAR-default} fallbacks) -- GPU passthrough +# is no longer optional here, matching docker-stack.sh's behavior. --enable-gpu +# now only customizes the driver/count via the env vars set above. + +extra_files=() if [[ "$enable_api" == true ]]; then export OLLAMA_WEBAPI_PORT="$api_port" - compose_files+=("-f" "docker-compose.api.yaml") + extra_files+=("-f" "docker-compose.api.yaml") fi if [[ -n "$data_dir" ]]; then export OLLAMA_DATA_DIR="$data_dir" - compose_files+=("-f" "docker-compose.data.yaml") + extra_files+=("-f" "docker-compose.data.yaml") fi if [[ "$enable_playwright" == true ]]; then - compose_files+=("-f" "docker-compose.playwright.yaml") + extra_files+=("-f" "docker-compose.playwright.yaml") fi export OPEN_WEBUI_PORT="$webui_port" @@ -163,7 +172,7 @@ fi # ── Launch ──────────────────────────────────────────────────────────────────── -if docker compose "${compose_files[@]}" "${up_args[@]}"; then +if ./docker-stack.sh "${extra_files[@]}" "${up_args[@]}"; then echo echo -e "${GREEN}${BOLD}${CHECK_MARK} Compose project started successfully.${RESET}" else diff --git a/docker-compose.audio.yaml b/docker-compose.audio.yaml new file mode 100644 index 0000000000..d0cdd0abe2 --- /dev/null +++ b/docker-compose.audio.yaml @@ -0,0 +1,85 @@ +services: + asr: + build: + context: ../asr-server + container_name: asr + restart: unless-stopped + ports: + - '8020:8020' + environment: + - 'ASR_AUTH_TOKEN=${ASR_AUTH_TOKEN}' + deploy: + resources: + reservations: + devices: + - driver: nvidia + count: 1 + capabilities: [gpu] + + tts: + build: + context: ../tts-server + container_name: tts + restart: unless-stopped + ports: + - '8010:8010' + environment: + - 'TTS_AUTH_TOKEN=${TTS_AUTH_TOKEN}' + - 'HF_HOME=/app/data' + # Fallback voice for direct /v1/audio/speech calls that use + # AUDIO_TTS_MODEL=sopro-v2-turbo without a `voice` field. Set to the + # basename (no .wav) of a clip dropped into services/tts-server/voices/. + # - 'TTS_SOPRO_DEFAULT_VOICE=your_voice_name' + volumes: + # Caches the downloaded Kokoro + Sopro weights across rebuilds. + - tts-data:/app/data + # Sopro voice reference clips (.wav); drop files in here on the + # host and reference them by name via AUDIO_TTS_VOICE / `voice`. + - ../tts-server/voices:/app/voices:ro + deploy: + resources: + reservations: + devices: + - driver: nvidia + count: 1 + capabilities: [gpu] + + open-webui: + depends_on: + - asr + - tts + environment: + - 'AUDIO_STT_ENGINE=openai' + - 'AUDIO_STT_OPENAI_API_BASE_URL=http://asr:8020/v1' + - 'AUDIO_STT_OPENAI_API_KEY=${ASR_AUTH_TOKEN}' + - 'AUDIO_STT_MODEL=nemotron-3.5-asr-streaming-0.6b' + - 'AUDIO_TTS_ENGINE=openai' + - 'AUDIO_TTS_OPENAI_API_BASE_URL=http://tts:8010/v1' + - 'AUDIO_TTS_OPENAI_API_KEY=${TTS_AUTH_TOKEN}' + # To switch back to Kokoro (fixed presets), set: + # AUDIO_TTS_MODEL=kokoro-82m + # AUDIO_TTS_VOICE=af_heart + - 'AUDIO_TTS_MODEL=sopro-v2-turbo' + - 'AUDIO_TTS_VOICE=jarvis-target' + + wyoming-openai: + image: ghcr.io/roryeckel/wyoming_openai:latest + container_name: wyoming-openai + restart: unless-stopped + depends_on: + - asr + - tts + ports: + - '10300:10300' + environment: + - 'STT_OPENAI_URL=http://asr:8020/v1' + - 'STT_OPENAI_KEY=${ASR_AUTH_TOKEN}' + - 'STT_MODELS=nemotron-3.5-asr-streaming-0.6b' + - 'TTS_OPENAI_URL=http://tts:8010/v1' + - 'TTS_OPENAI_KEY=${TTS_AUTH_TOKEN}' + - 'TTS_MODELS=sopro-v2-turbo' + - 'TTS_VOICES=jarvis-target' + - 'TTS_SPEED=1.2' + +volumes: + tts-data: {} diff --git a/docker-compose.yaml b/docker-compose.yaml index 349734a939..d68c76b535 100644 --- a/docker-compose.yaml +++ b/docker-compose.yaml @@ -2,11 +2,49 @@ services: ollama: volumes: - ollama:/root/.ollama + # Read-only mount for importing custom GGUF files (e.g. via `ollama create`) + - /home/brian-llm/models/gguf:/gguf-import:ro container_name: ollama pull_policy: always tty: true restart: unless-stopped image: ollama/ollama:${OLLAMA_DOCKER_TAG-latest} + environment: + # Never unload a loaded model on idle -- default is 5 minutes. + - 'OLLAMA_KEEP_ALIVE=-1' + - 'OLLAMA_NUM_PARALLEL=4' + deploy: + resources: + reservations: + devices: + - driver: nvidia + count: 1 + capabilities: [gpu] + # No ports published here -- Ollama has no auth of its own, so LAN access + # goes through the ollama-auth proxy below instead. open-webui still + # reaches it directly over the internal docker network (see + # OLLAMA_BASE_URL), which doesn't need the token. + + ollama-auth: + image: nginx:alpine + container_name: ollama-auth + restart: unless-stopped + depends_on: + - ollama + volumes: + - ./ollama-auth/default.conf.template:/etc/nginx/templates/default.conf.template:ro + environment: + - 'OLLAMA_AUTH_TOKEN=${OLLAMA_AUTH_TOKEN}' + ports: + # Reachable from the LAN directly; restrict to the LAN subnet with the + # host firewall (see ufw rules) as defense in depth, but the real + # boundary is now the Bearer token check in ollama-auth/default.conf.template. + # This is plain HTTP, so the token is readable by anything already on + # the LAN segment that can sniff traffic -- accepted here since the + # trust boundary is "the LAN subnet," not "the wire." Revisit with TLS + # if that assumption ever stops holding (untrusted/guest/IoT devices + # sharing this LAN, etc). + - '11434:11434' open-webui: build: @@ -19,14 +57,59 @@ services: depends_on: - ollama ports: - - ${OPEN_WEBUI_PORT-3000}:8080 + # Published on all interfaces since nginx runs on a separate LAN host + # (192.168.50.224) and must reach this port directly. Firewalling this + # to that host is still good hygiene (see ufw/DOCKER-USER notes) but is + # no longer the sole security boundary -- OIDC auth below is real + # authentication against Authentik regardless of network path. + - '${OPEN_WEBUI_PORT-3000}:8080' environment: - 'OLLAMA_BASE_URL=http://ollama:11434' - 'WEBUI_SECRET_KEY=' + - 'DEFAULT_MODELS=gemma-4-e2b-mtp' + - 'ENABLE_OAUTH_SIGNUP=true' + # Links Authentik login to the existing local account with the same + # email instead of rejecting it as a collision -- safe here since this + # is a single-admin personal instance, not an open-signup multi-tenant one. + - 'OAUTH_MERGE_ACCOUNTS_BY_EMAIL=true' + # Fill these in after creating the OAuth2/OpenID Provider + Application + # in Authentik (see instructions) -- OAUTH_CLIENT_ID/SECRET come from + # that provider. + - 'OAUTH_CLIENT_ID=${OAUTH_CLIENT_ID}' + - 'OAUTH_CLIENT_SECRET=${OAUTH_CLIENT_SECRET}' + - 'OPENID_PROVIDER_URL=https://auth.hetherman.cloud/application/o/ai-open-webui/.well-known/openid-configuration' + - 'OPENID_REDIRECT_URI=https://ai.hetherman.cloud/oauth/oidc/callback' + - 'OAUTH_SCOPES=openid email profile' + # SSO-only: Authentik becomes the sole way to authenticate, admin + # included. If Authentik is ever unreachable, nobody can log in until + # it's fixed -- no local password fallback. + - 'ENABLE_PASSWORD_AUTH=false' + - 'ENABLE_LOGIN_FORM=false' + - 'ENABLE_SIGNUP=false' extra_hosts: - host.docker.internal:host-gateway restart: unless-stopped + # ollama-keepwarm: + # image: curlimages/curl:latest + # container_name: ollama-keepwarm + # restart: unless-stopped + # depends_on: + # - ollama + # # Pings Ollama every 60s to load the default model and refresh its + # # keep_alive timer -- covers first boot and re-loads it any time the + # # ollama container itself restarts (which flushes it from memory). + # # Keep this model name in sync with DEFAULT_MODELS above. + # entrypoint: + # - sh + # - -c + # - | + # while true; do + # curl -s -o /dev/null -X POST http://ollama:11434/api/generate \ + # -d '{"model":"gemma-4-e2b","keep_alive":-1}' + # sleep 60 + # done + volumes: ollama: {} open-webui: {} diff --git a/docker-stack.sh b/docker-stack.sh new file mode 100755 index 0000000000..8821f4c5b7 --- /dev/null +++ b/docker-stack.sh @@ -0,0 +1,14 @@ +#!/usr/bin/env bash +set -euo pipefail +cd "$(dirname "${BASH_SOURCE[0]}")" + +# Wraps docker compose with the full file set for this stack (base + GPU +# passthrough + audio/ollama-auth services) so a plain "up"/"build" can't +# accidentally drop the ollama GPU override again. +readonly COMPOSE_FILES=(-f docker-compose.yaml -f docker-compose.gpu.yaml -f docker-compose.audio.yaml) + +if [[ $# -eq 0 ]]; then + exec docker compose "${COMPOSE_FILES[@]}" up -d --build +fi + +exec docker compose "${COMPOSE_FILES[@]}" "$@" diff --git a/ollama-auth/default.conf.template b/ollama-auth/default.conf.template new file mode 100644 index 0000000000..08092d61f4 --- /dev/null +++ b/ollama-auth/default.conf.template @@ -0,0 +1,33 @@ +map_hash_bucket_size 128; + +map $http_authorization $ollama_auth_ok { + "Bearer ${OLLAMA_AUTH_TOKEN}" 1; + default 0; +} + +server { + listen 11434; + + location / { + if ($ollama_auth_ok = 0) { + return 401; + } + + # Docker's embedded DNS (127.0.0.11) can reassign the "ollama" + # hostname to a new container IP on restart. proxy_pass to a + # literal upstream caches that IP for the container's lifetime; + # routing through a variable forces nginx to re-resolve via this + # resolver on each request instead, so a restarted ollama doesn't + # need a matching ollama-auth restart to be reachable again. + resolver 127.0.0.11 valid=10s; + set $ollama_upstream ollama; + proxy_pass http://$ollama_upstream:11434; + proxy_http_version 1.1; + proxy_set_header Connection ""; + proxy_set_header Host $host; + proxy_buffering off; + proxy_read_timeout 600s; + proxy_send_timeout 600s; + client_max_body_size 0; + } +}