services: ollama: volumes: - ollama:/root/.ollama # Read-only mount for importing custom GGUF files (e.g. via `ollama create`) - /home/brian-llm/models/gguf:/gguf-import:ro container_name: ollama pull_policy: always tty: true restart: unless-stopped image: ollama/ollama:${OLLAMA_DOCKER_TAG-latest} environment: # Never unload a loaded model on idle -- default is 5 minutes. - 'OLLAMA_KEEP_ALIVE=-1' - 'OLLAMA_NUM_PARALLEL=4' deploy: resources: reservations: devices: - driver: nvidia count: 1 capabilities: [gpu] # No ports published here -- Ollama has no auth of its own, so LAN access # goes through the ollama-auth proxy below instead. open-webui still # reaches it directly over the internal docker network (see # OLLAMA_BASE_URL), which doesn't need the token. ollama-auth: image: nginx:alpine container_name: ollama-auth restart: unless-stopped depends_on: - ollama volumes: - ./ollama-auth/default.conf.template:/etc/nginx/templates/default.conf.template:ro environment: - 'OLLAMA_AUTH_TOKEN=${OLLAMA_AUTH_TOKEN}' ports: # Reachable from the LAN directly; restrict to the LAN subnet with the # host firewall (see ufw rules) as defense in depth, but the real # boundary is now the Bearer token check in ollama-auth/default.conf.template. # This is plain HTTP, so the token is readable by anything already on # the LAN segment that can sniff traffic -- accepted here since the # trust boundary is "the LAN subnet," not "the wire." Revisit with TLS # if that assumption ever stops holding (untrusted/guest/IoT devices # sharing this LAN, etc). - '11434:11434' open-webui: build: context: . dockerfile: Dockerfile image: ghcr.io/open-webui/open-webui:${WEBUI_DOCKER_TAG-main} container_name: open-webui volumes: - open-webui:/app/backend/data depends_on: - ollama ports: # Published on all interfaces since nginx runs on a separate LAN host # (192.168.50.224) and must reach this port directly. Firewalling this # to that host is still good hygiene (see ufw/DOCKER-USER notes) but is # no longer the sole security boundary -- OIDC auth below is real # authentication against Authentik regardless of network path. - '${OPEN_WEBUI_PORT-3000}:8080' environment: - 'OLLAMA_BASE_URL=http://ollama:11434' - 'WEBUI_SECRET_KEY=' # Only takes effect on a fresh DB -- once ui.default_models exists in # the config table, that value wins (see open-webui's PersistentConfig # seed_defaults behavior). Currently gemma-4-e2b-mtp was removed from # Ollama in favor of the relaxed-MTP llama-mtp service; the live value # is 'gemma-4-e2b,/models/gemma-4-E2B-it-Q4_K_M.gguf'. - 'DEFAULT_MODELS=gemma-4-e2b,/models/gemma-4-E2B-it-Q4_K_M.gguf' - 'ENABLE_OAUTH_SIGNUP=true' # Links Authentik login to the existing local account with the same # email instead of rejecting it as a collision -- safe here since this # is a single-admin personal instance, not an open-signup multi-tenant one. - 'OAUTH_MERGE_ACCOUNTS_BY_EMAIL=true' # Fill these in after creating the OAuth2/OpenID Provider + Application # in Authentik (see instructions) -- OAUTH_CLIENT_ID/SECRET come from # that provider. - 'OAUTH_CLIENT_ID=${OAUTH_CLIENT_ID}' - 'OAUTH_CLIENT_SECRET=${OAUTH_CLIENT_SECRET}' - 'OPENID_PROVIDER_URL=https://auth.hetherman.cloud/application/o/ai-open-webui/.well-known/openid-configuration' - 'OPENID_REDIRECT_URI=https://ai.hetherman.cloud/oauth/oidc/callback' - 'OAUTH_SCOPES=openid email profile' # SSO-only: Authentik becomes the sole way to authenticate, admin # included. If Authentik is ever unreachable, nobody can log in until # it's fixed -- no local password fallback. - 'ENABLE_PASSWORD_AUTH=false' - 'ENABLE_LOGIN_FORM=false' - 'ENABLE_SIGNUP=false' extra_hosts: - host.docker.internal:host-gateway restart: unless-stopped # ollama-keepwarm: # image: curlimages/curl:latest # container_name: ollama-keepwarm # restart: unless-stopped # depends_on: # - ollama # # Pings Ollama every 60s to load the default model and refresh its # # keep_alive timer -- covers first boot and re-loads it any time the # # ollama container itself restarts (which flushes it from memory). # # Keep this model name in sync with DEFAULT_MODELS above. # entrypoint: # - sh # - -c # - | # while true; do # curl -s -o /dev/null -X POST http://ollama:11434/api/generate \ # -d '{"model":"gemma-4-e2b","keep_alive":-1}' # sleep 60 # done volumes: ollama: {} open-webui: {}