Initial voice-assistant-stack: open-webui + spotify-voice-assistant submodules, asr/tts sidecars

- open-webui: submodule, LAN OIDC/Authentik-fronted, GPU-passthrough ollama,
  ollama-auth proxy, asr/tts wired via docker-compose.audio.yaml
- spotify-voice-assistant: submodule, Home Assistant custom integration
- tts-server / asr-server: plain directories, built as sidecars by
  open-webui/docker-stack.sh (no separate git history)
This commit is contained in:
2026-08-31 00:47:35 -04:00
commit c2b92d96cb
14 changed files with 307 additions and 0 deletions
+1
View File
@@ -0,0 +1 @@
tts-server/audio-files/
+6
View File
@@ -0,0 +1,6 @@
[submodule "open-webui"]
path = open-webui
url = ssh://git@192.168.50.224:30009/bhetherman/open-webui.git
[submodule "spotify-voice-assistant"]
path = spotify-voice-assistant
url = https://git.hetherman.cloud/bhetherman/spotify-voice-assistant.git
+17
View File
@@ -0,0 +1,17 @@
FROM python:3.11-slim
RUN apt-get update && apt-get install -y --no-install-recommends \
libsndfile1 ffmpeg \
&& rm -rf /var/lib/apt/lists/*
WORKDIR /app
# The Titan Xp (Pascal, sm_61) needs a torch build old enough to still ship
# Pascal kernels; recent default PyPI wheels dropped sm_60/sm_61 entirely.
ENV PIP_EXTRA_INDEX_URL=https://download.pytorch.org/whl/cu124
COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt
COPY app.py .
EXPOSE 8020
CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "8020"]
+80
View File
@@ -0,0 +1,80 @@
import hmac
import logging
import os
import tempfile
import torch
from fastapi import Depends, FastAPI, File, Form, HTTPException, UploadFile
from fastapi.responses import JSONResponse, PlainTextResponse
from fastapi.security import HTTPAuthorizationCredentials, HTTPBearer
from transformers import AutoModelForRNNT, AutoProcessor
from transformers.audio_utils import load_audio
logging.basicConfig(level=logging.INFO)
log = logging.getLogger("asr-server")
AUTH_TOKEN = os.environ["ASR_AUTH_TOKEN"]
bearer_scheme = HTTPBearer()
def require_auth(creds: HTTPAuthorizationCredentials = Depends(bearer_scheme)) -> None:
if not hmac.compare_digest(creds.credentials, AUTH_TOKEN):
raise HTTPException(status_code=401, detail="invalid token")
MODEL_ID = os.environ.get("ASR_MODEL_ID", "nvidia/nemotron-3.5-asr-streaming-0.6b")
DEVICE = "cuda" if torch.cuda.is_available() else "cpu"
DTYPE = torch.float16 if DEVICE == "cuda" else torch.float32
# The RNN-T decoder's LSTM layers hit cuDNN's fused RNN kernel on `.to(device)`,
# which requires SM >= 7.5 in recent cuDNN builds. This GPU (Pascal, SM 6.1) is
# below that floor, so disable cuDNN and fall back to PyTorch's generic CUDA
# RNN kernels instead.
torch.backends.cudnn.enabled = False
app = FastAPI()
log.info("Loading %s on %s (%s)", MODEL_ID, DEVICE, DTYPE)
processor = AutoProcessor.from_pretrained(MODEL_ID)
model = AutoModelForRNNT.from_pretrained(MODEL_ID, dtype=DTYPE).to(DEVICE).eval()
SAMPLING_RATE = processor.feature_extractor.sampling_rate
log.info("Model ready (sampling_rate=%s)", SAMPLING_RATE)
@app.get("/health")
def health():
return {"status": "ok", "device": DEVICE}
@app.post("/v1/audio/transcriptions", dependencies=[Depends(require_auth)])
async def transcribe(
file: UploadFile = File(...),
model_name: str = Form("nemotron-3.5-asr-streaming-0.6b", alias="model"),
language: str | None = Form(None),
response_format: str = Form("json"),
):
suffix = os.path.splitext(file.filename or "")[1] or ".wav"
data = await file.read()
with tempfile.NamedTemporaryFile(suffix=suffix) as tmp:
tmp.write(data)
tmp.flush()
try:
audio = load_audio(tmp.name, sampling_rate=SAMPLING_RATE)
except Exception as exc:
raise HTTPException(status_code=400, detail=f"could not decode audio: {exc}") from exc
lang = language or "auto"
inputs = processor(audio, sampling_rate=SAMPLING_RATE, language=lang)
inputs = inputs.to(DEVICE, dtype=DTYPE)
with torch.inference_mode():
output = model.generate(**inputs, return_dict_in_generate=True)
text = processor.decode(output.sequences, skip_special_tokens=True)
if isinstance(text, list):
text = text[0] if text else ""
if response_format == "text":
return PlainTextResponse(text)
return JSONResponse({"text": text})
+8
View File
@@ -0,0 +1,8 @@
fastapi==0.115.*
uvicorn[standard]==0.32.*
python-multipart==0.0.*
torch==2.6.0
transformers>=5.13.0
soundfile
librosa
accelerate
+20
View File
@@ -0,0 +1,20 @@
{
"folders": [
{
"path": "open-webui"
},
{
"path": ".venv"
},
{
"path": "tts-server"
},
{
"path": "asr-server"
},
{
"path": "spotify-voice-assistant"
}
],
"settings": {}
}
Submodule
+1
Submodule open-webui added at 7eff527d16
+17
View File
@@ -0,0 +1,17 @@
FROM python:3.11-slim
RUN apt-get update && apt-get install -y --no-install-recommends \
libsndfile1 ffmpeg espeak-ng \
&& rm -rf /var/lib/apt/lists/*
WORKDIR /app
# The Titan Xp (Pascal, sm_61) needs a torch build old enough to still ship
# Pascal kernels; recent default PyPI wheels dropped sm_60/sm_61 entirely.
ENV PIP_EXTRA_INDEX_URL=https://download.pytorch.org/whl/cu121
COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt
COPY app.py .
EXPOSE 8010
CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "8010"]
+145
View File
@@ -0,0 +1,145 @@
import asyncio
import hmac
import io
import logging
import os
from concurrent.futures import ThreadPoolExecutor
from pathlib import Path
import numpy as np
import parselmouth
import soundfile as sf
import torch
from fastapi import Depends, FastAPI, HTTPException
from fastapi.responses import Response
from fastapi.security import HTTPAuthorizationCredentials, HTTPBearer
from kokoro import KPipeline
from pydantic import BaseModel
from sopro import SoproTTS
logging.basicConfig(level=logging.INFO)
log = logging.getLogger("tts-server")
AUTH_TOKEN = os.environ["TTS_AUTH_TOKEN"]
bearer_scheme = HTTPBearer()
def require_auth(creds: HTTPAuthorizationCredentials = Depends(bearer_scheme)) -> None:
if not hmac.compare_digest(creds.credentials, AUTH_TOKEN):
raise HTTPException(status_code=401, detail="invalid token")
# 'a' = American English. Kokoro's voices are fixed pretrained presets (not
# zero-shot cloning), so there's no per-request drift to work around like
# Audio8 needed -- picking a voice is just picking an ID.
LANG_CODE = os.environ.get("TTS_LANG_CODE", "a")
DEFAULT_VOICE = os.environ.get("TTS_DEFAULT_VOICE", "af_heart")
KOKORO_SAMPLE_RATE = 24000
# Sopro is zero-shot voice cloning, not preset voices -- a "voice" here is
# just a reference clip dropped into this directory as <name>.wav, picked by
# the `voice` request field.
VOICES_DIR = Path(os.environ.get("TTS_VOICES_DIR", "/app/voices"))
DEFAULT_SOPRO_VOICE = os.environ.get("TTS_SOPRO_DEFAULT_VOICE")
DEVICE = "cuda" if torch.cuda.is_available() else "cpu"
app = FastAPI()
log.info("Loading Kokoro (lang_code=%s) on %s", LANG_CODE, DEVICE)
kokoro_pipeline = KPipeline(lang_code=LANG_CODE, device=DEVICE)
log.info("Kokoro ready")
log.info("Loading Sopro on %s", DEVICE)
sopro_model = SoproTTS.from_pretrained("samuel-vitorino/sopro-v2-turbo", device=DEVICE)
log.info("Sopro ready")
# References are just resampled/cropped tensors of the voice clip -- cheap to
# keep around per voice name instead of redoing that work every request.
_sopro_ref_cache = {}
# Sopro caches a CUDA graph per decode length (sopro/nn/decode.py) captured on
# whichever thread first hits that length. Replaying -- or sampling after
# replaying -- from a different thread breaks torch's CUDA RNG bookkeeping
# ("Offset increment outside graph capture encountered unexpectedly"), so
# every Sopro call has to run on this same dedicated thread.
_sopro_executor = ThreadPoolExecutor(max_workers=1)
def _to_numpy(audio) -> np.ndarray:
if isinstance(audio, torch.Tensor):
return audio.detach().cpu().numpy()
return audio
def _time_stretch(audio: np.ndarray, sample_rate: int, speed: float) -> np.ndarray:
# Sopro has no native speed/duration control, unlike Kokoro's duration
# predictor. PSOLA (pitch-synchronous overlap-add) changes tempo without
# pitch-shifting, and handles speech transients/consonants far more
# cleanly than a generic STFT phase vocoder does.
snd = parselmouth.Sound(audio.astype(np.float64), sampling_frequency=sample_rate)
stretched = parselmouth.praat.call(snd, "Lengthen (overlap-add)", 75, 600, 1.0 / speed)
return stretched.values[0].astype(np.float32)
def _sopro_reference(name: str):
if name not in _sopro_ref_cache:
wav_path = VOICES_DIR / f"{name}.wav"
if not wav_path.exists():
raise HTTPException(status_code=400, detail=f"unknown sopro voice '{name}' (expected {wav_path})")
_sopro_ref_cache[name] = sopro_model.prepare_reference(ref_audio_path=str(wav_path))
return _sopro_ref_cache[name]
def _synthesize_kokoro(text: str, voice: str, speed: float) -> tuple[np.ndarray, int]:
# KPipeline yields one Result per chunk it splits the input into -- no
# manual chunking needed here, unlike the autoregressive model this
# replaced.
chunks = [_to_numpy(result.audio) for result in kokoro_pipeline(text, voice=voice, speed=speed)]
audio = np.concatenate(chunks) if len(chunks) > 1 else chunks[0]
return audio, KOKORO_SAMPLE_RATE
def _synthesize_sopro(text: str, voice: str | None, speed: float) -> tuple[np.ndarray, int]:
voice = voice or DEFAULT_SOPRO_VOICE
if not voice:
raise HTTPException(status_code=400, detail="no voice given and TTS_SOPRO_DEFAULT_VOICE is unset")
ref = _sopro_reference(voice)
audio = _to_numpy(sopro_model.synthesize(text, ref=ref))
if speed != 1.0:
audio = _time_stretch(audio, sopro_model.sample_rate, speed)
return audio, sopro_model.sample_rate
class SpeechRequest(BaseModel):
model: str | None = None
input: str
voice: str | None = None
response_format: str = "wav"
speed: float | None = None
@app.get("/health")
def health():
return {"status": "ok", "device": DEVICE}
@app.post("/v1/audio/speech", dependencies=[Depends(require_auth)])
async def synthesize(req: SpeechRequest):
if (req.model or "").lower().startswith("sopro"):
loop = asyncio.get_running_loop()
audio, sample_rate = await loop.run_in_executor(
_sopro_executor, _synthesize_sopro, req.input, req.voice, req.speed or 1.0
)
else:
audio, sample_rate = await asyncio.to_thread(
_synthesize_kokoro, req.input, req.voice or DEFAULT_VOICE, req.speed or 1.0
)
buf = io.BytesIO()
fmt = "WAV" if req.response_format in ("wav", None) else req.response_format.upper()
sf.write(buf, audio, sample_rate, format=fmt)
buf.seek(0)
media_type = "audio/wav" if fmt == "WAV" else f"audio/{req.response_format}"
return Response(content=buf.read(), media_type=media_type)
+11
View File
@@ -0,0 +1,11 @@
fastapi==0.115.*
uvicorn[standard]==0.32.*
torch==2.5.1
# Pinned to match torch==2.5.1+cu121 above -- an unpinned install pulls a
# newer torchaudio built against a CUDA runtime the Titan Xp's pinned torch
# doesn't ship, which fails with "libcudart.so.13: cannot open shared object".
torchaudio==2.5.1
kokoro>=0.9.2
sopro
soundfile
praat-parselmouth
Binary file not shown.
Binary file not shown.
Binary file not shown.