commit c2b92d96cb74611c14f111dcb6351958ac4cd17c Author: Brian Hetherman Date: Mon Aug 31 00:47:35 2026 -0400 Initial voice-assistant-stack: open-webui + spotify-voice-assistant submodules, asr/tts sidecars - open-webui: submodule, LAN OIDC/Authentik-fronted, GPU-passthrough ollama, ollama-auth proxy, asr/tts wired via docker-compose.audio.yaml - spotify-voice-assistant: submodule, Home Assistant custom integration - tts-server / asr-server: plain directories, built as sidecars by open-webui/docker-stack.sh (no separate git history) diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..678f28a --- /dev/null +++ b/.gitignore @@ -0,0 +1 @@ +tts-server/audio-files/ diff --git a/.gitmodules b/.gitmodules new file mode 100644 index 0000000..6b2dba9 --- /dev/null +++ b/.gitmodules @@ -0,0 +1,6 @@ +[submodule "open-webui"] + path = open-webui + url = ssh://git@192.168.50.224:30009/bhetherman/open-webui.git +[submodule "spotify-voice-assistant"] + path = spotify-voice-assistant + url = https://git.hetherman.cloud/bhetherman/spotify-voice-assistant.git diff --git a/asr-server/Dockerfile b/asr-server/Dockerfile new file mode 100644 index 0000000..917b1f9 --- /dev/null +++ b/asr-server/Dockerfile @@ -0,0 +1,17 @@ +FROM python:3.11-slim + +RUN apt-get update && apt-get install -y --no-install-recommends \ + libsndfile1 ffmpeg \ + && rm -rf /var/lib/apt/lists/* + +WORKDIR /app +# The Titan Xp (Pascal, sm_61) needs a torch build old enough to still ship +# Pascal kernels; recent default PyPI wheels dropped sm_60/sm_61 entirely. +ENV PIP_EXTRA_INDEX_URL=https://download.pytorch.org/whl/cu124 +COPY requirements.txt . +RUN pip install --no-cache-dir -r requirements.txt + +COPY app.py . + +EXPOSE 8020 +CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "8020"] diff --git a/asr-server/app.py b/asr-server/app.py new file mode 100644 index 0000000..567ffc0 --- /dev/null +++ b/asr-server/app.py @@ -0,0 +1,80 @@ +import hmac +import logging +import os +import tempfile + +import torch +from fastapi import Depends, FastAPI, File, Form, HTTPException, UploadFile +from fastapi.responses import JSONResponse, PlainTextResponse +from fastapi.security import HTTPAuthorizationCredentials, HTTPBearer +from transformers import AutoModelForRNNT, AutoProcessor +from transformers.audio_utils import load_audio + +logging.basicConfig(level=logging.INFO) +log = logging.getLogger("asr-server") + +AUTH_TOKEN = os.environ["ASR_AUTH_TOKEN"] +bearer_scheme = HTTPBearer() + + +def require_auth(creds: HTTPAuthorizationCredentials = Depends(bearer_scheme)) -> None: + if not hmac.compare_digest(creds.credentials, AUTH_TOKEN): + raise HTTPException(status_code=401, detail="invalid token") + + +MODEL_ID = os.environ.get("ASR_MODEL_ID", "nvidia/nemotron-3.5-asr-streaming-0.6b") +DEVICE = "cuda" if torch.cuda.is_available() else "cpu" +DTYPE = torch.float16 if DEVICE == "cuda" else torch.float32 + +# The RNN-T decoder's LSTM layers hit cuDNN's fused RNN kernel on `.to(device)`, +# which requires SM >= 7.5 in recent cuDNN builds. This GPU (Pascal, SM 6.1) is +# below that floor, so disable cuDNN and fall back to PyTorch's generic CUDA +# RNN kernels instead. +torch.backends.cudnn.enabled = False + +app = FastAPI() + +log.info("Loading %s on %s (%s)", MODEL_ID, DEVICE, DTYPE) +processor = AutoProcessor.from_pretrained(MODEL_ID) +model = AutoModelForRNNT.from_pretrained(MODEL_ID, dtype=DTYPE).to(DEVICE).eval() +SAMPLING_RATE = processor.feature_extractor.sampling_rate +log.info("Model ready (sampling_rate=%s)", SAMPLING_RATE) + + +@app.get("/health") +def health(): + return {"status": "ok", "device": DEVICE} + + +@app.post("/v1/audio/transcriptions", dependencies=[Depends(require_auth)]) +async def transcribe( + file: UploadFile = File(...), + model_name: str = Form("nemotron-3.5-asr-streaming-0.6b", alias="model"), + language: str | None = Form(None), + response_format: str = Form("json"), +): + suffix = os.path.splitext(file.filename or "")[1] or ".wav" + data = await file.read() + + with tempfile.NamedTemporaryFile(suffix=suffix) as tmp: + tmp.write(data) + tmp.flush() + try: + audio = load_audio(tmp.name, sampling_rate=SAMPLING_RATE) + except Exception as exc: + raise HTTPException(status_code=400, detail=f"could not decode audio: {exc}") from exc + + lang = language or "auto" + inputs = processor(audio, sampling_rate=SAMPLING_RATE, language=lang) + inputs = inputs.to(DEVICE, dtype=DTYPE) + + with torch.inference_mode(): + output = model.generate(**inputs, return_dict_in_generate=True) + + text = processor.decode(output.sequences, skip_special_tokens=True) + if isinstance(text, list): + text = text[0] if text else "" + + if response_format == "text": + return PlainTextResponse(text) + return JSONResponse({"text": text}) diff --git a/asr-server/requirements.txt b/asr-server/requirements.txt new file mode 100644 index 0000000..110f592 --- /dev/null +++ b/asr-server/requirements.txt @@ -0,0 +1,8 @@ +fastapi==0.115.* +uvicorn[standard]==0.32.* +python-multipart==0.0.* +torch==2.6.0 +transformers>=5.13.0 +soundfile +librosa +accelerate diff --git a/assistant-stack.code-workspace b/assistant-stack.code-workspace new file mode 100644 index 0000000..c819b0a --- /dev/null +++ b/assistant-stack.code-workspace @@ -0,0 +1,20 @@ +{ + "folders": [ + { + "path": "open-webui" + }, + { + "path": ".venv" + }, + { + "path": "tts-server" + }, + { + "path": "asr-server" + }, + { + "path": "spotify-voice-assistant" + } + ], + "settings": {} +} \ No newline at end of file diff --git a/open-webui b/open-webui new file mode 160000 index 0000000..7eff527 --- /dev/null +++ b/open-webui @@ -0,0 +1 @@ +Subproject commit 7eff527d16f2a143c28bd04f51cb388a7055966d diff --git a/spotify-voice-assistant b/spotify-voice-assistant new file mode 160000 index 0000000..7249919 --- /dev/null +++ b/spotify-voice-assistant @@ -0,0 +1 @@ +Subproject commit 72499190068f3db057d55013a45846bdccf11fc8 diff --git a/tts-server/Dockerfile b/tts-server/Dockerfile new file mode 100644 index 0000000..43166dc --- /dev/null +++ b/tts-server/Dockerfile @@ -0,0 +1,17 @@ +FROM python:3.11-slim + +RUN apt-get update && apt-get install -y --no-install-recommends \ + libsndfile1 ffmpeg espeak-ng \ + && rm -rf /var/lib/apt/lists/* + +WORKDIR /app +# The Titan Xp (Pascal, sm_61) needs a torch build old enough to still ship +# Pascal kernels; recent default PyPI wheels dropped sm_60/sm_61 entirely. +ENV PIP_EXTRA_INDEX_URL=https://download.pytorch.org/whl/cu121 +COPY requirements.txt . +RUN pip install --no-cache-dir -r requirements.txt + +COPY app.py . + +EXPOSE 8010 +CMD ["uvicorn", "app:app", "--host", "0.0.0.0", "--port", "8010"] diff --git a/tts-server/app.py b/tts-server/app.py new file mode 100644 index 0000000..27a6dc9 --- /dev/null +++ b/tts-server/app.py @@ -0,0 +1,145 @@ +import asyncio +import hmac +import io +import logging +import os +from concurrent.futures import ThreadPoolExecutor +from pathlib import Path + +import numpy as np +import parselmouth +import soundfile as sf +import torch +from fastapi import Depends, FastAPI, HTTPException +from fastapi.responses import Response +from fastapi.security import HTTPAuthorizationCredentials, HTTPBearer +from kokoro import KPipeline +from pydantic import BaseModel +from sopro import SoproTTS + +logging.basicConfig(level=logging.INFO) +log = logging.getLogger("tts-server") + +AUTH_TOKEN = os.environ["TTS_AUTH_TOKEN"] +bearer_scheme = HTTPBearer() + + +def require_auth(creds: HTTPAuthorizationCredentials = Depends(bearer_scheme)) -> None: + if not hmac.compare_digest(creds.credentials, AUTH_TOKEN): + raise HTTPException(status_code=401, detail="invalid token") + + +# 'a' = American English. Kokoro's voices are fixed pretrained presets (not +# zero-shot cloning), so there's no per-request drift to work around like +# Audio8 needed -- picking a voice is just picking an ID. +LANG_CODE = os.environ.get("TTS_LANG_CODE", "a") +DEFAULT_VOICE = os.environ.get("TTS_DEFAULT_VOICE", "af_heart") +KOKORO_SAMPLE_RATE = 24000 + +# Sopro is zero-shot voice cloning, not preset voices -- a "voice" here is +# just a reference clip dropped into this directory as .wav, picked by +# the `voice` request field. +VOICES_DIR = Path(os.environ.get("TTS_VOICES_DIR", "/app/voices")) +DEFAULT_SOPRO_VOICE = os.environ.get("TTS_SOPRO_DEFAULT_VOICE") + +DEVICE = "cuda" if torch.cuda.is_available() else "cpu" + +app = FastAPI() + +log.info("Loading Kokoro (lang_code=%s) on %s", LANG_CODE, DEVICE) +kokoro_pipeline = KPipeline(lang_code=LANG_CODE, device=DEVICE) +log.info("Kokoro ready") + +log.info("Loading Sopro on %s", DEVICE) +sopro_model = SoproTTS.from_pretrained("samuel-vitorino/sopro-v2-turbo", device=DEVICE) +log.info("Sopro ready") + +# References are just resampled/cropped tensors of the voice clip -- cheap to +# keep around per voice name instead of redoing that work every request. +_sopro_ref_cache = {} + +# Sopro caches a CUDA graph per decode length (sopro/nn/decode.py) captured on +# whichever thread first hits that length. Replaying -- or sampling after +# replaying -- from a different thread breaks torch's CUDA RNG bookkeeping +# ("Offset increment outside graph capture encountered unexpectedly"), so +# every Sopro call has to run on this same dedicated thread. +_sopro_executor = ThreadPoolExecutor(max_workers=1) + + +def _to_numpy(audio) -> np.ndarray: + if isinstance(audio, torch.Tensor): + return audio.detach().cpu().numpy() + return audio + + +def _time_stretch(audio: np.ndarray, sample_rate: int, speed: float) -> np.ndarray: + # Sopro has no native speed/duration control, unlike Kokoro's duration + # predictor. PSOLA (pitch-synchronous overlap-add) changes tempo without + # pitch-shifting, and handles speech transients/consonants far more + # cleanly than a generic STFT phase vocoder does. + snd = parselmouth.Sound(audio.astype(np.float64), sampling_frequency=sample_rate) + stretched = parselmouth.praat.call(snd, "Lengthen (overlap-add)", 75, 600, 1.0 / speed) + return stretched.values[0].astype(np.float32) + + +def _sopro_reference(name: str): + if name not in _sopro_ref_cache: + wav_path = VOICES_DIR / f"{name}.wav" + if not wav_path.exists(): + raise HTTPException(status_code=400, detail=f"unknown sopro voice '{name}' (expected {wav_path})") + _sopro_ref_cache[name] = sopro_model.prepare_reference(ref_audio_path=str(wav_path)) + return _sopro_ref_cache[name] + + +def _synthesize_kokoro(text: str, voice: str, speed: float) -> tuple[np.ndarray, int]: + # KPipeline yields one Result per chunk it splits the input into -- no + # manual chunking needed here, unlike the autoregressive model this + # replaced. + chunks = [_to_numpy(result.audio) for result in kokoro_pipeline(text, voice=voice, speed=speed)] + audio = np.concatenate(chunks) if len(chunks) > 1 else chunks[0] + return audio, KOKORO_SAMPLE_RATE + + +def _synthesize_sopro(text: str, voice: str | None, speed: float) -> tuple[np.ndarray, int]: + voice = voice or DEFAULT_SOPRO_VOICE + if not voice: + raise HTTPException(status_code=400, detail="no voice given and TTS_SOPRO_DEFAULT_VOICE is unset") + ref = _sopro_reference(voice) + audio = _to_numpy(sopro_model.synthesize(text, ref=ref)) + if speed != 1.0: + audio = _time_stretch(audio, sopro_model.sample_rate, speed) + return audio, sopro_model.sample_rate + + +class SpeechRequest(BaseModel): + model: str | None = None + input: str + voice: str | None = None + response_format: str = "wav" + speed: float | None = None + + +@app.get("/health") +def health(): + return {"status": "ok", "device": DEVICE} + + +@app.post("/v1/audio/speech", dependencies=[Depends(require_auth)]) +async def synthesize(req: SpeechRequest): + if (req.model or "").lower().startswith("sopro"): + loop = asyncio.get_running_loop() + audio, sample_rate = await loop.run_in_executor( + _sopro_executor, _synthesize_sopro, req.input, req.voice, req.speed or 1.0 + ) + else: + audio, sample_rate = await asyncio.to_thread( + _synthesize_kokoro, req.input, req.voice or DEFAULT_VOICE, req.speed or 1.0 + ) + + buf = io.BytesIO() + fmt = "WAV" if req.response_format in ("wav", None) else req.response_format.upper() + sf.write(buf, audio, sample_rate, format=fmt) + buf.seek(0) + + media_type = "audio/wav" if fmt == "WAV" else f"audio/{req.response_format}" + return Response(content=buf.read(), media_type=media_type) diff --git a/tts-server/requirements.txt b/tts-server/requirements.txt new file mode 100644 index 0000000..2338e62 --- /dev/null +++ b/tts-server/requirements.txt @@ -0,0 +1,11 @@ +fastapi==0.115.* +uvicorn[standard]==0.32.* +torch==2.5.1 +# Pinned to match torch==2.5.1+cu121 above -- an unpinned install pulls a +# newer torchaudio built against a CUDA runtime the Titan Xp's pinned torch +# doesn't ship, which fails with "libcudart.so.13: cannot open shared object". +torchaudio==2.5.1 +kokoro>=0.9.2 +sopro +soundfile +praat-parselmouth diff --git a/tts-server/voices/jarvis-target.orig.wav b/tts-server/voices/jarvis-target.orig.wav new file mode 100644 index 0000000..5a44c7e Binary files /dev/null and b/tts-server/voices/jarvis-target.orig.wav differ diff --git a/tts-server/voices/jarvis-target.silence-removed.wav b/tts-server/voices/jarvis-target.silence-removed.wav new file mode 100644 index 0000000..9460c1f Binary files /dev/null and b/tts-server/voices/jarvis-target.silence-removed.wav differ diff --git a/tts-server/voices/jarvis-target.wav b/tts-server/voices/jarvis-target.wav new file mode 100644 index 0000000..0661307 Binary files /dev/null and b/tts-server/voices/jarvis-target.wav differ