Add local TTS/STT wrapper servers (chatterbox, faster-whisper) as openai_compatible providers

Wraps the locally installed chatterbox-tts and faster-whisper packages in thin
FastAPI servers implementing OpenAI's audio API shape, since neither Ollama nor
OpenRouter support speech. Pinned to GPU 2 to avoid contending with Ollama's
resident models on GPU 1. Requires UFW rules for 8901/8902 (same pattern as
the existing 11434 rule) since UFW defaults to deny-incoming.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
Dieter Schlüter 2026-07-10 21:41:44 +02:00
commit fb1fbcfabe
4 changed files with 198 additions and 0 deletions

112
services/tts_server.py Normal file
View file

@ -0,0 +1,112 @@
#!/usr/bin/env python3
"""OpenAI-compatible TTS wrapper around chatterbox-tts, for Open Notebook.
Implements exactly the endpoint esperanto's OpenAICompatibleTextToSpeechModel calls:
POST /audio/speech {model, voice, input, response_format} -> raw audio bytes
Start:
~/miniforge3/envs/chatterbox/bin/python tts_server.py
Env vars:
TTS_HOST (default 0.0.0.0), TTS_PORT (default 8901)
CUDA_VISIBLE_DEVICES should be set by the caller (e.g. "1,2") to keep GPU 0 free.
"""
from __future__ import annotations
import os
import subprocess
import sys
import tempfile
from pathlib import Path
sys.path.insert(0, str(Path.home() / "chatterbox-tts-cli"))
import chatterbox_cli_v4 as tts # noqa: E402
import torch
import torchaudio as ta
from fastapi import FastAPI, HTTPException, Response
from pydantic import BaseModel
app = FastAPI(title="Chatterbox TTS (OpenAI-compatible)", version="1.0")
_DEVICE = tts.get_device(None)
_model_cache: dict[str, tuple] = {}
def _get_model(lang: str):
key = "en" if lang == "en" else "multi"
if key not in _model_cache:
_model_cache[key] = tts.load_model(lang, _DEVICE, t3_model="v3")
return _model_cache[key]
class SpeechRequest(BaseModel):
model: str | None = None
voice: str = "de"
input: str
response_format: str = "mp3"
_FORMAT_CONTENT_TYPE = {
"mp3": "audio/mpeg",
"wav": "audio/wav",
"opus": "audio/opus",
"flac": "audio/flac",
"aac": "audio/aac",
}
@app.get("/health")
def health():
return {"status": "ok", "device": _DEVICE}
@app.get("/audio/voices")
def voices():
return {"voices": [{"id": lang, "name": lang} for lang in sorted(tts.SUPPORTED_LANGS)]}
@app.post("/audio/speech")
def speech(req: SpeechRequest):
lang = req.voice if req.voice in tts.SUPPORTED_LANGS else "de"
raw = tts.clean_raw_text(req.input)
raw_chunks = tts.split_into_sentences(raw, max_len=400)
chunks = [tts.preprocess_tts_text(c, lang=lang, pronunciation_dict=None) for c in raw_chunks]
chunks = [c for c in chunks if c.strip()]
if not chunks:
raise HTTPException(status_code=422, detail="Kein synthetisierbarer Text übrig.")
model, model_kind, sr = _get_model(lang)
wavs = []
for chunk in chunks:
wavs.append(tts.generate_chunk(model, model_kind, chunk, lang, None))
final = wavs[0] if len(wavs) == 1 else torch.cat(wavs, dim=-1)
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as wav_tmp:
wav_path = wav_tmp.name
ta.save(wav_path, final, sr)
fmt = req.response_format if req.response_format in _FORMAT_CONTENT_TYPE else "wav"
try:
if fmt == "wav":
audio_bytes = Path(wav_path).read_bytes()
else:
out_path = wav_path.replace(".wav", f".{fmt}")
subprocess.run(
["ffmpeg", "-y", "-i", wav_path, out_path],
check=True, capture_output=True,
)
audio_bytes = Path(out_path).read_bytes()
os.unlink(out_path)
finally:
os.unlink(wav_path)
return Response(content=audio_bytes, media_type=_FORMAT_CONTENT_TYPE[fmt])
if __name__ == "__main__":
import uvicorn
uvicorn.run(app, host=os.environ.get("TTS_HOST", "0.0.0.0"),
port=int(os.environ.get("TTS_PORT", "8901")))