2026-07-10 21:41:44 +02:00
|
|
|
#!/usr/bin/env python3
|
|
|
|
|
"""OpenAI-compatible TTS wrapper around chatterbox-tts, for Open Notebook.
|
|
|
|
|
|
|
|
|
|
Implements exactly the endpoint esperanto's OpenAICompatibleTextToSpeechModel calls:
|
|
|
|
|
POST /audio/speech {model, voice, input, response_format} -> raw audio bytes
|
|
|
|
|
|
2026-07-10 22:38:03 +02:00
|
|
|
Voice cloning: `voice` selects a named reference clip from VOICES_DIR (each
|
|
|
|
|
<name>.wav becomes voice "<name>"). Unknown/omitted voices fall back to
|
|
|
|
|
"default". A recognized language code (see chatterbox_cli_v4.SUPPORTED_LANGS)
|
|
|
|
|
is still accepted as `voice` for the old behaviour: no cloning, built-in voice
|
|
|
|
|
for that language. All cloned voices are generated in German (CLONE_LANG) —
|
|
|
|
|
this deployment only has German reference clips; extend VOICE_LANG_OVERRIDES
|
|
|
|
|
below if you add clips in other languages.
|
|
|
|
|
|
2026-07-10 21:41:44 +02:00
|
|
|
Start:
|
|
|
|
|
~/miniforge3/envs/chatterbox/bin/python tts_server.py
|
|
|
|
|
|
|
|
|
|
Env vars:
|
|
|
|
|
TTS_HOST (default 0.0.0.0), TTS_PORT (default 8901)
|
2026-07-10 22:38:03 +02:00
|
|
|
CUDA_VISIBLE_DEVICES should be set by the caller (e.g. "2") to keep GPU 0/1 free.
|
2026-07-10 21:41:44 +02:00
|
|
|
"""
|
|
|
|
|
from __future__ import annotations
|
|
|
|
|
|
|
|
|
|
import os
|
|
|
|
|
import subprocess
|
|
|
|
|
import sys
|
|
|
|
|
import tempfile
|
|
|
|
|
from pathlib import Path
|
|
|
|
|
|
|
|
|
|
sys.path.insert(0, str(Path.home() / "chatterbox-tts-cli"))
|
|
|
|
|
import chatterbox_cli_v4 as tts # noqa: E402
|
|
|
|
|
|
|
|
|
|
import torch
|
|
|
|
|
import torchaudio as ta
|
|
|
|
|
from fastapi import FastAPI, HTTPException, Response
|
|
|
|
|
from pydantic import BaseModel
|
|
|
|
|
|
2026-07-10 22:38:03 +02:00
|
|
|
app = FastAPI(title="Chatterbox TTS (OpenAI-compatible, voice cloning)", version="2.0")
|
2026-07-10 21:41:44 +02:00
|
|
|
|
|
|
|
|
_DEVICE = tts.get_device(None)
|
|
|
|
|
_model_cache: dict[str, tuple] = {}
|
|
|
|
|
|
2026-07-10 22:38:03 +02:00
|
|
|
VOICES_DIR = Path(__file__).parent / "voices"
|
|
|
|
|
CLONE_LANG = "de" # language used for every cloned reference voice
|
|
|
|
|
VOICE_LANG_OVERRIDES: dict[str, str] = {} # e.g. {"john": "en"} if you add an English clip
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _discover_voices() -> dict[str, Path]:
|
|
|
|
|
if not VOICES_DIR.is_dir():
|
|
|
|
|
return {}
|
|
|
|
|
return {p.stem: p for p in sorted(VOICES_DIR.glob("*.wav"))}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
VOICES = _discover_voices()
|
|
|
|
|
|
2026-07-10 21:41:44 +02:00
|
|
|
|
|
|
|
|
def _get_model(lang: str):
|
|
|
|
|
key = "en" if lang == "en" else "multi"
|
|
|
|
|
if key not in _model_cache:
|
|
|
|
|
_model_cache[key] = tts.load_model(lang, _DEVICE, t3_model="v3")
|
|
|
|
|
return _model_cache[key]
|
|
|
|
|
|
|
|
|
|
|
2026-07-10 22:38:03 +02:00
|
|
|
def _resolve_voice(voice: str) -> tuple[str, str | None]:
|
|
|
|
|
"""Map the incoming `voice` string to (language, audio_prompt_path|None)."""
|
|
|
|
|
if voice in VOICES:
|
|
|
|
|
return VOICE_LANG_OVERRIDES.get(voice, CLONE_LANG), str(VOICES[voice])
|
|
|
|
|
if voice in tts.SUPPORTED_LANGS:
|
|
|
|
|
return voice, None
|
|
|
|
|
if "default" in VOICES:
|
|
|
|
|
return CLONE_LANG, str(VOICES["default"])
|
|
|
|
|
return "de", None
|
|
|
|
|
|
|
|
|
|
|
2026-07-10 21:41:44 +02:00
|
|
|
class SpeechRequest(BaseModel):
|
|
|
|
|
model: str | None = None
|
2026-07-10 22:38:03 +02:00
|
|
|
voice: str = "default"
|
2026-07-10 21:41:44 +02:00
|
|
|
input: str
|
|
|
|
|
response_format: str = "mp3"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
_FORMAT_CONTENT_TYPE = {
|
|
|
|
|
"mp3": "audio/mpeg",
|
|
|
|
|
"wav": "audio/wav",
|
|
|
|
|
"opus": "audio/opus",
|
|
|
|
|
"flac": "audio/flac",
|
|
|
|
|
"aac": "audio/aac",
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@app.get("/health")
|
|
|
|
|
def health():
|
2026-07-10 22:38:03 +02:00
|
|
|
return {"status": "ok", "device": _DEVICE, "voices": sorted(VOICES.keys())}
|
2026-07-10 21:41:44 +02:00
|
|
|
|
|
|
|
|
|
|
|
|
|
@app.get("/audio/voices")
|
|
|
|
|
def voices():
|
2026-07-10 22:38:03 +02:00
|
|
|
cloned = [
|
|
|
|
|
{
|
|
|
|
|
"id": name,
|
|
|
|
|
"name": name,
|
|
|
|
|
"gender": "NEUTRAL",
|
|
|
|
|
"language_code": VOICE_LANG_OVERRIDES.get(name, CLONE_LANG),
|
|
|
|
|
"description": f"Cloned voice from {path.name}",
|
|
|
|
|
}
|
|
|
|
|
for name, path in VOICES.items()
|
|
|
|
|
]
|
|
|
|
|
langs = [
|
|
|
|
|
{"id": lang, "name": lang, "gender": "NEUTRAL", "language_code": lang,
|
|
|
|
|
"description": "Built-in voice (no cloning)"}
|
|
|
|
|
for lang in sorted(tts.SUPPORTED_LANGS)
|
|
|
|
|
]
|
|
|
|
|
return {"voices": cloned + langs}
|
2026-07-10 21:41:44 +02:00
|
|
|
|
|
|
|
|
|
|
|
|
|
@app.post("/audio/speech")
|
|
|
|
|
def speech(req: SpeechRequest):
|
2026-07-10 22:38:03 +02:00
|
|
|
lang, voice_path = _resolve_voice(req.voice)
|
2026-07-10 21:41:44 +02:00
|
|
|
|
|
|
|
|
raw = tts.clean_raw_text(req.input)
|
|
|
|
|
raw_chunks = tts.split_into_sentences(raw, max_len=400)
|
|
|
|
|
chunks = [tts.preprocess_tts_text(c, lang=lang, pronunciation_dict=None) for c in raw_chunks]
|
|
|
|
|
chunks = [c for c in chunks if c.strip()]
|
|
|
|
|
if not chunks:
|
|
|
|
|
raise HTTPException(status_code=422, detail="Kein synthetisierbarer Text übrig.")
|
|
|
|
|
|
|
|
|
|
model, model_kind, sr = _get_model(lang)
|
|
|
|
|
|
|
|
|
|
wavs = []
|
|
|
|
|
for chunk in chunks:
|
2026-07-10 22:38:03 +02:00
|
|
|
wavs.append(tts.generate_chunk(model, model_kind, chunk, lang, voice_path))
|
2026-07-10 21:41:44 +02:00
|
|
|
final = wavs[0] if len(wavs) == 1 else torch.cat(wavs, dim=-1)
|
|
|
|
|
|
|
|
|
|
with tempfile.NamedTemporaryFile(suffix=".wav", delete=False) as wav_tmp:
|
|
|
|
|
wav_path = wav_tmp.name
|
|
|
|
|
ta.save(wav_path, final, sr)
|
|
|
|
|
|
|
|
|
|
fmt = req.response_format if req.response_format in _FORMAT_CONTENT_TYPE else "wav"
|
|
|
|
|
try:
|
|
|
|
|
if fmt == "wav":
|
|
|
|
|
audio_bytes = Path(wav_path).read_bytes()
|
|
|
|
|
else:
|
|
|
|
|
out_path = wav_path.replace(".wav", f".{fmt}")
|
|
|
|
|
subprocess.run(
|
|
|
|
|
["ffmpeg", "-y", "-i", wav_path, out_path],
|
|
|
|
|
check=True, capture_output=True,
|
|
|
|
|
)
|
|
|
|
|
audio_bytes = Path(out_path).read_bytes()
|
|
|
|
|
os.unlink(out_path)
|
|
|
|
|
finally:
|
|
|
|
|
os.unlink(wav_path)
|
|
|
|
|
|
|
|
|
|
return Response(content=audio_bytes, media_type=_FORMAT_CONTENT_TYPE[fmt])
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
if __name__ == "__main__":
|
|
|
|
|
import uvicorn
|
2026-07-10 22:38:03 +02:00
|
|
|
print(f"[chatterbox-tts] Registrierte Stimmen: {sorted(VOICES.keys()) or '(keine, nur Sprachcodes)'}")
|
2026-07-10 21:41:44 +02:00
|
|
|
uvicorn.run(app, host=os.environ.get("TTS_HOST", "0.0.0.0"),
|
|
|
|
|
port=int(os.environ.get("TTS_PORT", "8901")))
|