feat: initial voice assistant implementation
5-state machine (LAUSCH/AUFNAHME/GENERATOR/VORLESE/DIALOG) with: - Wake-word detection via openwakeword (hey_jarvis) - Speech-to-text via faster-whisper large-v3 (GPU) - LLM streaming via OpenAI-compatible backends (llama.cpp/ollama/openai) - TTS via piper with automatic language detection and voice selection - Selectable mic input (ReSpeaker, MOTU M2, camera, bluetooth, index) - Selectable audio output sink - Control words during playback (Stopp/Pause/Weiter/Noch einmal/…) - "Stopp" abort in all active states → LAUSCH Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
This commit is contained in:
commit
53b0dd4055
5 changed files with 1442 additions and 0 deletions
40
.gitignore
vendored
Normal file
40
.gitignore
vendored
Normal file
|
|
@ -0,0 +1,40 @@
|
|||
# Python
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
*.pyo
|
||||
*.pyd
|
||||
.Python
|
||||
*.egg-info/
|
||||
dist/
|
||||
build/
|
||||
*.egg
|
||||
.eggs/
|
||||
|
||||
# Virtuelle Umgebungen
|
||||
.venv/
|
||||
venv/
|
||||
env/
|
||||
.env
|
||||
|
||||
# IDE / Editor
|
||||
.idea/
|
||||
.vscode/
|
||||
*.swp
|
||||
*.swo
|
||||
*~
|
||||
|
||||
# Claude Code
|
||||
.claude/
|
||||
|
||||
# Piper TTS — Sprachmodelle (groß, nicht versionieren)
|
||||
*.onnx
|
||||
*.onnx.json
|
||||
|
||||
# Audio-Testdateien
|
||||
*.wav
|
||||
*.mp3
|
||||
*.flac
|
||||
*.ogg
|
||||
|
||||
# Logs
|
||||
*.log
|
||||
871
assistant.py
Executable file
871
assistant.py
Executable file
|
|
@ -0,0 +1,871 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Sprachassistent mit 5-Zustands-Maschine
|
||||
|
||||
Zustände:
|
||||
LAUSCH → wartet auf "Hey Jarvis"
|
||||
AUFNAHME → nimmt User-Sprache auf
|
||||
GENERATOR → sagt "Moment …", schickt Transkript an LLM
|
||||
VORLESE → liest Antwort vor (Steuer-Wörter aktiv)
|
||||
DIALOG → kurzes Zeitfenster für Rückfrage ohne Wake-Word
|
||||
|
||||
Steuer-Wörter (nur in VORLESE, "bitte" wird immer ignoriert):
|
||||
"Stopp" → Abbruch → DIALOG
|
||||
"Pause" → Wiedergabe anhalten (SIGSTOP)
|
||||
"Weiter" → Weiterspielen ab Satzanfang
|
||||
"Noch einmal" → letzten Satz wiederholen
|
||||
"Absatz noch einmal" → zum Absatzanfang
|
||||
"Alles von vorn" → von Anfang an
|
||||
|
||||
Verwendung:
|
||||
python3 assistant.py # llama.cpp, ReSpeaker
|
||||
python3 assistant.py --list-mics
|
||||
python3 assistant.py --no-wakeword --silence-sec 3.0
|
||||
python3 assistant.py --backend ollama --model llama3.2
|
||||
python3 assistant.py --backend openai --model gpt-4o-mini
|
||||
python3 assistant.py --mic motu --out motu
|
||||
|
||||
GPU-Belegung:
|
||||
GPU 1 (phys.) RTX 3090 — llama.cpp / Port 8001
|
||||
GPU 2 (phys.) RTX 3090 — Whisper (via --gpu 1 in CUDA_VISIBLE_DEVICES=1,2)
|
||||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
import sys
|
||||
import signal
|
||||
import queue
|
||||
import threading
|
||||
import warnings
|
||||
import argparse
|
||||
import subprocess
|
||||
import time
|
||||
from enum import Enum, auto
|
||||
|
||||
import numpy as np
|
||||
import sounddevice as sd
|
||||
from scipy.signal import resample_poly
|
||||
from faster_whisper import WhisperModel
|
||||
|
||||
warnings.filterwarnings("ignore", category=UserWarning, module="onnxruntime")
|
||||
from openwakeword.model import Model as WakeWordModel # noqa: E402
|
||||
|
||||
import openai
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from speak import (resolve_sink, get_voice, synth_bytes, # noqa: E402
|
||||
detect_lang, voice_for_lang, LANG_VOICES,
|
||||
set_en_variant, EN_VARIANT_VOICES)
|
||||
from piper.voice import PiperVoice # noqa: E402
|
||||
from mic import resolve_mic, print_mic_list, DEFAULT_MIC # noqa: E402
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Zustände
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
class State(Enum):
|
||||
LAUSCH = auto()
|
||||
AUFNAHME = auto()
|
||||
GENERATOR = auto()
|
||||
VORLESE = auto()
|
||||
DIALOG = auto()
|
||||
|
||||
STATE_LABEL = {
|
||||
State.LAUSCH: "LAUSCH",
|
||||
State.AUFNAHME: "AUFNAHME",
|
||||
State.GENERATOR: "GENERATOR",
|
||||
State.VORLESE: "VORLESE",
|
||||
State.DIALOG: "DIALOG",
|
||||
}
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Steuer-Befehle
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
CMD_STOP = "STOP"
|
||||
CMD_PAUSE = "PAUSE"
|
||||
CMD_RESUME = "RESUME"
|
||||
CMD_REPEAT = "REPEAT"
|
||||
CMD_PARAGRAPH = "PARAGRAPH"
|
||||
CMD_RESTART = "RESTART"
|
||||
|
||||
# Längste Muster zuerst (Präfix-Priorität)
|
||||
CMD_PATTERNS: list[tuple[str, str]] = [
|
||||
("absatz noch einmal", CMD_PARAGRAPH),
|
||||
("alles von vorn", CMD_RESTART),
|
||||
("noch einmal", CMD_REPEAT),
|
||||
("stopp", CMD_STOP),
|
||||
("stop", CMD_STOP),
|
||||
("pause", CMD_PAUSE),
|
||||
("weiter", CMD_RESUME),
|
||||
]
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Konstanten
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
WHISPER_RATE = 16000
|
||||
WW_CHUNK_16K = 1280 # openwakeword: 80 ms @ 16 kHz
|
||||
WW_COOLDOWN_SEC = 2.0 # Sekunden nach LAUSCH-Eintritt ohne Trigger
|
||||
SILENCE_RMS = 0.008
|
||||
MAX_RECORD_SEC = 60
|
||||
STOP_CHECK_SEC = 2.0 # Intervall Stop-Wort-Check in AUFNAHME
|
||||
STOP_WINDOW_SEC = 4.0 # Rollierendes Fenster Stop-Wort
|
||||
CTRL_CHECK_SEC = 0.5 # Intervall Steuer-Wort-Check in VORLESE
|
||||
CTRL_WINDOW_SEC = 3.0 # Rollierendes Fenster Steuer-Wort
|
||||
|
||||
BACKENDS: dict[str, dict] = {
|
||||
"llama": {"base_url": "http://localhost:8001/v1", "api_key": "none",
|
||||
"default_model": None},
|
||||
"ollama": {"base_url": "http://localhost:11434/v1", "api_key": "ollama",
|
||||
"default_model": "llama3"},
|
||||
"openai": {"base_url": None, "api_key": os.getenv("OPENAI_API_KEY", ""),
|
||||
"default_model": "gpt-4o-mini"},
|
||||
}
|
||||
|
||||
DEFAULT_SYSTEM = (
|
||||
"Du bist ein hilfreicher Sprachassistent. Antworte präzise und natürlich. "
|
||||
"Deine Antworten werden direkt als Sprache ausgegeben, also verwende "
|
||||
"keine Markdown-Formatierung, keine Aufzählungszeichen und keine Code-Blöcke."
|
||||
)
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Hilfsfunktionen
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def normalize_cmd(text: str) -> str:
|
||||
"""Entfernt 'bitte' und normalisiert für Befehlsvergleich."""
|
||||
text = re.sub(r'\bbitte\b', '', text, flags=re.IGNORECASE)
|
||||
return re.sub(r'\s+', ' ', text).strip().lower()
|
||||
|
||||
|
||||
def detect_command(text: str) -> str | None:
|
||||
norm = normalize_cmd(text)
|
||||
for pattern, cmd in CMD_PATTERNS:
|
||||
if pattern in norm:
|
||||
return cmd
|
||||
return None
|
||||
|
||||
|
||||
def show_status(state: State, info: str = "") -> None:
|
||||
suffix = f" {info}" if info else ""
|
||||
print(f"\r\033[K[{STATE_LABEL[state]}]{suffix}", end="", flush=True)
|
||||
|
||||
|
||||
def status_newline() -> None:
|
||||
print(flush=True)
|
||||
|
||||
|
||||
def quick_play(text: str, voice: PiperVoice, out: str) -> None:
|
||||
"""Spielt kurzen Text synchron (blockierend, kein Hintergrund-Thread)."""
|
||||
sink = resolve_sink(out)
|
||||
cmd = ["paplay", "--raw",
|
||||
f"--rate={voice.config.sample_rate}",
|
||||
"--channels=1", "--format=s16le"]
|
||||
if sink:
|
||||
cmd += [f"--device={sink}"]
|
||||
audio = synth_bytes(text, voice)
|
||||
proc = subprocess.Popen(cmd, stdin=subprocess.PIPE)
|
||||
proc.stdin.write(audio)
|
||||
proc.stdin.close()
|
||||
proc.wait()
|
||||
|
||||
|
||||
def play_chime(out: str) -> None:
|
||||
rate = 22050
|
||||
t = np.linspace(0, 0.2, int(rate * 0.2), endpoint=False)
|
||||
beep = (np.sin(2 * np.pi * 880 * t) * 0.35 * 32767).astype(np.int16)
|
||||
sink = resolve_sink(out)
|
||||
cmd = ["paplay", "--raw", f"--rate={rate}", "--channels=1", "--format=s16le"]
|
||||
if sink:
|
||||
cmd += [f"--device={sink}"]
|
||||
proc = subprocess.Popen(cmd, stdin=subprocess.PIPE)
|
||||
proc.stdin.write(beep.tobytes())
|
||||
proc.stdin.close()
|
||||
proc.wait()
|
||||
|
||||
|
||||
def _drain_audio(audio_q: queue.Queue, ctrl_audio_q: queue.Queue) -> None:
|
||||
"""Leert audio_q und leitet Chunks an ctrl_audio_q weiter."""
|
||||
while not audio_q.empty():
|
||||
try:
|
||||
ctrl_audio_q.put(audio_q.get_nowait())
|
||||
except queue.Empty:
|
||||
break
|
||||
|
||||
|
||||
def get_model_name(args: argparse.Namespace, client: openai.OpenAI) -> str:
|
||||
if args.model:
|
||||
return args.model
|
||||
if args.backend == "llama":
|
||||
try:
|
||||
models = client.models.list()
|
||||
if models.data:
|
||||
return models.data[0].id
|
||||
except Exception:
|
||||
pass
|
||||
return "default"
|
||||
return BACKENDS[args.backend]["default_model"] or "default"
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Steuer-Wort-Monitor (Hintergrund-Thread während VORLESE)
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def ctrl_monitor_fn(
|
||||
ctrl_audio_q: queue.Queue,
|
||||
ctrl_cmd_q: queue.Queue,
|
||||
stop_now: threading.Event,
|
||||
pause_now: threading.Event,
|
||||
stop_monitor: threading.Event,
|
||||
whisper: WhisperModel,
|
||||
capture_rate: int,
|
||||
) -> None:
|
||||
buffer = np.empty(0, dtype=np.float32)
|
||||
window_samples = int(CTRL_WINDOW_SEC * capture_rate)
|
||||
check_samples = int(CTRL_CHECK_SEC * capture_rate)
|
||||
accumulated = 0
|
||||
|
||||
while not stop_monitor.is_set():
|
||||
try:
|
||||
chunk = ctrl_audio_q.get(timeout=0.1)
|
||||
except queue.Empty:
|
||||
continue
|
||||
|
||||
buffer = np.concatenate([buffer, chunk])
|
||||
if len(buffer) > window_samples:
|
||||
buffer = buffer[-window_samples:]
|
||||
accumulated += len(chunk)
|
||||
|
||||
if accumulated < check_samples:
|
||||
continue
|
||||
accumulated = 0
|
||||
|
||||
window_16k = resample_poly(buffer, WHISPER_RATE, capture_rate).astype(np.float32)
|
||||
segs, _ = whisper.transcribe(window_16k, language="de",
|
||||
vad_filter=False, beam_size=1)
|
||||
text = " ".join(s.text for s in segs).strip()
|
||||
if not text:
|
||||
continue
|
||||
|
||||
cmd = detect_command(text)
|
||||
if cmd:
|
||||
print(f"\n[Steuerbefehl: {cmd} ← \"{normalize_cmd(text)}\"]", flush=True)
|
||||
ctrl_cmd_q.put(cmd)
|
||||
if cmd == CMD_STOP:
|
||||
stop_now.set()
|
||||
elif cmd == CMD_PAUSE and not pause_now.is_set():
|
||||
pause_now.set()
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Hauptschleife
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def run(args: argparse.Namespace) -> None:
|
||||
# ── Geräte & Modelle laden ────────────────────────────────────────────
|
||||
device_idx, capture_rate, device_label = resolve_mic(args.mic)
|
||||
capture_chunk = round(WW_CHUNK_16K * capture_rate / WHISPER_RATE)
|
||||
|
||||
print("Lade Whisper …", flush=True)
|
||||
whisper = WhisperModel(args.whisper_model, device="cuda",
|
||||
device_index=args.gpu, compute_type="float16")
|
||||
|
||||
print("Lade Piper-Stimme …", flush=True)
|
||||
tts_voice = get_voice(args.voice)
|
||||
|
||||
if not args.no_wakeword:
|
||||
print("Lade Wake-Word-Modell …", flush=True)
|
||||
ww_model = WakeWordModel()
|
||||
else:
|
||||
ww_model = None
|
||||
|
||||
backend_cfg = BACKENDS[args.backend]
|
||||
client = openai.OpenAI(
|
||||
base_url=backend_cfg["base_url"],
|
||||
api_key =backend_cfg["api_key"] or "none",
|
||||
)
|
||||
model_name = get_model_name(args, client)
|
||||
|
||||
print(f"Mikrofon: {device_label} ({capture_rate} Hz)")
|
||||
print(f"Backend : {args.backend} | Modell: {model_name}")
|
||||
print(f"Stimme : {args.voice} | Ausgabe: {args.out} | Englisch: {EN_VARIANT_VOICES[args.en_variant]}")
|
||||
print(f"Sprache : {args.lang} | Verlauf: {args.history} Turns")
|
||||
print(f"Stille : {args.silence_sec}s | Dialog: {args.dialog_timeout}s | Stop-Wort: {args.stop_word!r}")
|
||||
|
||||
set_en_variant(args.en_variant)
|
||||
|
||||
# ── Shared State ──────────────────────────────────────────────────────
|
||||
history: list[dict] = []
|
||||
audio_q: queue.Queue = queue.Queue()
|
||||
stop_word_detected: threading.Event = threading.Event()
|
||||
stop_check_busy: threading.Event = threading.Event()
|
||||
abort_to_lausch: threading.Event = threading.Event() # Stopp in AUFNAHME/GENERATOR/DIALOG
|
||||
abort_dialog_busy: threading.Event = threading.Event() # verhindert parallele DIALOG-Checks
|
||||
dialog_buf: list[np.ndarray] = [] # rollierende Aufnahme für DIALOG-Check
|
||||
dialog_check_n: list[int] = [0] # Chunk-Zähler für DIALOG-Check
|
||||
|
||||
# ── Audio-Callback (läuft in PortAudio-Thread) ────────────────────────
|
||||
def audio_callback(indata: np.ndarray, frames: int, t, status) -> None:
|
||||
if status and "input overflow" not in str(status):
|
||||
print(f"[audio] {status}", file=sys.stderr)
|
||||
audio_q.put(indata[:, 0].copy())
|
||||
|
||||
# ── Stop-Wort- / Stopp-Check für AUFNAHME (Hintergrund-Thread) ──────────
|
||||
def _check_stop_word(buf: np.ndarray) -> None:
|
||||
try:
|
||||
w16k = resample_poly(buf, WHISPER_RATE, capture_rate).astype(np.float32)
|
||||
segs, _ = whisper.transcribe(w16k, language=None, vad_filter=False, beam_size=1)
|
||||
text = " ".join(s.text for s in segs).strip()
|
||||
if detect_command(text) == CMD_STOP:
|
||||
print("\n[Stopp erkannt — Aufnahme abbrechen]", flush=True)
|
||||
abort_to_lausch.set()
|
||||
elif args.stop_word and args.stop_word.lower() in text.lower().split()[-5:]:
|
||||
print(f"\n[Stop-Wort '{args.stop_word}' erkannt]", flush=True)
|
||||
stop_word_detected.set()
|
||||
finally:
|
||||
stop_check_busy.clear()
|
||||
|
||||
# ── Stopp-Check für DIALOG (Hintergrund-Thread) ───────────────────────
|
||||
def _check_abort_dialog(buf: np.ndarray) -> None:
|
||||
try:
|
||||
w16k = resample_poly(buf, WHISPER_RATE, capture_rate).astype(np.float32)
|
||||
segs, _ = whisper.transcribe(w16k, language=None, vad_filter=False, beam_size=1)
|
||||
text = " ".join(s.text for s in segs).strip()
|
||||
if detect_command(text) == CMD_STOP:
|
||||
print("\n[Stopp erkannt in DIALOG]", flush=True)
|
||||
abort_to_lausch.set()
|
||||
finally:
|
||||
abort_dialog_busy.clear()
|
||||
|
||||
# ── GENERATOR + VORLESE (blockierend, läuft im Haupt-Thread) ─────────
|
||||
def _handle_turn(audio_data: np.ndarray) -> bool:
|
||||
"""Transkription → LLM → TTS. Blockiert bis Vorlesen beendet.
|
||||
Gibt True zurück wenn etwas verstanden wurde, sonst False."""
|
||||
nonlocal history
|
||||
|
||||
# Sofort-Abbruch: wurde "Stopp" schon in AUFNAHME erkannt?
|
||||
if abort_to_lausch.is_set():
|
||||
abort_to_lausch.clear()
|
||||
return False
|
||||
|
||||
# ── GENERATOR: Transkription ──────────────────────────────────────
|
||||
show_status(State.GENERATOR, "Transkription …")
|
||||
status_newline()
|
||||
|
||||
audio_16k = resample_poly(audio_data, WHISPER_RATE, capture_rate).astype(np.float32)
|
||||
segments, _ = whisper.transcribe(
|
||||
audio_16k, language=args.lang, vad_filter=True,
|
||||
vad_parameters={"min_silence_duration_ms": 500}, beam_size=5,
|
||||
)
|
||||
user_text = " ".join(s.text.strip() for s in segments).strip()
|
||||
|
||||
if args.stop_word and user_text:
|
||||
user_text = re.sub(
|
||||
r'[,.]?\s*\b' + re.escape(args.stop_word) + r'[.,!?]?\s*$',
|
||||
'', user_text, flags=re.IGNORECASE
|
||||
).strip()
|
||||
|
||||
# Nach Transkription: Stopp nochmals prüfen
|
||||
if abort_to_lausch.is_set():
|
||||
abort_to_lausch.clear()
|
||||
return False
|
||||
|
||||
if not user_text:
|
||||
print("(nichts verstanden)", flush=True)
|
||||
return False
|
||||
|
||||
print(f"Du: {user_text}", flush=True)
|
||||
|
||||
history.append({"role": "user", "content": user_text})
|
||||
if len(history) > args.history * 2:
|
||||
del history[: len(history) - args.history * 2]
|
||||
|
||||
messages = [{"role": "system", "content": args.system}] + history
|
||||
|
||||
# Quittungssatz sprechen
|
||||
show_status(State.GENERATOR, args.ack_text)
|
||||
quick_play(args.ack_text, tts_voice, args.out)
|
||||
status_newline()
|
||||
|
||||
# Abort-Monitor für LLM-Phase (liest audio_q, kein Whisper-Konflikt jetzt)
|
||||
_abort_gen_stop = threading.Event()
|
||||
def _abort_gen_monitor() -> None:
|
||||
buf_g = np.empty(0, dtype=np.float32)
|
||||
window = int(CTRL_WINDOW_SEC * capture_rate)
|
||||
check = int(CTRL_CHECK_SEC * capture_rate)
|
||||
acc = 0
|
||||
while not _abort_gen_stop.is_set():
|
||||
try:
|
||||
chunk = audio_q.get(timeout=0.1)
|
||||
except queue.Empty:
|
||||
continue
|
||||
buf_g = np.concatenate([buf_g, chunk])[-window:]
|
||||
acc += len(chunk)
|
||||
if acc < check:
|
||||
continue
|
||||
acc = 0
|
||||
w16k = resample_poly(buf_g, WHISPER_RATE, capture_rate).astype(np.float32)
|
||||
segs, _ = whisper.transcribe(w16k, language=None, vad_filter=False, beam_size=1)
|
||||
text = " ".join(s.text for s in segs).strip()
|
||||
if detect_command(text) == CMD_STOP:
|
||||
print("\n[Stopp erkannt in GENERATOR]", flush=True)
|
||||
abort_to_lausch.set()
|
||||
break
|
||||
abort_gen_th = threading.Thread(target=_abort_gen_monitor, daemon=True)
|
||||
abort_gen_th.start()
|
||||
|
||||
# ── LLM-Stream + Synthese (Hintergrund-Threads) ───────────────────
|
||||
sentences: list[str] = []
|
||||
sentence_bytes: list[bytes | None] = []
|
||||
paragraph_starts: list[int] = [0]
|
||||
full_parts: list[str] = []
|
||||
llm_done = threading.Event()
|
||||
|
||||
# Spracherkennung: sobald ≥60 Zeichen LLM-Text vorliegen
|
||||
response_voice = [tts_voice] # mutable holder
|
||||
voice_load_ok = [True] # False → Voice-Download fehlgeschlagen
|
||||
lang_detected_ev = threading.Event()
|
||||
|
||||
def _llm_thread() -> None:
|
||||
token_buffer = ""
|
||||
try:
|
||||
stream = client.chat.completions.create(
|
||||
model=model_name, messages=messages, stream=True
|
||||
)
|
||||
for llm_chunk in stream:
|
||||
delta = (llm_chunk.choices[0].delta.content or "") \
|
||||
if llm_chunk.choices else ""
|
||||
token_buffer += delta
|
||||
full_parts.append(delta)
|
||||
|
||||
# Spracherkennung auslösen sobald genug Text da ist
|
||||
if not lang_detected_ev.is_set():
|
||||
accumulated = "".join(full_parts)
|
||||
if len(accumulated) >= 60:
|
||||
lang = detect_lang(accumulated)
|
||||
try:
|
||||
v = voice_for_lang(lang, args.voice)
|
||||
if v is not tts_voice:
|
||||
voice_name = LANG_VOICES.get(lang or "", args.voice)
|
||||
print(f"[Sprache erkannt: {lang} → {voice_name}]",
|
||||
flush=True)
|
||||
response_voice[0] = v
|
||||
except Exception as exc:
|
||||
print(f"[Stimme nicht geladen: {exc}]",
|
||||
file=sys.stderr, flush=True)
|
||||
voice_load_ok[0] = False
|
||||
finally:
|
||||
lang_detected_ev.set()
|
||||
|
||||
# Absatz-Grenze → neuen Absatz markieren
|
||||
while "\n\n" in token_buffer:
|
||||
idx = token_buffer.index("\n\n")
|
||||
fragment = token_buffer[:idx].strip()
|
||||
token_buffer = token_buffer[idx + 2:]
|
||||
if len(fragment) > 3:
|
||||
print(f" → {fragment}")
|
||||
sentences.append(fragment)
|
||||
sentence_bytes.append(None)
|
||||
if sentences:
|
||||
paragraph_starts.append(len(sentences))
|
||||
|
||||
# Satz-Grenzen extrahieren
|
||||
while True:
|
||||
m = re.search(r'(?<=[.!?])\s+(?=[^\s])', token_buffer)
|
||||
if not m:
|
||||
break
|
||||
s = token_buffer[:m.start() + 1].strip()
|
||||
token_buffer = token_buffer[m.end():]
|
||||
if len(s) > 3:
|
||||
print(f" → {s}")
|
||||
sentences.append(s)
|
||||
sentence_bytes.append(None)
|
||||
|
||||
except Exception as exc:
|
||||
print(f"[LLM-Fehler] {exc}", file=sys.stderr)
|
||||
|
||||
remainder = token_buffer.strip()
|
||||
if len(remainder) > 3:
|
||||
print(f" → {remainder}")
|
||||
sentences.append(remainder)
|
||||
sentence_bytes.append(None)
|
||||
|
||||
# Fallback: LLM fertig aber <60 Zeichen → Default-Stimme
|
||||
if not lang_detected_ev.is_set():
|
||||
lang = detect_lang("".join(full_parts))
|
||||
try:
|
||||
response_voice[0] = voice_for_lang(lang, args.voice)
|
||||
except Exception as exc:
|
||||
print(f"[Stimme nicht geladen: {exc}]",
|
||||
file=sys.stderr, flush=True)
|
||||
voice_load_ok[0] = False
|
||||
finally:
|
||||
lang_detected_ev.set()
|
||||
|
||||
llm_done.set()
|
||||
|
||||
def _synth_thread() -> None:
|
||||
lang_detected_ev.wait() # wartet auf Spracherkennung
|
||||
if not voice_load_ok[0]:
|
||||
return # keine Synthese mit falscher Stimme
|
||||
voice_to_use = response_voice[0]
|
||||
idx = 0
|
||||
while True:
|
||||
if idx < len(sentences) and sentence_bytes[idx] is None:
|
||||
sentence_bytes[idx] = synth_bytes(sentences[idx], voice_to_use)
|
||||
idx += 1
|
||||
elif llm_done.is_set() and idx >= len(sentences):
|
||||
break
|
||||
else:
|
||||
time.sleep(0.01)
|
||||
|
||||
threading.Thread(target=_llm_thread, daemon=True).start()
|
||||
threading.Thread(target=_synth_thread, daemon=True).start()
|
||||
|
||||
# ── VORLESE ───────────────────────────────────────────────────────
|
||||
show_status(State.VORLESE)
|
||||
|
||||
ctrl_audio_q: queue.Queue = queue.Queue()
|
||||
ctrl_cmd_q: queue.Queue = queue.Queue()
|
||||
stop_now = threading.Event()
|
||||
pause_now = threading.Event()
|
||||
stop_mon = threading.Event()
|
||||
|
||||
threading.Thread(
|
||||
target=ctrl_monitor_fn,
|
||||
args=(ctrl_audio_q, ctrl_cmd_q, stop_now, pause_now,
|
||||
stop_mon, whisper, capture_rate),
|
||||
daemon=True,
|
||||
).start()
|
||||
|
||||
# paplay-Befehlsbasis: nach Spracherkennung (richtige Sample-Rate)
|
||||
lang_detected_ev.wait()
|
||||
|
||||
# Abort-Monitor beenden bevor ctrl_monitor Whisper nutzt (kein GPU-Konflikt)
|
||||
_abort_gen_stop.set()
|
||||
abort_gen_th.join(timeout=0.6)
|
||||
|
||||
if abort_to_lausch.is_set():
|
||||
abort_to_lausch.clear()
|
||||
stop_mon.set()
|
||||
status_newline()
|
||||
return False
|
||||
|
||||
# ── Voice-Load fehlgeschlagen → Vorlesen überspringen ────────────
|
||||
if not voice_load_ok[0]:
|
||||
llm_done.wait() # warten bis LLM fertig (vollständiger Text)
|
||||
stop_mon.set()
|
||||
status_newline()
|
||||
print("[Stimme nicht verfügbar — Text:]", flush=True)
|
||||
print("".join(full_parts), flush=True)
|
||||
quick_play(
|
||||
"Entschuldigung, die Stimme für diese Sprache konnte nicht geladen werden. "
|
||||
"Der Text ist auf dem Bildschirm zu sehen.",
|
||||
tts_voice, args.out,
|
||||
)
|
||||
history.append({"role": "assistant", "content": "".join(full_parts)})
|
||||
return True
|
||||
|
||||
sink = resolve_sink(args.out)
|
||||
paplay_base = [
|
||||
"paplay", "--raw",
|
||||
f"--rate={response_voice[0].config.sample_rate}",
|
||||
"--channels=1", "--format=s16le",
|
||||
]
|
||||
if sink:
|
||||
paplay_base += [f"--device={sink}"]
|
||||
|
||||
def _wait_for_resume_or_stop() -> str:
|
||||
"""Blockiert im Pause-Zustand bis RESUME oder STOP eintrifft."""
|
||||
while True:
|
||||
_drain_audio(audio_q, ctrl_audio_q)
|
||||
try:
|
||||
cmd2 = ctrl_cmd_q.get(timeout=0.1)
|
||||
except queue.Empty:
|
||||
continue
|
||||
if cmd2 in (CMD_RESUME, CMD_STOP):
|
||||
return cmd2
|
||||
|
||||
play_idx = 0
|
||||
|
||||
while True:
|
||||
# Fertig?
|
||||
if play_idx >= len(sentences) and llm_done.is_set():
|
||||
break
|
||||
|
||||
# Warte auf synthetisierte Bytes (draint dabei audio_q)
|
||||
if play_idx >= len(sentence_bytes) or sentence_bytes[play_idx] is None:
|
||||
_drain_audio(audio_q, ctrl_audio_q)
|
||||
time.sleep(0.02)
|
||||
continue
|
||||
|
||||
# Befehl zwischen Sätzen prüfen (non-blocking)
|
||||
try:
|
||||
cmd = ctrl_cmd_q.get_nowait()
|
||||
except queue.Empty:
|
||||
cmd = None
|
||||
|
||||
if cmd == CMD_STOP:
|
||||
break
|
||||
elif cmd == CMD_REPEAT:
|
||||
play_idx = max(0, play_idx - 1)
|
||||
continue
|
||||
elif cmd == CMD_PARAGRAPH:
|
||||
para = 0
|
||||
for ps in paragraph_starts:
|
||||
if ps <= play_idx:
|
||||
para = ps
|
||||
play_idx = para
|
||||
continue
|
||||
elif cmd == CMD_RESTART:
|
||||
play_idx = 0
|
||||
continue
|
||||
|
||||
# Satz abspielen
|
||||
label = sentences[play_idx][:50].replace('\n', ' ')
|
||||
show_status(State.VORLESE, f"[{play_idx + 1}/{len(sentences)}] {label}")
|
||||
|
||||
proc = subprocess.Popen(paplay_base, stdin=subprocess.PIPE)
|
||||
proc.stdin.write(sentence_bytes[play_idx])
|
||||
proc.stdin.close()
|
||||
|
||||
finished_normally = True
|
||||
while proc.poll() is None:
|
||||
_drain_audio(audio_q, ctrl_audio_q)
|
||||
if stop_now.is_set():
|
||||
proc.kill()
|
||||
finished_normally = False
|
||||
break
|
||||
if pause_now.is_set():
|
||||
proc.send_signal(signal.SIGSTOP)
|
||||
show_status(State.VORLESE, "[PAUSE — 'Weiter' zum Fortfahren]")
|
||||
cmd2 = _wait_for_resume_or_stop()
|
||||
proc.kill()
|
||||
pause_now.clear()
|
||||
if cmd2 == CMD_STOP:
|
||||
stop_now.set()
|
||||
# CMD_RESUME: finished_normally bleibt False → Satz von vorne
|
||||
finished_normally = False
|
||||
break
|
||||
time.sleep(0.03)
|
||||
|
||||
if finished_normally:
|
||||
play_idx += 1
|
||||
if stop_now.is_set():
|
||||
break
|
||||
|
||||
stop_mon.set()
|
||||
status_newline()
|
||||
history.append({"role": "assistant", "content": "".join(full_parts)})
|
||||
return True
|
||||
|
||||
# ── Zustands-Maschine ─────────────────────────────────────────────────
|
||||
state = State.AUFNAHME if args.no_wakeword else State.LAUSCH
|
||||
record_buffer: list[np.ndarray] = []
|
||||
silence_chunks = 0
|
||||
check_chunks = 0
|
||||
dialog_deadline = 0.0
|
||||
ww_cooldown_chunks = 0 # Chunks verbleibend ohne Wake-Word-Trigger
|
||||
|
||||
print("\nBereit." if args.no_wakeword
|
||||
else '\nWarte auf "Hey Jarvis" …', flush=True)
|
||||
|
||||
with sd.InputStream(
|
||||
device=device_idx, channels=1, samplerate=capture_rate,
|
||||
dtype="float32", blocksize=capture_chunk, callback=audio_callback,
|
||||
):
|
||||
while True:
|
||||
chunk_raw = audio_q.get()
|
||||
chunk_16k = resample_poly(chunk_raw, WHISPER_RATE, capture_rate).astype(np.float32)
|
||||
rms = float(np.sqrt(np.mean(chunk_raw ** 2)))
|
||||
|
||||
# ── LAUSCH ────────────────────────────────────────────────────
|
||||
if state == State.LAUSCH:
|
||||
show_status(state, 'Warte auf "Hey Jarvis" …')
|
||||
chunk_int16 = (chunk_16k * 32767).astype(np.int16)
|
||||
score = ww_model.predict(chunk_int16).get("hey_jarvis", 0)
|
||||
if ww_cooldown_chunks > 0:
|
||||
ww_cooldown_chunks -= 1
|
||||
elif score >= args.wakeword_threshold:
|
||||
status_newline()
|
||||
play_chime(args.out)
|
||||
state = State.AUFNAHME
|
||||
record_buffer = []
|
||||
silence_chunks = check_chunks = 0
|
||||
show_status(state, "Sprechen …")
|
||||
status_newline()
|
||||
|
||||
# ── AUFNAHME ──────────────────────────────────────────────────
|
||||
elif state == State.AUFNAHME:
|
||||
if abort_to_lausch.is_set():
|
||||
abort_to_lausch.clear()
|
||||
stop_check_busy.clear()
|
||||
record_buffer = []
|
||||
silence_chunks = check_chunks = 0
|
||||
status_newline()
|
||||
state = State.AUFNAHME if args.no_wakeword else State.LAUSCH
|
||||
if not args.no_wakeword:
|
||||
ww_cooldown_chunks = round(WW_COOLDOWN_SEC * capture_rate / capture_chunk)
|
||||
continue
|
||||
|
||||
record_buffer.append(chunk_raw)
|
||||
silence_chunks = (silence_chunks + 1) if rms < SILENCE_RMS else 0
|
||||
|
||||
silence_elapsed = silence_chunks * capture_chunk / capture_rate
|
||||
total_elapsed = len(record_buffer) * capture_chunk / capture_rate
|
||||
show_status(state, f"{total_elapsed:.1f}s (Stille {silence_elapsed:.1f}s)")
|
||||
|
||||
# Stop-Wort-Check im Hintergrund
|
||||
if args.stop_word:
|
||||
check_chunks += 1
|
||||
if check_chunks * capture_chunk / capture_rate >= STOP_CHECK_SEC:
|
||||
check_chunks = 0
|
||||
if not stop_check_busy.is_set():
|
||||
stop_check_busy.set()
|
||||
win = int(STOP_WINDOW_SEC * capture_rate / capture_chunk)
|
||||
snap = np.concatenate(record_buffer[-win:]).copy()
|
||||
threading.Thread(
|
||||
target=_check_stop_word, args=(snap,), daemon=True
|
||||
).start()
|
||||
|
||||
if silence_elapsed >= args.silence_sec \
|
||||
or total_elapsed >= MAX_RECORD_SEC \
|
||||
or stop_word_detected.is_set():
|
||||
stop_word_detected.clear()
|
||||
stop_check_busy.clear()
|
||||
status_newline()
|
||||
buf = np.concatenate(record_buffer)
|
||||
record_buffer = []
|
||||
silence_chunks = check_chunks = 0
|
||||
|
||||
understood = _handle_turn(buf) # blockiert während GENERATOR + VORLESE
|
||||
|
||||
# Audio aus der VORLESE-Phase verwerfen
|
||||
while not audio_q.empty():
|
||||
audio_q.get_nowait()
|
||||
|
||||
if understood:
|
||||
state = State.DIALOG
|
||||
dialog_deadline = time.time() + args.dialog_timeout
|
||||
show_status(state, f"Sprechen? ({args.dialog_timeout:.0f}s)")
|
||||
else:
|
||||
state = State.AUFNAHME if args.no_wakeword else State.LAUSCH
|
||||
if not args.no_wakeword:
|
||||
ww_cooldown_chunks = round(WW_COOLDOWN_SEC * capture_rate / capture_chunk)
|
||||
record_buffer = []
|
||||
silence_chunks = check_chunks = 0
|
||||
|
||||
# ── DIALOG ────────────────────────────────────────────────────
|
||||
elif state == State.DIALOG:
|
||||
# Sofort-Abbruch durch "Stopp"
|
||||
if abort_to_lausch.is_set():
|
||||
abort_to_lausch.clear()
|
||||
abort_dialog_busy.clear()
|
||||
dialog_buf.clear()
|
||||
dialog_check_n[0] = 0
|
||||
status_newline()
|
||||
state = State.AUFNAHME if args.no_wakeword else State.LAUSCH
|
||||
if not args.no_wakeword:
|
||||
ww_cooldown_chunks = round(WW_COOLDOWN_SEC * capture_rate / capture_chunk)
|
||||
continue
|
||||
|
||||
remaining = dialog_deadline - time.time()
|
||||
show_status(state, f"Sprechen? ({remaining:.1f}s)")
|
||||
|
||||
# Wakeword-Modell warm halten (kein Trigger im DIALOG)
|
||||
if not args.no_wakeword:
|
||||
chunk_int16 = (chunk_16k * 32767).astype(np.int16)
|
||||
ww_model.predict(chunk_int16)
|
||||
|
||||
# Periodischer Stopp-Check im Hintergrund
|
||||
dialog_buf.append(chunk_raw)
|
||||
max_dbuf = round(CTRL_WINDOW_SEC * capture_rate / capture_chunk)
|
||||
if len(dialog_buf) > max_dbuf:
|
||||
dialog_buf.pop(0)
|
||||
dialog_check_n[0] += 1
|
||||
if dialog_check_n[0] * capture_chunk / capture_rate >= CTRL_CHECK_SEC:
|
||||
dialog_check_n[0] = 0
|
||||
if not abort_dialog_busy.is_set():
|
||||
abort_dialog_busy.set()
|
||||
snap = np.concatenate(dialog_buf).copy()
|
||||
threading.Thread(
|
||||
target=_check_abort_dialog, args=(snap,), daemon=True
|
||||
).start()
|
||||
|
||||
if remaining <= 0:
|
||||
dialog_buf.clear()
|
||||
dialog_check_n[0] = 0
|
||||
abort_dialog_busy.clear()
|
||||
status_newline()
|
||||
# --no-wakeword: sofort wieder aufnahmebereit
|
||||
state = State.AUFNAHME if args.no_wakeword else State.LAUSCH
|
||||
if args.no_wakeword:
|
||||
record_buffer = []
|
||||
silence_chunks = check_chunks = 0
|
||||
show_status(state, "Sprechen …")
|
||||
status_newline()
|
||||
else:
|
||||
ww_cooldown_chunks = round(WW_COOLDOWN_SEC * capture_rate / capture_chunk)
|
||||
show_status(state, 'Warte auf "Hey Jarvis" …')
|
||||
|
||||
elif rms > SILENCE_RMS:
|
||||
dialog_buf.clear()
|
||||
dialog_check_n[0] = 0
|
||||
status_newline()
|
||||
state = State.AUFNAHME
|
||||
record_buffer = [chunk_raw]
|
||||
silence_chunks = check_chunks = 0
|
||||
show_status(state, "Sprechen …")
|
||||
status_newline()
|
||||
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
# Argumente
|
||||
# ─────────────────────────────────────────────────────────────────────────────
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
p = argparse.ArgumentParser(
|
||||
description="Sprachassistent: 5-Zustands-Maschine mit Steuer-Wörtern",
|
||||
formatter_class=argparse.ArgumentDefaultsHelpFormatter,
|
||||
)
|
||||
p.add_argument("--mic", default=DEFAULT_MIC,
|
||||
help="Eingabegerät: respeaker | motu | camera | default | bluetooth | <idx>")
|
||||
p.add_argument("--list-mics", action="store_true",
|
||||
help="Verfügbare Mikrofone anzeigen und beenden")
|
||||
p.add_argument("--backend", default="llama",
|
||||
choices=["llama", "ollama", "openai"], help="LLM-Backend")
|
||||
p.add_argument("--model", default=None,
|
||||
help="LLM-Modellname (default: auto-detect)")
|
||||
p.add_argument("--whisper-model", default="large-v3")
|
||||
p.add_argument("--gpu", type=int, default=1,
|
||||
help="CUDA-Index für Whisper (in CUDA_VISIBLE_DEVICES)")
|
||||
p.add_argument("--lang", default="de")
|
||||
p.add_argument("--voice", default="de_DE-thorsten-high")
|
||||
p.add_argument("--en-variant", default="us", choices=["us", "gb"],
|
||||
help="Englische TTS-Variante: us (en_US-ryan-high) | gb (en_GB-alan-medium)")
|
||||
p.add_argument("--out", default="default",
|
||||
help="Ausgabe-Sink: default | respeaker | hdmi | motu | bluetooth | <name>")
|
||||
p.add_argument("--system", default=DEFAULT_SYSTEM)
|
||||
p.add_argument("--history", type=int, default=10,
|
||||
help="Max. Konversations-Turns")
|
||||
p.add_argument("--wakeword-threshold", type=float, default=0.5)
|
||||
p.add_argument("--no-wakeword", action="store_true",
|
||||
help="Wake-Word deaktivieren — direkt in AUFNAHME starten")
|
||||
p.add_argument("--stop-word", default="over",
|
||||
help="Codewort zum Beenden der Aufnahme")
|
||||
p.add_argument("--silence-sec", type=float, default=5.0,
|
||||
help="Stille-Dauer bis Aufnahme endet")
|
||||
p.add_argument("--dialog-timeout", type=float, default=5.0,
|
||||
help="DIALOG-Fenster nach Vorlesen (Sekunden)")
|
||||
p.add_argument("--ack-text", default="Moment, ich antworte gleich.",
|
||||
help="Quittungssatz im GENERATOR-Zustand")
|
||||
return p.parse_args()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
args = parse_args()
|
||||
if args.list_mics:
|
||||
print_mic_list()
|
||||
sys.exit(0)
|
||||
try:
|
||||
run(args)
|
||||
except KeyboardInterrupt:
|
||||
print("\nBeendet.")
|
||||
127
mic.py
Normal file
127
mic.py
Normal file
|
|
@ -0,0 +1,127 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Mikrofon-Auswahl für sounddevice-Anwendungen.
|
||||
|
||||
Kurzname-Aliase:
|
||||
respeaker ReSpeaker XVF3800, 48 kHz (default)
|
||||
motu MOTU M2 Mikrofon-Eingang, 44100 Hz
|
||||
camera USB-Kamera-Mikrofon (VF0680), 44100 Hz
|
||||
default System-Standard-Eingabe (PipeWire)
|
||||
bluetooth erstes verbundenes Bluetooth-Eingabegerät
|
||||
<index> numerischer sounddevice-Index
|
||||
<substring> freie Substring-Suche im Gerätenamen
|
||||
|
||||
Verwendung als Bibliothek:
|
||||
from mic import resolve_mic, print_mic_list
|
||||
device_idx, capture_rate, label = resolve_mic("motu")
|
||||
|
||||
Verwendung als Skript:
|
||||
python3 mic.py # zeigt alle Eingabegeräte
|
||||
"""
|
||||
|
||||
import sys
|
||||
import sounddevice as sd
|
||||
|
||||
# Substring-Muster für bekannte Geräte (case-insensitive Suche)
|
||||
MIC_ALIASES: dict[str, str] = {
|
||||
"respeaker": "reSpeaker", # hw-Gerät kann in=0 melden, trotzdem nutzbar
|
||||
"motu": "M2:",
|
||||
"camera": "VF0680",
|
||||
}
|
||||
DEFAULT_MIC = "respeaker"
|
||||
|
||||
|
||||
def resolve_mic(spec: str) -> tuple[int, int, str]:
|
||||
"""
|
||||
Löst einen Mic-Bezeichner auf.
|
||||
Gibt (device_idx, native_samplerate, label) zurück.
|
||||
|
||||
Reihenfolge: numerischer Index → "default" → "bluetooth" → Alias → Substring.
|
||||
Alias-Suche filtert NICHT nach in-Kanälen (ReSpeaker hw meldet in=0, läuft dennoch).
|
||||
"""
|
||||
devs = sd.query_devices()
|
||||
|
||||
# Numerischer Index
|
||||
if spec.lstrip("-").isdigit():
|
||||
idx = int(spec)
|
||||
d = devs[idx]
|
||||
return idx, int(d["default_samplerate"]), d["name"]
|
||||
|
||||
# System-Standard-Eingabe
|
||||
if spec == "default":
|
||||
idx = sd.default.device[0]
|
||||
if idx < 0:
|
||||
raise ValueError("Kein System-Standard-Eingabegerät gefunden.")
|
||||
d = devs[idx]
|
||||
return idx, int(d["default_samplerate"]), f"{d['name']} (default)"
|
||||
|
||||
# Erstes Bluetooth-Eingabegerät
|
||||
if spec == "bluetooth":
|
||||
for i, d in enumerate(devs):
|
||||
if "bluez" in d["name"].lower() and d["max_input_channels"] > 0:
|
||||
return i, int(d["default_samplerate"]), d["name"]
|
||||
raise ValueError("Kein Bluetooth-Eingabegerät gefunden.")
|
||||
|
||||
# Alias → Substring-Suche ohne channels-Filter
|
||||
pattern = MIC_ALIASES.get(spec, spec)
|
||||
for i, d in enumerate(devs):
|
||||
if pattern.lower() in d["name"].lower():
|
||||
return i, int(d["default_samplerate"]), d["name"]
|
||||
|
||||
raise ValueError(
|
||||
f"Mikrofon '{spec}' nicht gefunden.\n"
|
||||
f"Verfügbare Geräte anzeigen: python3 mic.py"
|
||||
)
|
||||
|
||||
|
||||
def list_mics() -> list[dict]:
|
||||
"""Gibt alle Eingabegeräte mit Metadaten zurück (in-Kanäle > 0)."""
|
||||
return [
|
||||
{
|
||||
"index": i,
|
||||
"name": d["name"],
|
||||
"channels": d["max_input_channels"],
|
||||
"rate": int(d["default_samplerate"]),
|
||||
}
|
||||
for i, d in enumerate(sd.query_devices())
|
||||
if d["max_input_channels"] > 0
|
||||
]
|
||||
|
||||
|
||||
def print_mic_list() -> None:
|
||||
"""Gibt eine formatierte Liste aller Eingabegeräte auf stdout aus."""
|
||||
devs = list_mics()
|
||||
default_in = sd.default.device[0]
|
||||
|
||||
print("Verfügbare Mikrofone:\n")
|
||||
print(f" {'Kurzname':<12} {'Idx':>3} {'Rate':>6} {'Ch':>3} Gerätename")
|
||||
print(f" {'-'*12} {'---':>3} {'------':>6} {'---':>3} {'-'*52}")
|
||||
|
||||
shown: set[int] = set()
|
||||
for alias, pattern in MIC_ALIASES.items():
|
||||
try:
|
||||
idx, rate, name = resolve_mic(alias)
|
||||
except ValueError:
|
||||
continue
|
||||
marker = " *" if idx == default_in else ""
|
||||
print(f" {alias:<12} {idx:>3} {rate:>6} {'?':>3} {name}{marker}")
|
||||
shown.add(idx)
|
||||
|
||||
print(f" {'default':<12} {'':>3} {'':>6} {'':>3} System-Standard-Eingabe (PipeWire)")
|
||||
print(f" {'bluetooth':<12} {'':>3} {'':>6} {'':>3} erstes verbundenes BT-Eingabegerät (dynamisch)")
|
||||
|
||||
others = [d for d in devs if d["index"] not in shown]
|
||||
if others:
|
||||
print("\n Weitere (mit Index oder Namens-Substring ansprechen):")
|
||||
for d in others:
|
||||
marker = " *" if d["index"] == default_in else ""
|
||||
print(
|
||||
f" {'':12} {d['index']:>3} {d['rate']:>6}"
|
||||
f" {d['channels']:>3} {d['name']}{marker}"
|
||||
)
|
||||
|
||||
print("\n * = System-Standard")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
print_mic_list()
|
||||
289
speak.py
Executable file
289
speak.py
Executable file
|
|
@ -0,0 +1,289 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
TTS-Ausgabe via Piper — Text aus Argument oder stdin, Ausgabe auf wählbaren Kanal.
|
||||
|
||||
Verwendung:
|
||||
echo "Hallo Welt" | python3 speak.py
|
||||
python3 speak.py --text "Hallo Welt"
|
||||
python3 speak.py --text "Hallo" --out respeaker
|
||||
python3 speak.py --text "Hello" --voice en_US-ryan-high --out hdmi
|
||||
python3 speak.py --list
|
||||
|
||||
Ausgabe-Kurznamen (--out):
|
||||
default PipeWire-Default-Sink (aktuell: MOTU M2)
|
||||
respeaker ReSpeaker XVF3800 3,5mm-Klinke
|
||||
hdmi HDA NVidia HDMI
|
||||
bluetooth erstes verbundenes Bluetooth-Gerät (dynamisch)
|
||||
<sink-name> vollständiger PipeWire-Sink-Name
|
||||
|
||||
Sprechbar auch als Bibliothek:
|
||||
from speak import speak
|
||||
speak("Hallo Welt", out="respeaker")
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from pathlib import Path
|
||||
from piper.voice import PiperVoice
|
||||
|
||||
try:
|
||||
from langdetect import detect_langs as _ld_detect_langs, LangDetectException
|
||||
from langdetect.detector_factory import DetectorFactory as _DetectorFactory
|
||||
_DetectorFactory.seed = 42 # deterministisch
|
||||
_LANGDETECT_OK = True
|
||||
except ImportError:
|
||||
_LANGDETECT_OK = False
|
||||
|
||||
VOICES_DIR = Path.home() / ".local/share/piper/voices"
|
||||
DEFAULT_VOICE = "de_DE-thorsten-high"
|
||||
|
||||
# Beste verfügbare Piper-Stimme pro Sprache (ISO 639-1 → voice_name)
|
||||
LANG_VOICES: dict[str, str] = {
|
||||
"de": "de_DE-thorsten-high",
|
||||
"en": "en_US-ryan-high", # überschreibbar via set_en_variant()
|
||||
"fr": "fr_FR-upmc-medium",
|
||||
"es": "es_ES-davefx-medium",
|
||||
"it": "it_IT-paola-medium",
|
||||
"nl": "nl_NL-mls-medium",
|
||||
"pt": "pt_BR-faber-medium",
|
||||
"ru": "ru_RU-ruslan-medium",
|
||||
"pl": "pl_PL-bass-high",
|
||||
"uk": "uk_UA-lada-x_low",
|
||||
"cs": "cs_CZ-jirka-medium",
|
||||
}
|
||||
|
||||
EN_VARIANT_VOICES: dict[str, str] = {
|
||||
"us": "en_US-ryan-high",
|
||||
"gb": "en_GB-alan-medium",
|
||||
}
|
||||
|
||||
_voice_cache: dict[str, "PiperVoice"] = {}
|
||||
|
||||
# Feste Sink-Kurznamen
|
||||
SINK_ALIASES: dict[str, str] = {
|
||||
"respeaker": "alsa_output.usb-Seeed_Studio_reSpeaker_XVF3800_4-Mic_Array_114993700261100055-00.analog-stereo",
|
||||
"hdmi": "alsa_output.pci-0000_0b_00.1.hdmi-stereo",
|
||||
"motu": "alsa_output.usb-MOTU_M2_M20000046918-00.analog-stereo",
|
||||
}
|
||||
|
||||
|
||||
def list_sinks() -> dict[str, str]:
|
||||
"""Gibt {sink_name: beschreibung} für alle aktiven PipeWire-Sinks zurück."""
|
||||
result = subprocess.run(
|
||||
["pactl", "list", "sinks"],
|
||||
capture_output=True, text=True
|
||||
)
|
||||
sinks: dict[str, str] = {}
|
||||
name = ""
|
||||
for line in result.stdout.splitlines():
|
||||
line = line.strip()
|
||||
if line.startswith("Name:"):
|
||||
name = line.split(None, 1)[1]
|
||||
elif line.startswith("Beschreibung:") or line.startswith("Description:"):
|
||||
desc = line.split(None, 1)[1]
|
||||
if name:
|
||||
sinks[name] = desc
|
||||
return sinks
|
||||
|
||||
|
||||
def resolve_sink(out: str) -> str | None:
|
||||
"""Löst einen Kurznamen oder 'bluetooth' in einen echten Sink-Namen auf.
|
||||
Gibt None zurück → PipeWire-Default wird verwendet."""
|
||||
if out == "default":
|
||||
return None
|
||||
if out in SINK_ALIASES:
|
||||
return SINK_ALIASES[out]
|
||||
if out == "bluetooth":
|
||||
for name in list_sinks():
|
||||
if "bluez_output" in name:
|
||||
return name
|
||||
print("Kein Bluetooth-Gerät verbunden.", file=sys.stderr)
|
||||
return None
|
||||
# Vollständiger Sink-Name oder unbekannter Alias → direkt weitergeben
|
||||
return out
|
||||
|
||||
|
||||
def load_voice(voice_name: str) -> PiperVoice:
|
||||
onnx = VOICES_DIR / f"{voice_name}.onnx"
|
||||
if not onnx.exists():
|
||||
from piper.download_voices import download_voice
|
||||
VOICES_DIR.mkdir(parents=True, exist_ok=True)
|
||||
last_exc: Exception | None = None
|
||||
for attempt in range(1, 4):
|
||||
print(f"Stimme '{voice_name}' nicht lokal — lade herunter (Versuch {attempt}/3) …",
|
||||
file=sys.stderr, flush=True)
|
||||
try:
|
||||
download_voice(voice_name, VOICES_DIR)
|
||||
last_exc = None
|
||||
break
|
||||
except Exception as exc:
|
||||
last_exc = exc
|
||||
if attempt < 3:
|
||||
print(f"[Download-Fehler: {exc}] — wiederhole in {2 ** attempt}s …",
|
||||
file=sys.stderr, flush=True)
|
||||
time.sleep(2 ** attempt) # 2s, 4s
|
||||
if last_exc is not None or not onnx.exists():
|
||||
raise RuntimeError(
|
||||
f"Download von '{voice_name}' nach 3 Versuchen fehlgeschlagen"
|
||||
+ (f": {last_exc}" if last_exc else "")
|
||||
)
|
||||
print(f"Stimme '{voice_name}' gespeichert in {VOICES_DIR}.", file=sys.stderr)
|
||||
return PiperVoice.load(str(onnx))
|
||||
|
||||
|
||||
def get_voice(voice_name: str = DEFAULT_VOICE) -> PiperVoice:
|
||||
"""Gibt eine gecachte PiperVoice zurück — lädt nur beim ersten Aufruf."""
|
||||
if voice_name not in _voice_cache:
|
||||
_voice_cache[voice_name] = load_voice(voice_name)
|
||||
return _voice_cache[voice_name]
|
||||
|
||||
|
||||
def detect_lang(text: str) -> str | None:
|
||||
"""Erkennt die dominante Sprache. Gibt ISO-639-1-Code zurück oder None.
|
||||
|
||||
Akronyme (ALL-CAPS-Wörter wie GPT, KI, URL) und URLs werden vor der
|
||||
Erkennung entfernt, damit sie die Spracherkennung nicht verfälschen.
|
||||
Mindestkonfidenzschwelle: 0.85.
|
||||
"""
|
||||
if not _LANGDETECT_OK:
|
||||
return None
|
||||
# Akronyme (≥2 Großbuchstaben) und URLs entfernen
|
||||
cleaned = re.sub(r'\b[A-Z]{2,}\b', ' ', text)
|
||||
cleaned = re.sub(r'https?://\S+', ' ', cleaned)
|
||||
cleaned = re.sub(r'\s+', ' ', cleaned).strip()
|
||||
if len(cleaned) < 20:
|
||||
return None
|
||||
try:
|
||||
langs = _ld_detect_langs(cleaned)
|
||||
if langs and langs[0].prob >= 0.85:
|
||||
return langs[0].lang
|
||||
except LangDetectException:
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
def voice_for_lang(lang_code: str | None, fallback: str = DEFAULT_VOICE) -> PiperVoice:
|
||||
"""Gibt die passende (gecachte) PiperVoice für einen Sprachcode zurück.
|
||||
Lädt und cached die Stimme beim ersten Aufruf (inkl. Auto-Download)."""
|
||||
voice_name = LANG_VOICES.get(lang_code or "", fallback) if lang_code else fallback
|
||||
return get_voice(voice_name)
|
||||
|
||||
|
||||
def set_en_variant(variant: str) -> None:
|
||||
"""Wählt US- oder GB-Englisch für die automatische Spracherkennung.
|
||||
Muss vor dem ersten get_voice()-Aufruf gesetzt werden."""
|
||||
LANG_VOICES["en"] = EN_VARIANT_VOICES.get(variant, EN_VARIANT_VOICES["us"])
|
||||
|
||||
|
||||
def synth_bytes(text: str, voice: PiperVoice) -> bytes:
|
||||
"""Synthetisiert text vollständig in einen Byte-Puffer (ohne Subprocess)."""
|
||||
return b"".join(c.audio_int16_bytes for c in voice.synthesize(text))
|
||||
|
||||
|
||||
def speak(text: str, voice_name: str = DEFAULT_VOICE, out: str = "default") -> None:
|
||||
"""Synthetisiert text und gibt ihn auf dem gewünschten Ausgang aus."""
|
||||
voice = load_voice(voice_name)
|
||||
sink = resolve_sink(out)
|
||||
|
||||
# paplay liest raw-PCM von stdin
|
||||
cmd = ["paplay", "--raw",
|
||||
f"--rate={voice.config.sample_rate}",
|
||||
"--channels=1",
|
||||
"--format=s16le"]
|
||||
if sink:
|
||||
cmd += [f"--device={sink}"]
|
||||
|
||||
proc = subprocess.Popen(cmd, stdin=subprocess.PIPE)
|
||||
for chunk in voice.synthesize(text):
|
||||
proc.stdin.write(chunk.audio_int16_bytes)
|
||||
proc.stdin.close()
|
||||
proc.wait()
|
||||
|
||||
|
||||
def print_sink_list() -> None:
|
||||
sinks = list_sinks()
|
||||
default = subprocess.run(
|
||||
["pactl", "get-default-sink"], capture_output=True, text=True
|
||||
).stdout.strip()
|
||||
|
||||
print("Verfügbare Ausgänge:\n")
|
||||
print(f" {'Kurzname':<12} {'Sink-Name':<65} Beschreibung")
|
||||
print(f" {'-'*12} {'-'*65} {'-'*30}")
|
||||
|
||||
# Kurznamen zuerst
|
||||
for alias, sink_name in sorted(SINK_ALIASES.items()):
|
||||
desc = sinks.get(sink_name, "–")
|
||||
marker = " *" if sink_name == default else ""
|
||||
print(f" {alias:<12} {sink_name:<65} {desc}{marker}")
|
||||
|
||||
print(f" {'default':<12} {'(PipeWire-Default-Sink)':<65} → aktuell: {sinks.get(default, default)}")
|
||||
print(f" {'bluetooth':<12} {'(dynamisch, bluez_output.*)':<65} erstes verbundenes BT-Gerät")
|
||||
|
||||
# Weitere Sinks ohne Alias
|
||||
aliased = set(SINK_ALIASES.values())
|
||||
others = [(n, d) for n, d in sinks.items() if n not in aliased]
|
||||
if others:
|
||||
print("\n Weitere (kein Kurzname, mit vollem Namen ansprechen):")
|
||||
for name, desc in others:
|
||||
marker = " *" if name == default else ""
|
||||
print(f" {'':12} {name:<65} {desc}{marker}")
|
||||
print("\n * = aktueller Default")
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Piper TTS mit wählbarem Ausgang")
|
||||
parser.add_argument("--text", "-t", help="Auszugebender Text (sonst: stdin)")
|
||||
parser.add_argument("--voice", "-v", default=DEFAULT_VOICE,
|
||||
help=f"Piper-Stimme (default: {DEFAULT_VOICE})")
|
||||
parser.add_argument("--voice-lang", default="de",
|
||||
help="Sprachfilter für --list-voices, z.B. 'de', 'en', '' für alle")
|
||||
parser.add_argument("--out", "-o", default="default",
|
||||
help="Ausgabe: default | respeaker | hdmi | motu | bluetooth | <sink-name>")
|
||||
parser.add_argument("--list", "-l", action="store_true",
|
||||
help="Verfügbare Ausgänge anzeigen und beenden")
|
||||
parser.add_argument("--list-voices", "-L", action="store_true",
|
||||
help="Alle herunterladbaren Stimmen anzeigen (gefiltert nach --voice-lang)")
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.list:
|
||||
print_sink_list()
|
||||
return
|
||||
|
||||
if args.list_voices:
|
||||
import json
|
||||
from urllib.request import urlopen
|
||||
from piper.download_voices import VOICES_JSON
|
||||
with urlopen(VOICES_JSON) as resp:
|
||||
voices = json.load(resp)
|
||||
lang = args.voice_lang
|
||||
local = {f.stem for f in VOICES_DIR.glob("*.onnx")}
|
||||
matches = {k: v for k, v in sorted(voices.items())
|
||||
if not lang or k.startswith(lang)}
|
||||
print(f"{'Stimme':<40} {'Qualität':<8} Lokal")
|
||||
print(f"{'-'*40} {'-'*8} -----")
|
||||
for name, info in matches.items():
|
||||
q = info.get("quality", "?")
|
||||
marker = "✓" if name in local else ""
|
||||
print(f"{name:<40} {q:<8} {marker}")
|
||||
print(f"\n{len(matches)} Stimmen. Lokal vorhanden: {len([n for n in matches if n in local])}")
|
||||
return
|
||||
|
||||
if args.text:
|
||||
text = args.text
|
||||
elif not sys.stdin.isatty():
|
||||
text = sys.stdin.read().strip()
|
||||
else:
|
||||
parser.print_help()
|
||||
sys.exit(1)
|
||||
|
||||
if not text:
|
||||
sys.exit(0)
|
||||
|
||||
speak(text, voice_name=args.voice, out=args.out)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
115
transcribe.py
Executable file
115
transcribe.py
Executable file
|
|
@ -0,0 +1,115 @@
|
|||
#!/usr/bin/env python3
|
||||
"""
|
||||
Live-Transkription via faster-whisper (large-v3, GPU).
|
||||
|
||||
Aufnahme-Logik:
|
||||
- Puffert Audio in CHUNK_SEC-Sekunden-Blöcken
|
||||
- Faster-whisper VAD filtert Stille heraus
|
||||
- Ausgabe: fortlaufender Text auf stdout
|
||||
|
||||
Verwendung:
|
||||
python3 transcribe.py # ReSpeaker, GPU 2
|
||||
python3 transcribe.py --mic motu # MOTU M2
|
||||
python3 transcribe.py --mic camera # Kamera-Mikrofon
|
||||
python3 transcribe.py --mic 17 # sounddevice-Index direkt
|
||||
python3 transcribe.py --list-mics # alle Eingabegeräte anzeigen
|
||||
python3 transcribe.py [--lang de] [--model large-v3] [--gpu 1]
|
||||
|
||||
GPU-Belegung:
|
||||
GPU 0 NVIDIA T600 (4 GB) — reserviert
|
||||
GPU 1 RTX 3090 (24 GB) — llama.cpp / Port 8001
|
||||
GPU 2 RTX 3090 (24 GB) — Whisper (default)
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import queue
|
||||
import sys
|
||||
import numpy as np
|
||||
import sounddevice as sd
|
||||
from faster_whisper import WhisperModel
|
||||
from scipy.signal import resample_poly
|
||||
|
||||
WHISPER_RATE = 16000
|
||||
CHUNK_SEC = 5 # Sekunden pro Transkriptions-Durchlauf
|
||||
CHANNELS = 1
|
||||
|
||||
sys.path.insert(0, __import__("os").path.dirname(__file__))
|
||||
from mic import resolve_mic, print_mic_list, DEFAULT_MIC # noqa: E402
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
formatter_class=argparse.ArgumentDefaultsHelpFormatter
|
||||
)
|
||||
parser.add_argument("--mic", default=DEFAULT_MIC,
|
||||
help="Eingabegerät: respeaker | motu | camera | default | "
|
||||
"bluetooth | <index> | <substring>")
|
||||
parser.add_argument("--list-mics", action="store_true",
|
||||
help="Verfügbare Mikrofone anzeigen und beenden")
|
||||
parser.add_argument("--lang", default="de", help="Sprache (de/en/…)")
|
||||
parser.add_argument("--model", default="large-v3", help="Whisper-Modell")
|
||||
parser.add_argument("--gpu", type=int, default=1,
|
||||
help="CUDA-Index innerhalb CUDA_VISIBLE_DEVICES (default: 1 = phys. GPU 2)")
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.list_mics:
|
||||
print_mic_list()
|
||||
return
|
||||
|
||||
device_idx, capture_rate, device_label = resolve_mic(args.mic)
|
||||
chunk_samples = CHUNK_SEC * capture_rate
|
||||
|
||||
print(f"Mikrofon : {device_label} (Index {device_idx}, {capture_rate} Hz → resample → {WHISPER_RATE} Hz)")
|
||||
print(f"Modell : {args.model} | Sprache: {args.lang} | GPU: {args.gpu}")
|
||||
print("Lade Modell …", flush=True)
|
||||
|
||||
model = WhisperModel(
|
||||
args.model,
|
||||
device="cuda",
|
||||
device_index=args.gpu,
|
||||
compute_type="float16",
|
||||
)
|
||||
print("Modell geladen. Spreche …\n", flush=True)
|
||||
|
||||
audio_q: "queue.Queue[np.ndarray]" = queue.Queue()
|
||||
|
||||
def callback(indata, frames, time_info, status):
|
||||
if status:
|
||||
print(f"[audio status] {status}", file=sys.stderr)
|
||||
audio_q.put(indata[:, 0].copy())
|
||||
|
||||
try:
|
||||
with sd.InputStream(
|
||||
device=device_idx,
|
||||
channels=CHANNELS,
|
||||
samplerate=capture_rate,
|
||||
dtype="float32",
|
||||
blocksize=capture_rate // 4, # 250 ms Blöcke
|
||||
callback=callback,
|
||||
):
|
||||
buffer = np.empty(0, dtype="float32")
|
||||
while True:
|
||||
while buffer.shape[0] < chunk_samples:
|
||||
buffer = np.concatenate([buffer, audio_q.get()])
|
||||
|
||||
chunk = buffer[:chunk_samples].copy()
|
||||
buffer = buffer[chunk_samples:]
|
||||
|
||||
chunk_16k = resample_poly(chunk, WHISPER_RATE, capture_rate)
|
||||
|
||||
segments, _ = model.transcribe(
|
||||
chunk_16k,
|
||||
language=args.lang,
|
||||
vad_filter=True,
|
||||
vad_parameters={"min_silence_duration_ms": 500},
|
||||
beam_size=5,
|
||||
)
|
||||
text = " ".join(s.text.strip() for s in segments).strip()
|
||||
if text:
|
||||
print(text, flush=True)
|
||||
|
||||
except KeyboardInterrupt:
|
||||
print("\nBeendet.")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Loading…
Add table
Add a link
Reference in a new issue