From 37b14773f24c60b1a11dabdb55ff68b13f2c5c1f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Dieter=20Schl=C3=BCter?= Date: Wed, 17 Jun 2026 23:25:49 +0200 Subject: [PATCH] feat(tools): scripts/add_pronunciation.py zum bequemen Pflegen des Aussprache-Lexikons Fuegt 'wort:aussprache' in config/pronunciation..yaml ein (Default Sektion terms). Erhaelt Kommentare/Struktur, aktualisiert vorhandene Eintraege statt zu duplizieren, validiert das YAML vor dem Schreiben. Optional --verify zeigt die espeak-Phoneme vorher/nachher und warnt, wenn die Umschreibung nichts aendert. Doku-Hinweis in BEDIENUNGSANLEITUNG ergaenzt. Co-Authored-By: Claude Opus 4.8 --- BEDIENUNGSANLEITUNG.md | 7 ++ scripts/add_pronunciation.py | 142 +++++++++++++++++++++++++++++++++++ 2 files changed, 149 insertions(+) create mode 100644 scripts/add_pronunciation.py diff --git a/BEDIENUNGSANLEITUNG.md b/BEDIENUNGSANLEITUNG.md index 0fe9e7c..be63467 100644 --- a/BEDIENUNGSANLEITUNG.md +++ b/BEDIENUNGSANLEITUNG.md @@ -270,6 +270,13 @@ und Abkürzungen („Dr." → „Doktor", „z. B." → „zum Beispiel"). Reine Cloud-TTS `light` (Cloud spricht Zahlen/Abkürzungen selbst gut, daher schonend). - Eigene Begriffe/Fachwörter pflegst du in **`config/pronunciation.de.yaml`** (Abkürzungen, Einheiten, Terms) — erweitert die eingebauten Defaults, ohne Code zu ändern. +- **Bequem per Skript** (prüft auf Wunsch gleich die espeak-Phoneme): + ```bash + python scripts/add_pronunciation.py "strömt:ströhmt" # Wort:Aussprache + python scripts/add_pronunciation.py Mond Mohnd --verify # zeigt Phoneme vorher/nachher + python scripts/add_pronunciation.py kWh "Kilowattstunden" --section units + ``` + Danach den Server einmal neu starten. ## B2. Soundquelle & Ausgabe-Gerät wechseln (Mikrofon, Lautsprecher, Bluetooth, Handy) diff --git a/scripts/add_pronunciation.py b/scripts/add_pronunciation.py new file mode 100644 index 0000000..97b67be --- /dev/null +++ b/scripts/add_pronunciation.py @@ -0,0 +1,142 @@ +#!/usr/bin/env python3 +"""Fügt ein Wort + Aussprache ins Aussprache-Lexikon ein. + +Das Lexikon (config/pronunciation..yaml) wird vor dem lokalen TTS (piper) +angewandt: der geschriebene Begriff wird durch eine "sprechbare" Umschreibung +ersetzt (z. B. langer Vokal per eingefügtem 'h'). + +Beispiele: + python scripts/add_pronunciation.py strömen ströhmen + python scripts/add_pronunciation.py "strömen:ströhmen" + python scripts/add_pronunciation.py CPU "Ze Pe U" --section terms + python scripts/add_pronunciation.py kWh "Kilowattstunden" --section units + python scripts/add_pronunciation.py strömen ströhmen --verify # espeak-Phoneme zeigen + +Hinweis: Damit der laufende Server die Änderung nutzt, einmal neu starten. +""" + +from __future__ import annotations + +import argparse +import shutil +import subprocess +import sys +from pathlib import Path + +try: + import yaml +except ModuleNotFoundError: + sys.exit("Fehlt: Python-Paket 'PyYAML' (pip install pyyaml).") + +BASE_DIR = Path(__file__).resolve().parent.parent +SECTIONS = ("terms", "abbreviations", "units") + + +def _quote(s: str) -> str: + """YAML-sicher als doppelt-gequoteter String.""" + return '"' + s.replace("\\", "\\\\").replace('"', '\\"') + '"' + + +def _espeak(word: str, lang: str) -> str | None: + if not shutil.which("espeak-ng"): + return None + try: + out = subprocess.run( + ["espeak-ng", "-v", lang, "-q", "-x", word], + capture_output=True, text=True, timeout=10, + ) + return out.stdout.strip() or None + except (subprocess.SubprocessError, OSError): + return None + + +def _entry_line_index(lines: list[str], section: str, word: str) -> int | None: + """Index einer vorhandenen, aktiven Zeile ' "word": ...' in der Sektion.""" + in_section = False + target = word.lower() + for i, line in enumerate(lines): + stripped = line.strip() + if not line.startswith(" ") and stripped.rstrip(":") in SECTIONS: + in_section = stripped.rstrip(":") == section + continue + if in_section and stripped and not stripped.startswith("#"): + key = stripped.split(":", 1)[0].strip().strip('"').strip("'") + if key.lower() == target: + return i + return None + + +def _section_header_index(lines: list[str], section: str) -> int | None: + for i, line in enumerate(lines): + if not line.startswith(" ") and line.strip().rstrip(":") == section \ + and line.rstrip().endswith(":"): + return i + return None + + +def upsert(path: Path, section: str, word: str, pron: str) -> str: + text = path.read_text(encoding="utf-8") if path.exists() else "" + lines = text.splitlines() + new_line = f" {_quote(word)}: {_quote(pron)}" + + existing = _entry_line_index(lines, section, word) + if existing is not None: + action = "aktualisiert" + lines[existing] = new_line + else: + action = "hinzugefügt" + header = _section_header_index(lines, section) + if header is None: # Sektion fehlt -> am Ende anlegen + if lines and lines[-1].strip(): + lines.append("") + lines.append(f"{section}:") + lines.append(new_line) + else: + lines.insert(header + 1, new_line) + + result = "\n".join(lines) + "\n" + # Vor dem Schreiben validieren -> niemals kaputtes YAML hinterlassen. + parsed = yaml.safe_load(result) or {} + if (parsed.get(section) or {}).get(word) != pron: + raise SystemExit("Abbruch: Eintrag nach dem Schreiben nicht wie erwartet (YAML-Problem).") + path.write_text(result, encoding="utf-8") + return action + + +def main() -> None: + p = argparse.ArgumentParser(description="Wort + Aussprache ins Lexikon eintragen.") + p.add_argument("word", help="Wort wie geschrieben (oder 'wort:aussprache').") + p.add_argument("pron", nargs="?", help="sprechbare Umschreibung.") + p.add_argument("--section", choices=SECTIONS, default="terms") + p.add_argument("--lang", default="de") + p.add_argument("--file", default=None, help="Lexikon-Pfad (Default: config/pronunciation..yaml)") + p.add_argument("--verify", action="store_true", help="espeak-Phoneme vorher/nachher zeigen") + args = p.parse_args() + + word, pron = args.word, args.pron + if pron is None: + if ":" not in word: + p.error("Bitte 'wort aussprache' oder 'wort:aussprache' angeben.") + word, pron = (s.strip() for s in word.split(":", 1)) + if not word or not pron: + p.error("Wort und Aussprache dürfen nicht leer sein.") + + path = Path(args.file) if args.file else BASE_DIR / "config" / f"pronunciation.{args.lang}.yaml" + + if args.verify: + before, after = _espeak(word, args.lang), _espeak(pron, args.lang) + if before is None: + print(" (espeak-ng nicht gefunden – Verifikation übersprungen)") + else: + print(f" espeak {word!r:14} -> {before}") + print(f" espeak {pron!r:14} -> {after}") + if before == after: + print(" ⚠ Phoneme identisch – die Umschreibung ändert die Aussprache NICHT.") + + action = upsert(path, args.section, word, pron) + print(f"✓ {action}: [{args.section}] {word!r} -> {pron!r} ({path})") + print(" Hinweis: laufenden Server neu starten, damit die Änderung greift.") + + +if __name__ == "__main__": + main()