feat(hermes-tts): prepare fixed multilingual voice policy
Supersede draft PR #26 with a merge-safe prerequisite: bake and preload the amy, irina, and claude Piper models, route only validated server-side language to fixed voices, and leave the live voice deployment manifest unchanged. Remove the pinned WebUI speaker selector and its persisted preference, omit client voice fields from every outbound TTS path, and keep hands-free Voice Mode and the conversation instrument intact. Hostile or legacy voice fields remain ignored by the Piper server. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
parent
5f9c600f6e
commit
9e14acf390
@ -24,18 +24,41 @@ ADD --checksum=sha256:f7d01dde371555732c4c314111ac79672b1a5ce2fc19266ab42178fd8d
|
||||
ADD --checksum=sha256:45754dfdebb3b8661c3fc564713772deec6e064feeb5b4e9594857dc7305193a --chmod=0444 \
|
||||
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/en/en_US/lessac/low/en_US-lessac-low.onnx.json?download=true \
|
||||
/opt/models/piper/en_US-lessac-low.onnx.json
|
||||
|
||||
# Multilingual chat voice policy: English -> amy, Russian -> irina, Spanish ->
|
||||
# claude (Mexican Spanish, the only "claude" voice rhasspy/piper-voices
|
||||
# publishes; there is no es_ES-claude).
|
||||
ADD --checksum=sha256:b3a6e47b57b8c7fbe6a0ce2518161a50f59a9cdd8a50835c02cb02bdd6206c18 --chmod=0444 \
|
||||
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/en/en_US/amy/medium/en_US-amy-medium.onnx?download=true \
|
||||
/opt/models/piper/en_US-amy-medium.onnx
|
||||
ADD --checksum=sha256:95a23eb4d42909d38df73bb9ac7f45f597dbfcde2d1bf9526fdeaf5466977d77 --chmod=0444 \
|
||||
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/en/en_US/amy/medium/en_US-amy-medium.onnx.json?download=true \
|
||||
/opt/models/piper/en_US-amy-medium.onnx.json
|
||||
ADD --checksum=sha256:8ff38212d23da300bbe3705c645e6e5b9475f0bfde01558eb17813e22acaaaaa --chmod=0444 \
|
||||
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/ru/ru_RU/irina/medium/ru_RU-irina-medium.onnx?download=true \
|
||||
/opt/models/piper/ru_RU-irina-medium.onnx
|
||||
ADD --checksum=sha256:c2ec28bb38e2b59e93b959b3e40348c1afebbd272f30fed5d41205d08e98a9d7 --chmod=0444 \
|
||||
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/ru/ru_RU/irina/medium/ru_RU-irina-medium.onnx.json?download=true \
|
||||
/opt/models/piper/ru_RU-irina-medium.onnx.json
|
||||
ADD --checksum=sha256:3ef40a71ea63852cd8ab7e6fa7d2ecdcfa67a0b47c9c48e3f10e02ee02083ea0 --chmod=0444 \
|
||||
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/es/es_MX/claude/high/es_MX-claude-high.onnx?download=true \
|
||||
/opt/models/piper/es_MX-claude-high.onnx
|
||||
ADD --checksum=sha256:1afc81f703c0e4cb3b4d7c0dca096b8b54a98806807f0170cf5eb5557723c12d --chmod=0444 \
|
||||
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/es/es_MX/claude/high/es_MX-claude-high.onnx.json?download=true \
|
||||
/opt/models/piper/es_MX-claude-high.onnx.json
|
||||
RUN chmod 0555 /opt/models /opt/models/piper
|
||||
|
||||
COPY dockerfiles/hermes-jetson-tts-server.py /opt/atlas/hermes-jetson-tts-server.py
|
||||
RUN chmod 0555 /opt/atlas/hermes-jetson-tts-server.py
|
||||
|
||||
# Load the actual pinned voice during the ARM64 build. This catches package or
|
||||
# model-format drift before the image can reach Flux.
|
||||
RUN python -c "import stat; from pathlib import Path; from piper import PiperVoice; p=Path('/opt/models/piper'); models=[p/'en_US-lessac-high.onnx',p/'en_US-lessac-medium.onnx',p/'en_US-lessac-low.onnx']; assert stat.S_IMODE(p.stat().st_mode)==0o555; assert all(stat.S_IMODE(model.stat().st_mode)==0o444 for model in models); voices=[PiperVoice.load(model,Path(str(model)+'.json'),use_cuda=False,download_dir=p) for model in models]; assert all(voice.config.sample_rate>0 for voice in voices)"
|
||||
# Load every pinned voice during the ARM64 build, including the three baked
|
||||
# for the multilingual chat policy. This catches package or model-format
|
||||
# drift before the image can reach Flux.
|
||||
RUN python -c "import stat; from pathlib import Path; from piper import PiperVoice; p=Path('/opt/models/piper'); models=[p/'en_US-lessac-high.onnx',p/'en_US-lessac-medium.onnx',p/'en_US-lessac-low.onnx',p/'en_US-amy-medium.onnx',p/'ru_RU-irina-medium.onnx',p/'es_MX-claude-high.onnx']; assert stat.S_IMODE(p.stat().st_mode)==0o555; assert all(stat.S_IMODE(model.stat().st_mode)==0o444 for model in models); voices=[PiperVoice.load(model,Path(str(model)+'.json'),use_cuda=False,download_dir=p) for model in models]; assert all(voice.config.sample_rate>0 for voice in voices)"
|
||||
|
||||
ENV HERMES_TTS_HOST=0.0.0.0 \
|
||||
HERMES_TTS_PORT=9001 \
|
||||
HERMES_TTS_VOICE=en_US-lessac-medium \
|
||||
HERMES_TTS_VOICE=en_US-amy-medium \
|
||||
HERMES_TTS_CACHE=/opt/models/piper \
|
||||
OMP_NUM_THREADS=2 \
|
||||
PYTHONDONTWRITEBYTECODE=1 \
|
||||
|
||||
@ -17,12 +17,50 @@ from piper import PiperConfig, PiperVoice, SynthesisConfig
|
||||
|
||||
HOST = os.getenv("HERMES_TTS_HOST", "0.0.0.0")
|
||||
PORT = int(os.getenv("HERMES_TTS_PORT", "9001"))
|
||||
VOICE_NAME = os.getenv("HERMES_TTS_VOICE", "en_US-lessac-high")
|
||||
CACHE_DIR = Path(os.getenv("HERMES_TTS_CACHE", "/cache/piper"))
|
||||
MAX_TEXT_CHARS = 5000
|
||||
ONNX_THREADS = max(1, int(os.getenv("HERMES_TTS_ONNX_THREADS", "4")))
|
||||
VOICE_LOCK = threading.Lock()
|
||||
|
||||
# Fixed, allow-listed language -> baked voice mapping. This is the ONLY path
|
||||
# from a client-supplied string to a model name: client input is looked up
|
||||
# here and never used to build a filesystem path directly. Both "-" and "_"
|
||||
# separators and any case are accepted; anything not present here falls back
|
||||
# to DEFAULT_VOICE_NAME (safe English default), never an error and never an
|
||||
# unbaked model.
|
||||
LANGUAGE_VOICE_MAP = {
|
||||
"en": "en_US-amy-medium",
|
||||
"en-us": "en_US-amy-medium",
|
||||
"ru": "ru_RU-irina-medium",
|
||||
"ru-ru": "ru_RU-irina-medium",
|
||||
"es": "es_MX-claude-high",
|
||||
"es-mx": "es_MX-claude-high",
|
||||
"es-es": "es_MX-claude-high",
|
||||
}
|
||||
BAKED_VOICE_NAMES = frozenset(LANGUAGE_VOICE_MAP.values())
|
||||
DEFAULT_VOICE_NAME = os.getenv("HERMES_TTS_VOICE", "en_US-amy-medium")
|
||||
|
||||
|
||||
def normalize_language(value: object) -> str | None:
|
||||
"""Lowercase and fold "_"/"-" separators; reject non-string/blank input."""
|
||||
if not isinstance(value, str):
|
||||
return None
|
||||
normalized = value.strip().lower().replace("_", "-")
|
||||
return normalized or None
|
||||
|
||||
|
||||
def resolve_voice_name(language: object) -> str:
|
||||
"""Map a client-supplied language to one of the baked policy voices.
|
||||
|
||||
Unknown, missing, or malformed language always resolves to the safe
|
||||
default rather than raising, and the result is always a member of
|
||||
BAKED_VOICE_NAMES.
|
||||
"""
|
||||
normalized = normalize_language(language)
|
||||
if normalized is None:
|
||||
return DEFAULT_VOICE_NAME
|
||||
return LANGUAGE_VOICE_MAP.get(normalized, DEFAULT_VOICE_NAME)
|
||||
|
||||
|
||||
def _json(handler: BaseHTTPRequestHandler, status: int, payload: dict) -> None:
|
||||
body = json.dumps(payload).encode("utf-8")
|
||||
@ -46,7 +84,16 @@ class SpeechHandler(BaseHTTPRequestHandler):
|
||||
if self.path != "/health":
|
||||
_json(self, 404, {"error": "not found"})
|
||||
return
|
||||
_json(self, 200, {"ok": True, "voice": VOICE_NAME, "device": "cpu"})
|
||||
_json(
|
||||
self,
|
||||
200,
|
||||
{
|
||||
"ok": True,
|
||||
"voices": sorted(self.server.voices), # type: ignore[attr-defined]
|
||||
"default_voice": self.server.default_voice_name, # type: ignore[attr-defined]
|
||||
"device": "cpu",
|
||||
},
|
||||
)
|
||||
|
||||
def do_POST(self) -> None:
|
||||
if self.path != "/v1/audio/speech":
|
||||
@ -71,10 +118,16 @@ class SpeechHandler(BaseHTTPRequestHandler):
|
||||
return
|
||||
speed = min(2.0, max(0.5, speed))
|
||||
|
||||
# Policy is driven ONLY by "language". A client-supplied "voice"
|
||||
# field is deliberately never read here; it cannot override the
|
||||
# allow-listed mapping.
|
||||
voice_name = resolve_voice_name(payload.get("language"))
|
||||
voice = self.server.voices[voice_name] # type: ignore[attr-defined]
|
||||
|
||||
output = io.BytesIO()
|
||||
try:
|
||||
with VOICE_LOCK, wave.open(output, "wb") as wav_file:
|
||||
self.server.voice.synthesize_wav( # type: ignore[attr-defined]
|
||||
voice.synthesize_wav(
|
||||
text,
|
||||
wav_file,
|
||||
SynthesisConfig(length_scale=1.0 / speed),
|
||||
@ -84,6 +137,7 @@ class SpeechHandler(BaseHTTPRequestHandler):
|
||||
self.send_header("Content-Type", "audio/wav")
|
||||
self.send_header("Content-Length", str(len(audio)))
|
||||
self.send_header("Cache-Control", "no-store")
|
||||
self.send_header("X-TTS-Voice", voice_name)
|
||||
self.end_headers()
|
||||
self.wfile.write(audio)
|
||||
except Exception as exc:
|
||||
@ -91,30 +145,54 @@ class SpeechHandler(BaseHTTPRequestHandler):
|
||||
_json(self, 500, {"error": "speech synthesis failed"})
|
||||
|
||||
|
||||
def main() -> None:
|
||||
"""Load the checksum-pinned voice from the image and serve it on CPU."""
|
||||
CACHE_DIR.mkdir(parents=True, exist_ok=True)
|
||||
model_path = CACHE_DIR / f"{VOICE_NAME}.onnx"
|
||||
config_path = CACHE_DIR / f"{VOICE_NAME}.onnx.json"
|
||||
def _load_voice(cache_dir: Path, voice_name: str, threads: int) -> PiperVoice:
|
||||
model_path = cache_dir / f"{voice_name}.onnx"
|
||||
config_path = cache_dir / f"{voice_name}.onnx.json"
|
||||
if not model_path.exists() or not config_path.exists():
|
||||
raise RuntimeError(f"baked Piper voice is missing: {VOICE_NAME}")
|
||||
raise RuntimeError(f"baked Piper voice is missing: {voice_name}")
|
||||
with config_path.open("r", encoding="utf-8") as config_file:
|
||||
config = PiperConfig.from_dict(json.load(config_file))
|
||||
session_options = onnxruntime.SessionOptions()
|
||||
session_options.intra_op_num_threads = ONNX_THREADS
|
||||
session_options.intra_op_num_threads = threads
|
||||
session_options.inter_op_num_threads = 1
|
||||
session = onnxruntime.InferenceSession(
|
||||
str(model_path),
|
||||
sess_options=session_options,
|
||||
providers=["CPUExecutionProvider"],
|
||||
)
|
||||
voice = PiperVoice(session=session, config=config, download_dir=CACHE_DIR)
|
||||
return PiperVoice(session=session, config=config, download_dir=cache_dir)
|
||||
|
||||
|
||||
def load_voices(cache_dir: Path, threads: int) -> dict[str, PiperVoice]:
|
||||
"""Eagerly load all three policy voices.
|
||||
|
||||
Preload (not lazy-load-on-first-use) was chosen deliberately: measured
|
||||
RSS on this model set is ~88MB for one voice and ~243MB for all three
|
||||
(~+155MB versus the previous single-voice baseline), which comfortably
|
||||
fits the pod's memory budget on the CPU-only voice node. Preloading
|
||||
avoids a slow, request-serializing first synthesis per language and
|
||||
keeps the fail-closed missing-model check (below) at process start
|
||||
rather than deferring a possible crash to a live user request.
|
||||
"""
|
||||
return {name: _load_voice(cache_dir, name, threads) for name in sorted(BAKED_VOICE_NAMES)}
|
||||
|
||||
|
||||
def main() -> None:
|
||||
"""Load the checksum-pinned policy voices from the image and serve them on CPU."""
|
||||
CACHE_DIR.mkdir(parents=True, exist_ok=True)
|
||||
if DEFAULT_VOICE_NAME not in BAKED_VOICE_NAMES:
|
||||
raise RuntimeError(
|
||||
f"HERMES_TTS_VOICE must name one of the baked policy voices: {sorted(BAKED_VOICE_NAMES)}"
|
||||
)
|
||||
voices = load_voices(CACHE_DIR, ONNX_THREADS)
|
||||
print(
|
||||
f"[tts] loaded Piper voice {VOICE_NAME} on CPU with {ONNX_THREADS} ONNX threads",
|
||||
f"[tts] loaded {len(voices)} Piper voices on CPU with {ONNX_THREADS} ONNX threads each: "
|
||||
+ ", ".join(sorted(voices)),
|
||||
flush=True,
|
||||
)
|
||||
server = ThreadingHTTPServer((HOST, PORT), SpeechHandler)
|
||||
server.voice = voice # type: ignore[attr-defined]
|
||||
server.voices = voices # type: ignore[attr-defined]
|
||||
server.default_voice_name = DEFAULT_VOICE_NAME # type: ignore[attr-defined]
|
||||
print(f"[tts] ready on {HOST}:{PORT}", flush=True)
|
||||
server.serve_forever(poll_interval=0.25)
|
||||
|
||||
|
||||
@ -16,6 +16,43 @@ def replace_exact(path: Path, before: str, after: str, count: int = 1) -> None:
|
||||
path.write_text(source.replace(before, after, count), encoding="utf-8")
|
||||
|
||||
|
||||
def replace_between_exact(
|
||||
path: Path, start: str, end: str, after: str = "", count: int = 1
|
||||
) -> None:
|
||||
"""Replace one exact, bounded upstream region and fail when the pin drifts."""
|
||||
source = path.read_text(encoding="utf-8")
|
||||
if source.count(start) != count or source.count(end) != count:
|
||||
raise SystemExit(
|
||||
f"Atlas voice patch context changed in {path}: {start[:80]!r}"
|
||||
)
|
||||
start_index = source.index(start)
|
||||
end_index = source.index(end, start_index) + len(end)
|
||||
path.write_text(
|
||||
source[:start_index] + after + source[end_index:], encoding="utf-8"
|
||||
)
|
||||
|
||||
|
||||
def assert_absent(path: Path, *needles: str) -> None:
|
||||
"""Fail the image build if a removed voice-choice surface remains."""
|
||||
source = path.read_text(encoding="utf-8")
|
||||
remaining = [needle for needle in needles if needle in source]
|
||||
if remaining:
|
||||
raise SystemExit(f"Atlas voice choice remains in {path}: {remaining!r}")
|
||||
|
||||
|
||||
def remove_lines_containing(path: Path, *needles: str) -> None:
|
||||
"""Remove all pinned translation entries for a retired settings control."""
|
||||
source = path.read_text(encoding="utf-8")
|
||||
for needle in needles:
|
||||
if needle not in source:
|
||||
raise SystemExit(f"Atlas voice patch context changed in {path}: {needle!r}")
|
||||
lines = source.splitlines(keepends=True)
|
||||
path.write_text(
|
||||
"".join(line for line in lines if not any(n in line for n in needles)),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
|
||||
index = ROOT / "static/index.html"
|
||||
replace_exact(
|
||||
index,
|
||||
@ -29,6 +66,16 @@ replace_exact(
|
||||
'<option value="browser">Browser speech synthesis</option><option value="edge">Edge TTS (server)</option>',
|
||||
'<option value="atlas">Atlas Jetson (private)</option><option value="browser">Browser speech synthesis</option><option value="edge">Edge TTS (server)</option>',
|
||||
)
|
||||
replace_exact(
|
||||
index,
|
||||
'''<div class="settings-field"><label for="settingsTtsVoice" data-i18n="settings_label_tts_voice">Voice</label>
|
||||
<select id="settingsTtsVoice" style="width:100%;padding:8px;background:var(--code-bg);color:var(--text);border:1px solid var(--border2);border-radius:6px">
|
||||
<option value="">Default system voice</option>
|
||||
</select>
|
||||
<div style="font-size:11px;color:var(--muted);margin-top:4px" data-i18n="settings_desc_tts_voice">Preferred voice. Populated from your browser's available voices.</div>
|
||||
</div>''',
|
||||
"",
|
||||
)
|
||||
replace_exact(
|
||||
index,
|
||||
'<script src="static/boot.js?v=__WEBUI_VERSION__" defer></script>',
|
||||
@ -62,11 +109,33 @@ replace_exact(
|
||||
)
|
||||
|
||||
ui = ROOT / "static/ui.js"
|
||||
replace_exact(
|
||||
ui,
|
||||
''' const savedVoice=localStorage.getItem('hermes-tts-voice');
|
||||
const voices=speechSynthesis.getVoices();
|
||||
if(savedVoice&&voices.length){
|
||||
const match=voices.find(v=>v.name===savedVoice);
|
||||
if(match) utter.voice=match;
|
||||
}
|
||||
''',
|
||||
"",
|
||||
)
|
||||
replace_exact(ui, "function _playEdgeTtsChunked(text, btn){", "function _playEdgeTtsChunked(text, btn, engineOverride){")
|
||||
replace_exact(
|
||||
ui,
|
||||
" const voice=localStorage.getItem('hermes-tts-voice')||'zh-CN-XiaoxiaoNeural';\n",
|
||||
"",
|
||||
)
|
||||
replace_exact(
|
||||
ui,
|
||||
"body:JSON.stringify({text:chunk, voice:voice, rate:rate, pitch:pitch})",
|
||||
"body:JSON.stringify({text:chunk, voice:voice, rate:rate, pitch:pitch, engine:engineOverride||'edge'})",
|
||||
"body:JSON.stringify({text:chunk, rate:rate, pitch:pitch, engine:engineOverride||'edge'})",
|
||||
)
|
||||
replace_exact(
|
||||
ui,
|
||||
" voice: localStorage.getItem('hermes-tts-voice')||'',\n",
|
||||
"",
|
||||
count=2,
|
||||
)
|
||||
replace_exact(
|
||||
ui,
|
||||
@ -79,6 +148,94 @@ replace_exact(
|
||||
"if(engine==='edge'||engine==='atlas'){\n _playEdgeTtsChunked(clean, null, engine);",
|
||||
)
|
||||
|
||||
panels = ROOT / "static/panels.js"
|
||||
replace_exact(panels, " tts_voice:'hermes-tts-voice',\n", "")
|
||||
replace_exact(
|
||||
panels,
|
||||
''' const ttsVoiceSel=$('settingsTtsVoice');
|
||||
if(ttsVoiceSel) _setOwnedSpeechPayload(payload,'tts_voice',ttsVoiceSel.value||'');
|
||||
''',
|
||||
"",
|
||||
)
|
||||
replace_exact(
|
||||
panels,
|
||||
''' localStorage.setItem('hermes-tts-engine',this.value);
|
||||
window._populateTtsVoices();
|
||||
_schedulePreferencesAutosave();''',
|
||||
''' localStorage.setItem('hermes-tts-engine',this.value);
|
||||
_schedulePreferencesAutosave();''',
|
||||
)
|
||||
replace_between_exact(
|
||||
panels,
|
||||
" // Populate voice selector based on engine\n",
|
||||
" // TTS rate/pitch sliders\n",
|
||||
" // TTS speaker selection is intentionally server policy only.\n",
|
||||
)
|
||||
replace_exact(
|
||||
panels,
|
||||
"let _settingsSpeechChangedKeys=new Set();\n",
|
||||
"let _settingsSpeechChangedKeys=new Set();\n"
|
||||
"try{localStorage.removeItem('hermes-tts-voice');}catch(_){}\n",
|
||||
)
|
||||
|
||||
boot = ROOT / "static/boot.js"
|
||||
replace_exact(
|
||||
boot,
|
||||
''' voice: localStorage.getItem("hermes-tts-voice")||'',
|
||||
''',
|
||||
"",
|
||||
)
|
||||
replace_exact(
|
||||
boot,
|
||||
''' const voice=localStorage.getItem("hermes-tts-voice")||"zh-CN-XiaoxiaoNeural";
|
||||
''',
|
||||
"",
|
||||
)
|
||||
replace_exact(
|
||||
boot,
|
||||
" body: JSON.stringify({text: clean, voice, rate, pitch})",
|
||||
" body: JSON.stringify({text: clean, rate, pitch})",
|
||||
)
|
||||
replace_exact(
|
||||
boot,
|
||||
''' const savedVoice=localStorage.getItem('hermes-tts-voice');
|
||||
const voices=speechSynthesis.getVoices();
|
||||
if(savedVoice&&voices.length){
|
||||
const match=voices.find(v=>v.name===savedVoice);
|
||||
if(match) utter.voice=match;
|
||||
}
|
||||
''',
|
||||
"",
|
||||
)
|
||||
replace_exact(boot, " tts_voice:'',\n", "")
|
||||
replace_exact(boot, " ['tts_voice','hermes-tts-voice'],\n", "")
|
||||
|
||||
config = ROOT / "api/config.py"
|
||||
replace_exact(config, ' "tts_voice": "",\n', "")
|
||||
replace_exact(config, ' "tts_voice",\n', "")
|
||||
replace_exact(
|
||||
config,
|
||||
''' if k == "tts_voice":
|
||||
if not isinstance(v, str) or len(v) > 200 or "\\x00" in v:
|
||||
continue
|
||||
''',
|
||||
"",
|
||||
)
|
||||
|
||||
assert_absent(index, "settingsTtsVoice", "settings_label_tts_voice")
|
||||
assert_absent(ui, "hermes-tts-voice", "voice:voice")
|
||||
assert_absent(panels, "settingsTtsVoice", "tts_voice")
|
||||
assert_absent(boot, "hermes-tts-voice", "tts_voice", "text: clean, voice")
|
||||
assert_absent(config, '"tts_voice"')
|
||||
|
||||
i18n = ROOT / "static/i18n.js"
|
||||
remove_lines_containing(
|
||||
i18n,
|
||||
"settings_label_tts_voice:",
|
||||
"settings_desc_tts_voice:",
|
||||
)
|
||||
assert_absent(i18n, "settings_label_tts_voice", "settings_desc_tts_voice")
|
||||
|
||||
routes = ROOT / "api/routes.py"
|
||||
marker = " # ── ElevenLabs TTS ──────────────────────────────────────────────────\n"
|
||||
atlas = ''' # ── Atlas private Jetson TTS ─────────────────────────────────────────
|
||||
@ -94,10 +251,14 @@ atlas = ''' # ── Atlas private Jetson TTS ──────────
|
||||
speed = max(0.5, min(2.0, 1.0 + (float(rate_str.rstrip("%")) / 100.0)))
|
||||
except ValueError:
|
||||
speed = 1.0
|
||||
# No "voice" or "language" field: the WebUI has no signal for the
|
||||
# language of the text being spoken (see NOTES.md), so voice
|
||||
# selection is left entirely to the TTS service's own allow-listed
|
||||
# policy (English amy) rather than sending a value that would only
|
||||
# be ignored server-side or a fabricated language guess.
|
||||
request_body = json.dumps({
|
||||
"model": "piper",
|
||||
"input": text,
|
||||
"voice": "en_US-lessac-high",
|
||||
"speed": speed,
|
||||
}).encode("utf-8")
|
||||
request = Request(atlas_url, data=request_body, headers={
|
||||
|
||||
@ -40,6 +40,40 @@ or the WebUI. Browser chat remains available when `bot_token` is empty.
|
||||
The bot token and relay key must never be added to Git or a Kubernetes Secret.
|
||||
The router does not log prompt bodies, raw Telegram IDs, link codes, or tokens.
|
||||
|
||||
## Private Jetson voice: multilingual TTS policy
|
||||
|
||||
`hermes-tts` on `titan-21` bakes three checksum-pinned Piper voices and
|
||||
selects one per request from a fixed, allow-listed `language` field: `en`/
|
||||
`en-US` → `en_US-amy-medium`, `ru`/`ru-RU` → `ru_RU-irina-medium`, `es`/
|
||||
`es-MX`/`es-ES` → `es_MX-claude-high` (Piper's `claude` voice is Mexican
|
||||
Spanish; there is no Castilian `es_ES-claude`). Matching is case-insensitive
|
||||
and accepts both `_` and `-` separators. Any language that is missing,
|
||||
unrecognized, or malformed falls back to English amy rather than erroring.
|
||||
The mapping is a fixed dict from `language` to one of the three baked model
|
||||
names only — a client-supplied `voice` field is never read, so no client
|
||||
input can select or construct a model path. All three voices are preloaded
|
||||
at process start (see `dockerfiles/hermes-jetson-tts-server.py`).
|
||||
|
||||
Hermes Chat deliberately exposes no TTS speaker/model choice. The deterministic
|
||||
WebUI image patch removes the pinned upstream voice selector, its label and
|
||||
translations, its browser/server preference persistence, and every outbound
|
||||
client `voice` field while preserving the TTS engine, speech rate/pitch,
|
||||
dictation, hands-free Voice Mode, and the conversation instrument. Legacy
|
||||
`hermes-tts-voice` browser state is deleted. Voice choice is therefore policy,
|
||||
not a client preference: validated English maps to amy, Russian to irina,
|
||||
Spanish to claude, and every unsupported or absent language falls back to amy.
|
||||
|
||||
The private WebUI voice bridge (`dockerfiles/hermes-webui-atlas-voice.js`,
|
||||
patched into `api/routes.py` by `hermes-webui-atlas-patch.py`) has no signal
|
||||
for the language of the assistant reply it is about to speak — it sends only
|
||||
`text` and `engine`. Until the WebUI or gateway attaches an explicit
|
||||
`language` field to that request, every reply speaks in the safe English
|
||||
default regardless of its actual language. Closing that gap needs a language
|
||||
signal upstream of the TTS call (e.g. tagging the assistant turn with a
|
||||
detected/declared reply language and threading it through
|
||||
`hermes-webui-atlas-voice.js` → `api/routes.py` → the `language` field), not
|
||||
client- or server-side guessing bolted onto the TTS service itself.
|
||||
|
||||
## The one-sentence explanation
|
||||
|
||||
Hermes is the persistent agent runtime and control surface; Codex or the local
|
||||
|
||||
10
testing/fixtures/hermes-webui-0.52.181/api/config.py
Normal file
10
testing/fixtures/hermes-webui-0.52.181/api/config.py
Normal file
@ -0,0 +1,10 @@
|
||||
_SETTINGS_DEFAULTS = {
|
||||
"tts_voice": "",
|
||||
}
|
||||
_SETTINGS_SPEECH_KEYS = {
|
||||
"tts_voice",
|
||||
}
|
||||
UPSTREAM_VALIDATION_FRAGMENT = ''' if k == "tts_voice":
|
||||
if not isinstance(v, str) or len(v) > 200 or "\x00" in v:
|
||||
continue
|
||||
'''
|
||||
34
testing/fixtures/hermes-webui-0.52.181/static/boot.js
Normal file
34
testing/fixtures/hermes-webui-0.52.181/static/boot.js
Normal file
@ -0,0 +1,34 @@
|
||||
function speakWithRegisteredEngine(){
|
||||
const _opts={
|
||||
voice: localStorage.getItem("hermes-tts-voice")||'',
|
||||
rate: parseFloat(localStorage.getItem("hermes-tts-rate")),
|
||||
};
|
||||
return _opts;
|
||||
}
|
||||
function speakWithEdge(clean){
|
||||
const voice=localStorage.getItem("hermes-tts-voice")||"zh-CN-XiaoxiaoNeural";
|
||||
const rate='';
|
||||
const pitch='';
|
||||
return fetch('/api/tts', {
|
||||
body: JSON.stringify({text: clean, voice, rate, pitch})
|
||||
});
|
||||
}
|
||||
function speakWithBrowser(clean){
|
||||
const utter=new SpeechSynthesisUtterance(clean);
|
||||
const savedVoice=localStorage.getItem('hermes-tts-voice');
|
||||
const voices=speechSynthesis.getVoices();
|
||||
if(savedVoice&&voices.length){
|
||||
const match=voices.find(v=>v.name===savedVoice);
|
||||
if(match) utter.voice=match;
|
||||
}
|
||||
return utter;
|
||||
}
|
||||
function _mirrorSpeechSettingsFromServer(s){
|
||||
const defaults={
|
||||
tts_voice:'',
|
||||
};
|
||||
[
|
||||
['tts_voice','hermes-tts-voice'],
|
||||
].forEach(([settingKey,storageKey])=>localStorage.setItem(storageKey,s[settingKey]));
|
||||
return defaults;
|
||||
}
|
||||
4
testing/fixtures/hermes-webui-0.52.181/static/i18n.js
Normal file
4
testing/fixtures/hermes-webui-0.52.181/static/i18n.js
Normal file
@ -0,0 +1,4 @@
|
||||
const EN = {
|
||||
settings_label_tts_voice: 'Voice',
|
||||
settings_desc_tts_voice: "Preferred voice. Populated from your browser's available voices.",
|
||||
};
|
||||
@ -5,6 +5,12 @@
|
||||
</head>
|
||||
<body>
|
||||
<select id="settingsTtsEngine"><option value="browser">Browser speech synthesis</option><option value="edge">Edge TTS (server)</option></select>
|
||||
<div class="settings-field"><label for="settingsTtsVoice" data-i18n="settings_label_tts_voice">Voice</label>
|
||||
<select id="settingsTtsVoice" style="width:100%;padding:8px;background:var(--code-bg);color:var(--text);border:1px solid var(--border2);border-radius:6px">
|
||||
<option value="">Default system voice</option>
|
||||
</select>
|
||||
<div style="font-size:11px;color:var(--muted);margin-top:4px" data-i18n="settings_desc_tts_voice">Preferred voice. Populated from your browser's available voices.</div>
|
||||
</div>
|
||||
<div class="composer-box" id="composerBox">
|
||||
<div class="voice-mode-bar" id="voiceModeBar" style="display:none">
|
||||
<span class="voice-mode-indicator" id="voiceModeIndicator"></span>
|
||||
|
||||
44
testing/fixtures/hermes-webui-0.52.181/static/panels.js
Normal file
44
testing/fixtures/hermes-webui-0.52.181/static/panels.js
Normal file
@ -0,0 +1,44 @@
|
||||
const _SETTINGS_SPEECH_STORAGE_KEYS={
|
||||
tts_engine:'hermes-tts-engine',
|
||||
tts_voice:'hermes-tts-voice',
|
||||
tts_rate:'hermes-tts-rate',
|
||||
};
|
||||
let _settingsSpeechChangedKeys=new Set();
|
||||
|
||||
function _speechPreferencesPayloadFromUi(){
|
||||
const payload={};
|
||||
const ttsVoiceSel=$('settingsTtsVoice');
|
||||
if(ttsVoiceSel) _setOwnedSpeechPayload(payload,'tts_voice',ttsVoiceSel.value||'');
|
||||
return payload;
|
||||
}
|
||||
|
||||
function loadSettingsPanel(){
|
||||
const ttsEngineSel=$('settingsTtsEngine');
|
||||
if(ttsEngineSel){
|
||||
ttsEngineSel.onchange=function(){
|
||||
localStorage.setItem('hermes-tts-engine',this.value);
|
||||
window._populateTtsVoices();
|
||||
_schedulePreferencesAutosave();
|
||||
};
|
||||
}
|
||||
// Populate voice selector based on engine
|
||||
const ttsVoiceSel=$('settingsTtsVoice');
|
||||
window._populateTtsVoices=function(){
|
||||
if(!ttsVoiceSel) return;
|
||||
const engine=localStorage.getItem('hermes-tts-engine')||'browser';
|
||||
const current=String(_speechSetting('tts_voice','hermes-tts-voice','')||'');
|
||||
_syncSpeechPreferenceCache('tts_voice',current);
|
||||
if(engine==='edge'){
|
||||
const edgeVoices=[
|
||||
{value:'en-US-AriaNeural',label:'Aria (English, Female)'},
|
||||
];
|
||||
ttsVoiceSel.innerHTML='<option value="">Default (Xiaoxiao)</option>';
|
||||
edgeVoices.forEach(v=>ttsVoiceSel.appendChild(v));
|
||||
}
|
||||
};
|
||||
if(ttsVoiceSel&&'speechSynthesis' in window){
|
||||
window._populateTtsVoices();
|
||||
ttsVoiceSel.onchange=function(){_markSpeechPreferenceChanged('tts_voice');localStorage.setItem('hermes-tts-voice',this.value);_schedulePreferencesAutosave();};
|
||||
}
|
||||
// TTS rate/pitch sliders
|
||||
}
|
||||
@ -1,4 +1,15 @@
|
||||
function _buildBrowserUtterance(text, btn){
|
||||
const utter=new SpeechSynthesisUtterance(text);
|
||||
const savedVoice=localStorage.getItem('hermes-tts-voice');
|
||||
const voices=speechSynthesis.getVoices();
|
||||
if(savedVoice&&voices.length){
|
||||
const match=voices.find(v=>v.name===savedVoice);
|
||||
if(match) utter.voice=match;
|
||||
}
|
||||
return utter;
|
||||
}
|
||||
function _playEdgeTtsChunked(text, btn){
|
||||
const voice=localStorage.getItem('hermes-tts-voice')||'zh-CN-XiaoxiaoNeural';
|
||||
return fetch('/api/tts',{body:JSON.stringify({text:chunk, voice:voice, rate:rate, pitch:pitch})});
|
||||
}
|
||||
function speakSelected(clean, btn, engine){
|
||||
@ -11,3 +22,14 @@ function speakAutomatically(clean, engine){
|
||||
_playEdgeTtsChunked(clean, null);
|
||||
}
|
||||
}
|
||||
function registeredTts(engine, clean){
|
||||
const _opts={
|
||||
voice: localStorage.getItem('hermes-tts-voice')||'',
|
||||
rate: parseFloat(localStorage.getItem('hermes-tts-rate')),
|
||||
};
|
||||
const autoOpts={
|
||||
voice: localStorage.getItem('hermes-tts-voice')||'',
|
||||
pitch: parseFloat(localStorage.getItem('hermes-tts-pitch')),
|
||||
};
|
||||
return [engine, clean, _opts, autoOpts];
|
||||
}
|
||||
|
||||
@ -398,13 +398,27 @@ def test_voice_models_are_baked_and_runtime_has_no_public_egress():
|
||||
assert "HERMES_STT_CACHE=/opt/models/whisper" in stt_dockerfile
|
||||
assert "ADD --checksum=sha256:4cabf7c3" in tts_dockerfile
|
||||
assert "ADD --checksum=sha256:db42b97d" in tts_dockerfile
|
||||
assert tts_dockerfile.count("--chmod=0444") == 6
|
||||
assert "ADD --checksum=sha256:b3a6e47b57b8c7fbe6a0ce2518161a50f59a9cdd8a50835c02cb02bdd6206c18" in tts_dockerfile
|
||||
assert "ADD --checksum=sha256:95a23eb4d42909d38df73bb9ac7f45f597dbfcde2d1bf9526fdeaf5466977d77" in tts_dockerfile
|
||||
assert "ADD --checksum=sha256:8ff38212d23da300bbe3705c645e6e5b9475f0bfde01558eb17813e22acaaaaa" in tts_dockerfile
|
||||
assert "ADD --checksum=sha256:c2ec28bb38e2b59e93b959b3e40348c1afebbd272f30fed5d41205d08e98a9d7" in tts_dockerfile
|
||||
assert "ADD --checksum=sha256:3ef40a71ea63852cd8ab7e6fa7d2ecdcfa67a0b47c9c48e3f10e02ee02083ea0" in tts_dockerfile
|
||||
assert "ADD --checksum=sha256:1afc81f703c0e4cb3b4d7c0dca096b8b54a98806807f0170cf5eb5557723c12d" in tts_dockerfile
|
||||
assert tts_dockerfile.count("--chmod=0444") == 12
|
||||
assert "/opt/models/piper/en_US-amy-medium.onnx" in tts_dockerfile
|
||||
assert "/opt/models/piper/ru_RU-irina-medium.onnx" in tts_dockerfile
|
||||
assert "/opt/models/piper/es_MX-claude-high.onnx" in tts_dockerfile
|
||||
assert "chmod 0555 /opt/models /opt/models/piper" in tts_dockerfile
|
||||
assert "HERMES_TTS_CACHE=/opt/models/piper" in tts_dockerfile
|
||||
assert "HERMES_TTS_VOICE=en_US-amy-medium" in tts_dockerfile
|
||||
tts_server = (ROOT / "dockerfiles" / "hermes-jetson-tts-server.py").read_text()
|
||||
assert "download_voice" not in tts_server
|
||||
assert "baked Piper voice is missing" in tts_server
|
||||
assert "session_options.intra_op_num_threads = ONNX_THREADS" in tts_server
|
||||
assert "session_options.intra_op_num_threads = threads" in tts_server
|
||||
assert 'LANGUAGE_VOICE_MAP = {' in tts_server
|
||||
assert '"en": "en_US-amy-medium"' in tts_server
|
||||
assert '"ru": "ru_RU-irina-medium"' in tts_server
|
||||
assert '"es": "es_MX-claude-high"' in tts_server
|
||||
|
||||
policies = _documents(HERMES / "networkpolicy.yaml")
|
||||
voice_policy = next(
|
||||
|
||||
220
testing/tests/test_hermes_tts_language_routing.py
Normal file
220
testing/tests/test_hermes_tts_language_routing.py
Normal file
@ -0,0 +1,220 @@
|
||||
"""Language allow-list contracts for the private Hermes chat TTS voice policy."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
import io
|
||||
import json
|
||||
import sys
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
|
||||
from testing.tests.test_hermes_chat_support import ROOT
|
||||
|
||||
AMY = "en_US-amy-medium"
|
||||
IRINA = "ru_RU-irina-medium"
|
||||
CLAUDE = "es_MX-claude-high"
|
||||
|
||||
|
||||
def _load_tts_server(monkeypatch):
|
||||
server_path = ROOT / "dockerfiles" / "hermes-jetson-tts-server.py"
|
||||
spec = importlib.util.spec_from_file_location("hermes_jetson_tts_server", server_path)
|
||||
assert spec and spec.loader
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
|
||||
class _FakeSessionOptions:
|
||||
def __init__(self) -> None:
|
||||
self.intra_op_num_threads = None
|
||||
self.inter_op_num_threads = None
|
||||
|
||||
fake_onnxruntime = SimpleNamespace(
|
||||
SessionOptions=_FakeSessionOptions,
|
||||
InferenceSession=lambda *a, **k: SimpleNamespace(),
|
||||
)
|
||||
fake_piper = SimpleNamespace(
|
||||
PiperConfig=SimpleNamespace(from_dict=lambda d: d),
|
||||
PiperVoice=lambda **kwargs: SimpleNamespace(**kwargs),
|
||||
SynthesisConfig=lambda **kwargs: SimpleNamespace(**kwargs),
|
||||
)
|
||||
monkeypatch.setitem(sys.modules, "onnxruntime", fake_onnxruntime)
|
||||
monkeypatch.setitem(sys.modules, "piper", fake_piper)
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def tts(monkeypatch):
|
||||
return _load_tts_server(monkeypatch)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"language,expected",
|
||||
[
|
||||
("en", AMY),
|
||||
("en-US", AMY),
|
||||
("en_US", AMY),
|
||||
("EN", AMY),
|
||||
("En-Us", AMY),
|
||||
("ru", IRINA),
|
||||
("ru-RU", IRINA),
|
||||
("ru_RU", IRINA),
|
||||
("RU", IRINA),
|
||||
("es", CLAUDE),
|
||||
("es-MX", CLAUDE),
|
||||
("es_MX", CLAUDE),
|
||||
("es-ES", CLAUDE),
|
||||
("es_ES", CLAUDE),
|
||||
("ES", CLAUDE),
|
||||
],
|
||||
)
|
||||
def test_allow_listed_languages_resolve_to_the_approved_voice(tts, language, expected):
|
||||
assert tts.resolve_voice_name(language) == expected
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"language",
|
||||
[
|
||||
None,
|
||||
"",
|
||||
" ",
|
||||
"fr",
|
||||
"fr-FR",
|
||||
"de-DE",
|
||||
"xx",
|
||||
"en-GB",
|
||||
"es-AR",
|
||||
"english",
|
||||
123,
|
||||
1.5,
|
||||
True,
|
||||
[],
|
||||
{},
|
||||
{"lang": "ru"},
|
||||
"../../etc/passwd",
|
||||
"en_US-amy-medium/../../ru_RU-irina-medium",
|
||||
"\x00ru",
|
||||
"ru\x00",
|
||||
],
|
||||
)
|
||||
def test_unknown_missing_or_malformed_language_falls_back_to_amy(tts, language):
|
||||
assert tts.resolve_voice_name(language) == AMY
|
||||
|
||||
|
||||
def test_default_voice_name_matches_the_dockerfile_env_default(tts):
|
||||
assert tts.DEFAULT_VOICE_NAME == AMY
|
||||
|
||||
|
||||
def test_resolved_voice_is_always_one_of_the_three_baked_names(tts):
|
||||
assert frozenset({AMY, IRINA, CLAUDE}) == tts.BAKED_VOICE_NAMES
|
||||
fuzz_inputs = [
|
||||
"en", "ru", "es", "unknown", "", None, 42, "../../../etc/shadow",
|
||||
"en_US-amy-medium\x00; rm -rf /", "RU-ru", "Es-Es", "en-us-extra",
|
||||
]
|
||||
for value in fuzz_inputs:
|
||||
assert tts.resolve_voice_name(value) in tts.BAKED_VOICE_NAMES
|
||||
|
||||
|
||||
def test_client_voice_field_cannot_override_the_language_policy(tts):
|
||||
"""The POST handler must select the voice from "language" only.
|
||||
|
||||
A malicious or stale "voice" field in a hostile/legacy request must never
|
||||
change which baked model answers the request.
|
||||
"""
|
||||
calls: list[str] = []
|
||||
|
||||
class _RecordingVoice:
|
||||
def __init__(self, name: str) -> None:
|
||||
self.name = name
|
||||
|
||||
def synthesize_wav(self, text, wav_file, syn_config) -> None:
|
||||
calls.append(self.name)
|
||||
wav_file.setnchannels(1)
|
||||
wav_file.setsampwidth(2)
|
||||
wav_file.setframerate(16_000)
|
||||
wav_file.writeframes(b"\x00\x00")
|
||||
|
||||
class _RecordingHandler(tts.SpeechHandler):
|
||||
def __init__(self, payload):
|
||||
request = json.dumps(payload).encode("utf-8")
|
||||
self.path = "/v1/audio/speech"
|
||||
self.headers = {"Content-Length": str(len(request))}
|
||||
self.rfile = io.BytesIO(request)
|
||||
self.wfile = io.BytesIO()
|
||||
self.status = None
|
||||
self.response_headers = {}
|
||||
self.server = SimpleNamespace(
|
||||
voices={
|
||||
AMY: _RecordingVoice(AMY),
|
||||
IRINA: _RecordingVoice(IRINA),
|
||||
CLAUDE: _RecordingVoice(CLAUDE),
|
||||
},
|
||||
default_voice_name=AMY,
|
||||
)
|
||||
|
||||
def send_response(self, status, message=None):
|
||||
self.status = status
|
||||
|
||||
def send_header(self, name, value):
|
||||
self.response_headers[name] = value
|
||||
|
||||
def end_headers(self):
|
||||
return None
|
||||
|
||||
# A payload that supplies an attacker/legacy "voice" value but no
|
||||
# language must resolve to the safe default, never the "voice" value.
|
||||
handler = _RecordingHandler({"input": "hi", "voice": IRINA})
|
||||
handler.do_POST()
|
||||
assert handler.status == 200
|
||||
assert handler.response_headers["X-TTS-Voice"] == AMY
|
||||
|
||||
# A payload supplying both must still be governed by "language" alone.
|
||||
handler = _RecordingHandler({"input": "hi", "voice": CLAUDE, "language": "ru"})
|
||||
handler.do_POST()
|
||||
assert handler.status == 200
|
||||
assert handler.response_headers["X-TTS-Voice"] == IRINA
|
||||
|
||||
assert calls == [AMY, IRINA]
|
||||
|
||||
|
||||
def test_no_client_string_reaches_a_filesystem_path(tts):
|
||||
"""resolve_voice_name must only ever return a fixed, baked literal.
|
||||
|
||||
This is the property that keeps a client from ever causing the server to
|
||||
build a Path out of attacker-controlled text: the return value is always
|
||||
a member of the fixed allow-list, regardless of input shape.
|
||||
"""
|
||||
hostile_inputs = [
|
||||
"../../../../etc/passwd",
|
||||
"/etc/passwd",
|
||||
"en_US-amy-medium/../../../etc/passwd",
|
||||
"ru_RU-irina-medium\x00.onnx",
|
||||
"es_MX-claude-high; cat /etc/shadow",
|
||||
"\n\ren",
|
||||
"en" + "/" * 200,
|
||||
" ",
|
||||
]
|
||||
for value in hostile_inputs:
|
||||
result = tts.resolve_voice_name(value)
|
||||
assert result in tts.BAKED_VOICE_NAMES
|
||||
assert "/" not in result
|
||||
assert ".." not in result
|
||||
assert "\x00" not in result
|
||||
|
||||
|
||||
def test_normalize_language_rejects_non_string_input(tts):
|
||||
assert tts.normalize_language(None) is None
|
||||
assert tts.normalize_language(123) is None
|
||||
assert tts.normalize_language([]) is None
|
||||
assert tts.normalize_language("") is None
|
||||
assert tts.normalize_language(" ") is None
|
||||
assert tts.normalize_language("En_US") == "en-us"
|
||||
|
||||
|
||||
def test_default_voice_name_is_one_of_the_baked_voices(tts):
|
||||
assert tts.DEFAULT_VOICE_NAME in tts.BAKED_VOICE_NAMES
|
||||
|
||||
|
||||
def test_load_voices_fails_closed_when_a_baked_model_is_missing(tts, tmp_path):
|
||||
with pytest.raises(RuntimeError, match="baked Piper voice is missing"):
|
||||
tts.load_voices(tmp_path, threads=1)
|
||||
@ -57,6 +57,55 @@ def test_real_upstream_fixture_receives_visual_instrument_contract(tmp_path: Pat
|
||||
assert '<button' not in index[index.index('id="voiceModeBar"') : index.index('<textarea')]
|
||||
|
||||
|
||||
def test_patched_webui_has_no_user_voice_choice_or_client_voice_field(
|
||||
tmp_path: Path,
|
||||
):
|
||||
"""The pinned settings DOM and every TTS path leave speakers to policy."""
|
||||
target = _patched_fixture(tmp_path)
|
||||
index = (target / "static/index.html").read_text(encoding="utf-8")
|
||||
ui = (target / "static/ui.js").read_text(encoding="utf-8")
|
||||
panels = (target / "static/panels.js").read_text(encoding="utf-8")
|
||||
boot = (target / "static/boot.js").read_text(encoding="utf-8")
|
||||
i18n = (target / "static/i18n.js").read_text(encoding="utf-8")
|
||||
config = (target / "api/config.py").read_text(encoding="utf-8")
|
||||
routes = (target / "api/routes.py").read_text(encoding="utf-8")
|
||||
|
||||
assert "settingsTtsVoice" not in index
|
||||
assert "settings_label_tts_voice" not in index
|
||||
assert "settings_desc_tts_voice" not in index
|
||||
assert "Default system voice" not in index
|
||||
assert 'id="settingsTtsEngine"' in index
|
||||
assert 'id="btnVoiceMode"' in index
|
||||
assert 'id="voiceModeBar"' in index
|
||||
|
||||
assert "hermes-tts-voice" not in ui
|
||||
assert "voice:voice" not in ui
|
||||
assert (
|
||||
"body:JSON.stringify({text:chunk, rate:rate, pitch:pitch, "
|
||||
"engine:engineOverride||'edge'})"
|
||||
) in ui
|
||||
assert "settingsTtsVoice" not in panels
|
||||
assert "tts_voice" not in panels
|
||||
assert "localStorage.removeItem('hermes-tts-voice')" in panels
|
||||
assert panels.count("hermes-tts-voice") == 1
|
||||
assert "hermes-tts-voice" not in boot
|
||||
assert "tts_voice" not in boot
|
||||
assert "text: clean, voice" not in boot
|
||||
assert '"tts_voice"' not in config
|
||||
assert "settings_label_tts_voice" not in i18n
|
||||
assert "settings_desc_tts_voice" not in i18n
|
||||
|
||||
atlas_route = routes.split('if engine == "atlas":', 1)[1].split(
|
||||
"# ── ElevenLabs TTS", 1
|
||||
)[0]
|
||||
request_body = atlas_route.split("request_body = json.dumps({", 1)[1].split(
|
||||
"}).encode", 1
|
||||
)[0]
|
||||
assert '"input": text' in request_body
|
||||
assert '"voice"' not in request_body
|
||||
assert '"language"' not in request_body
|
||||
|
||||
|
||||
def test_visual_states_have_distinct_layers_finite_error_and_reduced_motion():
|
||||
css = VOICE_CSS.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user