Compare commits

...

2 Commits

Author SHA1 Message Date
Hermes Agent
6f1b0065f1 feat(hermes-voice): route Whisper language to multilingual Piper
Port the original #27 detected-language pipeline onto the verified PR #39 prerequisite while preserving the current-main conversation instrument and host continuity changes.

Keep voice selection server-side with no user selector or client voice field. Reuse 207c16ab only for its stricter exact-code trust boundary, omitting malformed or absent language so Piper defaults to Amy.
2026-08-21 11:52:49 +00:00
Hermes Agent
9e14acf390 feat(hermes-tts): prepare fixed multilingual voice policy
Supersede draft PR #26 with a merge-safe prerequisite: bake and preload the amy, irina, and claude Piper models, route only validated server-side language to fixed voices, and leave the live voice deployment manifest unchanged.

Remove the pinned WebUI speaker selector and its persisted preference, omit client voice fields from every outbound TTS path, and keep hands-free Voice Mode and the conversation instrument intact. Hostile or legacy voice fields remain ignored by the Piper server.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
2026-08-21 11:26:54 +00:00
22 changed files with 2186 additions and 41 deletions

View File

@ -24,18 +24,41 @@ ADD --checksum=sha256:f7d01dde371555732c4c314111ac79672b1a5ce2fc19266ab42178fd8d
ADD --checksum=sha256:45754dfdebb3b8661c3fc564713772deec6e064feeb5b4e9594857dc7305193a --chmod=0444 \
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/en/en_US/lessac/low/en_US-lessac-low.onnx.json?download=true \
/opt/models/piper/en_US-lessac-low.onnx.json
# Multilingual chat voice policy: English -> amy, Russian -> irina, Spanish ->
# claude (Mexican Spanish, the only "claude" voice rhasspy/piper-voices
# publishes; there is no es_ES-claude).
ADD --checksum=sha256:b3a6e47b57b8c7fbe6a0ce2518161a50f59a9cdd8a50835c02cb02bdd6206c18 --chmod=0444 \
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/en/en_US/amy/medium/en_US-amy-medium.onnx?download=true \
/opt/models/piper/en_US-amy-medium.onnx
ADD --checksum=sha256:95a23eb4d42909d38df73bb9ac7f45f597dbfcde2d1bf9526fdeaf5466977d77 --chmod=0444 \
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/en/en_US/amy/medium/en_US-amy-medium.onnx.json?download=true \
/opt/models/piper/en_US-amy-medium.onnx.json
ADD --checksum=sha256:8ff38212d23da300bbe3705c645e6e5b9475f0bfde01558eb17813e22acaaaaa --chmod=0444 \
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/ru/ru_RU/irina/medium/ru_RU-irina-medium.onnx?download=true \
/opt/models/piper/ru_RU-irina-medium.onnx
ADD --checksum=sha256:c2ec28bb38e2b59e93b959b3e40348c1afebbd272f30fed5d41205d08e98a9d7 --chmod=0444 \
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/ru/ru_RU/irina/medium/ru_RU-irina-medium.onnx.json?download=true \
/opt/models/piper/ru_RU-irina-medium.onnx.json
ADD --checksum=sha256:3ef40a71ea63852cd8ab7e6fa7d2ecdcfa67a0b47c9c48e3f10e02ee02083ea0 --chmod=0444 \
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/es/es_MX/claude/high/es_MX-claude-high.onnx?download=true \
/opt/models/piper/es_MX-claude-high.onnx
ADD --checksum=sha256:1afc81f703c0e4cb3b4d7c0dca096b8b54a98806807f0170cf5eb5557723c12d --chmod=0444 \
https://huggingface.co/rhasspy/piper-voices/resolve/ea046e8458f6acd997706d6e6066a022b42f6fb1/es/es_MX/claude/high/es_MX-claude-high.onnx.json?download=true \
/opt/models/piper/es_MX-claude-high.onnx.json
RUN chmod 0555 /opt/models /opt/models/piper
COPY dockerfiles/hermes-jetson-tts-server.py /opt/atlas/hermes-jetson-tts-server.py
RUN chmod 0555 /opt/atlas/hermes-jetson-tts-server.py
# Load the actual pinned voice during the ARM64 build. This catches package or
# model-format drift before the image can reach Flux.
RUN python -c "import stat; from pathlib import Path; from piper import PiperVoice; p=Path('/opt/models/piper'); models=[p/'en_US-lessac-high.onnx',p/'en_US-lessac-medium.onnx',p/'en_US-lessac-low.onnx']; assert stat.S_IMODE(p.stat().st_mode)==0o555; assert all(stat.S_IMODE(model.stat().st_mode)==0o444 for model in models); voices=[PiperVoice.load(model,Path(str(model)+'.json'),use_cuda=False,download_dir=p) for model in models]; assert all(voice.config.sample_rate>0 for voice in voices)"
# Load every pinned voice during the ARM64 build, including the three baked
# for the multilingual chat policy. This catches package or model-format
# drift before the image can reach Flux.
RUN python -c "import stat; from pathlib import Path; from piper import PiperVoice; p=Path('/opt/models/piper'); models=[p/'en_US-lessac-high.onnx',p/'en_US-lessac-medium.onnx',p/'en_US-lessac-low.onnx',p/'en_US-amy-medium.onnx',p/'ru_RU-irina-medium.onnx',p/'es_MX-claude-high.onnx']; assert stat.S_IMODE(p.stat().st_mode)==0o555; assert all(stat.S_IMODE(model.stat().st_mode)==0o444 for model in models); voices=[PiperVoice.load(model,Path(str(model)+'.json'),use_cuda=False,download_dir=p) for model in models]; assert all(voice.config.sample_rate>0 for voice in voices)"
ENV HERMES_TTS_HOST=0.0.0.0 \
HERMES_TTS_PORT=9001 \
HERMES_TTS_VOICE=en_US-lessac-medium \
HERMES_TTS_VOICE=en_US-amy-medium \
HERMES_TTS_CACHE=/opt/models/piper \
OMP_NUM_THREADS=2 \
PYTHONDONTWRITEBYTECODE=1 \

View File

@ -116,9 +116,16 @@ RUN /opt/hermes/.venv/bin/python -c 'import cryptography, yaml' \
&& grep -Fq 'routing_priority:priority' /opt/hermes-webui/static/atlas-router.js \
&& grep -Fq "'atlas/auto/fast':'Automatic · Fast'" /opt/hermes-webui/static/atlas-router.js \
&& grep -Fq 'explicit_reasoning_effort' /opt/hermes-webui/api/gateway_chat.py \
&& grep -Fq '"language": detected_language' /opt/hermes/tools/transcription_tools.py \
&& grep -Fq "'language': detected" /opt/hermes-webui/api/upload.py \
&& grep -Fq 'def _atlas_tts_language(body):' /opt/hermes-webui/api/routes.py \
&& grep -Fq 'request_payload["language"] = _atlas_language' /opt/hermes-webui/api/routes.py \
&& grep -Fq 'takeSttLanguage(token)' /opt/hermes-webui/static/atlas-voice.js \
&& /opt/hermes/.venv/bin/python -m py_compile \
/opt/hermes-webui/api/routes.py \
/opt/hermes-webui/api/gateway_chat.py
/opt/hermes-webui/api/upload.py \
/opt/hermes-webui/api/gateway_chat.py \
/opt/hermes/tools/transcription_tools.py
# Exercise the real server process in the target architecture before publish.
RUN set -eu; \

View File

@ -62,6 +62,34 @@ def _clean_transcript(result: dict) -> str:
return " ".join(kept).strip()
def _detected_language(result: object) -> str:
"""Return the bare ISO-639 code Whisper decoded with, or nothing at all.
``whisper.transcribe`` reports the language it auto-detected (or the one it
was told to use) as a plain lowercase token such as ``en``/``ru``/``yue``.
Anything that is not that exact shape is dropped rather than guessed at, so
a surprising model result can never become a downstream voice selector.
"""
if not isinstance(result, dict):
return ""
value = result.get("language")
if not isinstance(value, str):
return ""
code = value.strip().lower()
if not 2 <= len(code) <= 3 or not code.isascii() or not code.isalpha():
return ""
return code
def _transcription_payload(result: dict) -> dict:
"""Build the transcription contract: text plus the model's own language."""
return {
"text": _clean_transcript(result),
"model": MODEL_NAME,
"language": _detected_language(result),
}
def _json(handler: BaseHTTPRequestHandler, status: int, payload: dict) -> None:
body = json.dumps(payload).encode("utf-8")
handler.send_response(status)
@ -148,8 +176,7 @@ class SpeechHandler(BaseHTTPRequestHandler):
no_speech_threshold=0.5,
verbose=False,
)
transcript = _clean_transcript(result)
_json(self, 200, {"text": transcript, "model": MODEL_NAME})
_json(self, 200, _transcription_payload(result))
except Exception as exc:
print(f"[stt] transcription failed: {exc}", flush=True)
_json(self, 500, {"error": "transcription failed"})

View File

@ -17,12 +17,50 @@ from piper import PiperConfig, PiperVoice, SynthesisConfig
HOST = os.getenv("HERMES_TTS_HOST", "0.0.0.0")
PORT = int(os.getenv("HERMES_TTS_PORT", "9001"))
VOICE_NAME = os.getenv("HERMES_TTS_VOICE", "en_US-lessac-high")
CACHE_DIR = Path(os.getenv("HERMES_TTS_CACHE", "/cache/piper"))
MAX_TEXT_CHARS = 5000
ONNX_THREADS = max(1, int(os.getenv("HERMES_TTS_ONNX_THREADS", "4")))
VOICE_LOCK = threading.Lock()
# Fixed, allow-listed language -> baked voice mapping. This is the ONLY path
# from a client-supplied string to a model name: client input is looked up
# here and never used to build a filesystem path directly. Both "-" and "_"
# separators and any case are accepted; anything not present here falls back
# to DEFAULT_VOICE_NAME (safe English default), never an error and never an
# unbaked model.
LANGUAGE_VOICE_MAP = {
"en": "en_US-amy-medium",
"en-us": "en_US-amy-medium",
"ru": "ru_RU-irina-medium",
"ru-ru": "ru_RU-irina-medium",
"es": "es_MX-claude-high",
"es-mx": "es_MX-claude-high",
"es-es": "es_MX-claude-high",
}
BAKED_VOICE_NAMES = frozenset(LANGUAGE_VOICE_MAP.values())
DEFAULT_VOICE_NAME = os.getenv("HERMES_TTS_VOICE", "en_US-amy-medium")
def normalize_language(value: object) -> str | None:
"""Lowercase and fold "_"/"-" separators; reject non-string/blank input."""
if not isinstance(value, str):
return None
normalized = value.strip().lower().replace("_", "-")
return normalized or None
def resolve_voice_name(language: object) -> str:
"""Map a client-supplied language to one of the baked policy voices.
Unknown, missing, or malformed language always resolves to the safe
default rather than raising, and the result is always a member of
BAKED_VOICE_NAMES.
"""
normalized = normalize_language(language)
if normalized is None:
return DEFAULT_VOICE_NAME
return LANGUAGE_VOICE_MAP.get(normalized, DEFAULT_VOICE_NAME)
def _json(handler: BaseHTTPRequestHandler, status: int, payload: dict) -> None:
body = json.dumps(payload).encode("utf-8")
@ -46,7 +84,16 @@ class SpeechHandler(BaseHTTPRequestHandler):
if self.path != "/health":
_json(self, 404, {"error": "not found"})
return
_json(self, 200, {"ok": True, "voice": VOICE_NAME, "device": "cpu"})
_json(
self,
200,
{
"ok": True,
"voices": sorted(self.server.voices), # type: ignore[attr-defined]
"default_voice": self.server.default_voice_name, # type: ignore[attr-defined]
"device": "cpu",
},
)
def do_POST(self) -> None:
if self.path != "/v1/audio/speech":
@ -71,10 +118,16 @@ class SpeechHandler(BaseHTTPRequestHandler):
return
speed = min(2.0, max(0.5, speed))
# Policy is driven ONLY by "language". A client-supplied "voice"
# field is deliberately never read here; it cannot override the
# allow-listed mapping.
voice_name = resolve_voice_name(payload.get("language"))
voice = self.server.voices[voice_name] # type: ignore[attr-defined]
output = io.BytesIO()
try:
with VOICE_LOCK, wave.open(output, "wb") as wav_file:
self.server.voice.synthesize_wav( # type: ignore[attr-defined]
voice.synthesize_wav(
text,
wav_file,
SynthesisConfig(length_scale=1.0 / speed),
@ -84,6 +137,7 @@ class SpeechHandler(BaseHTTPRequestHandler):
self.send_header("Content-Type", "audio/wav")
self.send_header("Content-Length", str(len(audio)))
self.send_header("Cache-Control", "no-store")
self.send_header("X-TTS-Voice", voice_name)
self.end_headers()
self.wfile.write(audio)
except Exception as exc:
@ -91,30 +145,54 @@ class SpeechHandler(BaseHTTPRequestHandler):
_json(self, 500, {"error": "speech synthesis failed"})
def main() -> None:
"""Load the checksum-pinned voice from the image and serve it on CPU."""
CACHE_DIR.mkdir(parents=True, exist_ok=True)
model_path = CACHE_DIR / f"{VOICE_NAME}.onnx"
config_path = CACHE_DIR / f"{VOICE_NAME}.onnx.json"
def _load_voice(cache_dir: Path, voice_name: str, threads: int) -> PiperVoice:
model_path = cache_dir / f"{voice_name}.onnx"
config_path = cache_dir / f"{voice_name}.onnx.json"
if not model_path.exists() or not config_path.exists():
raise RuntimeError(f"baked Piper voice is missing: {VOICE_NAME}")
raise RuntimeError(f"baked Piper voice is missing: {voice_name}")
with config_path.open("r", encoding="utf-8") as config_file:
config = PiperConfig.from_dict(json.load(config_file))
session_options = onnxruntime.SessionOptions()
session_options.intra_op_num_threads = ONNX_THREADS
session_options.intra_op_num_threads = threads
session_options.inter_op_num_threads = 1
session = onnxruntime.InferenceSession(
str(model_path),
sess_options=session_options,
providers=["CPUExecutionProvider"],
)
voice = PiperVoice(session=session, config=config, download_dir=CACHE_DIR)
return PiperVoice(session=session, config=config, download_dir=cache_dir)
def load_voices(cache_dir: Path, threads: int) -> dict[str, PiperVoice]:
"""Eagerly load all three policy voices.
Preload (not lazy-load-on-first-use) was chosen deliberately: measured
RSS on this model set is ~88MB for one voice and ~243MB for all three
(~+155MB versus the previous single-voice baseline), which comfortably
fits the pod's memory budget on the CPU-only voice node. Preloading
avoids a slow, request-serializing first synthesis per language and
keeps the fail-closed missing-model check (below) at process start
rather than deferring a possible crash to a live user request.
"""
return {name: _load_voice(cache_dir, name, threads) for name in sorted(BAKED_VOICE_NAMES)}
def main() -> None:
"""Load the checksum-pinned policy voices from the image and serve them on CPU."""
CACHE_DIR.mkdir(parents=True, exist_ok=True)
if DEFAULT_VOICE_NAME not in BAKED_VOICE_NAMES:
raise RuntimeError(
f"HERMES_TTS_VOICE must name one of the baked policy voices: {sorted(BAKED_VOICE_NAMES)}"
)
voices = load_voices(CACHE_DIR, ONNX_THREADS)
print(
f"[tts] loaded Piper voice {VOICE_NAME} on CPU with {ONNX_THREADS} ONNX threads",
f"[tts] loaded {len(voices)} Piper voices on CPU with {ONNX_THREADS} ONNX threads each: "
+ ", ".join(sorted(voices)),
flush=True,
)
server = ThreadingHTTPServer((HOST, PORT), SpeechHandler)
server.voice = voice # type: ignore[attr-defined]
server.voices = voices # type: ignore[attr-defined]
server.default_voice_name = DEFAULT_VOICE_NAME # type: ignore[attr-defined]
print(f"[tts] ready on {HOST}:{PORT}", flush=True)
server.serve_forever(poll_interval=0.25)

View File

@ -6,6 +6,9 @@ from pathlib import Path
ROOT = Path(os.environ.get("HERMES_WEBUI_PATCH_ROOT", "/opt/hermes-webui"))
# The WebUI imports the pinned agent's STT tooling from the same image, so the
# local-command transcription envelope is patched alongside the WebUI itself.
AGENT_ROOT = Path(os.environ.get("HERMES_AGENT_PATCH_ROOT", "/opt/hermes"))
def replace_exact(path: Path, before: str, after: str, count: int = 1) -> None:
@ -16,6 +19,43 @@ def replace_exact(path: Path, before: str, after: str, count: int = 1) -> None:
path.write_text(source.replace(before, after, count), encoding="utf-8")
def replace_between_exact(
path: Path, start: str, end: str, after: str = "", count: int = 1
) -> None:
"""Replace one exact, bounded upstream region and fail when the pin drifts."""
source = path.read_text(encoding="utf-8")
if source.count(start) != count or source.count(end) != count:
raise SystemExit(
f"Atlas voice patch context changed in {path}: {start[:80]!r}"
)
start_index = source.index(start)
end_index = source.index(end, start_index) + len(end)
path.write_text(
source[:start_index] + after + source[end_index:], encoding="utf-8"
)
def assert_absent(path: Path, *needles: str) -> None:
"""Fail the image build if a removed voice-choice surface remains."""
source = path.read_text(encoding="utf-8")
remaining = [needle for needle in needles if needle in source]
if remaining:
raise SystemExit(f"Atlas voice choice remains in {path}: {remaining!r}")
def remove_lines_containing(path: Path, *needles: str) -> None:
"""Remove all pinned translation entries for a retired settings control."""
source = path.read_text(encoding="utf-8")
for needle in needles:
if needle not in source:
raise SystemExit(f"Atlas voice patch context changed in {path}: {needle!r}")
lines = source.splitlines(keepends=True)
path.write_text(
"".join(line for line in lines if not any(n in line for n in needles)),
encoding="utf-8",
)
index = ROOT / "static/index.html"
replace_exact(
index,
@ -29,6 +69,16 @@ replace_exact(
'<option value="browser">Browser speech synthesis</option><option value="edge">Edge TTS (server)</option>',
'<option value="atlas">Atlas Jetson (private)</option><option value="browser">Browser speech synthesis</option><option value="edge">Edge TTS (server)</option>',
)
replace_exact(
index,
'''<div class="settings-field"><label for="settingsTtsVoice" data-i18n="settings_label_tts_voice">Voice</label>
<select id="settingsTtsVoice" style="width:100%;padding:8px;background:var(--code-bg);color:var(--text);border:1px solid var(--border2);border-radius:6px">
<option value="">Default system voice</option>
</select>
<div style="font-size:11px;color:var(--muted);margin-top:4px" data-i18n="settings_desc_tts_voice">Preferred voice. Populated from your browser's available voices.</div>
</div>''',
"",
)
replace_exact(
index,
'<script src="static/boot.js?v=__WEBUI_VERSION__" defer></script>',
@ -62,11 +112,33 @@ replace_exact(
)
ui = ROOT / "static/ui.js"
replace_exact(
ui,
''' const savedVoice=localStorage.getItem('hermes-tts-voice');
const voices=speechSynthesis.getVoices();
if(savedVoice&&voices.length){
const match=voices.find(v=>v.name===savedVoice);
if(match) utter.voice=match;
}
''',
"",
)
replace_exact(ui, "function _playEdgeTtsChunked(text, btn){", "function _playEdgeTtsChunked(text, btn, engineOverride){")
replace_exact(
ui,
" const voice=localStorage.getItem('hermes-tts-voice')||'zh-CN-XiaoxiaoNeural';\n",
"",
)
replace_exact(
ui,
"body:JSON.stringify({text:chunk, voice:voice, rate:rate, pitch:pitch})",
"body:JSON.stringify({text:chunk, voice:voice, rate:rate, pitch:pitch, engine:engineOverride||'edge'})",
"body:JSON.stringify({text:chunk, rate:rate, pitch:pitch, engine:engineOverride||'edge'})",
)
replace_exact(
ui,
" voice: localStorage.getItem('hermes-tts-voice')||'',\n",
"",
count=2,
)
replace_exact(
ui,
@ -79,7 +151,177 @@ replace_exact(
"if(engine==='edge'||engine==='atlas'){\n _playEdgeTtsChunked(clean, null, engine);",
)
panels = ROOT / "static/panels.js"
replace_exact(panels, " tts_voice:'hermes-tts-voice',\n", "")
replace_exact(
panels,
''' const ttsVoiceSel=$('settingsTtsVoice');
if(ttsVoiceSel) _setOwnedSpeechPayload(payload,'tts_voice',ttsVoiceSel.value||'');
''',
"",
)
replace_exact(
panels,
''' localStorage.setItem('hermes-tts-engine',this.value);
window._populateTtsVoices();
_schedulePreferencesAutosave();''',
''' localStorage.setItem('hermes-tts-engine',this.value);
_schedulePreferencesAutosave();''',
)
replace_between_exact(
panels,
" // Populate voice selector based on engine\n",
" // TTS rate/pitch sliders\n",
" // TTS speaker selection is intentionally server policy only.\n",
)
replace_exact(
panels,
"let _settingsSpeechChangedKeys=new Set();\n",
"let _settingsSpeechChangedKeys=new Set();\n"
"try{localStorage.removeItem('hermes-tts-voice');}catch(_){}\n",
)
boot = ROOT / "static/boot.js"
replace_exact(
boot,
''' voice: localStorage.getItem("hermes-tts-voice")||'',
''',
"",
)
replace_exact(
boot,
''' const voice=localStorage.getItem("hermes-tts-voice")||"zh-CN-XiaoxiaoNeural";
''',
"",
)
replace_exact(
boot,
" body: JSON.stringify({text: clean, voice, rate, pitch})",
" body: JSON.stringify({text: clean, rate, pitch})",
)
replace_exact(
boot,
''' const savedVoice=localStorage.getItem('hermes-tts-voice');
const voices=speechSynthesis.getVoices();
if(savedVoice&&voices.length){
const match=voices.find(v=>v.name===savedVoice);
if(match) utter.voice=match;
}
''',
"",
)
replace_exact(boot, " tts_voice:'',\n", "")
replace_exact(boot, " ['tts_voice','hermes-tts-voice'],\n", "")
config = ROOT / "api/config.py"
replace_exact(config, ' "tts_voice": "",\n', "")
replace_exact(config, ' "tts_voice",\n', "")
replace_exact(
config,
''' if k == "tts_voice":
if not isinstance(v, str) or len(v) > 200 or "\\x00" in v:
continue
''',
"",
)
assert_absent(index, "settingsTtsVoice", "settings_label_tts_voice")
assert_absent(ui, "hermes-tts-voice", "voice:voice")
assert_absent(panels, "settingsTtsVoice", "tts_voice")
assert_absent(boot, "hermes-tts-voice", "tts_voice", "text: clean, voice")
assert_absent(config, '"tts_voice"')
i18n = ROOT / "static/i18n.js"
remove_lines_containing(
i18n,
"settings_label_tts_voice:",
"settings_desc_tts_voice:",
)
assert_absent(i18n, "settings_label_tts_voice", "settings_desc_tts_voice")
# The private Whisper service reports the language it decoded with. Carry that
# through the agent's local-command STT envelope so the WebUI can hand a voice
# hint to Piper instead of guessing the reply's language from its text.
transcription = AGENT_ROOT / "tools/transcription_tools.py"
replace_exact(
transcription,
''' transcript_text = txt_files[0].read_text(encoding="utf-8").strip()
logger.info(
"Transcribed %s via local STT command (%s, %d chars)",
Path(file_path).name,
normalized_model,
len(transcript_text),
)
return {"success": True, "transcript": transcript_text, "provider": "local_command"}
''',
''' transcript_text = txt_files[0].read_text(encoding="utf-8").strip()
logger.info(
"Transcribed %s via local STT command (%s, %d chars)",
Path(file_path).name,
normalized_model,
len(transcript_text),
)
detected_language = ""
language_files = sorted(Path(output_dir).glob("*.language"))
if language_files:
try:
candidate = language_files[0].read_text(encoding="utf-8").strip().lower()
except (OSError, ValueError):
candidate = ""
if 2 <= len(candidate) <= 3 and candidate.isascii() and candidate.isalpha():
detected_language = candidate
return {
"success": True,
"transcript": transcript_text,
"provider": "local_command",
"language": detected_language,
}
''',
)
upload = ROOT / "api/upload.py"
replace_exact(
upload,
""" transcript = str(result.get('transcript') or '').strip()
return j(handler, {'ok': True, 'transcript': transcript})
""",
""" transcript = str(result.get('transcript') or '').strip()
detected = str(result.get('language') or '').strip().lower()
if not (2 <= len(detected) <= 3 and detected.isascii() and detected.isalpha()):
detected = ''
return j(handler, {'ok': True, 'transcript': transcript, 'language': detected})
""",
)
routes = ROOT / "api/routes.py"
replace_exact(
routes,
"def _tts_open(req, *, timeout=30, opener_factory=None):",
'''ATLAS_TTS_LANGUAGES = ("en", "ru", "es")
def _atlas_tts_language(body):
"""Return a plain, allow-listed en/ru/es code, or "" to send no language.
This is a trust boundary, not a parser. Only the exact normalized codes the
private Piper deployment bakes a voice for are forwarded; a missing field,
a wrong type, a region tag, padding, control characters, a traversal or
injection string, an oversized value, an object, an array, a number or a
client-supplied "voice" all resolve to "" and the language field is then
omitted entirely, so the Jetson service applies its own English default.
Coercing a malformed value into a supported code would let a browser
describe hostile input as a language we support; omission cannot.
"""
if not isinstance(body, dict):
return ""
value = body.get("language")
if not isinstance(value, str):
return ""
return value if value in ATLAS_TTS_LANGUAGES else ""
def _tts_open(req, *, timeout=30, opener_factory=None):''',
)
marker = " # ── ElevenLabs TTS ──────────────────────────────────────────────────\n"
atlas = ''' # ── Atlas private Jetson TTS ─────────────────────────────────────────
if engine == "atlas":
@ -94,12 +336,19 @@ atlas = ''' # ── Atlas private Jetson TTS ──────────
speed = max(0.5, min(2.0, 1.0 + (float(rate_str.rstrip("%")) / 100.0)))
except ValueError:
speed = 1.0
request_body = json.dumps({
request_payload = {
"model": "piper",
"input": text,
"voice": "en_US-lessac-high",
"speed": speed,
}).encode("utf-8")
}
# Attach a language ONLY when the browser sent a plain allow-listed
# code. Omitting it is the fail-safe: the Jetson service then speaks
# its own English default, which is also what every partially rolled
# out combination of these components degrades to.
_atlas_language = _atlas_tts_language(data)
if _atlas_language:
request_payload["language"] = _atlas_language
request_body = json.dumps(request_payload).encode("utf-8")
request = Request(atlas_url, data=request_body, headers={
"Content-Type": "application/json",
"Accept": "audio/wav",

View File

@ -31,9 +31,37 @@
error:'Voice unavailable',
idle:'',
};
// The only language signal this file trusts is the one the private Whisper
// service returned for the audio of the turn currently being answered. It is
// bound to that turn's generation token and consumed exactly once.
let sttLanguage='';
let sttLanguageToken=-1;
const TTS_LANGUAGES=['en','ru','es'];
const originalAutoRead=window.autoReadLastAssistant;
const originalApplyPreference=window._applyVoiceModePref;
function normalizeSttLanguage(value){
if(typeof value!=='string') return '';
const code=value.trim().toLowerCase();
return TTS_LANGUAGES.indexOf(code)>=0?code:'';
}
function clearSttLanguage(){
sttLanguage='';
sttLanguageToken=-1;
}
function rememberSttLanguage(language, token){
sttLanguage=language||'';
sttLanguageToken=sttLanguage?token:-1;
}
function takeSttLanguage(token){
const language=sttLanguageToken===token?sttLanguage:'';
clearSttLanguage();
return language;
}
function toast(message){
if(typeof window.showToast==='function') window.showToast(message,3000);
}
@ -107,6 +135,7 @@
active=false;
thinkingSession=null;
clearErrorTimer();
clearSttLanguage();
stopCapture();
stopPlayback();
modeBtn.classList.remove('active');
@ -120,13 +149,14 @@
},delay||500);
}
function sendTranscript(transcript, token){
function sendTranscript(transcript, token, language){
if(!active||token!==generation) return;
const text=String(transcript||'').trim();
if(!text){restartSoon(token,350);return;}
if(!text){clearSttLanguage();restartSoon(token,350);return;}
composer.value=text;
if(typeof window.autoResize==='function') window.autoResize();
thinkingSession=(typeof S!=='undefined'&&S.session)?S.session.session_id:null;
rememberSttLanguage(language,token);
setState('thinking');
if(typeof window.send==='function') window.send();
}
@ -141,7 +171,7 @@
const response=await fetch('/api/transcribe',{method:'POST',body:form});
const payload=await response.json().catch(function(){return {};});
if(!response.ok) throw new Error(payload.error||('Whisper request failed: '+response.status));
sendTranscript(payload.transcript,token);
sendTranscript(payload.transcript,token,normalizeSttLanguage(payload.language));
}catch(error){
if(!active||token!==generation) return;
const message=errorMessage(error,'Private Whisper is unavailable');
@ -159,6 +189,7 @@
async function startListening(token){
if(!active||token!==generation) return;
stopCapture();
clearSttLanguage();
setState('listening');
try{
const capture=await navigator.mediaDevices.getUserMedia({
@ -273,11 +304,15 @@
});
}
async function fetchSpeech(chunk){
async function fetchSpeech(chunk, language){
// `language` is only ever the private STT result for this turn. When it is
// absent the field is omitted entirely and the server picks English.
const request={text:chunk,engine:'atlas'};
if(language) request.language=language;
const response=await fetch('/api/tts',{
method:'POST',
headers:{'Content-Type':'application/json'},
body:JSON.stringify({text:chunk,engine:'atlas'}),
body:JSON.stringify(request),
});
if(!response.ok){
const payload=await response.json().catch(function(){return {};});
@ -291,10 +326,12 @@
const currentSession=(typeof S!=='undefined'&&S.session)?S.session.session_id:null;
if(thinkingSession&&currentSession&&thinkingSession!==currentSession){
thinkingSession=null;
clearSttLanguage();
restartSoon(token,250);
return;
}
thinkingSession=null;
const language=takeSttLanguage(token);
const rows=document.querySelectorAll('.msg-row[data-role="assistant"], .assistant-segment[data-raw-text]');
if(!rows.length){restartSoon(token,250);return;}
const text=cleanForSpeech(rows[rows.length-1].dataset.rawText||'');
@ -302,11 +339,11 @@
setState('speaking');
const chunks=typeof window._splitForTTS==='function'?window._splitForTTS(text,280):[text];
try{
let pending=fetchSpeech(chunks[0]);
let pending=fetchSpeech(chunks[0],language);
for(let index=0;index<chunks.length;index+=1){
if(!active||token!==generation) return;
const blob=await pending;
if(index+1<chunks.length) pending=fetchSpeech(chunks[index+1]);
if(index+1<chunks.length) pending=fetchSpeech(chunks[index+1],language);
await playBlob(blob,token);
}
}catch(error){
@ -324,6 +361,7 @@
const token=generation;
active=true;
clearErrorTimer();
clearSttLanguage();
modeBtn.classList.add('active');
toast('Hands-free private voice mode on');
if(typeof window.stopTTS==='function') window.stopTTS();

View File

@ -40,6 +40,40 @@ or the WebUI. Browser chat remains available when `bot_token` is empty.
The bot token and relay key must never be added to Git or a Kubernetes Secret.
The router does not log prompt bodies, raw Telegram IDs, link codes, or tokens.
## Private Jetson voice: multilingual TTS policy
`hermes-tts` on `titan-21` bakes three checksum-pinned Piper voices and
selects one per request from a fixed, allow-listed `language` field: `en`/
`en-US``en_US-amy-medium`, `ru`/`ru-RU``ru_RU-irina-medium`, `es`/
`es-MX`/`es-ES``es_MX-claude-high` (Piper's `claude` voice is Mexican
Spanish; there is no Castilian `es_ES-claude`). Matching is case-insensitive
and accepts both `_` and `-` separators. Any language that is missing,
unrecognized, or malformed falls back to English amy rather than erroring.
The mapping is a fixed dict from `language` to one of the three baked model
names only — a client-supplied `voice` field is never read, so no client
input can select or construct a model path. All three voices are preloaded
at process start (see `dockerfiles/hermes-jetson-tts-server.py`).
Hermes Chat deliberately exposes no TTS speaker/model choice. The deterministic
WebUI image patch removes the pinned upstream voice selector, its label and
translations, its browser/server preference persistence, and every outbound
client `voice` field while preserving the TTS engine, speech rate/pitch,
dictation, hands-free Voice Mode, and the conversation instrument. Legacy
`hermes-tts-voice` browser state is deleted. Voice choice is therefore policy,
not a client preference: validated English maps to amy, Russian to irina,
Spanish to claude, and every unsupported or absent language falls back to amy.
The private WebUI voice bridge (`dockerfiles/hermes-webui-atlas-voice.js`,
patched into `api/routes.py` by `hermes-webui-atlas-patch.py`) has no signal
for the language of the assistant reply it is about to speak — it sends only
`text` and `engine`. Until the WebUI or gateway attaches an explicit
`language` field to that request, every reply speaks in the safe English
default regardless of its actual language. Closing that gap needs a language
signal upstream of the TTS call (e.g. tagging the assistant turn with a
detected/declared reply language and threading it through
`hermes-webui-atlas-voice.js``api/routes.py` → the `language` field), not
client- or server-side guessing bolted onto the TTS service itself.
## The one-sentence explanation
Hermes is the persistent agent runtime and control surface; Codex or the local
@ -347,6 +381,40 @@ Use this short explanation:
- `Use $tune-atlas-alerts. Trace one currently firing alert to its generated source and raw PromQL, but do not edit it.`
- `Use $master-hermes-on-atlas. Assess me on the request path and permission boundary. One question at a time.`
## Private voice: choosing the Piper voice from the STT-detected language
Hands-free voice mode in `chat.hermes.bstein.dev` selects the private Piper
voice from the language the private Jetson Whisper service reports for the
user's own speech. The signal travels one way only, and every hop narrows it:
1. `hermes-stt` returns `{text, model, language}`. `language` is whatever
Whisper decoded with, accepted only as a bare ISO-639 token (`en`, `ru`,
`es`, `yue`, …); anything else is reported as empty.
2. `hermes_stt_client.py` writes the usual `<stem>.txt` transcript plus a
`<stem>.language` sidecar. The transcript stays the only `.txt` in the
output directory, so the stock Hermes local-command contract is unchanged.
3. The patched local-command STT envelope reads that sidecar and adds
`language` to its result; `/api/transcribe` re-validates it and returns it
next to `transcript`.
4. `atlas-voice.js` keeps that value only for the turn it belongs to. It is
bound to the voice-mode generation token and the chat session id, consumed
exactly once by the reply that turn produced, and cleared on cancellation,
restart, session change, an empty transcript, or a transcription error.
5. `/api/tts` accepts `language` only from the fixed allow-list and otherwise
sends English. The Jetson TTS service applies the same allow-list again as
the final authority.
**What this does not claim.** The language is the language the *user spoke*,
not the language of the reply. A model asked a Russian question may answer in
English and will then be read aloud by the Russian voice, and vice versa; this
is a deliberate policy choice for hands-free mode, not a detection failure.
Nothing here detects the language of assistant text.
**Everything else stays English.** Typed messages, the manual read-aloud
button, and any assistant reply that was not produced by a hands-free spoken
turn carry no trusted STT signal, so they synthesize with `en_US-amy-medium`.
A `voice` field from a browser is never honoured at any hop.
## Honest limits
- Hermes does not currently apply production or cluster changes autonomously.

View File

@ -12,6 +12,34 @@ from pathlib import Path
from urllib.request import Request, urlopen
LANGUAGE_SUFFIX = ".language"
def _normalize_language(value: object) -> str:
"""Accept only a bare ISO-639 code from the private Whisper response."""
if not isinstance(value, str):
return ""
code = value.strip().lower()
if not 2 <= len(code) <= 3 or not code.isascii() or not code.isalpha():
return ""
return code
def _write_result(output_dir: Path, stem: str, transcript: str, language: str) -> Path:
"""Write the .txt Hermes reads, plus the language sidecar when we have one.
Hermes' local-command contract is "leave a .txt in --output-dir"; it globs
``*.txt`` and reads the first match. The sidecar deliberately uses another
suffix so the transcript stays the only ``.txt`` in the directory.
"""
output_dir.mkdir(parents=True, exist_ok=True)
transcript_path = output_dir / f"{stem}.txt"
transcript_path.write_text(transcript, encoding="utf-8")
if language:
(output_dir / f"{stem}{LANGUAGE_SUFFIX}").write_text(language, encoding="utf-8")
return transcript_path
def _multipart(audio: Path, language: str, model: str) -> tuple[bytes, str]:
boundary = f"atlas-hermes-{secrets.token_hex(12)}"
mime = mimetypes.guess_type(audio.name)[0] or "application/octet-stream"
@ -67,9 +95,12 @@ def main() -> None:
with urlopen(request, timeout=120) as response:
result = json.loads(response.read().decode("utf-8"))
transcript = str(result.get("text") or "").strip()
args.output_dir.mkdir(parents=True, exist_ok=True)
output = args.output_dir / f"{args.input_path.stem}.txt"
output.write_text(transcript, encoding="utf-8")
_write_result(
args.output_dir,
args.input_path.stem,
transcript,
_normalize_language(result.get("language")),
)
if __name__ == "__main__":

View File

@ -0,0 +1,28 @@
"""Pinned local-command STT envelope anchor used by image-patch tests."""
import contextlib
from pathlib import Path
class _Logger:
def info(self, *args):
return None
logger = _Logger()
def _transcribe_local_command(file_path, normalized_model, output_dir):
try:
with contextlib.nullcontext(output_dir):
txt_files = sorted(Path(output_dir).glob("*.txt"))
transcript_text = txt_files[0].read_text(encoding="utf-8").strip()
logger.info(
"Transcribed %s via local STT command (%s, %d chars)",
Path(file_path).name,
normalized_model,
len(transcript_text),
)
return {"success": True, "transcript": transcript_text, "provider": "local_command"}
except OSError as error:
return {"success": False, "transcript": "", "error": str(error)}

View File

@ -0,0 +1,10 @@
_SETTINGS_DEFAULTS = {
"tts_voice": "",
}
_SETTINGS_SPEECH_KEYS = {
"tts_voice",
}
UPSTREAM_VALIDATION_FRAGMENT = ''' if k == "tts_voice":
if not isinstance(v, str) or len(v) > 200 or "\x00" in v:
continue
'''

View File

@ -1,3 +1,41 @@
def tts(handler, engine):
"""Pinned Atlas TTS route anchors used by image-patch tests."""
import json
import os
from urllib.request import ProxyHandler, Request, build_opener
class _NoRedirectTtsHandler:
"""Placeholder for the upstream no-redirect opener handler."""
class _Logger:
def exception(self, message):
return None
logger = _Logger()
class _Upstream:
def read(self):
return b"RIFFsynthetic"
def __enter__(self):
return self
def __exit__(self, *exc_info):
return False
def _buffer_tts_audio_response(response):
return response.read()
def _tts_open(req, *, timeout=30, opener_factory=None):
return _Upstream()
def _handle_tts(handler, data, text, rate_str, engine):
# ── ElevenLabs TTS ──────────────────────────────────────────────────
return False
return None

View File

@ -0,0 +1,13 @@
"""Pinned upstream /api/transcribe response anchor used by image-patch tests."""
def j(handler, payload, status=200):
return {"status": status, "payload": payload}
def handle_transcribe(handler, result):
try:
transcript = str(result.get('transcript') or '').strip()
return j(handler, {'ok': True, 'transcript': transcript})
except ValueError as error:
return j(handler, {'error': str(error)}, status=400)

View File

@ -0,0 +1,34 @@
function speakWithRegisteredEngine(){
const _opts={
voice: localStorage.getItem("hermes-tts-voice")||'',
rate: parseFloat(localStorage.getItem("hermes-tts-rate")),
};
return _opts;
}
function speakWithEdge(clean){
const voice=localStorage.getItem("hermes-tts-voice")||"zh-CN-XiaoxiaoNeural";
const rate='';
const pitch='';
return fetch('/api/tts', {
body: JSON.stringify({text: clean, voice, rate, pitch})
});
}
function speakWithBrowser(clean){
const utter=new SpeechSynthesisUtterance(clean);
const savedVoice=localStorage.getItem('hermes-tts-voice');
const voices=speechSynthesis.getVoices();
if(savedVoice&&voices.length){
const match=voices.find(v=>v.name===savedVoice);
if(match) utter.voice=match;
}
return utter;
}
function _mirrorSpeechSettingsFromServer(s){
const defaults={
tts_voice:'',
};
[
['tts_voice','hermes-tts-voice'],
].forEach(([settingKey,storageKey])=>localStorage.setItem(storageKey,s[settingKey]));
return defaults;
}

View File

@ -0,0 +1,4 @@
const EN = {
settings_label_tts_voice: 'Voice',
settings_desc_tts_voice: "Preferred voice. Populated from your browser's available voices.",
};

View File

@ -5,6 +5,12 @@
</head>
<body>
<select id="settingsTtsEngine"><option value="browser">Browser speech synthesis</option><option value="edge">Edge TTS (server)</option></select>
<div class="settings-field"><label for="settingsTtsVoice" data-i18n="settings_label_tts_voice">Voice</label>
<select id="settingsTtsVoice" style="width:100%;padding:8px;background:var(--code-bg);color:var(--text);border:1px solid var(--border2);border-radius:6px">
<option value="">Default system voice</option>
</select>
<div style="font-size:11px;color:var(--muted);margin-top:4px" data-i18n="settings_desc_tts_voice">Preferred voice. Populated from your browser's available voices.</div>
</div>
<div class="composer-box" id="composerBox">
<div class="voice-mode-bar" id="voiceModeBar" style="display:none">
<span class="voice-mode-indicator" id="voiceModeIndicator"></span>

View File

@ -0,0 +1,44 @@
const _SETTINGS_SPEECH_STORAGE_KEYS={
tts_engine:'hermes-tts-engine',
tts_voice:'hermes-tts-voice',
tts_rate:'hermes-tts-rate',
};
let _settingsSpeechChangedKeys=new Set();
function _speechPreferencesPayloadFromUi(){
const payload={};
const ttsVoiceSel=$('settingsTtsVoice');
if(ttsVoiceSel) _setOwnedSpeechPayload(payload,'tts_voice',ttsVoiceSel.value||'');
return payload;
}
function loadSettingsPanel(){
const ttsEngineSel=$('settingsTtsEngine');
if(ttsEngineSel){
ttsEngineSel.onchange=function(){
localStorage.setItem('hermes-tts-engine',this.value);
window._populateTtsVoices();
_schedulePreferencesAutosave();
};
}
// Populate voice selector based on engine
const ttsVoiceSel=$('settingsTtsVoice');
window._populateTtsVoices=function(){
if(!ttsVoiceSel) return;
const engine=localStorage.getItem('hermes-tts-engine')||'browser';
const current=String(_speechSetting('tts_voice','hermes-tts-voice','')||'');
_syncSpeechPreferenceCache('tts_voice',current);
if(engine==='edge'){
const edgeVoices=[
{value:'en-US-AriaNeural',label:'Aria (English, Female)'},
];
ttsVoiceSel.innerHTML='<option value="">Default (Xiaoxiao)</option>';
edgeVoices.forEach(v=>ttsVoiceSel.appendChild(v));
}
};
if(ttsVoiceSel&&'speechSynthesis' in window){
window._populateTtsVoices();
ttsVoiceSel.onchange=function(){_markSpeechPreferenceChanged('tts_voice');localStorage.setItem('hermes-tts-voice',this.value);_schedulePreferencesAutosave();};
}
// TTS rate/pitch sliders
}

View File

@ -1,4 +1,15 @@
function _buildBrowserUtterance(text, btn){
const utter=new SpeechSynthesisUtterance(text);
const savedVoice=localStorage.getItem('hermes-tts-voice');
const voices=speechSynthesis.getVoices();
if(savedVoice&&voices.length){
const match=voices.find(v=>v.name===savedVoice);
if(match) utter.voice=match;
}
return utter;
}
function _playEdgeTtsChunked(text, btn){
const voice=localStorage.getItem('hermes-tts-voice')||'zh-CN-XiaoxiaoNeural';
return fetch('/api/tts',{body:JSON.stringify({text:chunk, voice:voice, rate:rate, pitch:pitch})});
}
function speakSelected(clean, btn, engine){
@ -11,3 +22,14 @@ function speakAutomatically(clean, engine){
_playEdgeTtsChunked(clean, null);
}
}
function registeredTts(engine, clean){
const _opts={
voice: localStorage.getItem('hermes-tts-voice')||'',
rate: parseFloat(localStorage.getItem('hermes-tts-rate')),
};
const autoOpts={
voice: localStorage.getItem('hermes-tts-voice')||'',
pitch: parseFloat(localStorage.getItem('hermes-tts-pitch')),
};
return [engine, clean, _opts, autoOpts];
}

View File

@ -0,0 +1,463 @@
// Deterministic browser stub that drives dockerfiles/hermes-webui-atlas-voice.js
// through complete hands-free turns with no microphone, audio device or GPU.
//
// The script under test is an IIFE with no exported seams, so the only honest
// way to assert what reaches POST /api/tts is to run it against a fake DOM and
// fake clock and record the requests it actually makes. Usage:
//
// node atlas_voice_language_probe.js <path-to-atlas-voice.js>
//
// It prints one JSON object describing every scenario to stdout.
'use strict';
const fs = require('fs');
const vm = require('vm');
const SCRIPT_PATH = process.argv[2];
if (!SCRIPT_PATH) {
throw new Error('usage: atlas_voice_language_probe.js <atlas-voice.js>');
}
const SOURCE = fs.readFileSync(SCRIPT_PATH, 'utf8');
function flush() {
// Four macrotask hops drain the promise chains the script builds around
// fetch()/json()/blob()/play() without ever waiting on wall-clock time.
return new Promise((resolve) => {
let hops = 0;
(function hop() {
hops += 1;
if (hops > 12) {
resolve();
return;
}
setImmediate(hop);
})();
});
}
function makeElement(id) {
return {
id,
style: {
values: new Map(),
setProperty(name, value) { this.values.set(name, String(value)); },
removeProperty(name) { this.values.delete(name); },
getPropertyValue(name) { return this.values.get(name) || ''; },
},
dataset: {},
attributes: {},
value: '',
textContent: '',
className: '',
classList: {
entries: new Set(),
add(name) { this.entries.add(name); },
remove(name) { this.entries.delete(name); },
contains(name) { return this.entries.has(name); },
},
listeners: [],
setAttribute(name, value) { this.attributes[name] = String(value); },
getAttribute(name) {
return Object.prototype.hasOwnProperty.call(this.attributes, name)
? this.attributes[name] : null;
},
addEventListener(type, handler) { this.listeners.push({ type, handler }); },
removeEventListener(type, handler) {
this.listeners = this.listeners.filter((entry) => entry.handler !== handler);
},
click() {
const event = { preventDefault() {}, stopImmediatePropagation() {} };
this.listeners
.filter((entry) => entry.type === 'click')
.forEach((entry) => entry.handler(event));
},
querySelector() { return null; },
insertBefore() {},
appendChild() {},
};
}
function makeHarness() {
const clock = { now: 1000000 };
const timeouts = [];
const intervals = new Map();
let timerId = 1;
const ttsRequests = [];
const transcribeCalls = [];
const toasts = [];
const sends = [];
let capability = { ok: true, available: true, provider: 'local_command' };
let transcribeResponse = { ok: true, transcript: 'hello', language: 'en' };
let transcribeStatus = 200;
let assistantRows = [];
let loud = false;
let recorder = null;
const storage = new Map();
const elements = {};
['btnVoiceMode', 'voiceModeBar', 'voiceModeIndicator', 'voiceModeLabel', 'msg']
.forEach((id) => { elements[id] = makeElement(id); });
function MediaRecorder() {
this.state = 'recording';
this.ondataavailable = null;
this.onstop = null;
recorder = this;
}
MediaRecorder.prototype.start = function start() { this.state = 'recording'; };
MediaRecorder.prototype.stop = function stop() {
if (this.state === 'inactive') return;
this.state = 'inactive';
if (this.onstop) this.onstop();
};
MediaRecorder.isTypeSupported = function isTypeSupported() { return true; };
function AudioContext() {
this.createAnalyser = () => ({
fftSize: 2048,
getByteTimeDomainData(samples) {
for (let i = 0; i < samples.length; i += 1) {
samples[i] = loud ? (i % 2 ? 200 : 56) : 128;
}
},
});
this.createBiquadFilter = () => ({
type: '', frequency: { value: 0 }, Q: { value: 0 }, connect() {},
});
this.createMediaStreamSource = () => ({ connect() {} });
this.close = () => {};
}
function AudioElement() {
this.currentTime = 0;
this.onended = null;
this.onerror = null;
this.pause = () => {};
this.play = () => {
setImmediate(() => { if (this.onended) this.onended(); });
return Promise.resolve();
};
}
async function fetchStub(url, init) {
if (url === '/api/transcribe/capability') {
return { ok: true, status: 200, json: async () => capability };
}
if (url === '/api/transcribe') {
transcribeCalls.push({ body: init && init.body });
return {
ok: transcribeStatus < 400,
status: transcribeStatus,
json: async () => transcribeResponse,
};
}
if (url === '/api/tts') {
ttsRequests.push(JSON.parse(init.body));
return {
ok: true,
status: 200,
blob: async () => ({ synthetic: true }),
json: async () => ({}),
};
}
throw new Error(`unexpected fetch: ${url}`);
}
const context = {
console,
Uint8Array,
Promise,
Math,
JSON,
String,
Number,
Error,
parseInt,
isNaN,
Set,
Map,
Array,
Object,
Date: { now: () => clock.now },
Blob: function Blob(parts, options) { this.parts = parts; this.type = (options || {}).type || ''; },
File: function File(parts, name, options) {
this.parts = parts; this.name = name; this.type = (options || {}).type || '';
},
FormData: function FormData() { this.entries = []; this.append = (k, v) => this.entries.push([k, v]); },
URL: { createObjectURL: () => 'blob:atlas-test', revokeObjectURL() {} },
Audio: AudioElement,
MediaRecorder,
AudioContext,
fetch: fetchStub,
localStorage: {
getItem: (key) => (storage.has(key) ? storage.get(key) : null),
setItem: (key, value) => { storage.set(key, String(value)); },
removeItem: (key) => { storage.delete(key); },
},
navigator: {
mediaDevices: {
getUserMedia: async () => ({ getTracks: () => [{ stop() {} }] }),
getSupportedConstraints: () => ({}),
},
},
document: {
getElementById: (id) => elements[id] || null,
querySelectorAll: () => assistantRows,
createElement: () => ({ value: '', textContent: '' }),
},
S: { session: { session_id: 'session-1' }, busy: false },
setTimeout: (fn, delay) => {
const id = timerId; timerId += 1;
timeouts.push({ id, fn, at: clock.now + (delay || 0) });
return id;
},
clearTimeout: (id) => {
const index = timeouts.findIndex((entry) => entry.id === id);
if (index >= 0) timeouts.splice(index, 1);
},
setInterval: (fn, delay) => {
const id = timerId; timerId += 1;
intervals.set(id, { fn, delay: delay || 0 });
return id;
},
clearInterval: (id) => { intervals.delete(id); },
};
context.window = context;
context.showToast = (message) => { toasts.push(message); };
context.send = () => { sends.push(elements.msg.value); };
context.autoResize = () => {};
vm.createContext(context);
vm.runInContext(SOURCE, context, { filename: 'atlas-voice.js' });
function runDueTimeouts() {
const due = timeouts.filter((entry) => entry.at <= clock.now);
due.forEach((entry) => {
const index = timeouts.indexOf(entry);
if (index >= 0) timeouts.splice(index, 1);
entry.fn();
});
}
function tick(ms) {
clock.now += ms;
Array.from(intervals.values()).forEach((entry) => entry.fn());
runDueTimeouts();
}
return {
context,
elements,
ttsRequests,
transcribeCalls,
toasts,
sends,
clock,
recorder: () => recorder,
setLoud: (value) => { loud = value; },
setCapability: (value) => { capability = value; },
setTranscribeResponse: (value, status) => {
transcribeResponse = value;
transcribeStatus = status === undefined ? 200 : status;
},
setAssistantReply: (text) => { assistantRows = [{ dataset: { rawText: text } }]; },
setSession: (id) => { context.S.session = { session_id: id }; },
advance: (ms) => { clock.now += ms; },
tick,
runDueTimeouts,
flush,
};
}
// Walk one capture window: pre-roll audio, three loud frames so the VAD latches
// speech, then silence past the hangover so MediaRecorder.stop() fires.
async function captureSpeech(harness) {
const active = harness.recorder();
if (!active) throw new Error('voice mode never created a recorder');
active.ondataavailable({ data: { size: 512 } });
harness.setLoud(true);
for (let i = 0; i < 4; i += 1) harness.tick(100);
active.ondataavailable({ data: { size: 512 } });
harness.setLoud(false);
harness.advance(2500);
harness.tick(100);
await harness.flush();
}
async function startVoiceMode(harness) {
await harness.flush();
harness.elements.btnVoiceMode.click();
await harness.flush();
}
// One complete turn: speak, transcribe, let the app "answer", read it back.
async function runTurn(harness, { transcript, language, reply }) {
const payload = { ok: true, transcript };
if (language !== undefined) payload.language = language;
harness.setTranscribeResponse(payload);
await captureSpeech(harness);
harness.setAssistantReply(reply || 'An answer.');
harness.context.autoReadLastAssistant();
await harness.flush();
}
async function restartListening(harness) {
harness.advance(1000);
harness.runDueTimeouts();
await harness.flush();
}
const scenarios = {};
scenarios.english_turn_speaks_english = async () => {
const harness = makeHarness();
await startVoiceMode(harness);
await runTurn(harness, { transcript: 'What is the weather?', language: 'en' });
return { tts: harness.ttsRequests };
};
scenarios.russian_turn_speaks_russian = async () => {
const harness = makeHarness();
await startVoiceMode(harness);
await runTurn(harness, { transcript: 'Как дела?', language: 'ru', reply: 'Всё хорошо.' });
return { tts: harness.ttsRequests };
};
scenarios.spanish_turn_speaks_spanish = async () => {
const harness = makeHarness();
await startVoiceMode(harness);
await runTurn(harness, { transcript: '¿Qué tal?', language: 'es', reply: 'Muy bien.' });
return { tts: harness.ttsRequests };
};
scenarios.missing_language_falls_back = async () => {
const harness = makeHarness();
await startVoiceMode(harness);
await runTurn(harness, { transcript: 'Hello there.' });
return { tts: harness.ttsRequests };
};
scenarios.unsupported_language_falls_back = async () => {
const harness = makeHarness();
await startVoiceMode(harness);
await runTurn(harness, { transcript: 'Bonjour tout le monde.', language: 'fr' });
return { tts: harness.ttsRequests };
};
scenarios.hostile_language_values_are_dropped = async () => {
const results = [];
const hostile = [
'ru; rm -rf /',
'../../ru_RU-irina-medium',
'ru',
'RUSSIAN',
{ language: 'ru' },
['ru'],
42,
null,
'r',
'ru ru',
'x'.repeat(4096),
];
for (const language of hostile) {
const harness = makeHarness();
// eslint-disable-next-line no-await-in-loop
await startVoiceMode(harness);
// eslint-disable-next-line no-await-in-loop
await runTurn(harness, { transcript: 'Say something.', language });
results.push({ sent: String(language), tts: harness.ttsRequests });
}
return { results };
};
scenarios.voice_field_is_never_sent = async () => {
const harness = makeHarness();
await startVoiceMode(harness);
await runTurn(harness, { transcript: 'Как дела?', language: 'ru' });
return { tts: harness.ttsRequests };
};
scenarios.language_does_not_leak_into_later_turn = async () => {
const harness = makeHarness();
await startVoiceMode(harness);
await runTurn(harness, { transcript: 'Как дела?', language: 'ru', reply: 'Всё хорошо.' });
await restartListening(harness);
await runTurn(harness, { transcript: 'And in English?', language: undefined });
await restartListening(harness);
await runTurn(harness, { transcript: '¿Y ahora?', language: 'es' });
return { tts: harness.ttsRequests };
};
scenarios.empty_transcript_does_not_arm_a_language = async () => {
const harness = makeHarness();
await startVoiceMode(harness);
harness.setTranscribeResponse({ ok: true, transcript: ' ', language: 'ru' });
await captureSpeech(harness);
const sendsAfterBlank = harness.sends.slice();
// A reply landing while the blank turn winds down must not inherit a
// language that transcript never earned.
harness.setAssistantReply('A stray answer.');
harness.context.autoReadLastAssistant();
await harness.flush();
await restartListening(harness);
await runTurn(harness, { transcript: 'Hello.', language: undefined });
return { sendsAfterBlank, tts: harness.ttsRequests, sends: harness.sends };
};
scenarios.session_change_discards_language = async () => {
const harness = makeHarness();
await startVoiceMode(harness);
harness.setTranscribeResponse({ ok: true, transcript: 'Как дела?', language: 'ru' });
await captureSpeech(harness);
harness.setSession('session-2');
harness.setAssistantReply('Reply that belongs to another chat.');
harness.context.autoReadLastAssistant();
await harness.flush();
const afterSwitch = harness.ttsRequests.slice();
await restartListening(harness);
await runTurn(harness, { transcript: 'Hello again.', language: undefined });
return { afterSwitch, tts: harness.ttsRequests };
};
scenarios.deactivation_discards_language = async () => {
const harness = makeHarness();
await startVoiceMode(harness);
harness.setTranscribeResponse({ ok: true, transcript: 'Как дела?', language: 'ru' });
await captureSpeech(harness);
harness.elements.btnVoiceMode.click();
await harness.flush();
harness.setAssistantReply('Late reply after the user left voice mode.');
harness.context.autoReadLastAssistant();
await harness.flush();
const afterDeactivate = harness.ttsRequests.slice();
harness.elements.btnVoiceMode.click();
await harness.flush();
await runTurn(harness, { transcript: 'Fresh start.', language: undefined });
return { afterDeactivate, tts: harness.ttsRequests };
};
scenarios.transcribe_error_speaks_nothing = async () => {
const harness = makeHarness();
await startVoiceMode(harness);
harness.setTranscribeResponse({ error: 'Whisper is down' }, 503);
await captureSpeech(harness);
harness.setAssistantReply('Some earlier answer.');
harness.context.autoReadLastAssistant();
await harness.flush();
return { tts: harness.ttsRequests, toasts: harness.toasts };
};
(async () => {
const output = {};
const names = Object.keys(scenarios);
for (const name of names) {
// eslint-disable-next-line no-await-in-loop
output[name] = await scenarios[name]();
}
process.stdout.write(JSON.stringify(output, null, 1));
})().catch((error) => {
process.stderr.write(String((error && error.stack) || error));
process.exit(1);
});

View File

@ -343,7 +343,7 @@ def test_chat_voice_uses_private_jetson_services_and_shared_auto_route():
assert "/api/tts" in voice_script
assert "speakResponse(generation)" in voice_script
assert "window._splitForTTS(text,280)" in voice_script
assert "pending=fetchSpeech(chunks[index+1])" in voice_script
assert "pending=fetchSpeech(chunks[index+1],language)" in voice_script
assert "restartSoon(token,450)" in voice_script
assert "constraints.voiceIsolation=true" in voice_script
assert "highpass.frequency.value=140" in voice_script
@ -398,13 +398,27 @@ def test_voice_models_are_baked_and_runtime_has_no_public_egress():
assert "HERMES_STT_CACHE=/opt/models/whisper" in stt_dockerfile
assert "ADD --checksum=sha256:4cabf7c3" in tts_dockerfile
assert "ADD --checksum=sha256:db42b97d" in tts_dockerfile
assert tts_dockerfile.count("--chmod=0444") == 6
assert "ADD --checksum=sha256:b3a6e47b57b8c7fbe6a0ce2518161a50f59a9cdd8a50835c02cb02bdd6206c18" in tts_dockerfile
assert "ADD --checksum=sha256:95a23eb4d42909d38df73bb9ac7f45f597dbfcde2d1bf9526fdeaf5466977d77" in tts_dockerfile
assert "ADD --checksum=sha256:8ff38212d23da300bbe3705c645e6e5b9475f0bfde01558eb17813e22acaaaaa" in tts_dockerfile
assert "ADD --checksum=sha256:c2ec28bb38e2b59e93b959b3e40348c1afebbd272f30fed5d41205d08e98a9d7" in tts_dockerfile
assert "ADD --checksum=sha256:3ef40a71ea63852cd8ab7e6fa7d2ecdcfa67a0b47c9c48e3f10e02ee02083ea0" in tts_dockerfile
assert "ADD --checksum=sha256:1afc81f703c0e4cb3b4d7c0dca096b8b54a98806807f0170cf5eb5557723c12d" in tts_dockerfile
assert tts_dockerfile.count("--chmod=0444") == 12
assert "/opt/models/piper/en_US-amy-medium.onnx" in tts_dockerfile
assert "/opt/models/piper/ru_RU-irina-medium.onnx" in tts_dockerfile
assert "/opt/models/piper/es_MX-claude-high.onnx" in tts_dockerfile
assert "chmod 0555 /opt/models /opt/models/piper" in tts_dockerfile
assert "HERMES_TTS_CACHE=/opt/models/piper" in tts_dockerfile
assert "HERMES_TTS_VOICE=en_US-amy-medium" in tts_dockerfile
tts_server = (ROOT / "dockerfiles" / "hermes-jetson-tts-server.py").read_text()
assert "download_voice" not in tts_server
assert "baked Piper voice is missing" in tts_server
assert "session_options.intra_op_num_threads = ONNX_THREADS" in tts_server
assert "session_options.intra_op_num_threads = threads" in tts_server
assert 'LANGUAGE_VOICE_MAP = {' in tts_server
assert '"en": "en_US-amy-medium"' in tts_server
assert '"ru": "ru_RU-irina-medium"' in tts_server
assert '"es": "es_MX-claude-high"' in tts_server
policies = _documents(HERMES / "networkpolicy.yaml")
voice_policy = next(

View File

@ -0,0 +1,220 @@
"""Language allow-list contracts for the private Hermes chat TTS voice policy."""
from __future__ import annotations
import importlib.util
import io
import json
import sys
from types import SimpleNamespace
import pytest
from testing.tests.test_hermes_chat_support import ROOT
AMY = "en_US-amy-medium"
IRINA = "ru_RU-irina-medium"
CLAUDE = "es_MX-claude-high"
def _load_tts_server(monkeypatch):
server_path = ROOT / "dockerfiles" / "hermes-jetson-tts-server.py"
spec = importlib.util.spec_from_file_location("hermes_jetson_tts_server", server_path)
assert spec and spec.loader
module = importlib.util.module_from_spec(spec)
class _FakeSessionOptions:
def __init__(self) -> None:
self.intra_op_num_threads = None
self.inter_op_num_threads = None
fake_onnxruntime = SimpleNamespace(
SessionOptions=_FakeSessionOptions,
InferenceSession=lambda *a, **k: SimpleNamespace(),
)
fake_piper = SimpleNamespace(
PiperConfig=SimpleNamespace(from_dict=lambda d: d),
PiperVoice=lambda **kwargs: SimpleNamespace(**kwargs),
SynthesisConfig=lambda **kwargs: SimpleNamespace(**kwargs),
)
monkeypatch.setitem(sys.modules, "onnxruntime", fake_onnxruntime)
monkeypatch.setitem(sys.modules, "piper", fake_piper)
spec.loader.exec_module(module)
return module
@pytest.fixture
def tts(monkeypatch):
return _load_tts_server(monkeypatch)
@pytest.mark.parametrize(
"language,expected",
[
("en", AMY),
("en-US", AMY),
("en_US", AMY),
("EN", AMY),
("En-Us", AMY),
("ru", IRINA),
("ru-RU", IRINA),
("ru_RU", IRINA),
("RU", IRINA),
("es", CLAUDE),
("es-MX", CLAUDE),
("es_MX", CLAUDE),
("es-ES", CLAUDE),
("es_ES", CLAUDE),
("ES", CLAUDE),
],
)
def test_allow_listed_languages_resolve_to_the_approved_voice(tts, language, expected):
assert tts.resolve_voice_name(language) == expected
@pytest.mark.parametrize(
"language",
[
None,
"",
" ",
"fr",
"fr-FR",
"de-DE",
"xx",
"en-GB",
"es-AR",
"english",
123,
1.5,
True,
[],
{},
{"lang": "ru"},
"../../etc/passwd",
"en_US-amy-medium/../../ru_RU-irina-medium",
"\x00ru",
"ru\x00",
],
)
def test_unknown_missing_or_malformed_language_falls_back_to_amy(tts, language):
assert tts.resolve_voice_name(language) == AMY
def test_default_voice_name_matches_the_dockerfile_env_default(tts):
assert tts.DEFAULT_VOICE_NAME == AMY
def test_resolved_voice_is_always_one_of_the_three_baked_names(tts):
assert frozenset({AMY, IRINA, CLAUDE}) == tts.BAKED_VOICE_NAMES
fuzz_inputs = [
"en", "ru", "es", "unknown", "", None, 42, "../../../etc/shadow",
"en_US-amy-medium\x00; rm -rf /", "RU-ru", "Es-Es", "en-us-extra",
]
for value in fuzz_inputs:
assert tts.resolve_voice_name(value) in tts.BAKED_VOICE_NAMES
def test_client_voice_field_cannot_override_the_language_policy(tts):
"""The POST handler must select the voice from "language" only.
A malicious or stale "voice" field in a hostile/legacy request must never
change which baked model answers the request.
"""
calls: list[str] = []
class _RecordingVoice:
def __init__(self, name: str) -> None:
self.name = name
def synthesize_wav(self, text, wav_file, syn_config) -> None:
calls.append(self.name)
wav_file.setnchannels(1)
wav_file.setsampwidth(2)
wav_file.setframerate(16_000)
wav_file.writeframes(b"\x00\x00")
class _RecordingHandler(tts.SpeechHandler):
def __init__(self, payload):
request = json.dumps(payload).encode("utf-8")
self.path = "/v1/audio/speech"
self.headers = {"Content-Length": str(len(request))}
self.rfile = io.BytesIO(request)
self.wfile = io.BytesIO()
self.status = None
self.response_headers = {}
self.server = SimpleNamespace(
voices={
AMY: _RecordingVoice(AMY),
IRINA: _RecordingVoice(IRINA),
CLAUDE: _RecordingVoice(CLAUDE),
},
default_voice_name=AMY,
)
def send_response(self, status, message=None):
self.status = status
def send_header(self, name, value):
self.response_headers[name] = value
def end_headers(self):
return None
# A payload that supplies an attacker/legacy "voice" value but no
# language must resolve to the safe default, never the "voice" value.
handler = _RecordingHandler({"input": "hi", "voice": IRINA})
handler.do_POST()
assert handler.status == 200
assert handler.response_headers["X-TTS-Voice"] == AMY
# A payload supplying both must still be governed by "language" alone.
handler = _RecordingHandler({"input": "hi", "voice": CLAUDE, "language": "ru"})
handler.do_POST()
assert handler.status == 200
assert handler.response_headers["X-TTS-Voice"] == IRINA
assert calls == [AMY, IRINA]
def test_no_client_string_reaches_a_filesystem_path(tts):
"""resolve_voice_name must only ever return a fixed, baked literal.
This is the property that keeps a client from ever causing the server to
build a Path out of attacker-controlled text: the return value is always
a member of the fixed allow-list, regardless of input shape.
"""
hostile_inputs = [
"../../../../etc/passwd",
"/etc/passwd",
"en_US-amy-medium/../../../etc/passwd",
"ru_RU-irina-medium\x00.onnx",
"es_MX-claude-high; cat /etc/shadow",
"\n\ren",
"en" + "/" * 200,
" ",
]
for value in hostile_inputs:
result = tts.resolve_voice_name(value)
assert result in tts.BAKED_VOICE_NAMES
assert "/" not in result
assert ".." not in result
assert "\x00" not in result
def test_normalize_language_rejects_non_string_input(tts):
assert tts.normalize_language(None) is None
assert tts.normalize_language(123) is None
assert tts.normalize_language([]) is None
assert tts.normalize_language("") is None
assert tts.normalize_language(" ") is None
assert tts.normalize_language("En_US") == "en-us"
def test_default_voice_name_is_one_of_the_baked_voices(tts):
assert tts.DEFAULT_VOICE_NAME in tts.BAKED_VOICE_NAMES
def test_load_voices_fails_closed_when_a_baked_model_is_missing(tts, tmp_path):
with pytest.raises(RuntimeError, match="baked Piper voice is missing"):
tts.load_voices(tmp_path, threads=1)

View File

@ -12,6 +12,7 @@ import sys
ROOT = Path(__file__).resolve().parents[2]
FIXTURE = ROOT / "testing/fixtures/hermes-webui-0.52.181"
AGENT_FIXTURE = ROOT / "testing/fixtures/hermes-agent"
PATCHER = ROOT / "dockerfiles/hermes-webui-atlas-patch.py"
VOICE_JS = ROOT / "dockerfiles/hermes-webui-atlas-voice.js"
VOICE_CSS = ROOT / "dockerfiles/hermes-webui-atlas-voice.css"
@ -20,9 +21,12 @@ DOM_PROBE = ROOT / "testing/probes/hermes_voice_instrument_probe.js"
def _patched_fixture(tmp_path: Path) -> Path:
target = tmp_path / "hermes-webui"
agent_target = tmp_path / "hermes-agent"
shutil.copytree(FIXTURE, target)
shutil.copytree(AGENT_FIXTURE, agent_target)
env = os.environ.copy()
env["HERMES_WEBUI_PATCH_ROOT"] = str(target)
env["HERMES_AGENT_PATCH_ROOT"] = str(agent_target)
subprocess.run(
[sys.executable, str(PATCHER)],
cwd=ROOT,
@ -57,6 +61,52 @@ def test_real_upstream_fixture_receives_visual_instrument_contract(tmp_path: Pat
assert '<button' not in index[index.index('id="voiceModeBar"') : index.index('<textarea')]
def test_patched_webui_has_no_user_voice_choice_or_client_voice_field(
tmp_path: Path,
):
"""The pinned settings DOM and every TTS path leave speakers to policy."""
target = _patched_fixture(tmp_path)
index = (target / "static/index.html").read_text(encoding="utf-8")
ui = (target / "static/ui.js").read_text(encoding="utf-8")
panels = (target / "static/panels.js").read_text(encoding="utf-8")
boot = (target / "static/boot.js").read_text(encoding="utf-8")
i18n = (target / "static/i18n.js").read_text(encoding="utf-8")
config = (target / "api/config.py").read_text(encoding="utf-8")
routes = (target / "api/routes.py").read_text(encoding="utf-8")
assert "settingsTtsVoice" not in index
assert "settings_label_tts_voice" not in index
assert "settings_desc_tts_voice" not in index
assert "Default system voice" not in index
assert 'id="settingsTtsEngine"' in index
assert 'id="btnVoiceMode"' in index
assert 'id="voiceModeBar"' in index
assert "hermes-tts-voice" not in ui
assert "voice:voice" not in ui
assert (
"body:JSON.stringify({text:chunk, rate:rate, pitch:pitch, "
"engine:engineOverride||'edge'})"
) in ui
assert "settingsTtsVoice" not in panels
assert "tts_voice" not in panels
assert "localStorage.removeItem('hermes-tts-voice')" in panels
assert panels.count("hermes-tts-voice") == 1
assert "hermes-tts-voice" not in boot
assert "tts_voice" not in boot
assert "text: clean, voice" not in boot
assert '"tts_voice"' not in config
assert "settings_label_tts_voice" not in i18n
assert "settings_desc_tts_voice" not in i18n
atlas_route = routes.split('if engine == "atlas":', 1)[1].split(
"# ── ElevenLabs TTS", 1
)[0]
assert '"input": text' in atlas_route
assert '"voice"' not in atlas_route
assert 'request_payload["language"] = _atlas_language' in atlas_route
def test_visual_states_have_distinct_layers_finite_error_and_reduced_motion():
css = VOICE_CSS.read_text(encoding="utf-8")
@ -101,8 +151,9 @@ def test_visual_slice_preserves_private_voice_request_and_capture_contract():
assert script.count("navigator.mediaDevices.getUserMedia(") == 1
assert "form.append('file',new File([blob],'voice-input.'+ext" in script
assert "fetch('/api/transcribe',{method:'POST',body:form})" in script
assert "body:JSON.stringify({text:chunk,engine:'atlas'})" in script
assert "const request={text:chunk,engine:'atlas'}" in script
assert "if(language) request.language=language" in script
assert "speakResponse(generation)" in script
assert "window._voiceModeImmediateSend" in script
assert "mute" not in script.lower()
assert "language" not in script.lower()
assert "const TTS_LANGUAGES=['en','ru','es']" in script

View File

@ -0,0 +1,677 @@
"""Voice-mode language routing: private Whisper STT decides the Piper voice.
Every assertion here runs without a GPU, a microphone or a cluster. The browser
contract is exercised by driving the real ``atlas-voice.js`` inside a stub DOM
(``testing/tests/data/atlas_voice_language_probe.js``), and the two server-side
trust boundaries are exercised by applying the real image patch to fixtures that
carry the exact upstream anchors and then importing the patched result.
"""
from __future__ import annotations
import importlib.util
import json
import os
import shutil
import subprocess
import sys
from pathlib import Path
from types import SimpleNamespace
import pytest
from testing.tests.test_hermes_chat_support import HERMES, ROOT
DOCKERFILES = ROOT / "dockerfiles"
ATLAS_PATCH = DOCKERFILES / "hermes-webui-atlas-patch.py"
VOICE_SCRIPT = DOCKERFILES / "hermes-webui-atlas-voice.js"
VOICE_PROBE = ROOT / "testing" / "tests" / "data" / "atlas_voice_language_probe.js"
WEBUI_FIXTURE = ROOT / "testing" / "fixtures" / "hermes-webui-0.52.181"
ATLAS_TTS_URL = "http://hermes-tts.hermes.svc.cluster.local:9001/v1/audio/speech"
# Voices baked by the multilingual Piper work (PR #26): en=amy, ru=irina,
# es=claude. Anything outside this set must resolve to English.
SUPPORTED = ("en", "ru", "es")
# ---------------------------------------------------------------------------
# Module loaders
# ---------------------------------------------------------------------------
def _load_stt_server(monkeypatch):
"""Import the Jetson Whisper service without CUDA, torch or whisper."""
path = DOCKERFILES / "hermes-jetson-stt-server.py"
spec = importlib.util.spec_from_file_location("hermes_jetson_stt_server", path)
assert spec and spec.loader
module = importlib.util.module_from_spec(spec)
monkeypatch.setitem(sys.modules, "cgi", SimpleNamespace())
monkeypatch.setitem(
sys.modules,
"torch",
SimpleNamespace(cuda=SimpleNamespace(is_available=lambda: False)),
)
monkeypatch.setitem(sys.modules, "whisper", SimpleNamespace())
spec.loader.exec_module(module)
return module
def _load_stt_client():
"""Import the local-command STT client that Hermes shells out to."""
path = HERMES / "scripts" / "hermes_stt_client.py"
spec = importlib.util.spec_from_file_location("hermes_stt_client", path)
assert spec and spec.loader
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
return module
# ---------------------------------------------------------------------------
# Patch fixtures — each file carries the exact upstream fragment the image
# patch pins, so importing the patched result exercises the inserted code.
# ---------------------------------------------------------------------------
INDEX_FIXTURE = (
'<select id="settingsTtsEngine">'
'<option value="browser">Browser speech synthesis</option>'
'<option value="edge">Edge TTS (server)</option></select>\n'
'<script src="static/boot.js?v=__WEBUI_VERSION__" defer></script>\n'
)
UI_FIXTURE = """function _playEdgeTtsChunked(text, btn){
fetch('/api/tts',{method:'POST',body:JSON.stringify({text:chunk, voice:voice, rate:rate, pitch:pitch})});
}
function readAloud(clean, btn, engine){
if(engine==='edge'){
_playEdgeTtsChunked(clean, btn);
}
}
function autoRead(clean, engine){
if(engine==='edge'){
_playEdgeTtsChunked(clean, null);
}
}
"""
HELPERS_FIXTURE = '''"""Stand-in for the WebUI helper module the patched code imports."""
def bad(handler, message, status=400):
return {"status": status, "error": message}
'''
ROUTES_FIXTURE = '''"""Stand-in carrying the exact upstream anchors the Atlas TTS patch pins."""
import json
import os
from urllib.request import ProxyHandler, Request, build_opener
class _NoRedirectTtsHandler:
"""Placeholder for the upstream no-redirect opener handler."""
class _Logger:
def __init__(self):
self.failures = []
def exception(self, message):
self.failures.append(message)
logger = _Logger()
UPSTREAM_REQUESTS = []
class _Upstream:
def __init__(self, payload):
self._payload = payload
def read(self):
return self._payload
def __enter__(self):
return self
def __exit__(self, *exc_info):
return False
def _buffer_tts_audio_response(response):
return response.read()
def _tts_open(req, *, timeout=30, opener_factory=None):
"""Thin network seam for the TTS upstream fetch so tests can intercept it."""
UPSTREAM_REQUESTS.append(json.loads(req.data.decode("utf-8")))
return _Upstream(b"RIFFsynthetic")
def _handle_tts(handler, data, text, rate_str, engine):
# ── ElevenLabs TTS ──────────────────────────────────────────────────
return None
'''
UPLOAD_FIXTURE = '''"""Stand-in carrying the exact upstream /api/transcribe response anchor."""
def j(handler, payload, status=200):
return {"status": status, "payload": payload}
def handle_transcribe(handler, result):
try:
transcript = str(result.get('transcript') or '').strip()
return j(handler, {'ok': True, 'transcript': transcript})
except ValueError as error:
return j(handler, {'error': str(error)}, status=400)
'''
TRANSCRIPTION_FIXTURE = '''"""Stand-in carrying the exact upstream local-command STT envelope anchor."""
import contextlib
from pathlib import Path
class _Logger:
def info(self, *args):
return None
logger = _Logger()
def _transcribe_local_command(file_path, normalized_model, output_dir):
try:
with contextlib.nullcontext(output_dir):
txt_files = sorted(Path(output_dir).glob("*.txt"))
transcript_text = txt_files[0].read_text(encoding="utf-8").strip()
logger.info(
"Transcribed %s via local STT command (%s, %d chars)",
Path(file_path).name,
normalized_model,
len(transcript_text),
)
return {"success": True, "transcript": transcript_text, "provider": "local_command"}
except OSError as error:
return {"success": False, "transcript": "", "error": str(error)}
'''
def _write(path: Path, text: str) -> None:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text(text, encoding="utf-8")
@pytest.fixture
def patched_webui(tmp_path, monkeypatch):
"""Apply the real Atlas image patch to fixture trees and import the result."""
webui = tmp_path / "hermes-webui"
agent = tmp_path / "hermes"
# Start with the pinned full-surface fixture introduced by PR #39 so this
# test proves the language pipeline composes with its voice-selector removal
# and conversation instrument, not merely with the older #27 anchors.
shutil.copytree(WEBUI_FIXTURE, webui)
_write(webui / "api" / "__init__.py", "")
_write(webui / "api" / "helpers.py", HELPERS_FIXTURE)
_write(webui / "api" / "routes.py", ROUTES_FIXTURE)
_write(webui / "api" / "upload.py", UPLOAD_FIXTURE)
_write(agent / "tools" / "__init__.py", "")
_write(agent / "tools" / "transcription_tools.py", TRANSCRIPTION_FIXTURE)
environment = dict(os.environ)
environment["HERMES_WEBUI_PATCH_ROOT"] = str(webui)
environment["HERMES_AGENT_PATCH_ROOT"] = str(agent)
completed = subprocess.run(
[sys.executable, str(ATLAS_PATCH)],
env=environment,
capture_output=True,
text=True,
)
assert completed.returncode == 0, completed.stderr or completed.stdout
for name in ("api", "api.helpers", "api.routes", "api.upload", "tools",
"tools.transcription_tools"):
sys.modules.pop(name, None)
monkeypatch.syspath_prepend(str(agent))
monkeypatch.syspath_prepend(str(webui))
import api.routes as routes # noqa: PLC0415
import api.upload as upload # noqa: PLC0415
import tools.transcription_tools as transcription # noqa: PLC0415
yield SimpleNamespace(
webui=webui,
agent=agent,
routes=routes,
upload=upload,
transcription=transcription,
)
for name in ("api", "api.helpers", "api.routes", "api.upload", "tools",
"tools.transcription_tools"):
sys.modules.pop(name, None)
class _Handler:
"""Just enough BaseHTTPRequestHandler surface for the Atlas TTS branch."""
def __init__(self):
self.status = None
self.headers_sent = {}
self.wfile = SimpleNamespace(write=self._write)
self.body = b""
def send_response(self, status):
self.status = status
def send_header(self, name, value):
self.headers_sent[name] = value
def end_headers(self):
return None
def _write(self, payload):
self.body += payload
def _atlas_tts(patched, monkeypatch, data):
"""Run the patched Atlas branch and return the JSON it sent to hermes-tts."""
monkeypatch.setenv("HERMES_WEBUI_ATLAS_TTS_URL", ATLAS_TTS_URL)
patched.routes.UPSTREAM_REQUESTS.clear()
handler = _Handler()
result = patched.routes._handle_tts(handler, data, "Some reply.", "", "atlas")
assert result is True, "the Atlas branch must own the response"
assert handler.status == 200
assert len(patched.routes.UPSTREAM_REQUESTS) == 1
return patched.routes.UPSTREAM_REQUESTS[0]
# ---------------------------------------------------------------------------
# 1. Whisper service reports the language it actually decoded with
# ---------------------------------------------------------------------------
def test_stt_response_carries_whisper_detected_language(monkeypatch):
module = _load_stt_server(monkeypatch)
payload = module._transcription_payload(
{
"language": "ru",
"segments": [
{"text": " Как дела?", "no_speech_prob": 0.1, "avg_logprob": -0.2}
],
}
)
assert payload["text"] == "Как дела?"
assert payload["language"] == "ru"
assert payload["model"] == module.MODEL_NAME
@pytest.mark.parametrize(
("raw", "expected"),
[
("en", "en"),
("RU", "ru"),
(" es ", "es"),
("yue", "yue"),
("fr", "fr"),
("en-US", ""),
("en_US", ""),
("e", ""),
("english", ""),
("", ""),
("../en", ""),
("en\x00", ""),
("ru; rm -rf /", ""),
("рус", ""),
(None, ""),
(7, ""),
(["ru"], ""),
({"language": "ru"}, ""),
],
)
def test_stt_language_field_is_shape_validated(monkeypatch, raw, expected):
module = _load_stt_server(monkeypatch)
assert module._detected_language({"language": raw}) == expected
def test_stt_language_absent_when_whisper_omits_it(monkeypatch):
module = _load_stt_server(monkeypatch)
assert module._detected_language({}) == ""
assert module._detected_language("not a result") == ""
assert module._transcription_payload({"text": "hi"})["language"] == ""
# ---------------------------------------------------------------------------
# 2. The local-command client carries the language without breaking the
# .txt contract Hermes reads the transcript from
# ---------------------------------------------------------------------------
def test_stt_client_writes_language_sidecar_beside_the_txt_contract(tmp_path):
module = _load_stt_client()
module._write_result(tmp_path, "voice-input", "Как дела?", "ru")
assert (tmp_path / "voice-input.txt").read_text(encoding="utf-8") == "Как дела?"
assert (tmp_path / "voice-input.language").read_text(encoding="utf-8") == "ru"
# Hermes globs *.txt and reads the first match: the sidecar must not join it.
assert sorted(p.name for p in tmp_path.glob("*.txt")) == ["voice-input.txt"]
def test_stt_client_omits_the_sidecar_when_no_language_was_detected(tmp_path):
module = _load_stt_client()
module._write_result(tmp_path, "voice-input", "Hello.", "")
assert (tmp_path / "voice-input.txt").exists()
assert not (tmp_path / "voice-input.language").exists()
@pytest.mark.parametrize(
("raw", "expected"),
[
("en", "en"),
("ES", "es"),
(" ru ", "ru"),
("en-US", ""),
("", ""),
("../../etc/passwd", ""),
("en\n", "en"),
("e", ""),
(None, ""),
(12, ""),
(["en"], ""),
],
)
def test_stt_client_normalises_the_service_language_field(raw, expected):
module = _load_stt_client()
assert module._normalize_language(raw) == expected
# ---------------------------------------------------------------------------
# 3. The patched agent envelope and /api/transcribe response carry it through
# ---------------------------------------------------------------------------
def test_patched_local_command_envelope_carries_the_sidecar_language(patched_webui, tmp_path):
output = tmp_path / "stt-out"
output.mkdir()
(output / "voice-input.txt").write_text("Как дела?", encoding="utf-8")
(output / "voice-input.language").write_text("ru\n", encoding="utf-8")
result = patched_webui.transcription._transcribe_local_command(
"/tmp/voice-input.wav", "small", output
)
assert result == {
"success": True,
"transcript": "Как дела?",
"provider": "local_command",
"language": "ru",
}
def test_patched_local_command_envelope_defaults_to_no_language(patched_webui, tmp_path):
output = tmp_path / "stt-out"
output.mkdir()
(output / "voice-input.txt").write_text("Hello.", encoding="utf-8")
result = patched_webui.transcription._transcribe_local_command(
"/tmp/voice-input.wav", "small", output
)
assert result["transcript"] == "Hello."
assert result["language"] == ""
@pytest.mark.parametrize(
"hostile",
["en-US", "../../en", "en; rm -rf /", "e", "english", "", "\x00en", "e n"],
)
def test_patched_local_command_envelope_rejects_malformed_sidecars(
patched_webui, tmp_path, hostile
):
output = tmp_path / "stt-out"
output.mkdir()
(output / "voice-input.txt").write_text("Hello.", encoding="utf-8")
(output / "voice-input.language").write_text(hostile, encoding="utf-8")
result = patched_webui.transcription._transcribe_local_command(
"/tmp/voice-input.wav", "small", output
)
assert result["language"] == ""
def test_patched_local_command_envelope_survives_an_undecodable_sidecar(
patched_webui, tmp_path
):
"""A corrupt sidecar must cost the language, never the transcript."""
output = tmp_path / "stt-out"
output.mkdir()
(output / "voice-input.txt").write_text("Hello.", encoding="utf-8")
(output / "voice-input.language").write_bytes(b"\xff\xfe\x00ru")
result = patched_webui.transcription._transcribe_local_command(
"/tmp/voice-input.wav", "small", output
)
assert result["success"] is True
assert result["transcript"] == "Hello."
assert result["language"] == ""
def test_patched_transcribe_response_reports_the_language(patched_webui):
response = patched_webui.upload.handle_transcribe(
None, {"success": True, "transcript": " Как дела? ", "language": "ru"}
)
assert response["payload"] == {"ok": True, "transcript": "Как дела?", "language": "ru"}
@pytest.mark.parametrize(
"hostile",
["", None, "en-US", "englishhh", "../en", 5, ["ru"], {"a": "b"}, "e"],
)
def test_patched_transcribe_response_blanks_untrusted_languages(patched_webui, hostile):
response = patched_webui.upload.handle_transcribe(
None, {"success": True, "transcript": "Hello.", "language": hostile}
)
assert response["payload"]["language"] == ""
assert response["payload"]["transcript"] == "Hello."
# ---------------------------------------------------------------------------
# 4. The /api/tts trust boundary: allow-list only, and never `voice`
# ---------------------------------------------------------------------------
@pytest.mark.parametrize("language", SUPPORTED)
def test_atlas_tts_forwards_allow_listed_languages(patched_webui, monkeypatch, language):
body = _atlas_tts(patched_webui, monkeypatch, {"engine": "atlas", "language": language})
assert body["language"] == language
assert body["model"] == "piper"
@pytest.mark.parametrize(
"hostile",
[
"fr",
"de",
"",
None,
"EN-GB",
" RU ",
"ru-RU",
"es_MX",
"../../ru_RU-irina-medium",
"ru; rm -rf /",
"ru\x00",
"ру",
5,
["ru"],
{"language": "ru"},
True,
"x" * 8192,
],
)
def test_atlas_tts_omits_untrusted_languages(
patched_webui, monkeypatch, hostile
):
body = _atlas_tts(patched_webui, monkeypatch, {"engine": "atlas", "language": hostile})
assert "language" not in body
def test_atlas_tts_language_is_absent_when_the_client_sends_none(patched_webui, monkeypatch):
body = _atlas_tts(patched_webui, monkeypatch, {"engine": "atlas"})
assert "language" not in body
def test_atlas_tts_voice_field_cannot_steer_synthesis(patched_webui, monkeypatch):
body = _atlas_tts(
patched_webui,
monkeypatch,
{"engine": "atlas", "voice": "ru_RU-irina-medium", "language": "en"},
)
assert body["language"] == "en"
body = _atlas_tts(
patched_webui,
monkeypatch,
{"engine": "atlas", "voice": "es_MX-claude-high"},
)
assert "language" not in body
def test_atlas_tts_language_helper_only_ever_returns_a_baked_voice_language(patched_webui):
resolve = patched_webui.routes._atlas_tts_language
hostile = [
None, 0, 1, -1, True, False, [], {}, set(), object(), b"ru",
"", " ", "\t\n", "en", "EN", "en-US", "en_us", "ru-RU", "es-MX",
"e", "eng", "english", "ru ru", "ru;es", "../ru", "ru\x00", "ру",
"x" * 65536, "en" * 4096,
]
for value in hostile:
expected = value if value in SUPPORTED else ""
assert resolve({"language": value}) == expected
for value in hostile:
assert resolve(value) == ""
assert resolve({"voice": "ru_RU-irina-medium"}) == ""
def test_manual_tts_button_body_still_carries_no_language(patched_webui):
"""The read-aloud button has no trusted STT signal, so it must stay Amy."""
ui = (patched_webui.webui / "static" / "ui.js").read_text(encoding="utf-8")
assert "engine:engineOverride||'edge'" in ui
assert "language" not in ui
# ---------------------------------------------------------------------------
# 5. Browser contract, driven through the real atlas-voice.js
# ---------------------------------------------------------------------------
@pytest.fixture(scope="module")
def voice_probe():
node = shutil.which("node")
if not node:
pytest.skip("node is required to drive the browser voice-mode contract")
completed = subprocess.run(
[node, str(VOICE_PROBE), str(VOICE_SCRIPT)],
capture_output=True,
text=True,
timeout=180,
)
assert completed.returncode == 0, completed.stderr
return json.loads(completed.stdout)
@pytest.mark.parametrize(
("scenario", "expected"),
[
("english_turn_speaks_english", "en"),
("russian_turn_speaks_russian", "ru"),
("spanish_turn_speaks_spanish", "es"),
],
)
def test_voice_mode_speaks_the_language_whisper_detected(voice_probe, scenario, expected):
requests = voice_probe[scenario]["tts"]
assert requests, "voice mode never reached /api/tts"
for request in requests:
assert request["engine"] == "atlas"
assert request["language"] == expected
@pytest.mark.parametrize(
"scenario", ["missing_language_falls_back", "unsupported_language_falls_back"]
)
def test_voice_mode_omits_language_without_a_trusted_signal(voice_probe, scenario):
requests = voice_probe[scenario]["tts"]
assert requests, "voice mode never reached /api/tts"
for request in requests:
assert "language" not in request
def test_voice_mode_drops_hostile_language_values(voice_probe):
for case in voice_probe["hostile_language_values_are_dropped"]["results"]:
for request in case["tts"]:
assert "language" not in request, case["sent"]
def test_voice_mode_never_sends_a_voice_field(voice_probe):
for request in voice_probe["voice_field_is_never_sent"]["tts"]:
assert set(request) <= {"text", "engine", "language"}
assert "voice" not in request
def test_voice_mode_does_not_reuse_a_previous_turn_language(voice_probe):
requests = voice_probe["language_does_not_leak_into_later_turn"]["tts"]
assert len(requests) == 3
assert requests[0]["language"] == "ru"
assert "language" not in requests[1]
assert requests[2]["language"] == "es"
def test_voice_mode_ignores_language_from_an_empty_transcript(voice_probe):
result = voice_probe["empty_transcript_does_not_arm_a_language"]
assert result["sendsAfterBlank"] == []
assert result["sends"] == ["Hello."]
assert result["tts"], "the follow-up turn should still be spoken"
for request in result["tts"]:
assert "language" not in request
def test_voice_mode_discards_language_when_the_session_changes(voice_probe):
result = voice_probe["session_change_discards_language"]
assert result["afterSwitch"] == []
for request in result["tts"]:
assert "language" not in request
def test_voice_mode_discards_language_when_voice_mode_is_turned_off(voice_probe):
result = voice_probe["deactivation_discards_language"]
assert result["afterDeactivate"] == []
for request in result["tts"]:
assert "language" not in request
def test_voice_mode_speaks_nothing_when_transcription_fails(voice_probe):
result = voice_probe["transcribe_error_speaks_nothing"]
assert result["tts"] == []
assert any("Whisper is down" in toast for toast in result["toasts"])
# ---------------------------------------------------------------------------
# 6. Build-time enforcement and documented semantics
# ---------------------------------------------------------------------------
def test_image_build_verifies_every_language_routing_patch():
dockerfile = (DOCKERFILES / "Dockerfile.hermes-webui").read_text(encoding="utf-8")
assert "'language': detected" in dockerfile
assert "def _atlas_tts_language(body):" in dockerfile
assert 'request_payload["language"] = _atlas_language' in dockerfile
assert '"language": detected_language' in dockerfile
assert "takeSttLanguage(token)" in dockerfile
assert "/opt/hermes-webui/api/upload.py" in dockerfile
assert "/opt/hermes/tools/transcription_tools.py" in dockerfile
def test_atlas_patch_roots_are_overridable_for_offline_verification():
patch = ATLAS_PATCH.read_text(encoding="utf-8")
assert 'os.environ.get("HERMES_WEBUI_PATCH_ROOT", "/opt/hermes-webui")' in patch
assert 'os.environ.get("HERMES_AGENT_PATCH_ROOT", "/opt/hermes")' in patch
def test_notes_document_the_stt_driven_voice_selection_and_its_limits():
notes = (HERMES / "NOTES.md").read_text(encoding="utf-8")
assert "STT-detected language" in notes
for marker in ("hands-free", "Typed messages", "en_US-amy-medium"):
assert marker in notes