From 876b3dbee3cce3caa87f036181a367f88932d334 Mon Sep 17 00:00:00 2001 From: Hermes Agent Date: Thu, 20 Aug 2026 19:55:52 +0000 Subject: [PATCH] feat(hermes-voice): pick the Piper voice from the private Whisper language Hands-free voice mode had no language signal at all, so every spoken reply was synthesized with the English voice no matter what the user actually said. The multilingual Piper work (PR #26) added server-side routing for a "language" field but nothing ever sent one. Carry the language the private Jetson Whisper service already detects through to the TTS request for the reply that speech produced, and only for that reply. hermes-stt returns {text, model, language}, accepted only as a bare ISO-639 token; hermes_stt_client.py writes a .language sidecar next to the .txt transcript Hermes reads, leaving the local-command contract intact; the patched local-command envelope and /api/transcribe re-validate it and surface it; atlas-voice.js binds it to the voice-mode generation token and chat session, consumes it exactly once, and clears it on cancellation, restart, session change, empty transcript or transcription error; /api/tts honours it only from the fixed en/ru/es allow-list and otherwise sends English. A client "voice" field is never read at any hop, and typed messages, the manual read-aloud button, and any reply not produced by a spoken turn carry no trusted signal and stay on the English voice. The two WebUI-side and one agent-side edits are fail-closed replace_exact patches; both patch roots are now env-overridable so the contract can be verified offline without a GPU or an image build. --- dockerfiles/Dockerfile.hermes-webui | 9 +- dockerfiles/hermes-jetson-stt-server.py | 31 +- dockerfiles/hermes-webui-atlas-patch.py | 87 ++- dockerfiles/hermes-webui-atlas-voice.js | 52 +- services/hermes/NOTES.md | 34 + services/hermes/scripts/hermes_stt_client.py | 37 +- .../tests/data/atlas_voice_language_probe.js | 452 ++++++++++++ testing/tests/test_hermes_chat_quality.py | 2 +- .../test_hermes_voice_language_routing.py | 670 ++++++++++++++++++ 9 files changed, 1359 insertions(+), 15 deletions(-) create mode 100644 testing/tests/data/atlas_voice_language_probe.js create mode 100644 testing/tests/test_hermes_voice_language_routing.py diff --git a/dockerfiles/Dockerfile.hermes-webui b/dockerfiles/Dockerfile.hermes-webui index 52803505..4ba3fe57 100644 --- a/dockerfiles/Dockerfile.hermes-webui +++ b/dockerfiles/Dockerfile.hermes-webui @@ -113,9 +113,16 @@ RUN /opt/hermes/.venv/bin/python -c 'import cryptography, yaml' \ && grep -Fq 'routing_priority:priority' /opt/hermes-webui/static/atlas-router.js \ && grep -Fq "'atlas/auto/fast':'Automatic · Fast'" /opt/hermes-webui/static/atlas-router.js \ && grep -Fq 'explicit_reasoning_effort' /opt/hermes-webui/api/gateway_chat.py \ + && grep -Fq '"language": detected_language' /opt/hermes/tools/transcription_tools.py \ + && grep -Fq "'language': detected" /opt/hermes-webui/api/upload.py \ + && grep -Fq 'def _atlas_tts_language(body):' /opt/hermes-webui/api/routes.py \ + && grep -Fq '"language": _atlas_tts_language(data)' /opt/hermes-webui/api/routes.py \ + && grep -Fq 'takeSttLanguage(token)' /opt/hermes-webui/static/atlas-voice.js \ && /opt/hermes/.venv/bin/python -m py_compile \ /opt/hermes-webui/api/routes.py \ - /opt/hermes-webui/api/gateway_chat.py + /opt/hermes-webui/api/upload.py \ + /opt/hermes-webui/api/gateway_chat.py \ + /opt/hermes/tools/transcription_tools.py # Exercise the real server process in the target architecture before publish. RUN set -eu; \ diff --git a/dockerfiles/hermes-jetson-stt-server.py b/dockerfiles/hermes-jetson-stt-server.py index 1ecb9374..6e0efd08 100644 --- a/dockerfiles/hermes-jetson-stt-server.py +++ b/dockerfiles/hermes-jetson-stt-server.py @@ -62,6 +62,34 @@ def _clean_transcript(result: dict) -> str: return " ".join(kept).strip() +def _detected_language(result: object) -> str: + """Return the bare ISO-639 code Whisper decoded with, or nothing at all. + + ``whisper.transcribe`` reports the language it auto-detected (or the one it + was told to use) as a plain lowercase token such as ``en``/``ru``/``yue``. + Anything that is not that exact shape is dropped rather than guessed at, so + a surprising model result can never become a downstream voice selector. + """ + if not isinstance(result, dict): + return "" + value = result.get("language") + if not isinstance(value, str): + return "" + code = value.strip().lower() + if not 2 <= len(code) <= 3 or not code.isascii() or not code.isalpha(): + return "" + return code + + +def _transcription_payload(result: dict) -> dict: + """Build the transcription contract: text plus the model's own language.""" + return { + "text": _clean_transcript(result), + "model": MODEL_NAME, + "language": _detected_language(result), + } + + def _json(handler: BaseHTTPRequestHandler, status: int, payload: dict) -> None: body = json.dumps(payload).encode("utf-8") handler.send_response(status) @@ -148,8 +176,7 @@ class SpeechHandler(BaseHTTPRequestHandler): no_speech_threshold=0.5, verbose=False, ) - transcript = _clean_transcript(result) - _json(self, 200, {"text": transcript, "model": MODEL_NAME}) + _json(self, 200, _transcription_payload(result)) except Exception as exc: print(f"[stt] transcription failed: {exc}", flush=True) _json(self, 500, {"error": "transcription failed"}) diff --git a/dockerfiles/hermes-webui-atlas-patch.py b/dockerfiles/hermes-webui-atlas-patch.py index 20d04399..4d5062a8 100644 --- a/dockerfiles/hermes-webui-atlas-patch.py +++ b/dockerfiles/hermes-webui-atlas-patch.py @@ -1,10 +1,14 @@ #!/usr/bin/env python3 """Apply fail-closed Atlas voice integration patches to pinned Hermes WebUI.""" +import os from pathlib import Path -ROOT = Path("/opt/hermes-webui") +ROOT = Path(os.environ.get("HERMES_WEBUI_PATCH_ROOT", "/opt/hermes-webui")) +# The WebUI imports the pinned agent's STT tooling from the same image, so the +# local-command transcription envelope is patched alongside the WebUI itself. +AGENT_ROOT = Path(os.environ.get("HERMES_AGENT_PATCH_ROOT", "/opt/hermes")) def replace_exact(path: Path, before: str, after: str, count: int = 1) -> None: @@ -45,7 +49,87 @@ replace_exact( "if(engine==='edge'||engine==='atlas'){\n _playEdgeTtsChunked(clean, null, engine);", ) +# The private Whisper service reports the language it decoded with. Carry that +# through the agent's local-command STT envelope so the WebUI can hand a voice +# hint to Piper instead of guessing the reply's language from its text. +transcription = AGENT_ROOT / "tools/transcription_tools.py" +replace_exact( + transcription, + ''' transcript_text = txt_files[0].read_text(encoding="utf-8").strip() + logger.info( + "Transcribed %s via local STT command (%s, %d chars)", + Path(file_path).name, + normalized_model, + len(transcript_text), + ) + return {"success": True, "transcript": transcript_text, "provider": "local_command"} +''', + ''' transcript_text = txt_files[0].read_text(encoding="utf-8").strip() + logger.info( + "Transcribed %s via local STT command (%s, %d chars)", + Path(file_path).name, + normalized_model, + len(transcript_text), + ) + detected_language = "" + language_files = sorted(Path(output_dir).glob("*.language")) + if language_files: + try: + candidate = language_files[0].read_text(encoding="utf-8").strip().lower() + except (OSError, ValueError): + candidate = "" + if 2 <= len(candidate) <= 3 and candidate.isascii() and candidate.isalpha(): + detected_language = candidate + return { + "success": True, + "transcript": transcript_text, + "provider": "local_command", + "language": detected_language, + } +''', +) + +upload = ROOT / "api/upload.py" +replace_exact( + upload, + """ transcript = str(result.get('transcript') or '').strip() + return j(handler, {'ok': True, 'transcript': transcript}) +""", + """ transcript = str(result.get('transcript') or '').strip() + detected = str(result.get('language') or '').strip().lower() + if not (2 <= len(detected) <= 3 and detected.isascii() and detected.isalpha()): + detected = '' + return j(handler, {'ok': True, 'transcript': transcript, 'language': detected}) +""", +) + routes = ROOT / "api/routes.py" +replace_exact( + routes, + "def _tts_open(req, *, timeout=30, opener_factory=None):", + '''ATLAS_TTS_LANGUAGES = ("en", "ru", "es") + + +def _atlas_tts_language(body): + """Resolve the private TTS language from a request body, English by default. + + Only a language the private Piper deployment actually bakes a voice for is + honoured. Everything else — a missing field, a hostile string, a wrong type, + or a client-supplied "voice" — resolves to English, so nothing a browser + sends can steer synthesis outside the fixed policy. The Jetson service + applies the same allow-list again as the final authority. + """ + if not isinstance(body, dict): + return "en" + value = body.get("language") + if not isinstance(value, str): + return "en" + code = value.strip().lower().replace("_", "-").split("-", 1)[0] + return code if code in ATLAS_TTS_LANGUAGES else "en" + + +def _tts_open(req, *, timeout=30, opener_factory=None):''', +) marker = " # ── ElevenLabs TTS ──────────────────────────────────────────────────\n" atlas = ''' # ── Atlas private Jetson TTS ───────────────────────────────────────── if engine == "atlas": @@ -65,6 +149,7 @@ atlas = ''' # ── Atlas private Jetson TTS ────────── "input": text, "voice": "en_US-lessac-high", "speed": speed, + "language": _atlas_tts_language(data), }).encode("utf-8") request = Request(atlas_url, data=request_body, headers={ "Content-Type": "application/json", diff --git a/dockerfiles/hermes-webui-atlas-voice.js b/dockerfiles/hermes-webui-atlas-voice.js index e83c35b7..042eb1af 100644 --- a/dockerfiles/hermes-webui-atlas-voice.js +++ b/dockerfiles/hermes-webui-atlas-voice.js @@ -19,9 +19,37 @@ let vadTimer=null; let currentAudio=null; let thinkingSession=null; + // The only language signal this file trusts is the one the private Whisper + // service returned for the audio of the turn currently being answered. It is + // bound to that turn's generation token and consumed exactly once. + let sttLanguage=''; + let sttLanguageToken=-1; + const TTS_LANGUAGES=['en','ru','es']; const originalAutoRead=window.autoReadLastAssistant; const originalApplyPreference=window._applyVoiceModePref; + function normalizeSttLanguage(value){ + if(typeof value!=='string') return ''; + const code=value.trim().toLowerCase(); + return TTS_LANGUAGES.indexOf(code)>=0?code:''; + } + + function clearSttLanguage(){ + sttLanguage=''; + sttLanguageToken=-1; + } + + function rememberSttLanguage(language, token){ + sttLanguage=language||''; + sttLanguageToken=sttLanguage?token:-1; + } + + function takeSttLanguage(token){ + const language=sttLanguageToken===token?sttLanguage:''; + clearSttLanguage(); + return language; + } + function toast(message){ if(typeof window.showToast==='function') window.showToast(message,3000); } @@ -54,6 +82,7 @@ active=false; state='idle'; thinkingSession=null; + clearSttLanguage(); stopCapture(); stopPlayback(); modeBtn.classList.remove('active'); @@ -67,13 +96,14 @@ },delay||500); } - function sendTranscript(transcript, token){ + function sendTranscript(transcript, token, language){ if(!active||token!==generation) return; const text=String(transcript||'').trim(); - if(!text){restartSoon(token,350);return;} + if(!text){clearSttLanguage();restartSoon(token,350);return;} composer.value=text; if(typeof window.autoResize==='function') window.autoResize(); thinkingSession=(typeof S!=='undefined'&&S.session)?S.session.session_id:null; + rememberSttLanguage(language,token); setState('thinking'); if(typeof window.send==='function') window.send(); } @@ -88,7 +118,7 @@ const response=await fetch('/api/transcribe',{method:'POST',body:form}); const payload=await response.json().catch(function(){return {};}); if(!response.ok) throw new Error(payload.error||('Whisper request failed: '+response.status)); - sendTranscript(payload.transcript,token); + sendTranscript(payload.transcript,token,normalizeSttLanguage(payload.language)); }catch(error){ if(!active||token!==generation) return; deactivate(false); @@ -105,6 +135,7 @@ async function startListening(token){ if(!active||token!==generation) return; stopCapture(); + clearSttLanguage(); setState('listening'); try{ const capture=await navigator.mediaDevices.getUserMedia({ @@ -214,11 +245,15 @@ }); } - async function fetchSpeech(chunk){ + async function fetchSpeech(chunk, language){ + // `language` is only ever the private STT result for this turn. When it is + // absent the field is omitted entirely and the server picks English. + const request={text:chunk,engine:'atlas'}; + if(language) request.language=language; const response=await fetch('/api/tts',{ method:'POST', headers:{'Content-Type':'application/json'}, - body:JSON.stringify({text:chunk,engine:'atlas'}), + body:JSON.stringify(request), }); if(!response.ok){ const payload=await response.json().catch(function(){return {};}); @@ -232,10 +267,12 @@ const currentSession=(typeof S!=='undefined'&&S.session)?S.session.session_id:null; if(thinkingSession&¤tSession&&thinkingSession!==currentSession){ thinkingSession=null; + clearSttLanguage(); restartSoon(token,250); return; } thinkingSession=null; + const language=takeSttLanguage(token); const rows=document.querySelectorAll('.msg-row[data-role="assistant"], .assistant-segment[data-raw-text]'); if(!rows.length){restartSoon(token,250);return;} const text=cleanForSpeech(rows[rows.length-1].dataset.rawText||''); @@ -243,11 +280,11 @@ setState('speaking'); const chunks=typeof window._splitForTTS==='function'?window._splitForTTS(text,280):[text]; try{ - let pending=fetchSpeech(chunks[0]); + let pending=fetchSpeech(chunks[0],language); for(let index=0;index.txt` transcript plus a + `.language` sidecar. The transcript stays the only `.txt` in the + output directory, so the stock Hermes local-command contract is unchanged. +3. The patched local-command STT envelope reads that sidecar and adds + `language` to its result; `/api/transcribe` re-validates it and returns it + next to `transcript`. +4. `atlas-voice.js` keeps that value only for the turn it belongs to. It is + bound to the voice-mode generation token and the chat session id, consumed + exactly once by the reply that turn produced, and cleared on cancellation, + restart, session change, an empty transcript, or a transcription error. +5. `/api/tts` accepts `language` only from the fixed allow-list and otherwise + sends English. The Jetson TTS service applies the same allow-list again as + the final authority. + +**What this does not claim.** The language is the language the *user spoke*, +not the language of the reply. A model asked a Russian question may answer in +English and will then be read aloud by the Russian voice, and vice versa; this +is a deliberate policy choice for hands-free mode, not a detection failure. +Nothing here detects the language of assistant text. + +**Everything else stays English.** Typed messages, the manual read-aloud +button, and any assistant reply that was not produced by a hands-free spoken +turn carry no trusted STT signal, so they synthesize with `en_US-amy-medium`. +A `voice` field from a browser is never honoured at any hop. + ## Honest limits - Hermes does not currently apply production or cluster changes autonomously. diff --git a/services/hermes/scripts/hermes_stt_client.py b/services/hermes/scripts/hermes_stt_client.py index 639fc7b1..96f7081d 100644 --- a/services/hermes/scripts/hermes_stt_client.py +++ b/services/hermes/scripts/hermes_stt_client.py @@ -12,6 +12,34 @@ from pathlib import Path from urllib.request import Request, urlopen +LANGUAGE_SUFFIX = ".language" + + +def _normalize_language(value: object) -> str: + """Accept only a bare ISO-639 code from the private Whisper response.""" + if not isinstance(value, str): + return "" + code = value.strip().lower() + if not 2 <= len(code) <= 3 or not code.isascii() or not code.isalpha(): + return "" + return code + + +def _write_result(output_dir: Path, stem: str, transcript: str, language: str) -> Path: + """Write the .txt Hermes reads, plus the language sidecar when we have one. + + Hermes' local-command contract is "leave a .txt in --output-dir"; it globs + ``*.txt`` and reads the first match. The sidecar deliberately uses another + suffix so the transcript stays the only ``.txt`` in the directory. + """ + output_dir.mkdir(parents=True, exist_ok=True) + transcript_path = output_dir / f"{stem}.txt" + transcript_path.write_text(transcript, encoding="utf-8") + if language: + (output_dir / f"{stem}{LANGUAGE_SUFFIX}").write_text(language, encoding="utf-8") + return transcript_path + + def _multipart(audio: Path, language: str, model: str) -> tuple[bytes, str]: boundary = f"atlas-hermes-{secrets.token_hex(12)}" mime = mimetypes.guess_type(audio.name)[0] or "application/octet-stream" @@ -67,9 +95,12 @@ def main() -> None: with urlopen(request, timeout=120) as response: result = json.loads(response.read().decode("utf-8")) transcript = str(result.get("text") or "").strip() - args.output_dir.mkdir(parents=True, exist_ok=True) - output = args.output_dir / f"{args.input_path.stem}.txt" - output.write_text(transcript, encoding="utf-8") + _write_result( + args.output_dir, + args.input_path.stem, + transcript, + _normalize_language(result.get("language")), + ) if __name__ == "__main__": diff --git a/testing/tests/data/atlas_voice_language_probe.js b/testing/tests/data/atlas_voice_language_probe.js new file mode 100644 index 00000000..2db09f92 --- /dev/null +++ b/testing/tests/data/atlas_voice_language_probe.js @@ -0,0 +1,452 @@ +// Deterministic browser stub that drives dockerfiles/hermes-webui-atlas-voice.js +// through complete hands-free turns with no microphone, audio device or GPU. +// +// The script under test is an IIFE with no exported seams, so the only honest +// way to assert what reaches POST /api/tts is to run it against a fake DOM and +// fake clock and record the requests it actually makes. Usage: +// +// node atlas_voice_language_probe.js +// +// It prints one JSON object describing every scenario to stdout. +'use strict'; + +const fs = require('fs'); +const vm = require('vm'); + +const SCRIPT_PATH = process.argv[2]; +if (!SCRIPT_PATH) { + throw new Error('usage: atlas_voice_language_probe.js '); +} +const SOURCE = fs.readFileSync(SCRIPT_PATH, 'utf8'); + +function flush() { + // Four macrotask hops drain the promise chains the script builds around + // fetch()/json()/blob()/play() without ever waiting on wall-clock time. + return new Promise((resolve) => { + let hops = 0; + (function hop() { + hops += 1; + if (hops > 12) { + resolve(); + return; + } + setImmediate(hop); + })(); + }); +} + +function makeElement(id) { + return { + id, + style: {}, + dataset: {}, + value: '', + textContent: '', + className: '', + classList: { + entries: new Set(), + add(name) { this.entries.add(name); }, + remove(name) { this.entries.delete(name); }, + contains(name) { return this.entries.has(name); }, + }, + listeners: [], + addEventListener(type, handler) { this.listeners.push({ type, handler }); }, + removeEventListener(type, handler) { + this.listeners = this.listeners.filter((entry) => entry.handler !== handler); + }, + click() { + const event = { preventDefault() {}, stopImmediatePropagation() {} }; + this.listeners + .filter((entry) => entry.type === 'click') + .forEach((entry) => entry.handler(event)); + }, + querySelector() { return null; }, + insertBefore() {}, + appendChild() {}, + }; +} + +function makeHarness() { + const clock = { now: 1000000 }; + const timeouts = []; + const intervals = new Map(); + let timerId = 1; + + const ttsRequests = []; + const transcribeCalls = []; + const toasts = []; + const sends = []; + + let capability = { ok: true, available: true, provider: 'local_command' }; + let transcribeResponse = { ok: true, transcript: 'hello', language: 'en' }; + let transcribeStatus = 200; + let assistantRows = []; + let loud = false; + let recorder = null; + + const storage = new Map(); + const elements = {}; + ['btnVoiceMode', 'voiceModeBar', 'voiceModeIndicator', 'voiceModeLabel', 'msg'] + .forEach((id) => { elements[id] = makeElement(id); }); + + function MediaRecorder() { + this.state = 'recording'; + this.ondataavailable = null; + this.onstop = null; + recorder = this; + } + MediaRecorder.prototype.start = function start() { this.state = 'recording'; }; + MediaRecorder.prototype.stop = function stop() { + if (this.state === 'inactive') return; + this.state = 'inactive'; + if (this.onstop) this.onstop(); + }; + MediaRecorder.isTypeSupported = function isTypeSupported() { return true; }; + + function AudioContext() { + this.createAnalyser = () => ({ + fftSize: 2048, + getByteTimeDomainData(samples) { + for (let i = 0; i < samples.length; i += 1) { + samples[i] = loud ? (i % 2 ? 200 : 56) : 128; + } + }, + }); + this.createBiquadFilter = () => ({ + type: '', frequency: { value: 0 }, Q: { value: 0 }, connect() {}, + }); + this.createMediaStreamSource = () => ({ connect() {} }); + this.close = () => {}; + } + + function AudioElement() { + this.currentTime = 0; + this.onended = null; + this.onerror = null; + this.pause = () => {}; + this.play = () => { + setImmediate(() => { if (this.onended) this.onended(); }); + return Promise.resolve(); + }; + } + + async function fetchStub(url, init) { + if (url === '/api/transcribe/capability') { + return { ok: true, status: 200, json: async () => capability }; + } + if (url === '/api/transcribe') { + transcribeCalls.push({ body: init && init.body }); + return { + ok: transcribeStatus < 400, + status: transcribeStatus, + json: async () => transcribeResponse, + }; + } + if (url === '/api/tts') { + ttsRequests.push(JSON.parse(init.body)); + return { + ok: true, + status: 200, + blob: async () => ({ synthetic: true }), + json: async () => ({}), + }; + } + throw new Error(`unexpected fetch: ${url}`); + } + + const context = { + console, + Uint8Array, + Promise, + Math, + JSON, + String, + Number, + Error, + parseInt, + isNaN, + Set, + Map, + Array, + Object, + Date: { now: () => clock.now }, + Blob: function Blob(parts, options) { this.parts = parts; this.type = (options || {}).type || ''; }, + File: function File(parts, name, options) { + this.parts = parts; this.name = name; this.type = (options || {}).type || ''; + }, + FormData: function FormData() { this.entries = []; this.append = (k, v) => this.entries.push([k, v]); }, + URL: { createObjectURL: () => 'blob:atlas-test', revokeObjectURL() {} }, + Audio: AudioElement, + MediaRecorder, + AudioContext, + fetch: fetchStub, + localStorage: { + getItem: (key) => (storage.has(key) ? storage.get(key) : null), + setItem: (key, value) => { storage.set(key, String(value)); }, + removeItem: (key) => { storage.delete(key); }, + }, + navigator: { + mediaDevices: { + getUserMedia: async () => ({ getTracks: () => [{ stop() {} }] }), + getSupportedConstraints: () => ({}), + }, + }, + document: { + getElementById: (id) => elements[id] || null, + querySelectorAll: () => assistantRows, + createElement: () => ({ value: '', textContent: '' }), + }, + S: { session: { session_id: 'session-1' }, busy: false }, + setTimeout: (fn, delay) => { + const id = timerId; timerId += 1; + timeouts.push({ id, fn, at: clock.now + (delay || 0) }); + return id; + }, + clearTimeout: (id) => { + const index = timeouts.findIndex((entry) => entry.id === id); + if (index >= 0) timeouts.splice(index, 1); + }, + setInterval: (fn, delay) => { + const id = timerId; timerId += 1; + intervals.set(id, { fn, delay: delay || 0 }); + return id; + }, + clearInterval: (id) => { intervals.delete(id); }, + }; + context.window = context; + context.showToast = (message) => { toasts.push(message); }; + context.send = () => { sends.push(elements.msg.value); }; + context.autoResize = () => {}; + + vm.createContext(context); + vm.runInContext(SOURCE, context, { filename: 'atlas-voice.js' }); + + function runDueTimeouts() { + const due = timeouts.filter((entry) => entry.at <= clock.now); + due.forEach((entry) => { + const index = timeouts.indexOf(entry); + if (index >= 0) timeouts.splice(index, 1); + entry.fn(); + }); + } + + function tick(ms) { + clock.now += ms; + Array.from(intervals.values()).forEach((entry) => entry.fn()); + runDueTimeouts(); + } + + return { + context, + elements, + ttsRequests, + transcribeCalls, + toasts, + sends, + clock, + recorder: () => recorder, + setLoud: (value) => { loud = value; }, + setCapability: (value) => { capability = value; }, + setTranscribeResponse: (value, status) => { + transcribeResponse = value; + transcribeStatus = status === undefined ? 200 : status; + }, + setAssistantReply: (text) => { assistantRows = [{ dataset: { rawText: text } }]; }, + setSession: (id) => { context.S.session = { session_id: id }; }, + advance: (ms) => { clock.now += ms; }, + tick, + runDueTimeouts, + flush, + }; +} + +// Walk one capture window: pre-roll audio, three loud frames so the VAD latches +// speech, then silence past the hangover so MediaRecorder.stop() fires. +async function captureSpeech(harness) { + const active = harness.recorder(); + if (!active) throw new Error('voice mode never created a recorder'); + active.ondataavailable({ data: { size: 512 } }); + harness.setLoud(true); + for (let i = 0; i < 4; i += 1) harness.tick(100); + active.ondataavailable({ data: { size: 512 } }); + harness.setLoud(false); + harness.advance(2500); + harness.tick(100); + await harness.flush(); +} + +async function startVoiceMode(harness) { + await harness.flush(); + harness.elements.btnVoiceMode.click(); + await harness.flush(); +} + +// One complete turn: speak, transcribe, let the app "answer", read it back. +async function runTurn(harness, { transcript, language, reply }) { + const payload = { ok: true, transcript }; + if (language !== undefined) payload.language = language; + harness.setTranscribeResponse(payload); + await captureSpeech(harness); + harness.setAssistantReply(reply || 'An answer.'); + harness.context.autoReadLastAssistant(); + await harness.flush(); +} + +async function restartListening(harness) { + harness.advance(1000); + harness.runDueTimeouts(); + await harness.flush(); +} + +const scenarios = {}; + +scenarios.english_turn_speaks_english = async () => { + const harness = makeHarness(); + await startVoiceMode(harness); + await runTurn(harness, { transcript: 'What is the weather?', language: 'en' }); + return { tts: harness.ttsRequests }; +}; + +scenarios.russian_turn_speaks_russian = async () => { + const harness = makeHarness(); + await startVoiceMode(harness); + await runTurn(harness, { transcript: 'Как дела?', language: 'ru', reply: 'Всё хорошо.' }); + return { tts: harness.ttsRequests }; +}; + +scenarios.spanish_turn_speaks_spanish = async () => { + const harness = makeHarness(); + await startVoiceMode(harness); + await runTurn(harness, { transcript: '¿Qué tal?', language: 'es', reply: 'Muy bien.' }); + return { tts: harness.ttsRequests }; +}; + +scenarios.missing_language_falls_back = async () => { + const harness = makeHarness(); + await startVoiceMode(harness); + await runTurn(harness, { transcript: 'Hello there.' }); + return { tts: harness.ttsRequests }; +}; + +scenarios.unsupported_language_falls_back = async () => { + const harness = makeHarness(); + await startVoiceMode(harness); + await runTurn(harness, { transcript: 'Bonjour tout le monde.', language: 'fr' }); + return { tts: harness.ttsRequests }; +}; + +scenarios.hostile_language_values_are_dropped = async () => { + const results = []; + const hostile = [ + 'ru; rm -rf /', + '../../ru_RU-irina-medium', + 'ru', + 'RUSSIAN', + { language: 'ru' }, + ['ru'], + 42, + null, + 'r', + 'ru ru', + 'x'.repeat(4096), + ]; + for (const language of hostile) { + const harness = makeHarness(); + // eslint-disable-next-line no-await-in-loop + await startVoiceMode(harness); + // eslint-disable-next-line no-await-in-loop + await runTurn(harness, { transcript: 'Say something.', language }); + results.push({ sent: String(language), tts: harness.ttsRequests }); + } + return { results }; +}; + +scenarios.voice_field_is_never_sent = async () => { + const harness = makeHarness(); + await startVoiceMode(harness); + await runTurn(harness, { transcript: 'Как дела?', language: 'ru' }); + return { tts: harness.ttsRequests }; +}; + +scenarios.language_does_not_leak_into_later_turn = async () => { + const harness = makeHarness(); + await startVoiceMode(harness); + await runTurn(harness, { transcript: 'Как дела?', language: 'ru', reply: 'Всё хорошо.' }); + await restartListening(harness); + await runTurn(harness, { transcript: 'And in English?', language: undefined }); + await restartListening(harness); + await runTurn(harness, { transcript: '¿Y ahora?', language: 'es' }); + return { tts: harness.ttsRequests }; +}; + +scenarios.empty_transcript_does_not_arm_a_language = async () => { + const harness = makeHarness(); + await startVoiceMode(harness); + harness.setTranscribeResponse({ ok: true, transcript: ' ', language: 'ru' }); + await captureSpeech(harness); + const sendsAfterBlank = harness.sends.slice(); + // A reply landing while the blank turn winds down must not inherit a + // language that transcript never earned. + harness.setAssistantReply('A stray answer.'); + harness.context.autoReadLastAssistant(); + await harness.flush(); + await restartListening(harness); + await runTurn(harness, { transcript: 'Hello.', language: undefined }); + return { sendsAfterBlank, tts: harness.ttsRequests, sends: harness.sends }; +}; + +scenarios.session_change_discards_language = async () => { + const harness = makeHarness(); + await startVoiceMode(harness); + harness.setTranscribeResponse({ ok: true, transcript: 'Как дела?', language: 'ru' }); + await captureSpeech(harness); + harness.setSession('session-2'); + harness.setAssistantReply('Reply that belongs to another chat.'); + harness.context.autoReadLastAssistant(); + await harness.flush(); + const afterSwitch = harness.ttsRequests.slice(); + await restartListening(harness); + await runTurn(harness, { transcript: 'Hello again.', language: undefined }); + return { afterSwitch, tts: harness.ttsRequests }; +}; + +scenarios.deactivation_discards_language = async () => { + const harness = makeHarness(); + await startVoiceMode(harness); + harness.setTranscribeResponse({ ok: true, transcript: 'Как дела?', language: 'ru' }); + await captureSpeech(harness); + harness.elements.btnVoiceMode.click(); + await harness.flush(); + harness.setAssistantReply('Late reply after the user left voice mode.'); + harness.context.autoReadLastAssistant(); + await harness.flush(); + const afterDeactivate = harness.ttsRequests.slice(); + harness.elements.btnVoiceMode.click(); + await harness.flush(); + await runTurn(harness, { transcript: 'Fresh start.', language: undefined }); + return { afterDeactivate, tts: harness.ttsRequests }; +}; + +scenarios.transcribe_error_speaks_nothing = async () => { + const harness = makeHarness(); + await startVoiceMode(harness); + harness.setTranscribeResponse({ error: 'Whisper is down' }, 503); + await captureSpeech(harness); + harness.setAssistantReply('Some earlier answer.'); + harness.context.autoReadLastAssistant(); + await harness.flush(); + return { tts: harness.ttsRequests, toasts: harness.toasts }; +}; + +(async () => { + const output = {}; + const names = Object.keys(scenarios); + for (const name of names) { + // eslint-disable-next-line no-await-in-loop + output[name] = await scenarios[name](); + } + process.stdout.write(JSON.stringify(output, null, 1)); +})().catch((error) => { + process.stderr.write(String((error && error.stack) || error)); + process.exit(1); +}); diff --git a/testing/tests/test_hermes_chat_quality.py b/testing/tests/test_hermes_chat_quality.py index 4cf45ffb..99054909 100644 --- a/testing/tests/test_hermes_chat_quality.py +++ b/testing/tests/test_hermes_chat_quality.py @@ -343,7 +343,7 @@ def test_chat_voice_uses_private_jetson_services_and_shared_auto_route(): assert "/api/tts" in voice_script assert "speakResponse(generation)" in voice_script assert "window._splitForTTS(text,280)" in voice_script - assert "pending=fetchSpeech(chunks[index+1])" in voice_script + assert "pending=fetchSpeech(chunks[index+1],language)" in voice_script assert "restartSoon(token,450)" in voice_script assert "constraints.voiceIsolation=true" in voice_script assert "highpass.frequency.value=140" in voice_script diff --git a/testing/tests/test_hermes_voice_language_routing.py b/testing/tests/test_hermes_voice_language_routing.py new file mode 100644 index 00000000..bc000958 --- /dev/null +++ b/testing/tests/test_hermes_voice_language_routing.py @@ -0,0 +1,670 @@ +"""Voice-mode language routing: private Whisper STT decides the Piper voice. + +Every assertion here runs without a GPU, a microphone or a cluster. The browser +contract is exercised by driving the real ``atlas-voice.js`` inside a stub DOM +(``testing/tests/data/atlas_voice_language_probe.js``), and the two server-side +trust boundaries are exercised by applying the real image patch to fixtures that +carry the exact upstream anchors and then importing the patched result. +""" + +from __future__ import annotations + +import importlib.util +import json +import os +import shutil +import subprocess +import sys +from pathlib import Path +from types import SimpleNamespace + +import pytest + +from testing.tests.test_hermes_chat_support import HERMES, ROOT + +DOCKERFILES = ROOT / "dockerfiles" +ATLAS_PATCH = DOCKERFILES / "hermes-webui-atlas-patch.py" +VOICE_SCRIPT = DOCKERFILES / "hermes-webui-atlas-voice.js" +VOICE_PROBE = ROOT / "testing" / "tests" / "data" / "atlas_voice_language_probe.js" +ATLAS_TTS_URL = "http://hermes-tts.hermes.svc.cluster.local:9001/v1/audio/speech" + +# Voices baked by the multilingual Piper work (PR #26): en=amy, ru=irina, +# es=claude. Anything outside this set must resolve to English. +SUPPORTED = ("en", "ru", "es") + + +# --------------------------------------------------------------------------- +# Module loaders +# --------------------------------------------------------------------------- + + +def _load_stt_server(monkeypatch): + """Import the Jetson Whisper service without CUDA, torch or whisper.""" + path = DOCKERFILES / "hermes-jetson-stt-server.py" + spec = importlib.util.spec_from_file_location("hermes_jetson_stt_server", path) + assert spec and spec.loader + module = importlib.util.module_from_spec(spec) + monkeypatch.setitem(sys.modules, "cgi", SimpleNamespace()) + monkeypatch.setitem( + sys.modules, + "torch", + SimpleNamespace(cuda=SimpleNamespace(is_available=lambda: False)), + ) + monkeypatch.setitem(sys.modules, "whisper", SimpleNamespace()) + spec.loader.exec_module(module) + return module + + +def _load_stt_client(): + """Import the local-command STT client that Hermes shells out to.""" + path = HERMES / "scripts" / "hermes_stt_client.py" + spec = importlib.util.spec_from_file_location("hermes_stt_client", path) + assert spec and spec.loader + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +# --------------------------------------------------------------------------- +# Patch fixtures — each file carries the exact upstream fragment the image +# patch pins, so importing the patched result exercises the inserted code. +# --------------------------------------------------------------------------- + + +INDEX_FIXTURE = ( + '\n' + '\n' +) + +UI_FIXTURE = """function _playEdgeTtsChunked(text, btn){ + fetch('/api/tts',{method:'POST',body:JSON.stringify({text:chunk, voice:voice, rate:rate, pitch:pitch})}); +} +function readAloud(clean, btn, engine){ + if(engine==='edge'){ + _playEdgeTtsChunked(clean, btn); + } +} +function autoRead(clean, engine){ + if(engine==='edge'){ + _playEdgeTtsChunked(clean, null); + } +} +""" + +HELPERS_FIXTURE = '''"""Stand-in for the WebUI helper module the patched code imports.""" + + +def bad(handler, message, status=400): + return {"status": status, "error": message} +''' + +ROUTES_FIXTURE = '''"""Stand-in carrying the exact upstream anchors the Atlas TTS patch pins.""" + +import json +import os +from urllib.request import ProxyHandler, Request, build_opener + + +class _NoRedirectTtsHandler: + """Placeholder for the upstream no-redirect opener handler.""" + + +class _Logger: + def __init__(self): + self.failures = [] + + def exception(self, message): + self.failures.append(message) + + +logger = _Logger() +UPSTREAM_REQUESTS = [] + + +class _Upstream: + def __init__(self, payload): + self._payload = payload + + def read(self): + return self._payload + + def __enter__(self): + return self + + def __exit__(self, *exc_info): + return False + + +def _buffer_tts_audio_response(response): + return response.read() + + +def _tts_open(req, *, timeout=30, opener_factory=None): + """Thin network seam for the TTS upstream fetch so tests can intercept it.""" + UPSTREAM_REQUESTS.append(json.loads(req.data.decode("utf-8"))) + return _Upstream(b"RIFFsynthetic") + + +def _handle_tts(handler, data, text, rate_str, engine): + # ── ElevenLabs TTS ────────────────────────────────────────────────── + return None +''' + +UPLOAD_FIXTURE = '''"""Stand-in carrying the exact upstream /api/transcribe response anchor.""" + + +def j(handler, payload, status=200): + return {"status": status, "payload": payload} + + +def handle_transcribe(handler, result): + try: + transcript = str(result.get('transcript') or '').strip() + return j(handler, {'ok': True, 'transcript': transcript}) + except ValueError as error: + return j(handler, {'error': str(error)}, status=400) +''' + +TRANSCRIPTION_FIXTURE = '''"""Stand-in carrying the exact upstream local-command STT envelope anchor.""" + +import contextlib +from pathlib import Path + + +class _Logger: + def info(self, *args): + return None + + +logger = _Logger() + + +def _transcribe_local_command(file_path, normalized_model, output_dir): + try: + with contextlib.nullcontext(output_dir): + txt_files = sorted(Path(output_dir).glob("*.txt")) + transcript_text = txt_files[0].read_text(encoding="utf-8").strip() + logger.info( + "Transcribed %s via local STT command (%s, %d chars)", + Path(file_path).name, + normalized_model, + len(transcript_text), + ) + return {"success": True, "transcript": transcript_text, "provider": "local_command"} + except OSError as error: + return {"success": False, "transcript": "", "error": str(error)} +''' + + +def _write(path: Path, text: str) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(text, encoding="utf-8") + + +@pytest.fixture +def patched_webui(tmp_path, monkeypatch): + """Apply the real Atlas image patch to fixture trees and import the result.""" + webui = tmp_path / "hermes-webui" + agent = tmp_path / "hermes" + _write(webui / "static" / "index.html", INDEX_FIXTURE) + _write(webui / "static" / "ui.js", UI_FIXTURE) + _write(webui / "api" / "__init__.py", "") + _write(webui / "api" / "helpers.py", HELPERS_FIXTURE) + _write(webui / "api" / "routes.py", ROUTES_FIXTURE) + _write(webui / "api" / "upload.py", UPLOAD_FIXTURE) + _write(agent / "tools" / "__init__.py", "") + _write(agent / "tools" / "transcription_tools.py", TRANSCRIPTION_FIXTURE) + + environment = dict(os.environ) + environment["HERMES_WEBUI_PATCH_ROOT"] = str(webui) + environment["HERMES_AGENT_PATCH_ROOT"] = str(agent) + completed = subprocess.run( + [sys.executable, str(ATLAS_PATCH)], + env=environment, + capture_output=True, + text=True, + ) + assert completed.returncode == 0, completed.stderr or completed.stdout + + for name in ("api", "api.helpers", "api.routes", "api.upload", "tools", + "tools.transcription_tools"): + sys.modules.pop(name, None) + monkeypatch.syspath_prepend(str(agent)) + monkeypatch.syspath_prepend(str(webui)) + import api.routes as routes # noqa: PLC0415 + import api.upload as upload # noqa: PLC0415 + import tools.transcription_tools as transcription # noqa: PLC0415 + + yield SimpleNamespace( + webui=webui, + agent=agent, + routes=routes, + upload=upload, + transcription=transcription, + ) + + for name in ("api", "api.helpers", "api.routes", "api.upload", "tools", + "tools.transcription_tools"): + sys.modules.pop(name, None) + + +class _Handler: + """Just enough BaseHTTPRequestHandler surface for the Atlas TTS branch.""" + + def __init__(self): + self.status = None + self.headers_sent = {} + self.wfile = SimpleNamespace(write=self._write) + self.body = b"" + + def send_response(self, status): + self.status = status + + def send_header(self, name, value): + self.headers_sent[name] = value + + def end_headers(self): + return None + + def _write(self, payload): + self.body += payload + + +def _atlas_tts(patched, monkeypatch, data): + """Run the patched Atlas branch and return the JSON it sent to hermes-tts.""" + monkeypatch.setenv("HERMES_WEBUI_ATLAS_TTS_URL", ATLAS_TTS_URL) + patched.routes.UPSTREAM_REQUESTS.clear() + handler = _Handler() + result = patched.routes._handle_tts(handler, data, "Some reply.", "", "atlas") + assert result is True, "the Atlas branch must own the response" + assert handler.status == 200 + assert len(patched.routes.UPSTREAM_REQUESTS) == 1 + return patched.routes.UPSTREAM_REQUESTS[0] + + +# --------------------------------------------------------------------------- +# 1. Whisper service reports the language it actually decoded with +# --------------------------------------------------------------------------- + + +def test_stt_response_carries_whisper_detected_language(monkeypatch): + module = _load_stt_server(monkeypatch) + payload = module._transcription_payload( + { + "language": "ru", + "segments": [ + {"text": " Как дела?", "no_speech_prob": 0.1, "avg_logprob": -0.2} + ], + } + ) + assert payload["text"] == "Как дела?" + assert payload["language"] == "ru" + assert payload["model"] == module.MODEL_NAME + + +@pytest.mark.parametrize( + ("raw", "expected"), + [ + ("en", "en"), + ("RU", "ru"), + (" es ", "es"), + ("yue", "yue"), + ("fr", "fr"), + ("en-US", ""), + ("en_US", ""), + ("e", ""), + ("english", ""), + ("", ""), + ("../en", ""), + ("en\x00", ""), + ("ru; rm -rf /", ""), + ("рус", ""), + (None, ""), + (7, ""), + (["ru"], ""), + ({"language": "ru"}, ""), + ], +) +def test_stt_language_field_is_shape_validated(monkeypatch, raw, expected): + module = _load_stt_server(monkeypatch) + assert module._detected_language({"language": raw}) == expected + + +def test_stt_language_absent_when_whisper_omits_it(monkeypatch): + module = _load_stt_server(monkeypatch) + assert module._detected_language({}) == "" + assert module._detected_language("not a result") == "" + assert module._transcription_payload({"text": "hi"})["language"] == "" + + +# --------------------------------------------------------------------------- +# 2. The local-command client carries the language without breaking the +# .txt contract Hermes reads the transcript from +# --------------------------------------------------------------------------- + + +def test_stt_client_writes_language_sidecar_beside_the_txt_contract(tmp_path): + module = _load_stt_client() + module._write_result(tmp_path, "voice-input", "Как дела?", "ru") + assert (tmp_path / "voice-input.txt").read_text(encoding="utf-8") == "Как дела?" + assert (tmp_path / "voice-input.language").read_text(encoding="utf-8") == "ru" + # Hermes globs *.txt and reads the first match: the sidecar must not join it. + assert sorted(p.name for p in tmp_path.glob("*.txt")) == ["voice-input.txt"] + + +def test_stt_client_omits_the_sidecar_when_no_language_was_detected(tmp_path): + module = _load_stt_client() + module._write_result(tmp_path, "voice-input", "Hello.", "") + assert (tmp_path / "voice-input.txt").exists() + assert not (tmp_path / "voice-input.language").exists() + + +@pytest.mark.parametrize( + ("raw", "expected"), + [ + ("en", "en"), + ("ES", "es"), + (" ru ", "ru"), + ("en-US", ""), + ("", ""), + ("../../etc/passwd", ""), + ("en\n", "en"), + ("e", ""), + (None, ""), + (12, ""), + (["en"], ""), + ], +) +def test_stt_client_normalises_the_service_language_field(raw, expected): + module = _load_stt_client() + assert module._normalize_language(raw) == expected + + +# --------------------------------------------------------------------------- +# 3. The patched agent envelope and /api/transcribe response carry it through +# --------------------------------------------------------------------------- + + +def test_patched_local_command_envelope_carries_the_sidecar_language(patched_webui, tmp_path): + output = tmp_path / "stt-out" + output.mkdir() + (output / "voice-input.txt").write_text("Как дела?", encoding="utf-8") + (output / "voice-input.language").write_text("ru\n", encoding="utf-8") + result = patched_webui.transcription._transcribe_local_command( + "/tmp/voice-input.wav", "small", output + ) + assert result == { + "success": True, + "transcript": "Как дела?", + "provider": "local_command", + "language": "ru", + } + + +def test_patched_local_command_envelope_defaults_to_no_language(patched_webui, tmp_path): + output = tmp_path / "stt-out" + output.mkdir() + (output / "voice-input.txt").write_text("Hello.", encoding="utf-8") + result = patched_webui.transcription._transcribe_local_command( + "/tmp/voice-input.wav", "small", output + ) + assert result["transcript"] == "Hello." + assert result["language"] == "" + + +@pytest.mark.parametrize( + "hostile", + ["en-US", "../../en", "en; rm -rf /", "e", "english", "", "\x00en", "e n"], +) +def test_patched_local_command_envelope_rejects_malformed_sidecars( + patched_webui, tmp_path, hostile +): + output = tmp_path / "stt-out" + output.mkdir() + (output / "voice-input.txt").write_text("Hello.", encoding="utf-8") + (output / "voice-input.language").write_text(hostile, encoding="utf-8") + result = patched_webui.transcription._transcribe_local_command( + "/tmp/voice-input.wav", "small", output + ) + assert result["language"] == "" + + +def test_patched_local_command_envelope_survives_an_undecodable_sidecar( + patched_webui, tmp_path +): + """A corrupt sidecar must cost the language, never the transcript.""" + output = tmp_path / "stt-out" + output.mkdir() + (output / "voice-input.txt").write_text("Hello.", encoding="utf-8") + (output / "voice-input.language").write_bytes(b"\xff\xfe\x00ru") + result = patched_webui.transcription._transcribe_local_command( + "/tmp/voice-input.wav", "small", output + ) + assert result["success"] is True + assert result["transcript"] == "Hello." + assert result["language"] == "" + + +def test_patched_transcribe_response_reports_the_language(patched_webui): + response = patched_webui.upload.handle_transcribe( + None, {"success": True, "transcript": " Как дела? ", "language": "ru"} + ) + assert response["payload"] == {"ok": True, "transcript": "Как дела?", "language": "ru"} + + +@pytest.mark.parametrize( + "hostile", + ["", None, "en-US", "englishhh", "../en", 5, ["ru"], {"a": "b"}, "e"], +) +def test_patched_transcribe_response_blanks_untrusted_languages(patched_webui, hostile): + response = patched_webui.upload.handle_transcribe( + None, {"success": True, "transcript": "Hello.", "language": hostile} + ) + assert response["payload"]["language"] == "" + assert response["payload"]["transcript"] == "Hello." + + +# --------------------------------------------------------------------------- +# 4. The /api/tts trust boundary: allow-list only, and never `voice` +# --------------------------------------------------------------------------- + + +@pytest.mark.parametrize("language", SUPPORTED) +def test_atlas_tts_forwards_allow_listed_languages(patched_webui, monkeypatch, language): + body = _atlas_tts(patched_webui, monkeypatch, {"engine": "atlas", "language": language}) + assert body["language"] == language + assert body["model"] == "piper" + + +@pytest.mark.parametrize( + "hostile", + [ + "fr", + "de", + "", + None, + "EN-GB", + "../../ru_RU-irina-medium", + "ru; rm -rf /", + "ru\x00", + "ру", + 5, + ["ru"], + {"language": "ru"}, + True, + "x" * 8192, + ], +) +def test_atlas_tts_falls_back_to_english_for_untrusted_languages( + patched_webui, monkeypatch, hostile +): + body = _atlas_tts(patched_webui, monkeypatch, {"engine": "atlas", "language": hostile}) + assert body["language"] == "en" + + +def test_atlas_tts_language_is_absent_when_the_client_sends_none(patched_webui, monkeypatch): + body = _atlas_tts(patched_webui, monkeypatch, {"engine": "atlas"}) + assert body["language"] == "en" + + +def test_atlas_tts_voice_field_cannot_steer_synthesis(patched_webui, monkeypatch): + body = _atlas_tts( + patched_webui, + monkeypatch, + {"engine": "atlas", "voice": "ru_RU-irina-medium", "language": "en"}, + ) + assert body["language"] == "en" + body = _atlas_tts( + patched_webui, + monkeypatch, + {"engine": "atlas", "voice": "es_MX-claude-high"}, + ) + assert body["language"] == "en" + + +def test_atlas_tts_language_helper_only_ever_returns_a_baked_voice_language(patched_webui): + resolve = patched_webui.routes._atlas_tts_language + hostile = [ + None, 0, 1, -1, True, False, [], {}, set(), object(), b"ru", + "", " ", "\t\n", "en", "EN", "en-US", "en_us", "ru-RU", "es-MX", + "e", "eng", "english", "ru ru", "ru;es", "../ru", "ru\x00", "ру", + "x" * 65536, "en" * 4096, + ] + for value in hostile: + assert resolve({"language": value}) in SUPPORTED + for value in hostile: + assert resolve(value) in SUPPORTED + assert resolve({"voice": "ru_RU-irina-medium"}) == "en" + + +def test_manual_tts_button_body_still_carries_no_language(patched_webui): + """The read-aloud button has no trusted STT signal, so it must stay Amy.""" + ui = (patched_webui.webui / "static" / "ui.js").read_text(encoding="utf-8") + assert "engine:engineOverride||'edge'" in ui + assert "language" not in ui + + +# --------------------------------------------------------------------------- +# 5. Browser contract, driven through the real atlas-voice.js +# --------------------------------------------------------------------------- + + +@pytest.fixture(scope="module") +def voice_probe(): + node = shutil.which("node") + if not node: + pytest.skip("node is required to drive the browser voice-mode contract") + completed = subprocess.run( + [node, str(VOICE_PROBE), str(VOICE_SCRIPT)], + capture_output=True, + text=True, + timeout=180, + ) + assert completed.returncode == 0, completed.stderr + return json.loads(completed.stdout) + + +@pytest.mark.parametrize( + ("scenario", "expected"), + [ + ("english_turn_speaks_english", "en"), + ("russian_turn_speaks_russian", "ru"), + ("spanish_turn_speaks_spanish", "es"), + ], +) +def test_voice_mode_speaks_the_language_whisper_detected(voice_probe, scenario, expected): + requests = voice_probe[scenario]["tts"] + assert requests, "voice mode never reached /api/tts" + for request in requests: + assert request["engine"] == "atlas" + assert request["language"] == expected + + +@pytest.mark.parametrize( + "scenario", ["missing_language_falls_back", "unsupported_language_falls_back"] +) +def test_voice_mode_omits_language_without_a_trusted_signal(voice_probe, scenario): + requests = voice_probe[scenario]["tts"] + assert requests, "voice mode never reached /api/tts" + for request in requests: + assert "language" not in request + + +def test_voice_mode_drops_hostile_language_values(voice_probe): + for case in voice_probe["hostile_language_values_are_dropped"]["results"]: + for request in case["tts"]: + assert "language" not in request, case["sent"] + + +def test_voice_mode_never_sends_a_voice_field(voice_probe): + for request in voice_probe["voice_field_is_never_sent"]["tts"]: + assert set(request) <= {"text", "engine", "language"} + assert "voice" not in request + + +def test_voice_mode_does_not_reuse_a_previous_turn_language(voice_probe): + requests = voice_probe["language_does_not_leak_into_later_turn"]["tts"] + assert len(requests) == 3 + assert requests[0]["language"] == "ru" + assert "language" not in requests[1] + assert requests[2]["language"] == "es" + + +def test_voice_mode_ignores_language_from_an_empty_transcript(voice_probe): + result = voice_probe["empty_transcript_does_not_arm_a_language"] + assert result["sendsAfterBlank"] == [] + assert result["sends"] == ["Hello."] + assert result["tts"], "the follow-up turn should still be spoken" + for request in result["tts"]: + assert "language" not in request + + +def test_voice_mode_discards_language_when_the_session_changes(voice_probe): + result = voice_probe["session_change_discards_language"] + assert result["afterSwitch"] == [] + for request in result["tts"]: + assert "language" not in request + + +def test_voice_mode_discards_language_when_voice_mode_is_turned_off(voice_probe): + result = voice_probe["deactivation_discards_language"] + assert result["afterDeactivate"] == [] + for request in result["tts"]: + assert "language" not in request + + +def test_voice_mode_speaks_nothing_when_transcription_fails(voice_probe): + result = voice_probe["transcribe_error_speaks_nothing"] + assert result["tts"] == [] + assert any("Whisper is down" in toast for toast in result["toasts"]) + + +# --------------------------------------------------------------------------- +# 6. Build-time enforcement and documented semantics +# --------------------------------------------------------------------------- + + +def test_image_build_verifies_every_language_routing_patch(): + dockerfile = (DOCKERFILES / "Dockerfile.hermes-webui").read_text(encoding="utf-8") + assert "'language': detected" in dockerfile + assert "def _atlas_tts_language(body):" in dockerfile + assert '"language": _atlas_tts_language(data)' in dockerfile + assert '"language": detected_language' in dockerfile + assert "takeSttLanguage(token)" in dockerfile + assert "/opt/hermes-webui/api/upload.py" in dockerfile + assert "/opt/hermes/tools/transcription_tools.py" in dockerfile + + +def test_atlas_patch_roots_are_overridable_for_offline_verification(): + patch = ATLAS_PATCH.read_text(encoding="utf-8") + assert 'os.environ.get("HERMES_WEBUI_PATCH_ROOT", "/opt/hermes-webui")' in patch + assert 'os.environ.get("HERMES_AGENT_PATCH_ROOT", "/opt/hermes")' in patch + + +def test_notes_document_the_stt_driven_voice_selection_and_its_limits(): + notes = (HERMES / "NOTES.md").read_text(encoding="utf-8") + assert "STT-detected language" in notes + for marker in ("hands-free", "Typed messages", "en_US-amy-medium"): + assert marker in notes