#!/usr/bin/env python3 """Apply fail-closed Atlas voice integration patches to pinned Hermes WebUI.""" import os from pathlib import Path ROOT = Path(os.environ.get("HERMES_WEBUI_PATCH_ROOT", "/opt/hermes-webui")) # The WebUI imports the pinned agent's STT tooling from the same image, so the # local-command transcription envelope is patched alongside the WebUI itself. AGENT_ROOT = Path(os.environ.get("HERMES_AGENT_PATCH_ROOT", "/opt/hermes")) def replace_exact(path: Path, before: str, after: str, count: int = 1) -> None: """Replace an exact upstream fragment and fail when the pin has drifted.""" source = path.read_text(encoding="utf-8") if source.count(before) != count: raise SystemExit(f"Atlas voice patch context changed in {path}: {before[:80]!r}") path.write_text(source.replace(before, after, count), encoding="utf-8") def replace_between_exact( path: Path, start: str, end: str, after: str = "", count: int = 1 ) -> None: """Replace one exact, bounded upstream region and fail when the pin drifts.""" source = path.read_text(encoding="utf-8") if source.count(start) != count or source.count(end) != count: raise SystemExit( f"Atlas voice patch context changed in {path}: {start[:80]!r}" ) start_index = source.index(start) end_index = source.index(end, start_index) + len(end) path.write_text( source[:start_index] + after + source[end_index:], encoding="utf-8" ) def assert_absent(path: Path, *needles: str) -> None: """Fail the image build if a removed voice-choice surface remains.""" source = path.read_text(encoding="utf-8") remaining = [needle for needle in needles if needle in source] if remaining: raise SystemExit(f"Atlas voice choice remains in {path}: {remaining!r}") def remove_lines_containing(path: Path, *needles: str) -> None: """Remove all pinned translation entries for a retired settings control.""" source = path.read_text(encoding="utf-8") for needle in needles: if needle not in source: raise SystemExit(f"Atlas voice patch context changed in {path}: {needle!r}") lines = source.splitlines(keepends=True) path.write_text( "".join(line for line in lines if not any(n in line for n in needles)), encoding="utf-8", ) index = ROOT / "static/index.html" replace_exact( index, '', '\n' '', ) replace_exact( index, '', '', ) replace_exact( index, '''
Preferred voice. Populated from your browser's available voices.
''', "", ) replace_exact( index, '', '\n', ) replace_exact( index, ''' ''', ''' ''', ) ui = ROOT / "static/ui.js" replace_exact( ui, ''' const savedVoice=localStorage.getItem('hermes-tts-voice'); const voices=speechSynthesis.getVoices(); if(savedVoice&&voices.length){ const match=voices.find(v=>v.name===savedVoice); if(match) utter.voice=match; } ''', "", ) replace_exact(ui, "function _playEdgeTtsChunked(text, btn){", "function _playEdgeTtsChunked(text, btn, engineOverride){") replace_exact( ui, " const voice=localStorage.getItem('hermes-tts-voice')||'zh-CN-XiaoxiaoNeural';\n", "", ) replace_exact( ui, "body:JSON.stringify({text:chunk, voice:voice, rate:rate, pitch:pitch})", "body:JSON.stringify({text:chunk, rate:rate, pitch:pitch, engine:engineOverride||'edge'})", ) replace_exact( ui, " voice: localStorage.getItem('hermes-tts-voice')||'',\n", "", count=2, ) replace_exact( ui, "if(engine==='edge'){\n _playEdgeTtsChunked(clean, btn);", "if(engine==='edge'||engine==='atlas'){\n _playEdgeTtsChunked(clean, btn, engine);", ) replace_exact( ui, "if(engine==='edge'){\n _playEdgeTtsChunked(clean, null);", "if(engine==='edge'||engine==='atlas'){\n _playEdgeTtsChunked(clean, null, engine);", ) panels = ROOT / "static/panels.js" replace_exact(panels, " tts_voice:'hermes-tts-voice',\n", "") replace_exact( panels, ''' const ttsVoiceSel=$('settingsTtsVoice'); if(ttsVoiceSel) _setOwnedSpeechPayload(payload,'tts_voice',ttsVoiceSel.value||''); ''', "", ) replace_exact( panels, ''' localStorage.setItem('hermes-tts-engine',this.value); window._populateTtsVoices(); _schedulePreferencesAutosave();''', ''' localStorage.setItem('hermes-tts-engine',this.value); _schedulePreferencesAutosave();''', ) replace_between_exact( panels, " // Populate voice selector based on engine\n", " // TTS rate/pitch sliders\n", " // TTS speaker selection is intentionally server policy only.\n", ) replace_exact( panels, "let _settingsSpeechChangedKeys=new Set();\n", "let _settingsSpeechChangedKeys=new Set();\n" "try{localStorage.removeItem('hermes-tts-voice');}catch(_){}\n", ) boot = ROOT / "static/boot.js" replace_exact( boot, ''' voice: localStorage.getItem("hermes-tts-voice")||'', ''', "", ) replace_exact( boot, ''' const voice=localStorage.getItem("hermes-tts-voice")||"zh-CN-XiaoxiaoNeural"; ''', "", ) replace_exact( boot, " body: JSON.stringify({text: clean, voice, rate, pitch})", " body: JSON.stringify({text: clean, rate, pitch})", ) replace_exact( boot, ''' const savedVoice=localStorage.getItem('hermes-tts-voice'); const voices=speechSynthesis.getVoices(); if(savedVoice&&voices.length){ const match=voices.find(v=>v.name===savedVoice); if(match) utter.voice=match; } ''', "", ) replace_exact(boot, " tts_voice:'',\n", "") replace_exact(boot, " ['tts_voice','hermes-tts-voice'],\n", "") config = ROOT / "api/config.py" replace_exact(config, ' "tts_voice": "",\n', "") replace_exact(config, ' "tts_voice",\n', "") replace_exact( config, ''' if k == "tts_voice": if not isinstance(v, str) or len(v) > 200 or "\\x00" in v: continue ''', "", ) assert_absent(index, "settingsTtsVoice", "settings_label_tts_voice") assert_absent(ui, "hermes-tts-voice", "voice:voice") assert_absent(panels, "settingsTtsVoice", "tts_voice") assert_absent(boot, "hermes-tts-voice", "tts_voice", "text: clean, voice") assert_absent(config, '"tts_voice"') i18n = ROOT / "static/i18n.js" remove_lines_containing( i18n, "settings_label_tts_voice:", "settings_desc_tts_voice:", ) assert_absent(i18n, "settings_label_tts_voice", "settings_desc_tts_voice") # The private Whisper service reports the language it decoded with. Carry that # through the agent's local-command STT envelope so the WebUI can hand a voice # hint to Piper instead of guessing the reply's language from its text. transcription = AGENT_ROOT / "tools/transcription_tools.py" replace_exact( transcription, ''' transcript_text = txt_files[0].read_text(encoding="utf-8").strip() logger.info( "Transcribed %s via local STT command (%s, %d chars)", Path(file_path).name, normalized_model, len(transcript_text), ) return {"success": True, "transcript": transcript_text, "provider": "local_command"} ''', ''' transcript_text = txt_files[0].read_text(encoding="utf-8").strip() logger.info( "Transcribed %s via local STT command (%s, %d chars)", Path(file_path).name, normalized_model, len(transcript_text), ) detected_language = "" language_files = sorted(Path(output_dir).glob("*.language")) if language_files: try: candidate = language_files[0].read_text(encoding="utf-8").strip().lower() except (OSError, ValueError): candidate = "" if 2 <= len(candidate) <= 3 and candidate.isascii() and candidate.isalpha(): detected_language = candidate return { "success": True, "transcript": transcript_text, "provider": "local_command", "language": detected_language, } ''', ) upload = ROOT / "api/upload.py" replace_exact( upload, """ transcript = str(result.get('transcript') or '').strip() return j(handler, {'ok': True, 'transcript': transcript}) """, """ transcript = str(result.get('transcript') or '').strip() detected = str(result.get('language') or '').strip().lower() if not (2 <= len(detected) <= 3 and detected.isascii() and detected.isalpha()): detected = '' return j(handler, {'ok': True, 'transcript': transcript, 'language': detected}) """, ) routes = ROOT / "api/routes.py" replace_exact( routes, "def _tts_open(req, *, timeout=30, opener_factory=None):", '''ATLAS_TTS_LANGUAGES = ("en", "ru", "es") def _atlas_tts_language(body): """Return a plain, allow-listed en/ru/es code, or "" to send no language. This is a trust boundary, not a parser. Only the exact normalized codes the private Piper deployment bakes a voice for are forwarded; a missing field, a wrong type, a region tag, padding, control characters, a traversal or injection string, an oversized value, an object, an array, a number or a client-supplied "voice" all resolve to "" and the language field is then omitted entirely, so the Jetson service applies its own English default. Coercing a malformed value into a supported code would let a browser describe hostile input as a language we support; omission cannot. """ if not isinstance(body, dict): return "" value = body.get("language") if not isinstance(value, str): return "" return value if value in ATLAS_TTS_LANGUAGES else "" def _tts_open(req, *, timeout=30, opener_factory=None):''', ) marker = " # ── ElevenLabs TTS ──────────────────────────────────────────────────\n" atlas = ''' # ── Atlas private Jetson TTS ───────────────────────────────────────── if engine == "atlas": atlas_url = os.getenv("HERMES_WEBUI_ATLAS_TTS_URL", "").strip() expected_url = "http://hermes-tts.hermes.svc.cluster.local:9001/v1/audio/speech" if atlas_url != expected_url: from api.helpers import bad as _bad return _bad(handler, "Atlas private TTS is not configured", 503) speed = 1.0 if rate_str: try: speed = max(0.5, min(2.0, 1.0 + (float(rate_str.rstrip("%")) / 100.0))) except ValueError: speed = 1.0 request_payload = { "model": "piper", "input": text, "speed": speed, } # Attach a language ONLY when the browser sent a plain allow-listed # code. Omitting it is the fail-safe: the Jetson service then speaks # its own English default, which is also what every partially rolled # out combination of these components degrades to. _atlas_language = _atlas_tts_language(data) if _atlas_language: request_payload["language"] = _atlas_language request_body = json.dumps(request_payload).encode("utf-8") request = Request(atlas_url, data=request_body, headers={ "Content-Type": "application/json", "Accept": "audio/wav", }) try: with _tts_open( request, timeout=45, opener_factory=lambda: build_opener(ProxyHandler({}), _NoRedirectTtsHandler()), ) as response: audio_data = _buffer_tts_audio_response(response) except Exception: logger.exception("Atlas private TTS generation failed") from api.helpers import bad as _bad return _bad(handler, "Atlas private TTS generation failed", 502) handler.send_response(200) handler.send_header("Content-Type", "audio/wav") handler.send_header("Cache-Control", "no-store") handler.send_header("Content-Length", str(len(audio_data))) handler.end_headers() try: handler.wfile.write(audio_data) except (BrokenPipeError, ConnectionResetError): pass return True ''' replace_exact(routes, marker, atlas + marker)