2026-08-21 11:52:49 +00:00
|
|
|
|
"""Voice-mode language routing: private Whisper STT decides the Piper voice.
|
|
|
|
|
|
|
|
|
|
|
|
Every assertion here runs without a GPU, a microphone or a cluster. The browser
|
|
|
|
|
|
contract is exercised by driving the real ``atlas-voice.js`` inside a stub DOM
|
|
|
|
|
|
(``testing/tests/data/atlas_voice_language_probe.js``), and the two server-side
|
|
|
|
|
|
trust boundaries are exercised by applying the real image patch to fixtures that
|
|
|
|
|
|
carry the exact upstream anchors and then importing the patched result.
|
|
|
|
|
|
"""
|
|
|
|
|
|
|
|
|
|
|
|
from __future__ import annotations
|
|
|
|
|
|
|
|
|
|
|
|
import importlib.util
|
|
|
|
|
|
import json
|
|
|
|
|
|
import os
|
|
|
|
|
|
import shutil
|
|
|
|
|
|
import subprocess
|
|
|
|
|
|
import sys
|
|
|
|
|
|
from pathlib import Path
|
|
|
|
|
|
from types import SimpleNamespace
|
|
|
|
|
|
|
|
|
|
|
|
import pytest
|
|
|
|
|
|
|
|
|
|
|
|
from testing.tests.test_hermes_chat_support import HERMES, ROOT
|
|
|
|
|
|
|
|
|
|
|
|
DOCKERFILES = ROOT / "dockerfiles"
|
|
|
|
|
|
ATLAS_PATCH = DOCKERFILES / "hermes-webui-atlas-patch.py"
|
|
|
|
|
|
VOICE_SCRIPT = DOCKERFILES / "hermes-webui-atlas-voice.js"
|
|
|
|
|
|
VOICE_PROBE = ROOT / "testing" / "tests" / "data" / "atlas_voice_language_probe.js"
|
|
|
|
|
|
WEBUI_FIXTURE = ROOT / "testing" / "fixtures" / "hermes-webui-0.52.181"
|
|
|
|
|
|
ATLAS_TTS_URL = "http://hermes-tts.hermes.svc.cluster.local:9001/v1/audio/speech"
|
|
|
|
|
|
|
|
|
|
|
|
# Voices baked by the multilingual Piper work (PR #26): en=amy, ru=irina,
|
|
|
|
|
|
# es=claude. Anything outside this set must resolve to English.
|
|
|
|
|
|
SUPPORTED = ("en", "ru", "es")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
# Module loaders
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _load_stt_server(monkeypatch):
|
|
|
|
|
|
"""Import the Jetson Whisper service without CUDA, torch or whisper."""
|
|
|
|
|
|
path = DOCKERFILES / "hermes-jetson-stt-server.py"
|
|
|
|
|
|
spec = importlib.util.spec_from_file_location("hermes_jetson_stt_server", path)
|
|
|
|
|
|
assert spec and spec.loader
|
|
|
|
|
|
module = importlib.util.module_from_spec(spec)
|
|
|
|
|
|
monkeypatch.setitem(sys.modules, "cgi", SimpleNamespace())
|
|
|
|
|
|
monkeypatch.setitem(
|
|
|
|
|
|
sys.modules,
|
|
|
|
|
|
"torch",
|
|
|
|
|
|
SimpleNamespace(cuda=SimpleNamespace(is_available=lambda: False)),
|
|
|
|
|
|
)
|
|
|
|
|
|
monkeypatch.setitem(sys.modules, "whisper", SimpleNamespace())
|
|
|
|
|
|
spec.loader.exec_module(module)
|
|
|
|
|
|
return module
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _load_stt_client():
|
|
|
|
|
|
"""Import the local-command STT client that Hermes shells out to."""
|
|
|
|
|
|
path = HERMES / "scripts" / "hermes_stt_client.py"
|
|
|
|
|
|
spec = importlib.util.spec_from_file_location("hermes_stt_client", path)
|
|
|
|
|
|
assert spec and spec.loader
|
|
|
|
|
|
module = importlib.util.module_from_spec(spec)
|
|
|
|
|
|
spec.loader.exec_module(module)
|
|
|
|
|
|
return module
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
# Patch fixtures — each file carries the exact upstream fragment the image
|
|
|
|
|
|
# patch pins, so importing the patched result exercises the inserted code.
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
INDEX_FIXTURE = (
|
|
|
|
|
|
'<select id="settingsTtsEngine">'
|
|
|
|
|
|
'<option value="browser">Browser speech synthesis</option>'
|
|
|
|
|
|
'<option value="edge">Edge TTS (server)</option></select>\n'
|
|
|
|
|
|
'<script src="static/boot.js?v=__WEBUI_VERSION__" defer></script>\n'
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
UI_FIXTURE = """function _playEdgeTtsChunked(text, btn){
|
|
|
|
|
|
fetch('/api/tts',{method:'POST',body:JSON.stringify({text:chunk, voice:voice, rate:rate, pitch:pitch})});
|
|
|
|
|
|
}
|
|
|
|
|
|
function readAloud(clean, btn, engine){
|
|
|
|
|
|
if(engine==='edge'){
|
|
|
|
|
|
_playEdgeTtsChunked(clean, btn);
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
function autoRead(clean, engine){
|
|
|
|
|
|
if(engine==='edge'){
|
|
|
|
|
|
_playEdgeTtsChunked(clean, null);
|
|
|
|
|
|
}
|
|
|
|
|
|
}
|
|
|
|
|
|
"""
|
|
|
|
|
|
|
|
|
|
|
|
HELPERS_FIXTURE = '''"""Stand-in for the WebUI helper module the patched code imports."""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def bad(handler, message, status=400):
|
|
|
|
|
|
return {"status": status, "error": message}
|
|
|
|
|
|
'''
|
|
|
|
|
|
|
|
|
|
|
|
ROUTES_FIXTURE = '''"""Stand-in carrying the exact upstream anchors the Atlas TTS patch pins."""
|
|
|
|
|
|
|
2026-08-23 18:29:02 -03:00
|
|
|
|
import html as _html
|
2026-08-21 11:52:49 +00:00
|
|
|
|
import json
|
|
|
|
|
|
import os
|
2026-08-23 18:29:02 -03:00
|
|
|
|
import re
|
2026-08-21 11:52:49 +00:00
|
|
|
|
from urllib.request import ProxyHandler, Request, build_opener
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
class _NoRedirectTtsHandler:
|
|
|
|
|
|
"""Placeholder for the upstream no-redirect opener handler."""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
class _Logger:
|
|
|
|
|
|
def __init__(self):
|
|
|
|
|
|
self.failures = []
|
|
|
|
|
|
|
|
|
|
|
|
def exception(self, message):
|
|
|
|
|
|
self.failures.append(message)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
logger = _Logger()
|
|
|
|
|
|
UPSTREAM_REQUESTS = []
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
class _Upstream:
|
|
|
|
|
|
def __init__(self, payload):
|
|
|
|
|
|
self._payload = payload
|
|
|
|
|
|
|
|
|
|
|
|
def read(self):
|
|
|
|
|
|
return self._payload
|
|
|
|
|
|
|
|
|
|
|
|
def __enter__(self):
|
|
|
|
|
|
return self
|
|
|
|
|
|
|
|
|
|
|
|
def __exit__(self, *exc_info):
|
|
|
|
|
|
return False
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _buffer_tts_audio_response(response):
|
|
|
|
|
|
return response.read()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _tts_open(req, *, timeout=30, opener_factory=None):
|
|
|
|
|
|
"""Thin network seam for the TTS upstream fetch so tests can intercept it."""
|
|
|
|
|
|
UPSTREAM_REQUESTS.append(json.loads(req.data.decode("utf-8")))
|
|
|
|
|
|
return _Upstream(b"RIFFsynthetic")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _handle_tts(handler, data, text, rate_str, engine):
|
|
|
|
|
|
# ── ElevenLabs TTS ──────────────────────────────────────────────────
|
|
|
|
|
|
return None
|
2026-08-23 18:29:02 -03:00
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def handle_get(handler, parsed) -> bool:
|
|
|
|
|
|
"""Handle all GET routes. Returns True if handled, False for 404."""
|
|
|
|
|
|
return False
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def handle_post(handler, parsed) -> bool:
|
|
|
|
|
|
"""Pinned POST route anchors for the Atlas voice patch."""
|
|
|
|
|
|
if parsed.path == "/api/transcribe":
|
|
|
|
|
|
return handle_transcribe(handler)
|
|
|
|
|
|
|
|
|
|
|
|
if parsed.path == "/api/tts":
|
|
|
|
|
|
return _handle_tts(handler, parsed)
|
|
|
|
|
|
return False
|
2026-08-21 11:52:49 +00:00
|
|
|
|
'''
|
|
|
|
|
|
|
|
|
|
|
|
UPLOAD_FIXTURE = '''"""Stand-in carrying the exact upstream /api/transcribe response anchor."""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def j(handler, payload, status=200):
|
|
|
|
|
|
return {"status": status, "payload": payload}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def handle_transcribe(handler, result):
|
|
|
|
|
|
try:
|
|
|
|
|
|
transcript = str(result.get('transcript') or '').strip()
|
|
|
|
|
|
return j(handler, {'ok': True, 'transcript': transcript})
|
|
|
|
|
|
except ValueError as error:
|
|
|
|
|
|
return j(handler, {'error': str(error)}, status=400)
|
|
|
|
|
|
'''
|
|
|
|
|
|
|
|
|
|
|
|
TRANSCRIPTION_FIXTURE = '''"""Stand-in carrying the exact upstream local-command STT envelope anchor."""
|
|
|
|
|
|
|
|
|
|
|
|
import contextlib
|
|
|
|
|
|
from pathlib import Path
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
class _Logger:
|
|
|
|
|
|
def info(self, *args):
|
|
|
|
|
|
return None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
logger = _Logger()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _transcribe_local_command(file_path, normalized_model, output_dir):
|
|
|
|
|
|
try:
|
|
|
|
|
|
with contextlib.nullcontext(output_dir):
|
|
|
|
|
|
txt_files = sorted(Path(output_dir).glob("*.txt"))
|
|
|
|
|
|
transcript_text = txt_files[0].read_text(encoding="utf-8").strip()
|
|
|
|
|
|
logger.info(
|
|
|
|
|
|
"Transcribed %s via local STT command (%s, %d chars)",
|
|
|
|
|
|
Path(file_path).name,
|
|
|
|
|
|
normalized_model,
|
|
|
|
|
|
len(transcript_text),
|
|
|
|
|
|
)
|
|
|
|
|
|
return {"success": True, "transcript": transcript_text, "provider": "local_command"}
|
|
|
|
|
|
except OSError as error:
|
|
|
|
|
|
return {"success": False, "transcript": "", "error": str(error)}
|
|
|
|
|
|
'''
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _write(path: Path, text: str) -> None:
|
|
|
|
|
|
path.parent.mkdir(parents=True, exist_ok=True)
|
|
|
|
|
|
path.write_text(text, encoding="utf-8")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.fixture
|
|
|
|
|
|
def patched_webui(tmp_path, monkeypatch):
|
|
|
|
|
|
"""Apply the real Atlas image patch to fixture trees and import the result."""
|
|
|
|
|
|
webui = tmp_path / "hermes-webui"
|
|
|
|
|
|
agent = tmp_path / "hermes"
|
|
|
|
|
|
# Start with the pinned full-surface fixture introduced by PR #39 so this
|
|
|
|
|
|
# test proves the language pipeline composes with its voice-selector removal
|
|
|
|
|
|
# and conversation instrument, not merely with the older #27 anchors.
|
|
|
|
|
|
shutil.copytree(WEBUI_FIXTURE, webui)
|
|
|
|
|
|
_write(webui / "api" / "__init__.py", "")
|
|
|
|
|
|
_write(webui / "api" / "helpers.py", HELPERS_FIXTURE)
|
|
|
|
|
|
_write(webui / "api" / "routes.py", ROUTES_FIXTURE)
|
|
|
|
|
|
_write(webui / "api" / "upload.py", UPLOAD_FIXTURE)
|
|
|
|
|
|
_write(agent / "tools" / "__init__.py", "")
|
|
|
|
|
|
_write(agent / "tools" / "transcription_tools.py", TRANSCRIPTION_FIXTURE)
|
|
|
|
|
|
|
|
|
|
|
|
environment = dict(os.environ)
|
|
|
|
|
|
environment["HERMES_WEBUI_PATCH_ROOT"] = str(webui)
|
|
|
|
|
|
environment["HERMES_AGENT_PATCH_ROOT"] = str(agent)
|
|
|
|
|
|
completed = subprocess.run(
|
|
|
|
|
|
[sys.executable, str(ATLAS_PATCH)],
|
|
|
|
|
|
env=environment,
|
|
|
|
|
|
capture_output=True,
|
|
|
|
|
|
text=True,
|
|
|
|
|
|
)
|
|
|
|
|
|
assert completed.returncode == 0, completed.stderr or completed.stdout
|
|
|
|
|
|
|
|
|
|
|
|
for name in ("api", "api.helpers", "api.routes", "api.upload", "tools",
|
|
|
|
|
|
"tools.transcription_tools"):
|
|
|
|
|
|
sys.modules.pop(name, None)
|
|
|
|
|
|
monkeypatch.syspath_prepend(str(agent))
|
|
|
|
|
|
monkeypatch.syspath_prepend(str(webui))
|
|
|
|
|
|
import api.routes as routes # noqa: PLC0415
|
|
|
|
|
|
import api.upload as upload # noqa: PLC0415
|
|
|
|
|
|
import tools.transcription_tools as transcription # noqa: PLC0415
|
|
|
|
|
|
|
|
|
|
|
|
yield SimpleNamespace(
|
|
|
|
|
|
webui=webui,
|
|
|
|
|
|
agent=agent,
|
|
|
|
|
|
routes=routes,
|
|
|
|
|
|
upload=upload,
|
|
|
|
|
|
transcription=transcription,
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
for name in ("api", "api.helpers", "api.routes", "api.upload", "tools",
|
|
|
|
|
|
"tools.transcription_tools"):
|
|
|
|
|
|
sys.modules.pop(name, None)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
class _Handler:
|
|
|
|
|
|
"""Just enough BaseHTTPRequestHandler surface for the Atlas TTS branch."""
|
|
|
|
|
|
|
|
|
|
|
|
def __init__(self):
|
|
|
|
|
|
self.status = None
|
|
|
|
|
|
self.headers_sent = {}
|
|
|
|
|
|
self.wfile = SimpleNamespace(write=self._write)
|
|
|
|
|
|
self.body = b""
|
|
|
|
|
|
|
|
|
|
|
|
def send_response(self, status):
|
|
|
|
|
|
self.status = status
|
|
|
|
|
|
|
|
|
|
|
|
def send_header(self, name, value):
|
|
|
|
|
|
self.headers_sent[name] = value
|
|
|
|
|
|
|
|
|
|
|
|
def end_headers(self):
|
|
|
|
|
|
return None
|
|
|
|
|
|
|
|
|
|
|
|
def _write(self, payload):
|
|
|
|
|
|
self.body += payload
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _atlas_tts(patched, monkeypatch, data):
|
|
|
|
|
|
"""Run the patched Atlas branch and return the JSON it sent to hermes-tts."""
|
|
|
|
|
|
monkeypatch.setenv("HERMES_WEBUI_ATLAS_TTS_URL", ATLAS_TTS_URL)
|
|
|
|
|
|
patched.routes.UPSTREAM_REQUESTS.clear()
|
|
|
|
|
|
handler = _Handler()
|
|
|
|
|
|
result = patched.routes._handle_tts(handler, data, "Some reply.", "", "atlas")
|
|
|
|
|
|
assert result is True, "the Atlas branch must own the response"
|
|
|
|
|
|
assert handler.status == 200
|
|
|
|
|
|
assert len(patched.routes.UPSTREAM_REQUESTS) == 1
|
|
|
|
|
|
return patched.routes.UPSTREAM_REQUESTS[0]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
# 1. Whisper service reports the language it actually decoded with
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_stt_response_carries_whisper_detected_language(monkeypatch):
|
|
|
|
|
|
module = _load_stt_server(monkeypatch)
|
|
|
|
|
|
payload = module._transcription_payload(
|
|
|
|
|
|
{
|
|
|
|
|
|
"language": "ru",
|
|
|
|
|
|
"segments": [
|
|
|
|
|
|
{"text": " Как дела?", "no_speech_prob": 0.1, "avg_logprob": -0.2}
|
|
|
|
|
|
],
|
|
|
|
|
|
}
|
|
|
|
|
|
)
|
|
|
|
|
|
assert payload["text"] == "Как дела?"
|
|
|
|
|
|
assert payload["language"] == "ru"
|
|
|
|
|
|
assert payload["model"] == module.MODEL_NAME
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
|
|
|
|
("raw", "expected"),
|
|
|
|
|
|
[
|
|
|
|
|
|
("en", "en"),
|
|
|
|
|
|
("RU", "ru"),
|
|
|
|
|
|
(" es ", "es"),
|
|
|
|
|
|
("yue", "yue"),
|
|
|
|
|
|
("fr", "fr"),
|
|
|
|
|
|
("en-US", ""),
|
|
|
|
|
|
("en_US", ""),
|
|
|
|
|
|
("e", ""),
|
|
|
|
|
|
("english", ""),
|
|
|
|
|
|
("", ""),
|
|
|
|
|
|
("../en", ""),
|
|
|
|
|
|
("en\x00", ""),
|
|
|
|
|
|
("ru; rm -rf /", ""),
|
|
|
|
|
|
("рус", ""),
|
|
|
|
|
|
(None, ""),
|
|
|
|
|
|
(7, ""),
|
|
|
|
|
|
(["ru"], ""),
|
|
|
|
|
|
({"language": "ru"}, ""),
|
|
|
|
|
|
],
|
|
|
|
|
|
)
|
|
|
|
|
|
def test_stt_language_field_is_shape_validated(monkeypatch, raw, expected):
|
|
|
|
|
|
module = _load_stt_server(monkeypatch)
|
|
|
|
|
|
assert module._detected_language({"language": raw}) == expected
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_stt_language_absent_when_whisper_omits_it(monkeypatch):
|
|
|
|
|
|
module = _load_stt_server(monkeypatch)
|
|
|
|
|
|
assert module._detected_language({}) == ""
|
|
|
|
|
|
assert module._detected_language("not a result") == ""
|
|
|
|
|
|
assert module._transcription_payload({"text": "hi"})["language"] == ""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
# 2. The local-command client carries the language without breaking the
|
|
|
|
|
|
# .txt contract Hermes reads the transcript from
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-08-24 20:16:57 -03:00
|
|
|
|
def test_stt_client_retries_a_brief_outage_then_fails_cleanly(monkeypatch, tmp_path):
|
|
|
|
|
|
"""A momentary connection refusal (service roll) retries; a persistent
|
|
|
|
|
|
outage raises a plain RuntimeError, never a urllib traceback."""
|
|
|
|
|
|
module = _load_stt_client()
|
|
|
|
|
|
monkeypatch.setattr(module.time, "sleep", lambda *_a, **_k: None)
|
|
|
|
|
|
|
|
|
|
|
|
calls = {"n": 0}
|
|
|
|
|
|
|
|
|
|
|
|
class _Resp:
|
|
|
|
|
|
def __enter__(self):
|
|
|
|
|
|
return self
|
|
|
|
|
|
|
|
|
|
|
|
def __exit__(self, *a):
|
|
|
|
|
|
return False
|
|
|
|
|
|
|
|
|
|
|
|
def read(self):
|
|
|
|
|
|
return b'{"text": "recovered", "language": "en"}'
|
|
|
|
|
|
|
|
|
|
|
|
def flaky(_request, timeout=None):
|
|
|
|
|
|
calls["n"] += 1
|
|
|
|
|
|
if calls["n"] < 3:
|
|
|
|
|
|
raise ConnectionRefusedError(111, "Connection refused")
|
|
|
|
|
|
return _Resp()
|
|
|
|
|
|
|
|
|
|
|
|
monkeypatch.setattr(module, "urlopen", flaky)
|
|
|
|
|
|
req = module.Request("http://stt.invalid/", data=b"", method="POST")
|
|
|
|
|
|
assert module._transcribe_with_retry(req)["text"] == "recovered"
|
|
|
|
|
|
assert calls["n"] == 3
|
|
|
|
|
|
|
|
|
|
|
|
def always_refused(_request, timeout=None):
|
|
|
|
|
|
raise ConnectionRefusedError(111, "Connection refused")
|
|
|
|
|
|
|
|
|
|
|
|
monkeypatch.setattr(module, "urlopen", always_refused)
|
|
|
|
|
|
import pytest as _pytest
|
|
|
|
|
|
|
|
|
|
|
|
with _pytest.raises(RuntimeError, match="temporarily unavailable"):
|
|
|
|
|
|
module._transcribe_with_retry(req)
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-08-21 11:52:49 +00:00
|
|
|
|
def test_stt_client_writes_language_sidecar_beside_the_txt_contract(tmp_path):
|
|
|
|
|
|
module = _load_stt_client()
|
|
|
|
|
|
module._write_result(tmp_path, "voice-input", "Как дела?", "ru")
|
|
|
|
|
|
assert (tmp_path / "voice-input.txt").read_text(encoding="utf-8") == "Как дела?"
|
|
|
|
|
|
assert (tmp_path / "voice-input.language").read_text(encoding="utf-8") == "ru"
|
|
|
|
|
|
# Hermes globs *.txt and reads the first match: the sidecar must not join it.
|
|
|
|
|
|
assert sorted(p.name for p in tmp_path.glob("*.txt")) == ["voice-input.txt"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_stt_client_omits_the_sidecar_when_no_language_was_detected(tmp_path):
|
|
|
|
|
|
module = _load_stt_client()
|
|
|
|
|
|
module._write_result(tmp_path, "voice-input", "Hello.", "")
|
|
|
|
|
|
assert (tmp_path / "voice-input.txt").exists()
|
|
|
|
|
|
assert not (tmp_path / "voice-input.language").exists()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
|
|
|
|
("raw", "expected"),
|
|
|
|
|
|
[
|
|
|
|
|
|
("en", "en"),
|
|
|
|
|
|
("ES", "es"),
|
|
|
|
|
|
(" ru ", "ru"),
|
|
|
|
|
|
("en-US", ""),
|
|
|
|
|
|
("", ""),
|
|
|
|
|
|
("../../etc/passwd", ""),
|
|
|
|
|
|
("en\n", "en"),
|
|
|
|
|
|
("e", ""),
|
|
|
|
|
|
(None, ""),
|
|
|
|
|
|
(12, ""),
|
|
|
|
|
|
(["en"], ""),
|
|
|
|
|
|
],
|
|
|
|
|
|
)
|
|
|
|
|
|
def test_stt_client_normalises_the_service_language_field(raw, expected):
|
|
|
|
|
|
module = _load_stt_client()
|
|
|
|
|
|
assert module._normalize_language(raw) == expected
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
# 3. The patched agent envelope and /api/transcribe response carry it through
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_patched_local_command_envelope_carries_the_sidecar_language(patched_webui, tmp_path):
|
|
|
|
|
|
output = tmp_path / "stt-out"
|
|
|
|
|
|
output.mkdir()
|
|
|
|
|
|
(output / "voice-input.txt").write_text("Как дела?", encoding="utf-8")
|
|
|
|
|
|
(output / "voice-input.language").write_text("ru\n", encoding="utf-8")
|
|
|
|
|
|
result = patched_webui.transcription._transcribe_local_command(
|
|
|
|
|
|
"/tmp/voice-input.wav", "small", output
|
|
|
|
|
|
)
|
|
|
|
|
|
assert result == {
|
|
|
|
|
|
"success": True,
|
|
|
|
|
|
"transcript": "Как дела?",
|
|
|
|
|
|
"provider": "local_command",
|
|
|
|
|
|
"language": "ru",
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_patched_local_command_envelope_defaults_to_no_language(patched_webui, tmp_path):
|
|
|
|
|
|
output = tmp_path / "stt-out"
|
|
|
|
|
|
output.mkdir()
|
|
|
|
|
|
(output / "voice-input.txt").write_text("Hello.", encoding="utf-8")
|
|
|
|
|
|
result = patched_webui.transcription._transcribe_local_command(
|
|
|
|
|
|
"/tmp/voice-input.wav", "small", output
|
|
|
|
|
|
)
|
|
|
|
|
|
assert result["transcript"] == "Hello."
|
|
|
|
|
|
assert result["language"] == ""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
|
|
|
|
"hostile",
|
|
|
|
|
|
["en-US", "../../en", "en; rm -rf /", "e", "english", "", "\x00en", "e n"],
|
|
|
|
|
|
)
|
|
|
|
|
|
def test_patched_local_command_envelope_rejects_malformed_sidecars(
|
|
|
|
|
|
patched_webui, tmp_path, hostile
|
|
|
|
|
|
):
|
|
|
|
|
|
output = tmp_path / "stt-out"
|
|
|
|
|
|
output.mkdir()
|
|
|
|
|
|
(output / "voice-input.txt").write_text("Hello.", encoding="utf-8")
|
|
|
|
|
|
(output / "voice-input.language").write_text(hostile, encoding="utf-8")
|
|
|
|
|
|
result = patched_webui.transcription._transcribe_local_command(
|
|
|
|
|
|
"/tmp/voice-input.wav", "small", output
|
|
|
|
|
|
)
|
|
|
|
|
|
assert result["language"] == ""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_patched_local_command_envelope_survives_an_undecodable_sidecar(
|
|
|
|
|
|
patched_webui, tmp_path
|
|
|
|
|
|
):
|
|
|
|
|
|
"""A corrupt sidecar must cost the language, never the transcript."""
|
|
|
|
|
|
output = tmp_path / "stt-out"
|
|
|
|
|
|
output.mkdir()
|
|
|
|
|
|
(output / "voice-input.txt").write_text("Hello.", encoding="utf-8")
|
|
|
|
|
|
(output / "voice-input.language").write_bytes(b"\xff\xfe\x00ru")
|
|
|
|
|
|
result = patched_webui.transcription._transcribe_local_command(
|
|
|
|
|
|
"/tmp/voice-input.wav", "small", output
|
|
|
|
|
|
)
|
|
|
|
|
|
assert result["success"] is True
|
|
|
|
|
|
assert result["transcript"] == "Hello."
|
|
|
|
|
|
assert result["language"] == ""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_patched_transcribe_response_reports_the_language(patched_webui):
|
|
|
|
|
|
response = patched_webui.upload.handle_transcribe(
|
|
|
|
|
|
None, {"success": True, "transcript": " Как дела? ", "language": "ru"}
|
|
|
|
|
|
)
|
|
|
|
|
|
assert response["payload"] == {"ok": True, "transcript": "Как дела?", "language": "ru"}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
|
|
|
|
"hostile",
|
|
|
|
|
|
["", None, "en-US", "englishhh", "../en", 5, ["ru"], {"a": "b"}, "e"],
|
|
|
|
|
|
)
|
|
|
|
|
|
def test_patched_transcribe_response_blanks_untrusted_languages(patched_webui, hostile):
|
|
|
|
|
|
response = patched_webui.upload.handle_transcribe(
|
|
|
|
|
|
None, {"success": True, "transcript": "Hello.", "language": hostile}
|
|
|
|
|
|
)
|
|
|
|
|
|
assert response["payload"]["language"] == ""
|
|
|
|
|
|
assert response["payload"]["transcript"] == "Hello."
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
# 4. The /api/tts trust boundary: allow-list only, and never `voice`
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize("language", SUPPORTED)
|
|
|
|
|
|
def test_atlas_tts_forwards_allow_listed_languages(patched_webui, monkeypatch, language):
|
|
|
|
|
|
body = _atlas_tts(patched_webui, monkeypatch, {"engine": "atlas", "language": language})
|
|
|
|
|
|
assert body["language"] == language
|
|
|
|
|
|
assert body["model"] == "piper"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
|
|
|
|
"hostile",
|
|
|
|
|
|
[
|
|
|
|
|
|
"fr",
|
|
|
|
|
|
"de",
|
|
|
|
|
|
"",
|
|
|
|
|
|
None,
|
|
|
|
|
|
"EN-GB",
|
|
|
|
|
|
" RU ",
|
|
|
|
|
|
"ru-RU",
|
|
|
|
|
|
"es_MX",
|
|
|
|
|
|
"../../ru_RU-irina-medium",
|
|
|
|
|
|
"ru; rm -rf /",
|
|
|
|
|
|
"ru\x00",
|
|
|
|
|
|
"ру",
|
|
|
|
|
|
5,
|
|
|
|
|
|
["ru"],
|
|
|
|
|
|
{"language": "ru"},
|
|
|
|
|
|
True,
|
|
|
|
|
|
"x" * 8192,
|
|
|
|
|
|
],
|
|
|
|
|
|
)
|
|
|
|
|
|
def test_atlas_tts_omits_untrusted_languages(
|
|
|
|
|
|
patched_webui, monkeypatch, hostile
|
|
|
|
|
|
):
|
|
|
|
|
|
body = _atlas_tts(patched_webui, monkeypatch, {"engine": "atlas", "language": hostile})
|
|
|
|
|
|
assert "language" not in body
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_atlas_tts_language_is_absent_when_the_client_sends_none(patched_webui, monkeypatch):
|
|
|
|
|
|
body = _atlas_tts(patched_webui, monkeypatch, {"engine": "atlas"})
|
|
|
|
|
|
assert "language" not in body
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_atlas_tts_voice_field_cannot_steer_synthesis(patched_webui, monkeypatch):
|
|
|
|
|
|
body = _atlas_tts(
|
|
|
|
|
|
patched_webui,
|
|
|
|
|
|
monkeypatch,
|
|
|
|
|
|
{"engine": "atlas", "voice": "ru_RU-irina-medium", "language": "en"},
|
|
|
|
|
|
)
|
|
|
|
|
|
assert body["language"] == "en"
|
|
|
|
|
|
body = _atlas_tts(
|
|
|
|
|
|
patched_webui,
|
|
|
|
|
|
monkeypatch,
|
|
|
|
|
|
{"engine": "atlas", "voice": "es_MX-claude-high"},
|
|
|
|
|
|
)
|
|
|
|
|
|
assert "language" not in body
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_atlas_tts_language_helper_only_ever_returns_a_baked_voice_language(patched_webui):
|
|
|
|
|
|
resolve = patched_webui.routes._atlas_tts_language
|
|
|
|
|
|
hostile = [
|
|
|
|
|
|
None, 0, 1, -1, True, False, [], {}, set(), object(), b"ru",
|
|
|
|
|
|
"", " ", "\t\n", "en", "EN", "en-US", "en_us", "ru-RU", "es-MX",
|
|
|
|
|
|
"e", "eng", "english", "ru ru", "ru;es", "../ru", "ru\x00", "ру",
|
|
|
|
|
|
"x" * 65536, "en" * 4096,
|
|
|
|
|
|
]
|
|
|
|
|
|
for value in hostile:
|
|
|
|
|
|
expected = value if value in SUPPORTED else ""
|
|
|
|
|
|
assert resolve({"language": value}) == expected
|
|
|
|
|
|
for value in hostile:
|
|
|
|
|
|
assert resolve(value) == ""
|
|
|
|
|
|
assert resolve({"voice": "ru_RU-irina-medium"}) == ""
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_manual_tts_button_body_still_carries_no_language(patched_webui):
|
|
|
|
|
|
"""The read-aloud button has no trusted STT signal, so it must stay Amy."""
|
|
|
|
|
|
ui = (patched_webui.webui / "static" / "ui.js").read_text(encoding="utf-8")
|
|
|
|
|
|
assert "engine:engineOverride||'edge'" in ui
|
|
|
|
|
|
assert "language" not in ui
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
# 5. Browser contract, driven through the real atlas-voice.js
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.fixture(scope="module")
|
|
|
|
|
|
def voice_probe():
|
|
|
|
|
|
node = shutil.which("node")
|
|
|
|
|
|
if not node:
|
|
|
|
|
|
pytest.skip("node is required to drive the browser voice-mode contract")
|
|
|
|
|
|
completed = subprocess.run(
|
|
|
|
|
|
[node, str(VOICE_PROBE), str(VOICE_SCRIPT)],
|
|
|
|
|
|
capture_output=True,
|
|
|
|
|
|
text=True,
|
|
|
|
|
|
timeout=180,
|
|
|
|
|
|
)
|
|
|
|
|
|
assert completed.returncode == 0, completed.stderr
|
|
|
|
|
|
return json.loads(completed.stdout)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
|
|
|
|
("scenario", "expected"),
|
|
|
|
|
|
[
|
|
|
|
|
|
("english_turn_speaks_english", "en"),
|
|
|
|
|
|
("russian_turn_speaks_russian", "ru"),
|
|
|
|
|
|
("spanish_turn_speaks_spanish", "es"),
|
|
|
|
|
|
],
|
|
|
|
|
|
)
|
|
|
|
|
|
def test_voice_mode_speaks_the_language_whisper_detected(voice_probe, scenario, expected):
|
|
|
|
|
|
requests = voice_probe[scenario]["tts"]
|
|
|
|
|
|
assert requests, "voice mode never reached /api/tts"
|
|
|
|
|
|
for request in requests:
|
|
|
|
|
|
assert request["engine"] == "atlas"
|
|
|
|
|
|
assert request["language"] == expected
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
|
|
|
|
"scenario", ["missing_language_falls_back", "unsupported_language_falls_back"]
|
|
|
|
|
|
)
|
|
|
|
|
|
def test_voice_mode_omits_language_without_a_trusted_signal(voice_probe, scenario):
|
|
|
|
|
|
requests = voice_probe[scenario]["tts"]
|
|
|
|
|
|
assert requests, "voice mode never reached /api/tts"
|
|
|
|
|
|
for request in requests:
|
|
|
|
|
|
assert "language" not in request
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_voice_mode_drops_hostile_language_values(voice_probe):
|
|
|
|
|
|
for case in voice_probe["hostile_language_values_are_dropped"]["results"]:
|
|
|
|
|
|
for request in case["tts"]:
|
|
|
|
|
|
assert "language" not in request, case["sent"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_voice_mode_never_sends_a_voice_field(voice_probe):
|
|
|
|
|
|
for request in voice_probe["voice_field_is_never_sent"]["tts"]:
|
hermes(voice): continuous mic, barge stitching, 1.15x speech
Three conversational fixes for hands-free chat:
- The microphone now stays hot for the whole session: capture runs on
its own epoch, re-arms immediately after each utterance endpoints,
and keeps recording through transcribing/thinking/speaking - speech
is never lost to Hermes being busy. Speech onset during a response
cancels it through the live capture path (echo-guarded exactly like
the old monitor) without touching the running recorder.
- When the user talks over Hermes before any visible reply appeared,
the interrupted utterance and the follow-up are stitched into one
message (20s window), so the response addresses the whole thought.
- TTS speaks 15% faster by default (server-side length_scale, no pitch
shift), user-tunable via hermes-voice-tts-speed (0.5-2.0), honored on
streaming, WAV fallback and thinking-cue paths.
245 voice-lane tests pass; single getUserMedia site preserved;
Dockerfile grep guards verified.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 06:27:38 -03:00
|
|
|
|
assert set(request) <= {"text", "engine", "language", "turn_id", "speed"}
|
2026-08-21 11:52:49 +00:00
|
|
|
|
assert "voice" not in request
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_voice_mode_does_not_reuse_a_previous_turn_language(voice_probe):
|
|
|
|
|
|
requests = voice_probe["language_does_not_leak_into_later_turn"]["tts"]
|
|
|
|
|
|
assert len(requests) == 3
|
|
|
|
|
|
assert requests[0]["language"] == "ru"
|
|
|
|
|
|
assert "language" not in requests[1]
|
|
|
|
|
|
assert requests[2]["language"] == "es"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_voice_mode_ignores_language_from_an_empty_transcript(voice_probe):
|
|
|
|
|
|
result = voice_probe["empty_transcript_does_not_arm_a_language"]
|
|
|
|
|
|
assert result["sendsAfterBlank"] == []
|
|
|
|
|
|
assert result["sends"] == ["Hello."]
|
|
|
|
|
|
assert result["tts"], "the follow-up turn should still be spoken"
|
|
|
|
|
|
for request in result["tts"]:
|
|
|
|
|
|
assert "language" not in request
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_voice_mode_discards_language_when_the_session_changes(voice_probe):
|
|
|
|
|
|
result = voice_probe["session_change_discards_language"]
|
|
|
|
|
|
assert result["afterSwitch"] == []
|
|
|
|
|
|
for request in result["tts"]:
|
|
|
|
|
|
assert "language" not in request
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_voice_mode_discards_language_when_voice_mode_is_turned_off(voice_probe):
|
|
|
|
|
|
result = voice_probe["deactivation_discards_language"]
|
|
|
|
|
|
assert result["afterDeactivate"] == []
|
|
|
|
|
|
for request in result["tts"]:
|
|
|
|
|
|
assert "language" not in request
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_voice_mode_speaks_nothing_when_transcription_fails(voice_probe):
|
|
|
|
|
|
result = voice_probe["transcribe_error_speaks_nothing"]
|
|
|
|
|
|
assert result["tts"] == []
|
|
|
|
|
|
assert any("Whisper is down" in toast for toast in result["toasts"])
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-08-23 18:29:02 -03:00
|
|
|
|
def test_adaptive_chunks_wait_for_sentence_then_change_size(voice_probe):
|
|
|
|
|
|
result = voice_probe["adaptive_chunks_are_sentence_gated"]
|
|
|
|
|
|
assert result["partial"] == []
|
|
|
|
|
|
chunks = result["complete"]
|
|
|
|
|
|
assert 40 <= len(chunks[0]) <= 60
|
|
|
|
|
|
assert all(100 <= len(chunk) <= 140 for chunk in chunks[1:-1])
|
|
|
|
|
|
assert "".join(chunks).replace(" ", "")
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-08-23 22:13:52 -03:00
|
|
|
|
def test_spoken_http_urls_are_skipped_without_mangling_prose(voice_probe):
|
|
|
|
|
|
result = voice_probe["spoken_urls_are_skipped_without_damaging_text"]
|
|
|
|
|
|
assert result["sentence"] == "Read this. Then continue."
|
|
|
|
|
|
assert result["wrapped"] == "Open. Next."
|
|
|
|
|
|
assert result["punctuated"] == "Try, or!"
|
|
|
|
|
|
assert result["domains"] == "Keep example.com and sub.example.org exactly as written."
|
|
|
|
|
|
assert result["prose"] == "No links here; keep this sentence exactly as written."
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_url_elision_happens_before_every_localized_voice_route(voice_probe):
|
|
|
|
|
|
cases = voice_probe["spoken_urls_are_elided_before_all_voice_routes"]["results"]
|
|
|
|
|
|
assert [case["language"] for case in cases] == ["en", "ru", "es"]
|
|
|
|
|
|
for case in cases:
|
|
|
|
|
|
assert case["tts"]
|
|
|
|
|
|
spoken = " ".join(request["text"] for request in case["tts"])
|
|
|
|
|
|
assert "http://" not in spoken
|
|
|
|
|
|
assert "https://" not in spoken
|
|
|
|
|
|
assert "private.example" not in spoken
|
|
|
|
|
|
assert "example.com" in spoken
|
|
|
|
|
|
assert all(request["language"] == case["language"] for request in case["tts"])
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_canonical_pcm_fallback_wav_is_runtime_valid(voice_probe):
|
|
|
|
|
|
result = voice_probe["canonical_pcm_fallback_builds_a_valid_wav"]
|
|
|
|
|
|
assert result == {
|
|
|
|
|
|
"type": "audio/wav",
|
|
|
|
|
|
"size": 50,
|
|
|
|
|
|
"riff": "RIFF",
|
|
|
|
|
|
"wave": "WAVE",
|
|
|
|
|
|
"format": 1,
|
|
|
|
|
|
"channels": 1,
|
|
|
|
|
|
"rate": 16000,
|
|
|
|
|
|
"bits": 16,
|
|
|
|
|
|
"data": 6,
|
|
|
|
|
|
"pcm": [0, 0, 255, 127, 0, 128],
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_applied_aec_settings_are_runtime_fail_safe(voice_probe):
|
|
|
|
|
|
assert voice_probe["applied_aec_settings_are_fail_safe"] == {
|
|
|
|
|
|
"applied": True,
|
|
|
|
|
|
"rejected": False,
|
|
|
|
|
|
"unknown": True,
|
|
|
|
|
|
"unsupported": True,
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-08-23 18:29:02 -03:00
|
|
|
|
def test_first_complete_sentence_reaches_tts_before_completion_callback(voice_probe):
|
|
|
|
|
|
result = voice_probe["first_sentence_speaks_before_stream_completion"]
|
|
|
|
|
|
assert result["beforeBoundary"] == 0
|
|
|
|
|
|
assert result["afterBoundary"] == 1
|
|
|
|
|
|
assert len(result["tts"]) >= 2
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_final_renderer_revision_closes_queue_without_replaying_prefix(voice_probe):
|
|
|
|
|
|
result = voice_probe["final_renderer_revision_closes_speech_queue"]
|
|
|
|
|
|
assert result["beforeRevision"] == 1
|
|
|
|
|
|
assert result["afterRevision"] >= 2
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_streaming_tts_failure_falls_back_to_complete_wav(voice_probe):
|
|
|
|
|
|
result = voice_probe["streaming_tts_failure_falls_back_to_wav"]
|
|
|
|
|
|
assert len(result["stream"]) == 1
|
|
|
|
|
|
assert len(result["wav"]) == 1
|
|
|
|
|
|
assert result["stream"][0]["turn_id"] == result["wav"][0]["turn_id"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_one_ahead_bound_and_turn_cancellation(voice_probe):
|
|
|
|
|
|
result = voice_probe["one_ahead_is_bounded_and_turn_cancel_stops_audio"]
|
|
|
|
|
|
assert result["beforeFirstEnds"] == 2
|
|
|
|
|
|
assert result["afterCancel"] == 2
|
|
|
|
|
|
assert result["active"] is False
|
|
|
|
|
|
assert result["firstAudioPaused"] is True
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_authenticated_voice_websocket_uses_rendered_csrf_subprotocol(
|
|
|
|
|
|
patched_webui, monkeypatch
|
|
|
|
|
|
):
|
|
|
|
|
|
routes = patched_webui.routes
|
|
|
|
|
|
monkeypatch.setattr(
|
|
|
|
|
|
routes, "_check_same_origin_browser_request", lambda _handler: True, raising=False
|
|
|
|
|
|
)
|
|
|
|
|
|
monkeypatch.setitem(
|
|
|
|
|
|
sys.modules,
|
|
|
|
|
|
"api.auth",
|
|
|
|
|
|
SimpleNamespace(
|
|
|
|
|
|
csrf_token_for_session=lambda _cookie: "csrf-token",
|
|
|
|
|
|
is_auth_enabled=lambda: True,
|
|
|
|
|
|
parse_cookie=lambda _handler: None,
|
|
|
|
|
|
verify_session=lambda value: value == "session-cookie",
|
|
|
|
|
|
),
|
|
|
|
|
|
)
|
|
|
|
|
|
handler = SimpleNamespace(
|
|
|
|
|
|
_trusted_auth_session_cookie_value="session-cookie",
|
|
|
|
|
|
headers={
|
|
|
|
|
|
"Origin": "https://chat.bstein.dev",
|
|
|
|
|
|
"Sec-WebSocket-Protocol": (
|
|
|
|
|
|
"hermes-voice-v1, hermes-csrf.csrf-token"
|
|
|
|
|
|
),
|
|
|
|
|
|
}
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
assert routes._atlas_ws_authorized(handler) is True
|
|
|
|
|
|
handler.headers["Sec-WebSocket-Protocol"] = "hermes-voice-v1, hermes-csrf.wrong"
|
|
|
|
|
|
assert routes._atlas_ws_authorized(handler) is False
|
|
|
|
|
|
handler.headers["Origin"] = ""
|
|
|
|
|
|
assert routes._atlas_ws_authorized(handler) is False
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_streaming_tts_payload_is_narrow_and_turn_bound(patched_webui):
|
|
|
|
|
|
payload = patched_webui.routes._atlas_tts_stream_payload(
|
|
|
|
|
|
{
|
|
|
|
|
|
"text": "A safe sentence.",
|
|
|
|
|
|
"language": "ru",
|
|
|
|
|
|
"turn_id": "voice-turn-7",
|
|
|
|
|
|
"voice": "../../untrusted",
|
|
|
|
|
|
"speed": 99,
|
|
|
|
|
|
}
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
assert payload == {
|
|
|
|
|
|
"model": "piper",
|
|
|
|
|
|
"input": "A safe sentence.",
|
hermes(voice): continuous mic, barge stitching, 1.15x speech
Three conversational fixes for hands-free chat:
- The microphone now stays hot for the whole session: capture runs on
its own epoch, re-arms immediately after each utterance endpoints,
and keeps recording through transcribing/thinking/speaking - speech
is never lost to Hermes being busy. Speech onset during a response
cancels it through the live capture path (echo-guarded exactly like
the old monitor) without touching the running recorder.
- When the user talks over Hermes before any visible reply appeared,
the interrupted utterance and the follow-up are stitched into one
message (20s window), so the response addresses the whole thought.
- TTS speaks 15% faster by default (server-side length_scale, no pitch
shift), user-tunable via hermes-voice-tts-speed (0.5-2.0), honored on
streaming, WAV fallback and thinking-cue paths.
245 voice-lane tests pass; single getUserMedia site preserved;
Dockerfile grep guards verified.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 06:27:38 -03:00
|
|
|
|
"speed": 2.0,
|
2026-08-23 18:29:02 -03:00
|
|
|
|
"language": "ru",
|
|
|
|
|
|
"turn_id": "voice-turn-7",
|
|
|
|
|
|
}
|
hermes(voice): continuous mic, barge stitching, 1.15x speech
Three conversational fixes for hands-free chat:
- The microphone now stays hot for the whole session: capture runs on
its own epoch, re-arms immediately after each utterance endpoints,
and keeps recording through transcribing/thinking/speaking - speech
is never lost to Hermes being busy. Speech onset during a response
cancels it through the live capture path (echo-guarded exactly like
the old monitor) without touching the running recorder.
- When the user talks over Hermes before any visible reply appeared,
the interrupted utterance and the follow-up are stitched into one
message (20s window), so the response addresses the whole thought.
- TTS speaks 15% faster by default (server-side length_scale, no pitch
shift), user-tunable via hermes-voice-tts-speed (0.5-2.0), honored on
streaming, WAV fallback and thinking-cue paths.
245 voice-lane tests pass; single getUserMedia site preserved;
Dockerfile grep guards verified.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 06:27:38 -03:00
|
|
|
|
assert patched_webui.routes._atlas_tts_stream_payload(
|
|
|
|
|
|
{"text": "A safe sentence.", "speed": 1.15}
|
|
|
|
|
|
)["speed"] == 1.15
|
|
|
|
|
|
assert patched_webui.routes._atlas_tts_stream_payload(
|
|
|
|
|
|
{"text": "A safe sentence."}
|
|
|
|
|
|
)["speed"] == 1.0
|
|
|
|
|
|
for hostile in ("2", True, None, [1.5], {"speed": 1.5}, float("nan")):
|
|
|
|
|
|
assert patched_webui.routes._atlas_tts_stream_payload(
|
|
|
|
|
|
{"text": "A safe sentence.", "speed": hostile}
|
|
|
|
|
|
)["speed"] == 1.0
|
2026-08-23 18:29:02 -03:00
|
|
|
|
|
|
|
|
|
|
|
2026-08-23 22:13:52 -03:00
|
|
|
|
def test_streaming_tts_payload_forwards_only_allowlisted_localized_cues(patched_webui):
|
|
|
|
|
|
routes = patched_webui.routes
|
|
|
|
|
|
assert routes._atlas_tts_stream_payload(
|
|
|
|
|
|
{
|
|
|
|
|
|
"text": "client text is ignored by the cue cache",
|
|
|
|
|
|
"language": "es",
|
|
|
|
|
|
"cue_id": "still_working",
|
|
|
|
|
|
"turn_id": "voice-turn-7:thinking-cue:2",
|
|
|
|
|
|
}
|
|
|
|
|
|
) == {
|
|
|
|
|
|
"model": "piper",
|
|
|
|
|
|
"input": "client text is ignored by the cue cache",
|
|
|
|
|
|
"speed": 1.0,
|
|
|
|
|
|
"language": "es",
|
|
|
|
|
|
"cue_id": "still_working",
|
|
|
|
|
|
"turn_id": "voice-turn-7:thinking-cue:2",
|
|
|
|
|
|
}
|
|
|
|
|
|
for hostile in ("invented", "../thinking", 7, None):
|
|
|
|
|
|
with pytest.raises(ValueError, match="invalid thinking cue"):
|
|
|
|
|
|
routes._atlas_tts_stream_payload(
|
|
|
|
|
|
{"text": "ignored", "language": "en", "cue_id": hostile}
|
|
|
|
|
|
)
|
|
|
|
|
|
with pytest.raises(ValueError, match="invalid thinking cue"):
|
|
|
|
|
|
routes._atlas_tts_stream_payload(
|
|
|
|
|
|
{"text": "ignored", "language": "fr", "cue_id": "thinking"}
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-08-21 11:52:49 +00:00
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
# 6. Build-time enforcement and documented semantics
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_image_build_verifies_every_language_routing_patch():
|
|
|
|
|
|
dockerfile = (DOCKERFILES / "Dockerfile.hermes-webui").read_text(encoding="utf-8")
|
|
|
|
|
|
assert "'language': detected" in dockerfile
|
|
|
|
|
|
assert "def _atlas_tts_language(body):" in dockerfile
|
|
|
|
|
|
assert 'request_payload["language"] = _atlas_language' in dockerfile
|
|
|
|
|
|
assert '"language": detected_language' in dockerfile
|
|
|
|
|
|
assert "takeSttLanguage(token)" in dockerfile
|
|
|
|
|
|
assert "/opt/hermes-webui/api/upload.py" in dockerfile
|
|
|
|
|
|
assert "/opt/hermes/tools/transcription_tools.py" in dockerfile
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_atlas_patch_roots_are_overridable_for_offline_verification():
|
|
|
|
|
|
patch = ATLAS_PATCH.read_text(encoding="utf-8")
|
|
|
|
|
|
assert 'os.environ.get("HERMES_WEBUI_PATCH_ROOT", "/opt/hermes-webui")' in patch
|
|
|
|
|
|
assert 'os.environ.get("HERMES_AGENT_PATCH_ROOT", "/opt/hermes")' in patch
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_notes_document_the_stt_driven_voice_selection_and_its_limits():
|
|
|
|
|
|
notes = (HERMES / "NOTES.md").read_text(encoding="utf-8")
|
|
|
|
|
|
assert "STT-detected language" in notes
|
|
|
|
|
|
for marker in ("hands-free", "Typed messages", "en_US-amy-medium"):
|
|
|
|
|
|
assert marker in notes
|
2026-08-24 13:58:35 -03:00
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_reply_language_heuristic_routes_detectionless_spanish(voice_probe):
|
|
|
|
|
|
"""A Spanish reply on a detection-less turn still gets the Spanish voice."""
|
|
|
|
|
|
requests = voice_probe["spanish_reply_without_detection_uses_reply_heuristic"]["tts"]
|
|
|
|
|
|
assert requests, "voice mode never reached /api/tts"
|
|
|
|
|
|
for request in requests:
|
|
|
|
|
|
assert request["language"] == "es"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_reply_script_evidence_corrects_wrong_detection(voice_probe):
|
|
|
|
|
|
"""Script-level evidence in the reply text overrides a wrong STT hint, so
|
|
|
|
|
|
a Spanish reply is never spoken by the English Amy voice."""
|
|
|
|
|
|
requests = voice_probe["reply_script_evidence_corrects_wrong_detection"]["tts"]
|
|
|
|
|
|
assert requests, "voice mode never reached /api/tts"
|
|
|
|
|
|
for request in requests:
|
|
|
|
|
|
assert request["language"] == "es"
|
hermes(voice): workspace nav home, character orb, conversation rename, voice-lang fix
Final conversation-mode polish from mobile testing:
- The Workspace toggle now sits with the chat/Telegram nav at every
width: nav.rail on desktop, the top app titlebar on mobile. The
floating pill that pushed the mobile composer's control row (and the
conversation-mode button) off screen is gone - a fallback exists only
for headless DOMs and is pinned to a top corner, never over the
composer.
- The conversation orb watermark is now the Hermes character avatar
(static/hermes-agent-192.png) instead of the caduceus staff.
- User-facing 'hands-free' copy renamed to 'Conversation mode'.
- Wrong-voice fix: strongReplyLanguage flagged Spanish on a single
accented char, so an English reply naming European cities (Zürich,
Málaga) overrode the correct English STT detection and was spoken by
the Spanish voice. Detection now requires density (Cyrillic >=4 at
>=50%, or inverted punctuation / >=2 accents corroborated by Spanish
stopwords); plain English always speaks English, forced language wins,
accent-free Spanish still routes via trusted STT. 294 voice tests.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 19:22:20 -03:00
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_english_reply_full_of_european_names_stays_on_the_english_voice(voice_probe):
|
|
|
|
|
|
"""FIX 4: an English paragraph packed with accented European proper nouns
|
|
|
|
|
|
(Zürich, München, Málaga, café, Kraków) carries no decisive Spanish/Russian
|
|
|
|
|
|
evidence, so it never trips a foreign voice. A single accented place name used
|
|
|
|
|
|
to make strongReplyLanguage return 'es' and outrank a correct 'en' STT — the
|
|
|
|
|
|
root cause of the live "English reply about Europe spoken in a foreign voice"
|
|
|
|
|
|
bug."""
|
|
|
|
|
|
r = voice_probe["reply_language_resolution"]
|
|
|
|
|
|
assert r["europeStrong"] == ""
|
|
|
|
|
|
assert r["europeDetectionless"] == "" # English default (Amy) — no language field
|
|
|
|
|
|
assert r["europeWithEnglishStt"] == "en"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_reply_language_resolution_routes_cyrillic_and_spanish(voice_probe):
|
|
|
|
|
|
"""FIX 4: decisive script/orthography evidence still routes correctly —
|
|
|
|
|
|
a Cyrillic paragraph to the Russian voice, a clearly-Spanish paragraph to the
|
|
|
|
|
|
Spanish voice."""
|
|
|
|
|
|
r = voice_probe["reply_language_resolution"]
|
|
|
|
|
|
assert r["cyrillicStrong"] == "ru"
|
|
|
|
|
|
assert r["cyrillicDetectionless"] == "ru"
|
|
|
|
|
|
assert r["spanishStrong"] == "es"
|
|
|
|
|
|
assert r["spanishDetectionless"] == "es"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_reply_language_resolution_forced_language_always_wins(voice_probe):
|
|
|
|
|
|
"""FIX 4: an explicit conversation-mode force beats every auto signal —
|
|
|
|
|
|
including plain English text and decisive Cyrillic evidence."""
|
|
|
|
|
|
r = voice_probe["reply_language_resolution"]
|
|
|
|
|
|
assert r["forcedOverridesEnglishText"] == "ru"
|
|
|
|
|
|
assert r["forcedOverridesCyrillic"] == "en"
|