atlas-iac/testing/tests/test_hermes_voice_capture_continuity.py
jenkins d041f1d1ee hermes(voice): round-3 fixes, language routing, session-toast root cause
- Captions read message bodies only (the scraper was concatenating
  avatar, author and worklog chips); multi-segment interim turns are
  now speakable and drive clean speak-to-thinking-to-speak cycles when
  playback drains mid-turn.
- Dynamic endpointing: complete-looking partials (3+ words or terminal
  punctuation) endpoint at the base window; the long hold remains only
  for one-two-word fragments. A stale-busy 10s settle wait on every
  post-error send is gone.
- Both overlay captions are bounded, touch-scrollable regions with
  follow-tail; caps raised for long turns.
- Error envelopes are never spoken or captioned; errored turns run
  resyncCapture (fresh STT session on the hot mic).
- Workspace toggle now lives in the sidebar rail (floating button only
  below the rail breakpoint).
- False 'session unavailable' toast root-caused: the router continuity
  guard shows it on a 409 that fired when a transient profile-listing
  failure failed closed into a fake cross-profile mismatch; the patcher
  now answers from the alias cache and never claims a default-vs-named
  mismatch while aliases are unconfirmed.
- Barge-in sends carry a one-line cut-point marker with the last spoken
  sentence; visible-history truncation judged infeasible client-side.
- Language switching works end-to-end: sticky per-session STT language
  hint (restarting an unused next session on switch), reply voice from
  script evidence, STT detection, then stopword heuristic; cues and WAV
  fallback share the turn language.
- Legacy CI guards: node skip for the DOM probe, ffmpeg/codec skips for
  the container-fallback test. 272 voice-lane tests green.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
2026-08-24 13:58:35 -03:00

265 lines
11 KiB
Python

"""Continuous-capture contracts: no first-word clipping, stitching, overlay.
The scenarios drive the real ``atlas-voice.js`` against a stub streaming STT
WebSocket whose VAD, speculative EOS-freeze, resume and commit semantics
mirror ``hermes-jetson-stt-server.py``. Words are encoded as distinct PCM
amplitudes, so the transcript of a commit is exactly the words whose audio
survived endpointing — the live "only the first word was sent" regression is
directly observable in the composer text the probe records.
"""
from __future__ import annotations
import json
import shutil
import subprocess
from pathlib import Path
import pytest
ROOT = Path(__file__).resolve().parents[2]
VOICE_SCRIPT = ROOT / "dockerfiles" / "hermes-webui-atlas-voice.js"
PROBE = ROOT / "testing" / "probes" / "hermes_voice_capture_probe.js"
@pytest.fixture(scope="module")
def probe_results() -> dict:
node = shutil.which("node")
if not node:
pytest.skip("node is required to drive the capture continuity contract")
completed = subprocess.run(
[node, str(PROBE), str(VOICE_SCRIPT)],
check=False,
capture_output=True,
text=True,
timeout=180,
)
assert completed.returncode == 0, completed.stderr
return json.loads(completed.stdout)
def test_multiword_utterance_with_interword_pause_is_complete(probe_results):
scenario = probe_results["multiword_utterance_with_interword_pause"]
assert scenario["sends"] == ["alpha bravo"]
assert scenario["state"] == "thinking"
def test_thinking_pause_after_first_word_does_not_split(probe_results):
"""A >1100ms pause right after the first word stays inside one utterance."""
scenario = probe_results["thinking_pause_after_first_word_does_not_split"]
assert scenario["sends"] == ["alpha bravo"]
# The pause froze a first-word EOS snapshot on the server; resumed speech
# must have cleared it before the commit.
events = scenario["serverEvents"][0]
assert "resume-clear" in events or "recover-by-rms" in events
def test_barge_in_stitches_regardless_of_partial_assistant_output(probe_results):
scenario = probe_results["second_utterance_during_response_is_complete"]
assert scenario["firstSends"] == ["alpha"]
# Assistant text was already visible (and being spoken) when the user
# talked over the response: the interrupted thought and the follow-up form
# one stitched message, and the cut marker names the sentence that was
# playing so the model knows where its reply stopped being heard.
assert scenario["sends"] == [
"alpha",
"alpha bravo charlie\n[voice interruption: you were cut off after "
'"Partial answer already visible."]',
]
def test_normal_completion_clears_stitch_and_mic_stays_hot(probe_results):
scenario = probe_results["back_to_back_utterances_stay_hot"]
assert scenario["sendsAfterFirst"] == ["alpha"]
assert scenario["sends"] == ["alpha", "bravo"]
# The whole session runs on one getUserMedia lease.
assert scenario["micAcquisitions"] == 1
def test_conversation_overlay_lifecycle(probe_results):
scenario = probe_results["conversation_overlay_lifecycle"]
assert scenario["overlayPresent"] is True
assert scenario["role"] == "dialog"
assert scenario["ariaModal"] == "true"
assert scenario["captionsLive"] == "polite"
assert scenario["listeningState"] == "listening"
assert scenario["thinkingState"] in {"thinking", "speaking"}
assert scenario["userCaption"] == "alpha"
assert scenario["assistantCaption"] == "A visible answer."
# Mute drives the real capture track and is fully reversible.
assert scenario["mutedPressed"] == "true"
assert scenario["mutedTracks"] == [False]
assert scenario["unmutedTracks"] == [True]
# Both exit paths remove the overlay completely and end hands-free mode.
assert scenario["removedOnExit"] is True
assert scenario["inactiveAfterExit"] is True
assert scenario["removedOnEscape"] is True
def test_overlay_source_contract():
source = VOICE_SCRIPT.read_text(encoding="utf-8")
css = (ROOT / "dockerfiles" / "hermes-webui-atlas-voice.css").read_text(
encoding="utf-8"
)
# Lazily created, fully removed, storage-free, single-capture overlay.
assert "function openConversationOverlay()" in source
assert "function removeConversationOverlay()" in source
assert source.count("navigator.mediaDevices.getUserMedia(") == 1
overlay_region = source.split("Conversation mode overlay", 1)[1].split(
"function clearErrorTimer", 1
)[0]
assert "localStorage" not in overlay_region
assert "track.enabled=!conversation.muted" in overlay_region
assert "'aria-label':'Voice conversation'" in overlay_region
assert "'aria-live':'polite'" in overlay_region
assert "event.key==='Escape'" in overlay_region
assert "event.key==='Tab'" in overlay_region
# Orb styling: breathing, level-driven, playback-driven, reduced-motion.
assert ".voice-conversation" in css
assert "--conversation-level" in css
assert "voice-conversation-breathe" in css
assert ".voice-conversation.is-playing .voice-conversation-orb-halo" in css
reduced = css.split("@media (prefers-reduced-motion: reduce)", 1)[1]
assert ".voice-conversation *" in reduced
def test_endpointing_and_streaming_hardening_source_contract():
source = VOICE_SCRIPT.read_text(encoding="utf-8")
# Young utterances hold their endpoint past a thinking pause — unless the
# streaming partial already reads as a plausibly complete utterance
# (>=3 words or terminal punctuation), which endpoints at the base window.
assert "const VAD_EARLY_SILENCE_MS=1800" in source
assert "const VAD_COMMITTED_SPEECH_MS=1200" in source
assert "latestPartial:function(){return lastPartialText;}" in source
assert "const partialComplete=partialWords>=3||(partialWords>0" in source
assert (
"const endpointSilenceMs=(speechMs<VAD_COMMITTED_SPEECH_MS&&!partialComplete)"
"?Math.max(silenceMs,VAD_EARLY_SILENCE_MS):silenceMs" in source
)
# Resume always reaches the wire after a server-length silence gap.
assert "const SERVER_EOS_SILENCE_MS=650" in source
assert "speechGapMs>=SERVER_EOS_SILENCE_MS" in source
assert "if(settled||committed||!speculative) return;" not in source.split(
"resume:function()", 1
)[1].split("speculate:function()", 1)[0]
# No speculative Whisper pass on a one-word fragment.
assert "const SPECULATE_MIN_SPEECH_MS=700" in source
assert "speechMs>=SPECULATE_MIN_SPEECH_MS" in source
# Echo residue can no longer ratchet the onset threshold above speech.
assert "Math.min(rms,speechThreshold)*0.06" in source
def test_stitch_gate_ignores_partial_assistant_output():
source = VOICE_SCRIPT.read_text(encoding="utf-8")
region = source.split("function recordPendingStitch()", 1)[1].split(
"async function sendTranscript", 1
)[0]
assert "currentAssistantText()" not in region
assert "pendingStitch={text:sent.text,at:Date.now(),cut:cut}" in region
# Normal completion still clears the stitch context.
assert (
"if(isFinal){\n // The completion callback marks a normally "
"completed response" in source
)
def test_three_word_partial_endpoints_at_base_silence(probe_results):
scenario = probe_results["three_word_partial_endpoints_at_base_silence"]
assert scenario["sends"] == ["alpha bravo charlie"]
assert scenario["state"] == "thinking"
def test_two_word_young_utterance_keeps_the_hold(probe_results):
scenario = probe_results["two_word_young_utterance_keeps_the_hold"]
assert scenario["sendsEarly"] == []
assert scenario["sends"] == ["alpha bravo"]
def test_errored_turn_is_never_spoken_and_capture_resyncs(probe_results):
scenario = probe_results["errored_turn_is_not_spoken_and_capture_resyncs"]
assert scenario["stateAfterError"] == "listening"
assert scenario["labelAfterError"] == "Something went wrong — listening"
assert scenario["ttsDuringError"] == 0
assert scenario["freshSessions"] >= 1
# The follow-up utterance is sent alone: no stitch with the errored turn.
assert scenario["sends"] == ["alpha", "bravo"]
def test_barge_cut_marker_is_one_bounded_line(probe_results):
scenario = probe_results["barge_cut_marker_records_spoken_tail"]
assert scenario["ttsTexts"][0] == "The first point is ready."
assert scenario["sends"][1] == (
"alpha bravo\n[voice interruption: you were cut off after "
'"The first point is ready."]'
)
marker = scenario["sends"][1].split("\n", 1)[1]
assert "\n" not in marker
def test_sticky_language_biases_next_stt_session(probe_results):
scenario = probe_results["sticky_language_biases_next_stt_session"]
assert scenario["startLanguages"][0] == "auto"
assert "en" in scenario["startLanguages"]
def test_cut_marker_error_resync_and_speaking_cycles_source_contract():
source = VOICE_SCRIPT.read_text(encoding="utf-8")
# Cut marker: exactly one appended line, tail bounded to 120 characters.
assert "function voiceCutMarker(cut)" in source
assert "tail.length>120?tail.slice(tail.length-120):tail" in source
assert '\\n[voice interruption: you were cut off after "' in source
# Error envelopes are detected structurally and trigger a capture resync.
assert "function handleAssistantResponseError(token)" in source
assert "function resyncCapture(token,statusLabel)" in source
assert "'Something went wrong — listening'" in source
assert "segment.dataset.error==='1'" in source
assert ".provider-error-details" in source
# A cancellation with no in-flight stream never gates the send.
assert "if(!cancellation.streamId){" in source
# Speaking ⇄ thinking cycles inside one interim-message turn.
assert "function scheduleSpeakingIdleFallback(turn,session)" in source
assert "if(state==='speaking') setState('thinking')" in source
assert "function playbackAudible()" in source
# Body-only scraping: never the avatar letter, author name or status chip.
assert "querySelectorAll('.msg-body')" in source
assert "function readSegmentBody(segment)" in source
assert "function readAssistantTurn(turn)" in source
def test_caption_regions_are_bounded_scrollable_and_follow_tail():
source = VOICE_SCRIPT.read_text(encoding="utf-8")
css = (ROOT / "dockerfiles" / "hermes-webui-atlas-voice.css").read_text(
encoding="utf-8"
)
assert "function updateCaptionRegion(element,text,limit)" in source
assert "function attachCaptionScroll(element)" in source
assert "element.dataset.follow=gap<=24?'1':'0'" in source
for token in (
"overflow-y: auto",
"overscroll-behavior: contain",
"-webkit-overflow-scrolling: touch",
"touch-action: pan-y",
"max-height: 22vh",
"max-height: 38vh",
"mask-image",
):
assert token in css
def test_reply_language_stickiness_source_contract():
source = VOICE_SCRIPT.read_text(encoding="utf-8")
assert "let sessionLanguage=''" in source
assert "language:sessionLanguage||'auto'" in source
assert "function strongReplyLanguage(text)" in source
assert "function detectReplyLanguage(text)" in source
assert "scheduleThinkingCues(token,language||sessionLanguage,thinkingTurnId)" in source
# The sticky language resets with each hands-free session.
assert source.count("sessionLanguage='';") >= 2