- Captions read message bodies only (the scraper was concatenating avatar, author and worklog chips); multi-segment interim turns are now speakable and drive clean speak-to-thinking-to-speak cycles when playback drains mid-turn. - Dynamic endpointing: complete-looking partials (3+ words or terminal punctuation) endpoint at the base window; the long hold remains only for one-two-word fragments. A stale-busy 10s settle wait on every post-error send is gone. - Both overlay captions are bounded, touch-scrollable regions with follow-tail; caps raised for long turns. - Error envelopes are never spoken or captioned; errored turns run resyncCapture (fresh STT session on the hot mic). - Workspace toggle now lives in the sidebar rail (floating button only below the rail breakpoint). - False 'session unavailable' toast root-caused: the router continuity guard shows it on a 409 that fired when a transient profile-listing failure failed closed into a fake cross-profile mismatch; the patcher now answers from the alias cache and never claims a default-vs-named mismatch while aliases are unconfirmed. - Barge-in sends carry a one-line cut-point marker with the last spoken sentence; visible-history truncation judged infeasible client-side. - Language switching works end-to-end: sticky per-session STT language hint (restarting an unused next session on switch), reply voice from script evidence, STT detection, then stopword heuristic; cues and WAV fallback share the turn language. - Legacy CI guards: node skip for the DOM probe, ffmpeg/codec skips for the container-fallback test. 272 voice-lane tests green. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
265 lines
11 KiB
Python
265 lines
11 KiB
Python
"""Continuous-capture contracts: no first-word clipping, stitching, overlay.
|
|
|
|
The scenarios drive the real ``atlas-voice.js`` against a stub streaming STT
|
|
WebSocket whose VAD, speculative EOS-freeze, resume and commit semantics
|
|
mirror ``hermes-jetson-stt-server.py``. Words are encoded as distinct PCM
|
|
amplitudes, so the transcript of a commit is exactly the words whose audio
|
|
survived endpointing — the live "only the first word was sent" regression is
|
|
directly observable in the composer text the probe records.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import shutil
|
|
import subprocess
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
|
|
ROOT = Path(__file__).resolve().parents[2]
|
|
VOICE_SCRIPT = ROOT / "dockerfiles" / "hermes-webui-atlas-voice.js"
|
|
PROBE = ROOT / "testing" / "probes" / "hermes_voice_capture_probe.js"
|
|
|
|
|
|
@pytest.fixture(scope="module")
|
|
def probe_results() -> dict:
|
|
node = shutil.which("node")
|
|
if not node:
|
|
pytest.skip("node is required to drive the capture continuity contract")
|
|
completed = subprocess.run(
|
|
[node, str(PROBE), str(VOICE_SCRIPT)],
|
|
check=False,
|
|
capture_output=True,
|
|
text=True,
|
|
timeout=180,
|
|
)
|
|
assert completed.returncode == 0, completed.stderr
|
|
return json.loads(completed.stdout)
|
|
|
|
|
|
def test_multiword_utterance_with_interword_pause_is_complete(probe_results):
|
|
scenario = probe_results["multiword_utterance_with_interword_pause"]
|
|
assert scenario["sends"] == ["alpha bravo"]
|
|
assert scenario["state"] == "thinking"
|
|
|
|
|
|
def test_thinking_pause_after_first_word_does_not_split(probe_results):
|
|
"""A >1100ms pause right after the first word stays inside one utterance."""
|
|
scenario = probe_results["thinking_pause_after_first_word_does_not_split"]
|
|
assert scenario["sends"] == ["alpha bravo"]
|
|
# The pause froze a first-word EOS snapshot on the server; resumed speech
|
|
# must have cleared it before the commit.
|
|
events = scenario["serverEvents"][0]
|
|
assert "resume-clear" in events or "recover-by-rms" in events
|
|
|
|
|
|
def test_barge_in_stitches_regardless_of_partial_assistant_output(probe_results):
|
|
scenario = probe_results["second_utterance_during_response_is_complete"]
|
|
assert scenario["firstSends"] == ["alpha"]
|
|
# Assistant text was already visible (and being spoken) when the user
|
|
# talked over the response: the interrupted thought and the follow-up form
|
|
# one stitched message, and the cut marker names the sentence that was
|
|
# playing so the model knows where its reply stopped being heard.
|
|
assert scenario["sends"] == [
|
|
"alpha",
|
|
"alpha bravo charlie\n[voice interruption: you were cut off after "
|
|
'"Partial answer already visible."]',
|
|
]
|
|
|
|
|
|
def test_normal_completion_clears_stitch_and_mic_stays_hot(probe_results):
|
|
scenario = probe_results["back_to_back_utterances_stay_hot"]
|
|
assert scenario["sendsAfterFirst"] == ["alpha"]
|
|
assert scenario["sends"] == ["alpha", "bravo"]
|
|
# The whole session runs on one getUserMedia lease.
|
|
assert scenario["micAcquisitions"] == 1
|
|
|
|
|
|
def test_conversation_overlay_lifecycle(probe_results):
|
|
scenario = probe_results["conversation_overlay_lifecycle"]
|
|
assert scenario["overlayPresent"] is True
|
|
assert scenario["role"] == "dialog"
|
|
assert scenario["ariaModal"] == "true"
|
|
assert scenario["captionsLive"] == "polite"
|
|
assert scenario["listeningState"] == "listening"
|
|
assert scenario["thinkingState"] in {"thinking", "speaking"}
|
|
assert scenario["userCaption"] == "alpha"
|
|
assert scenario["assistantCaption"] == "A visible answer."
|
|
# Mute drives the real capture track and is fully reversible.
|
|
assert scenario["mutedPressed"] == "true"
|
|
assert scenario["mutedTracks"] == [False]
|
|
assert scenario["unmutedTracks"] == [True]
|
|
# Both exit paths remove the overlay completely and end hands-free mode.
|
|
assert scenario["removedOnExit"] is True
|
|
assert scenario["inactiveAfterExit"] is True
|
|
assert scenario["removedOnEscape"] is True
|
|
|
|
|
|
def test_overlay_source_contract():
|
|
source = VOICE_SCRIPT.read_text(encoding="utf-8")
|
|
css = (ROOT / "dockerfiles" / "hermes-webui-atlas-voice.css").read_text(
|
|
encoding="utf-8"
|
|
)
|
|
|
|
# Lazily created, fully removed, storage-free, single-capture overlay.
|
|
assert "function openConversationOverlay()" in source
|
|
assert "function removeConversationOverlay()" in source
|
|
assert source.count("navigator.mediaDevices.getUserMedia(") == 1
|
|
overlay_region = source.split("Conversation mode overlay", 1)[1].split(
|
|
"function clearErrorTimer", 1
|
|
)[0]
|
|
assert "localStorage" not in overlay_region
|
|
assert "track.enabled=!conversation.muted" in overlay_region
|
|
assert "'aria-label':'Voice conversation'" in overlay_region
|
|
assert "'aria-live':'polite'" in overlay_region
|
|
assert "event.key==='Escape'" in overlay_region
|
|
assert "event.key==='Tab'" in overlay_region
|
|
|
|
# Orb styling: breathing, level-driven, playback-driven, reduced-motion.
|
|
assert ".voice-conversation" in css
|
|
assert "--conversation-level" in css
|
|
assert "voice-conversation-breathe" in css
|
|
assert ".voice-conversation.is-playing .voice-conversation-orb-halo" in css
|
|
reduced = css.split("@media (prefers-reduced-motion: reduce)", 1)[1]
|
|
assert ".voice-conversation *" in reduced
|
|
|
|
|
|
def test_endpointing_and_streaming_hardening_source_contract():
|
|
source = VOICE_SCRIPT.read_text(encoding="utf-8")
|
|
|
|
# Young utterances hold their endpoint past a thinking pause — unless the
|
|
# streaming partial already reads as a plausibly complete utterance
|
|
# (>=3 words or terminal punctuation), which endpoints at the base window.
|
|
assert "const VAD_EARLY_SILENCE_MS=1800" in source
|
|
assert "const VAD_COMMITTED_SPEECH_MS=1200" in source
|
|
assert "latestPartial:function(){return lastPartialText;}" in source
|
|
assert "const partialComplete=partialWords>=3||(partialWords>0" in source
|
|
assert (
|
|
"const endpointSilenceMs=(speechMs<VAD_COMMITTED_SPEECH_MS&&!partialComplete)"
|
|
"?Math.max(silenceMs,VAD_EARLY_SILENCE_MS):silenceMs" in source
|
|
)
|
|
# Resume always reaches the wire after a server-length silence gap.
|
|
assert "const SERVER_EOS_SILENCE_MS=650" in source
|
|
assert "speechGapMs>=SERVER_EOS_SILENCE_MS" in source
|
|
assert "if(settled||committed||!speculative) return;" not in source.split(
|
|
"resume:function()", 1
|
|
)[1].split("speculate:function()", 1)[0]
|
|
# No speculative Whisper pass on a one-word fragment.
|
|
assert "const SPECULATE_MIN_SPEECH_MS=700" in source
|
|
assert "speechMs>=SPECULATE_MIN_SPEECH_MS" in source
|
|
# Echo residue can no longer ratchet the onset threshold above speech.
|
|
assert "Math.min(rms,speechThreshold)*0.06" in source
|
|
|
|
|
|
def test_stitch_gate_ignores_partial_assistant_output():
|
|
source = VOICE_SCRIPT.read_text(encoding="utf-8")
|
|
region = source.split("function recordPendingStitch()", 1)[1].split(
|
|
"async function sendTranscript", 1
|
|
)[0]
|
|
assert "currentAssistantText()" not in region
|
|
assert "pendingStitch={text:sent.text,at:Date.now(),cut:cut}" in region
|
|
# Normal completion still clears the stitch context.
|
|
assert (
|
|
"if(isFinal){\n // The completion callback marks a normally "
|
|
"completed response" in source
|
|
)
|
|
|
|
|
|
def test_three_word_partial_endpoints_at_base_silence(probe_results):
|
|
scenario = probe_results["three_word_partial_endpoints_at_base_silence"]
|
|
assert scenario["sends"] == ["alpha bravo charlie"]
|
|
assert scenario["state"] == "thinking"
|
|
|
|
|
|
def test_two_word_young_utterance_keeps_the_hold(probe_results):
|
|
scenario = probe_results["two_word_young_utterance_keeps_the_hold"]
|
|
assert scenario["sendsEarly"] == []
|
|
assert scenario["sends"] == ["alpha bravo"]
|
|
|
|
|
|
def test_errored_turn_is_never_spoken_and_capture_resyncs(probe_results):
|
|
scenario = probe_results["errored_turn_is_not_spoken_and_capture_resyncs"]
|
|
assert scenario["stateAfterError"] == "listening"
|
|
assert scenario["labelAfterError"] == "Something went wrong — listening"
|
|
assert scenario["ttsDuringError"] == 0
|
|
assert scenario["freshSessions"] >= 1
|
|
# The follow-up utterance is sent alone: no stitch with the errored turn.
|
|
assert scenario["sends"] == ["alpha", "bravo"]
|
|
|
|
|
|
def test_barge_cut_marker_is_one_bounded_line(probe_results):
|
|
scenario = probe_results["barge_cut_marker_records_spoken_tail"]
|
|
assert scenario["ttsTexts"][0] == "The first point is ready."
|
|
assert scenario["sends"][1] == (
|
|
"alpha bravo\n[voice interruption: you were cut off after "
|
|
'"The first point is ready."]'
|
|
)
|
|
marker = scenario["sends"][1].split("\n", 1)[1]
|
|
assert "\n" not in marker
|
|
|
|
|
|
def test_sticky_language_biases_next_stt_session(probe_results):
|
|
scenario = probe_results["sticky_language_biases_next_stt_session"]
|
|
assert scenario["startLanguages"][0] == "auto"
|
|
assert "en" in scenario["startLanguages"]
|
|
|
|
|
|
def test_cut_marker_error_resync_and_speaking_cycles_source_contract():
|
|
source = VOICE_SCRIPT.read_text(encoding="utf-8")
|
|
|
|
# Cut marker: exactly one appended line, tail bounded to 120 characters.
|
|
assert "function voiceCutMarker(cut)" in source
|
|
assert "tail.length>120?tail.slice(tail.length-120):tail" in source
|
|
assert '\\n[voice interruption: you were cut off after "' in source
|
|
# Error envelopes are detected structurally and trigger a capture resync.
|
|
assert "function handleAssistantResponseError(token)" in source
|
|
assert "function resyncCapture(token,statusLabel)" in source
|
|
assert "'Something went wrong — listening'" in source
|
|
assert "segment.dataset.error==='1'" in source
|
|
assert ".provider-error-details" in source
|
|
# A cancellation with no in-flight stream never gates the send.
|
|
assert "if(!cancellation.streamId){" in source
|
|
# Speaking ⇄ thinking cycles inside one interim-message turn.
|
|
assert "function scheduleSpeakingIdleFallback(turn,session)" in source
|
|
assert "if(state==='speaking') setState('thinking')" in source
|
|
assert "function playbackAudible()" in source
|
|
# Body-only scraping: never the avatar letter, author name or status chip.
|
|
assert "querySelectorAll('.msg-body')" in source
|
|
assert "function readSegmentBody(segment)" in source
|
|
assert "function readAssistantTurn(turn)" in source
|
|
|
|
|
|
def test_caption_regions_are_bounded_scrollable_and_follow_tail():
|
|
source = VOICE_SCRIPT.read_text(encoding="utf-8")
|
|
css = (ROOT / "dockerfiles" / "hermes-webui-atlas-voice.css").read_text(
|
|
encoding="utf-8"
|
|
)
|
|
|
|
assert "function updateCaptionRegion(element,text,limit)" in source
|
|
assert "function attachCaptionScroll(element)" in source
|
|
assert "element.dataset.follow=gap<=24?'1':'0'" in source
|
|
for token in (
|
|
"overflow-y: auto",
|
|
"overscroll-behavior: contain",
|
|
"-webkit-overflow-scrolling: touch",
|
|
"touch-action: pan-y",
|
|
"max-height: 22vh",
|
|
"max-height: 38vh",
|
|
"mask-image",
|
|
):
|
|
assert token in css
|
|
|
|
|
|
def test_reply_language_stickiness_source_contract():
|
|
source = VOICE_SCRIPT.read_text(encoding="utf-8")
|
|
|
|
assert "let sessionLanguage=''" in source
|
|
assert "language:sessionLanguage||'auto'" in source
|
|
assert "function strongReplyLanguage(text)" in source
|
|
assert "function detectReplyLanguage(text)" in source
|
|
assert "scheduleThinkingCues(token,language||sessionLanguage,thinkingTurnId)" in source
|
|
# The sticky language resets with each hands-free session.
|
|
assert source.count("sessionLanguage='';") >= 2
|