hermes(voice): continuous mic, barge stitching, 1.15x speech

Three conversational fixes for hands-free chat:
- The microphone now stays hot for the whole session: capture runs on
  its own epoch, re-arms immediately after each utterance endpoints,
  and keeps recording through transcribing/thinking/speaking - speech
  is never lost to Hermes being busy. Speech onset during a response
  cancels it through the live capture path (echo-guarded exactly like
  the old monitor) without touching the running recorder.
- When the user talks over Hermes before any visible reply appeared,
  the interrupted utterance and the follow-up are stitched into one
  message (20s window), so the response addresses the whole thought.
- TTS speaks 15% faster by default (server-side length_scale, no pitch
  shift), user-tunable via hermes-voice-tts-speed (0.5-2.0), honored on
  streaming, WAV fallback and thinking-cue paths.

245 voice-lane tests pass; single getUserMedia site preserved;
Dockerfile grep guards verified.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
This commit is contained in:
jenkins 2026-08-24 06:27:38 -03:00
parent 17037773bf
commit 91cfb901a2
4 changed files with 276 additions and 47 deletions

View File

@ -456,6 +456,25 @@ def _handle_atlas_voice_preflight(handler):
)
def _atlas_tts_speed(body):
"""Clamp the optional client speech-rate to Piper's supported 0.5-2.0.
The value is a UX preference, not a trust decision: a missing, boolean,
non-numeric or NaN value falls back to the neutral 1.0 this proxy always
sent before the hands-free speed became client-tunable. Infinities clamp
to the range bounds like any other out-of-range number.
"""
if not isinstance(body, dict):
return 1.0
value = body.get("speed")
if isinstance(value, bool) or not isinstance(value, (int, float)):
return 1.0
speed = float(value)
if speed != speed:
return 1.0
return max(0.5, min(2.0, speed))
def _atlas_tts_stream_payload(data):
"""Build the narrow Piper payload used by the raw-PCM stream endpoint."""
if not isinstance(data, dict):
@ -466,7 +485,7 @@ def _atlas_tts_stream_payload(data):
text = text.strip()
if len(text) > 500:
raise ValueError("text too long (max 500 characters)")
payload = {"model": "piper", "input": text, "speed": 1.0}
payload = {"model": "piper", "input": text, "speed": _atlas_tts_speed(data)}
language = _atlas_tts_language(data)
if language:
payload["language"] = language
@ -745,6 +764,10 @@ atlas = ''' # ── Atlas private Jetson TTS ──────────
speed = max(0.5, min(2.0, 1.0 + (float(rate_str.rstrip("%")) / 100.0)))
except ValueError:
speed = 1.0
if isinstance(data, dict) and "speed" in data:
# Hands-free clients send an explicit validated speed; it wins
# over the legacy percentage rate string.
speed = _atlas_tts_speed(data)
request_payload = {
"model": "piper",
"input": text,

View File

@ -36,6 +36,17 @@
let visualInputLevel=0;
let streamingCapability={tts:null,stt:null,preflight:null};
let voicePreflight=null;
// Continuous-capture state. The microphone stream and its AudioContext stay
// hot for the whole hands-free session; each utterance is one "capture turn"
// guarded by captureGeneration so a response-side barge (which bumps
// `generation`) never interrupts the running microphone pipeline.
let captureGeneration=0;
let captureActive=false;
let captureGraph=null;
// Barge-in stitching: transcript whose model response was cancelled before
// any visible assistant output, plus the transcript most recently sent.
let pendingStitch=null;
let lastSentTranscript=null;
const voiceTabNonce=(function(){
try{
const bytes=new Uint8Array(16);
@ -58,6 +69,12 @@
const BARGE_LOOKBACK_MS=900;
const BARGE_DUCK_FRAMES=2;
const BARGE_TRIGGER_FRAMES=4;
// A response cancelled by barge-in leaves the user mid-thought: the next
// finalized utterance inside this window is sent as one stitched message.
const STITCH_WINDOW_MS=20000;
const TTS_SPEED_DEFAULT=1.15;
const TTS_SPEED_MIN=0.5;
const TTS_SPEED_MAX=2;
const THINKING_CUE_FIRST_MS=1900;
const THINKING_CUE_INTERVAL_MS=6500;
const THINKING_CUE_POOLS={
@ -304,11 +321,13 @@
thinkingSession=null;
thinkingTurnId='';
suppressAutoRead=false;
pendingStitch=null;
lastSentTranscript=null;
clearBargeCancellation();
stopResponseObserver();
cancelThinkingCues();
cancelSpeechTurn();
stopCapture();
releaseMicrophone();
stopPlayback();
modeBtn.classList.remove('active');
setState('error',message);
@ -320,6 +339,13 @@
}
function stopCapture(){
// Tear down the per-utterance capture pipeline (recorder, VAD, streaming
// STT, WebAudio graph) but deliberately KEEP the microphone stream and its
// AudioContext, so the next capture turn starts instantly and speech
// during transcribing/thinking is never lost. Only releaseMicrophone()
// actually ends the session's single getUserMedia lease.
captureGeneration+=1;
captureActive=false;
cancelVoicePreflight(captureTurnId);
stopBargeMonitor();
if(vadTimer){clearInterval(vadTimer);vadTimer=null;}
@ -327,8 +353,24 @@
try{recorder.stop();}catch(_){ }
}
recorder=null;
if(captureNode){try{captureNode.disconnect();}catch(_){ }}
captureNode=null;
if(captureGraph){
// The AudioContext persists across turns, so every per-turn node must be
// detached here or the shared graph would grow with each utterance.
try{if(captureGraph.mediaSource) captureGraph.mediaSource.disconnect();}catch(_){ }
try{if(captureGraph.highpass) captureGraph.highpass.disconnect();}catch(_){ }
try{if(captureGraph.analyser) captureGraph.analyser.disconnect();}catch(_){ }
try{if(captureGraph.silentGain) captureGraph.silentGain.disconnect();}catch(_){ }
captureGraph=null;
}
cancelStreamingStt();
}
function releaseMicrophone(){
// Full microphone teardown. Used only by deactivate(), showUnavailable()
// and fatal capture errors; every other path retains the hot microphone.
stopCapture();
if(stream){stream.getTracks().forEach(function(track){track.stop();});stream=null;}
if(audioContext){try{audioContext.close();}catch(_){ }audioContext=null;}
}
@ -369,11 +411,13 @@
clearErrorTimer();
clearSttLanguage();
suppressAutoRead=false;
pendingStitch=null;
lastSentTranscript=null;
clearBargeCancellation();
stopResponseObserver();
cancelThinkingCues();
cancelSpeechTurn();
stopCapture();
releaseMicrophone();
stopPlayback();
modeBtn.classList.remove('active');
setState('idle');
@ -382,7 +426,12 @@
function restartSoon(token,delay){
window.setTimeout(function(){
if(active&&token===generation) startListening(token);
if(!active||token!==generation) return;
// Continuous capture: while the microphone pipeline is already running,
// finishing a response only needs the display returned to Listening.
// Rebuilding capture here would drop speech already being collected.
if(captureActive){setState('listening');return;}
startListening(token);
},delay||500);
}
@ -422,31 +471,56 @@
return false;
}
async function sendTranscript(transcript,token,language){
function recordPendingStitch(){
// Called at the moment a model turn is cancelled by user speech. Stitching
// only applies when the cancelled turn produced NO visible assistant text:
// the model never answered, so the next utterance restates the complete
// thought as one message. Any partial answer means the interruption stands
// on its own and the next utterance is sent alone.
const sent=lastSentTranscript;
lastSentTranscript=null;
if(!sent||!sent.text) return;
if(currentAssistantText()){pendingStitch=null;return;}
pendingStitch={text:sent.text,at:Date.now()};
}
async function sendTranscript(transcript,token,language,turnId){
if(!active||token!==generation) return;
cancelVoicePreflight(captureTurnId);
cancelVoicePreflight(turnId||captureTurnId);
const text=String(transcript||'').trim();
if(!text){clearSttLanguage();restartSoon(token,350);return;}
composer.value=text;
if(typeof window.autoResize==='function') window.autoResize();
setState('thinking');
startBargeMonitor(token);
if(!bargeCancelPromise&&typeof S!=='undefined'&&(S.busy||S.activeStreamId)){
// A new utterance finished while the previous model turn was still in
// flight (continuous capture makes this a normal interruption): remember
// the interrupted transcript for stitching, then cancel the stale turn.
recordPendingStitch();
suppressAutoRead=true;
bargeCancelPromise=cancelActiveModelTurn();
}
// A turn whose response was barge-cancelled before any assistant output is
// restated as one stitched message, so the model answers the full thought.
const stitch=(pendingStitch&&pendingStitch.text&&(Date.now()-pendingStitch.at)<=STITCH_WINDOW_MS)?pendingStitch:null;
pendingStitch=null;
composer.value=stitch?stitch.text+' '+text:text;
if(typeof window.autoResize==='function') window.autoResize();
setState('thinking');
// The live VAD/STT capture is the barge-in detector while it runs; the
// energy-only monitor remains only as a fallback when capture is down.
if(!captureActive) startBargeMonitor(token);
const cancellationSettled=await settleBargeCancellation(token);
if(!active||token!==generation) return;
if(!cancellationSettled){
pendingStitch=stitch;
toast('The previous response did not stop. Please repeat your interruption.');
restartSoon(token,250);
return;
}
thinkingSession=(typeof S!=='undefined'&&S.session)?S.session.session_id:null;
thinkingTurnId=captureTurnId||String(token)+'-'+String(++turnSequence);
thinkingTurnId=turnId||captureTurnId||String(token)+'-'+String(++turnSequence);
rememberAssistantBaseline();
rememberSttLanguage(language,token);
if(typeof window.send==='function'){
lastSentTranscript={text:composer.value,token:token};
window.send();
suppressAutoRead=false;
startResponseObserver(token);
@ -462,7 +536,7 @@
return 'webm';
}
async function transcribe(blob,token){
async function transcribe(blob,token,turnId){
if(!active||token!==generation) return;
setState('transcribing');
const ext=audioExtension(blob.type);
@ -472,7 +546,7 @@
const response=await fetch('/api/transcribe',{method:'POST',body:form});
const payload=await response.json().catch(function(){return {};});
if(!response.ok) throw new Error(payload.error||('Whisper request failed: '+response.status));
sendTranscript(payload.transcript,token,normalizeSttLanguage(payload.language));
sendTranscript(payload.transcript,token,normalizeSttLanguage(payload.language),turnId);
}catch(error){
if(!active||token!==generation) return;
const message=errorMessage(error,'Private Whisper is unavailable');
@ -754,6 +828,7 @@
node.connect(silentGain);
silentGain.connect(context.destination);
captureNode=node;
if(captureGraph) captureGraph.silentGain=silentGain;
session.setWorkletReady(true);
return true;
}catch(_){
@ -811,6 +886,7 @@
// during AudioWorklet/session setup.
monitor.handoff=true;
const oldToken=monitor.token;
recordPendingStitch();
generation+=1;
const token=generation;
thinkingSession=null;
@ -836,6 +912,30 @@
});
}
function bargeFromLiveCapture(){
// Live-capture barge-in: the always-on VAD/STT listener detected a real
// speech onset while Hermes was mid-response. Cancel the RESPONSE side
// only — cues, playback, the model stream — and bump `generation` so every
// stale response-side continuation dies. The already-running capture is
// guarded by captureGeneration and keeps collecting the interrupting
// utterance without losing a syllable.
if(!active||(state!=='thinking'&&state!=='speaking')) return;
recordPendingStitch();
generation+=1;
thinkingSession=null;
thinkingTurnId='';
suppressAutoRead=true;
clearSttLanguage();
stopResponseObserver();
cancelThinkingCues();
cancelSpeechTurn();
stopPlayback();
if(typeof window.stopTTS==='function') window.stopTTS();
clearBargeCancellation();
bargeCancelPromise=cancelActiveModelTurn();
setState('listening','Listening — interrupted');
}
async function startBargeMonitor(token){
if(!active||token!==generation||(state!=='thinking'&&state!=='speaking')||bargeMonitor) return;
const monitor={token:token,cancelled:false,handoff:false,aecUsable:true,timer:null,stream:null,context:null,source:null,node:null,silentGain:null,lookback:[],lookbackFrames:0};
@ -915,7 +1015,7 @@
}
}
async function transcribeStreamingOrFallback(blob,token,session,allowContainerFallback){
async function transcribeStreamingOrFallback(blob,token,session,allowContainerFallback,turnId){
if(!active||token!==generation) return;
setState('transcribing');
let pcmFallback=null;
@ -924,8 +1024,8 @@
const result=await session.commit();
if(!String(result.transcript||'').trim()) throw new Error('Streaming transcription returned no final text');
if(active&&token===generation){
streamingStt=null;
sendTranscript(result.transcript,token,result.language);
if(streamingStt===session) streamingStt=null;
sendTranscript(result.transcript,token,result.language,turnId);
return;
}
}catch(_){
@ -936,7 +1036,7 @@
if(streamingStt===session) streamingStt=null;
}
if(pcmFallback){
transcribe(pcmFallback,token);
transcribe(pcmFallback,token,turnId);
return;
}
if(allowContainerFallback===false){
@ -944,29 +1044,54 @@
restartSoon(token,250);
return;
}
transcribe(blob,token);
transcribe(blob,token,turnId);
}
async function startListening(token,reusedCapture){
if(!active||token!==generation) return;
stopCapture();
stopPlayback();
cancelSpeechTurn();
clearSttLanguage();
// A continuous-capture restart (reusedCapture.preserveDisplay) happens the
// instant an utterance finalizes, while the display legitimately shows
// Hermes's own activity (transcribing/thinking/speaking). Capture is
// therefore tracked by captureActive, never by the visible state, and the
// response-side teardown below is skipped so that turn is not disturbed.
const preserveDisplay=!!(reusedCapture&&reusedCapture.preserveDisplay);
if(!preserveDisplay){
stopPlayback();
cancelSpeechTurn();
clearSttLanguage();
}
captureTurnId=(voiceTabNonce?voiceTabNonce+'-':'')+String(token)+'-'+String(++turnSequence);
setState('listening');
// Capture turns carry their own epoch: a response-side barge bumps
// `generation` but never this counter, so capture survives it seamlessly.
const captureToken=++captureGeneration;
captureActive=true;
if(!preserveDisplay) setState('listening');
try{
const capture=reusedCapture&&reusedCapture.stream?reusedCapture.stream:await acquireMicrophone();
if(!active||token!==generation){
// A barge-monitor handoff supplies its own stream/context; otherwise the
// retained session microphone is reused. Only a fresh hands-free session
// actually asks for the microphone again, keeping acquireMicrophone()
// the single getUserMedia call site.
if(reusedCapture&&reusedCapture.stream&&stream&&reusedCapture.stream!==stream){
stream.getTracks().forEach(function(track){track.stop();});
stream=null;
}
if(reusedCapture&&reusedCapture.context&&audioContext&&reusedCapture.context!==audioContext){
try{audioContext.close();}catch(_){ }
audioContext=null;
}
const capture=(reusedCapture&&reusedCapture.stream)||stream||await acquireMicrophone();
if(!active||captureToken!==captureGeneration){
if(reusedCapture&&reusedCapture.handoffMonitor) disposeBargeResources(reusedCapture.handoffMonitor,false);
else capture.getTracks().forEach(function(track){track.stop();});
else if(capture!==stream) capture.getTracks().forEach(function(track){track.stop();});
return;
}
stream=capture;
const captureAec=captureAecIsUsable(stream);
const Context=window.AudioContext||window.webkitAudioContext;
if(reusedCapture&&reusedCapture.context){
audioContext=reusedCapture.context;
}else{
}else if(!audioContext){
try{
// Let the browser's native resampler produce Whisper's 16 kHz input.
audioContext=new Context({sampleRate:16000,latencyHint:'interactive'});
@ -983,6 +1108,7 @@
const mediaSource=audioContext.createMediaStreamSource(stream);
mediaSource.connect(highpass);
highpass.connect(analyser);
captureGraph={mediaSource:mediaSource,highpass:highpass,analyser:analyser,silentGain:null};
streamingStt=createStreamingSttSession(captureTurnId,audioContext);
// Whisper receives the browser's full-band processed microphone signal;
// the 140 Hz high-pass remains a VAD-only aid so low voices are not
@ -1010,6 +1136,9 @@
let noiseFloor=0.008;
let lastSpeech=Date.now();
let speculative=false;
let ducked=false;
let bargeArmAt=0;
let playbackWasLive=false;
const started=Date.now();
recorder=new MediaRecorder(stream,mime?{mimeType:mime}:undefined);
let recordedMime=recorder.mimeType||mime||'';
@ -1027,23 +1156,39 @@
new Promise(function(resolve){window.setTimeout(resolve,100);}),
]);
}
const recordedStream=stream;
stream=null;
if(recordedStream) recordedStream.getTracks().forEach(function(track){track.stop();});
const context=audioContext;
audioContext=null;
if(context){try{context.close();}catch(_){ }}
captureNode=null;
recorder=null;
if(!active||token!==generation){cancelStreamingStt();return;}
if(!heardSpeech||!chunks.length){cancelStreamingStt();restartSoon(token,300);return;}
// A stale onstop (teardown by stopCapture/releaseMicrophone, or a
// newer capture turn already running) must not touch the globals: by
// now they can belong to the NEXT turn, and its session was already
// cancelled by stopCapture. Real recorders fire onstop asynchronously,
// so this guard has to come before anything else is read.
if(!active||captureToken!==captureGeneration) return;
// Continuous capture: the microphone stream and AudioContext stay hot.
// Detach only this utterance's session, then re-enter capture below so
// speech during transcribing/thinking becomes the next utterance.
const utteranceTurnId=captureTurnId;
const session=streamingStt;
streamingStt=null;
recorder=null;
if(!heardSpeech||!chunks.length){
if(session) session.cancel();
// Nothing worth transcribing: recycle the recorder on the hot mic.
startListening(generation,{preserveDisplay:true});
return;
}
// The utterance is dispatched under the CURRENT response epoch: when a
// live-capture barge just cancelled a model turn, this transcript is
// the interruption that replaces it.
transcribeStreamingOrFallback(
new Blob(chunks,{type:recordedMime||'audio/webm'}),
token,
generation,
session,
!(reusedCapture&&reusedCapture.requireStreamingLookback)
!(reusedCapture&&reusedCapture.requireStreamingLookback),
utteranceTurnId
);
// Re-enter capture immediately — never wait for transcription or the
// response. The display may show transcribing/thinking while the next
// capture turn is already live underneath.
startListening(generation,{preserveDisplay:true});
};
// Ask the browser for one finalized container at stop. Android Chromium
// can emit timeslice fragments without a reusable EBML initialization
@ -1054,7 +1199,7 @@
const silenceMs=Math.max(900,parseInt(localStorage.getItem('hermes-voice-silence-ms')||'1100',10)||1100);
const speculateMs=Math.min(silenceMs-250,Math.max(450,Math.round(silenceMs*0.55)));
vadTimer=window.setInterval(function(){
if(!active||token!==generation||!recorder||recorder.state==='inactive') return;
if(!active||captureToken!==captureGeneration||!recorder||recorder.state==='inactive') return;
analyser.getByteTimeDomainData(samples);
let energy=0;
for(let index=0;index<samples.length;index+=1){
@ -1064,13 +1209,39 @@
const rms=Math.sqrt(energy/samples.length);
updateInputLevel(rms);
const now=Date.now();
// Continuous capture stays live while Hermes thinks and speaks, so the
// same VAD that endpoints utterances is also the barge-in detector.
// While our own audio plays the microphone hears the speakers: without
// AEC nothing counts as speech at all, and with AEC a post-playback
// arming delay plus the sustained-frame requirement below mirror the
// conservative energy monitor this path supersedes.
const playbackLive=indicator.classList.contains('is-playing')||!!currentAudio||!!(playbackSession&&playbackSession.node)||!!(thinkingCue&&thinkingCue.node);
if(playbackLive&&!playbackWasLive){
bargeArmAt=now+(state==='thinking'?150:450);
}else if(!playbackLive&&playbackWasLive){
bargeArmAt=now+100;
}
playbackWasLive=playbackLive;
const speechThreshold=Math.max(0.04,noiseFloor*2.4+0.006);
const voiceNow=rms>speechThreshold;
const voiceNow=rms>speechThreshold&&(!playbackLive||(captureAec&&now>=bargeArmAt));
if(!heardSpeech&&!voiceNow){noiseFloor=(noiseFloor*0.94)+(rms*0.06);}
voiceFrames=voiceNow?Math.min(voiceFrames+1,5):Math.max(voiceFrames-1,0);
if(!heardSpeech&&voiceFrames>=3){
if(playbackLive){
if(voiceFrames>=BARGE_DUCK_FRAMES&&!ducked){ducked=true;setPlaybackDucked(true);}
if(!voiceFrames&&ducked){ducked=false;setPlaybackDucked(false);}
}else if(ducked){
ducked=false;
setPlaybackDucked(false);
}
// Speech onset. While playback runs an echo-safe barge needs the same
// sustained evidence (BARGE_TRIGGER_FRAMES) as the old energy monitor.
const onsetFrames=playbackLive?BARGE_TRIGGER_FRAMES:3;
if(!heardSpeech&&voiceFrames>=onsetFrames){
heardSpeech=true;
lastSpeech=now;
// The user started a new utterance while Hermes was mid-response:
// cancel the response side only; this capture keeps running.
if(state==='thinking'||state==='speaking') bargeFromLiveCapture();
}else if(heardSpeech&&voiceNow){
lastSpeech=now;
if(speculative&&streamingStt){streamingStt.resume();speculative=false;}
@ -1092,7 +1263,10 @@
disposeBargeResources(reusedCapture.handoffMonitor,true);
reusedCapture.handoffMonitor=null;
}
if(!active||token!==generation) return;
if(!active||captureToken!==captureGeneration) return;
// This capture turn is dead. A fatal microphone error ends the session
// and releases the retained microphone via showUnavailable().
captureActive=false;
const message=errorMessage(error,'Microphone permission is required');
showUnavailable(message);
toast(message);
@ -1232,8 +1406,17 @@
});
}
function ttsSpeed(){
// Hands-free speech-rate preference. Piper accepts 0.5-2.0 (length_scale
// 1/speed, no pitch shift); hands-free defaults to a slightly brisk 1.15,
// seeded once at initialize() and user-tunable via localStorage.
const stored=parseFloat(localStorage.getItem('hermes-voice-tts-speed')||'');
if(!Number.isFinite(stored)) return TTS_SPEED_DEFAULT;
return Math.min(TTS_SPEED_MAX,Math.max(TTS_SPEED_MIN,stored));
}
function ttsRequest(chunk,language,turnId){
const request={text:chunk,engine:'atlas',turn_id:turnId};
const request={text:chunk,engine:'atlas',turn_id:turnId,speed:ttsSpeed()};
if(language) request.language=language;
return request;
}
@ -1624,6 +1807,11 @@
function pumpAssistantResponse(token,isFinal){
if(!active||token!==generation||(state!=='thinking'&&state!=='speaking')) return;
if(isFinal){
// The completion callback marks a normally completed response: any
// interrupted-thought stitch from an earlier barge is now stale.
pendingStitch=null;
}
const currentSession=(typeof S!=='undefined'&&S.session)?S.session.session_id:null;
if(thinkingSession&&currentSession&&thinkingSession!==currentSession){
thinkingSession=null;
@ -1758,6 +1946,11 @@
localStorage.setItem('hermes-atlas-voice-latency-v2','1');
localStorage.setItem('hermes-voice-silence-ms','1100');
}
// Seed the hands-free speech-rate default once; later user changes to
// hermes-voice-tts-speed (clamped to 0.5-2.0) are respected as-is.
if(localStorage.getItem('hermes-voice-tts-speed')===null){
localStorage.setItem('hermes-voice-tts-speed',String(TTS_SPEED_DEFAULT));
}
const selector=document.getElementById('settingsTtsEngine');
if(selector&&!selector.querySelector('option[value="atlas"]')){
const option=document.createElement('option');

View File

@ -158,7 +158,10 @@ def test_visual_slice_preserves_private_voice_request_and_capture_contract():
assert script.count("navigator.mediaDevices.getUserMedia(") == 1
assert "form.append('file',new File([blob],'voice-input.'+ext" in script
assert "fetch('/api/transcribe',{method:'POST',body:form})" in script
assert "const request={text:chunk,engine:'atlas',turn_id:turnId}" in script
assert (
"const request={text:chunk,engine:'atlas',turn_id:turnId,speed:ttsSpeed()}"
in script
)
assert "if(language) request.language=language" in script
assert "speakResponse(generation)" in script
assert "window._voiceModeImmediateSend" in script
@ -200,4 +203,4 @@ def test_release_candidate_keeps_finalized_webm_as_quality_fallback():
assert "recorder.start(250)" not in script
assert "new Blob(chunks,{type:recordedMime||'audio/webm'})" in script
assert "transcribeStreamingOrFallback" in script
assert "transcribe(blob,token)" in script
assert "transcribe(blob,token,turnId)" in script

View File

@ -624,7 +624,7 @@ def test_voice_mode_drops_hostile_language_values(voice_probe):
def test_voice_mode_never_sends_a_voice_field(voice_probe):
for request in voice_probe["voice_field_is_never_sent"]["tts"]:
assert set(request) <= {"text", "engine", "language", "turn_id"}
assert set(request) <= {"text", "engine", "language", "turn_id", "speed"}
assert "voice" not in request
@ -797,10 +797,20 @@ def test_streaming_tts_payload_is_narrow_and_turn_bound(patched_webui):
assert payload == {
"model": "piper",
"input": "A safe sentence.",
"speed": 1.0,
"speed": 2.0,
"language": "ru",
"turn_id": "voice-turn-7",
}
assert patched_webui.routes._atlas_tts_stream_payload(
{"text": "A safe sentence.", "speed": 1.15}
)["speed"] == 1.15
assert patched_webui.routes._atlas_tts_stream_payload(
{"text": "A safe sentence."}
)["speed"] == 1.0
for hostile in ("2", True, None, [1.5], {"speed": 1.5}, float("nan")):
assert patched_webui.routes._atlas_tts_stream_payload(
{"text": "A safe sentence.", "speed": hostile}
)["speed"] == 1.0
def test_streaming_tts_payload_forwards_only_allowlisted_localized_cues(patched_webui):