hermes(voice): interim-ack tail, natural fillers, unified audio + output picker
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 72, passed: 3724
Some checks failed
Tests / Declarative: Post Actions failed: 49, skipped: 72, passed: 3724
Three conversation-mode fixes in one pass: - Interim-acknowledgement truncation: when an interim ack folds into the hidden worklog segment mid-speech, the retained unspoken tail is now flushed and spoken in full before the Thinking transition, and a distinct follow-up message is chunked from its own start and queued after the interim drains (no more 'stops after the first clause, rest resurfaces with the next message'). - Natural thinking fillers: brief per-language interjections (Umm/Hmm/ One sec; Mmm/A ver; Хм/Секунду) on genuine >1.9s thinking gaps only, non-repeating, answer-preempting, mute-aware. - One unified audio sink for every spoken output (reply, cues, fillers, WAV fallback) - fixes cues playing the loudspeaker while the reply used a different output - plus a tidy corner output-device selector (enumerateDevices + setSinkId, feature-detected, session-only) styled like the language selector. Also realigns two STT-server decode-param assertions to the dict form from the STT tuning commit. 284 voice tests. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
This commit is contained in:
parent
64b7bc55cf
commit
181a7517c2
@ -688,6 +688,120 @@
|
||||
color: rgb(var(--voice-accent));
|
||||
}
|
||||
|
||||
/* Output-device selector: a twin of the language chooser, seated just left of
|
||||
it in the same corner. Feature-detected and hidden entirely when the browser
|
||||
cannot enumerate/route outputs, so it never appears as a dead control. */
|
||||
.voice-conversation-out {
|
||||
position: absolute;
|
||||
top: calc(16px + env(safe-area-inset-top, 0px));
|
||||
right: calc(64px + env(safe-area-inset-right, 0px));
|
||||
z-index: 2;
|
||||
}
|
||||
|
||||
.voice-conversation-out-btn {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: center;
|
||||
width: 40px;
|
||||
height: 40px;
|
||||
padding: 0;
|
||||
color: rgba(226, 238, 250, 0.72);
|
||||
background: rgba(255, 255, 255, 0.06);
|
||||
border: 1px solid rgba(var(--voice-accent), 0.28);
|
||||
border-radius: 50%;
|
||||
cursor: pointer;
|
||||
transition: color 160ms ease, background 160ms ease, border-color 160ms ease;
|
||||
}
|
||||
|
||||
.voice-conversation-out-btn:hover {
|
||||
color: rgba(240, 248, 255, 0.95);
|
||||
background: rgba(255, 255, 255, 0.12);
|
||||
}
|
||||
|
||||
.voice-conversation-out-btn:focus-visible {
|
||||
outline: 2px solid rgb(var(--voice-accent));
|
||||
outline-offset: 2px;
|
||||
}
|
||||
|
||||
/* A non-default output device is marked with the same accent dot. */
|
||||
.voice-conversation-out-btn.is-forced {
|
||||
color: rgb(var(--voice-accent));
|
||||
border-color: rgba(var(--voice-accent), 0.7);
|
||||
}
|
||||
|
||||
.voice-conversation-out-btn.is-forced::after {
|
||||
content: "";
|
||||
position: absolute;
|
||||
top: 2px;
|
||||
right: 2px;
|
||||
width: 8px;
|
||||
height: 8px;
|
||||
border-radius: 50%;
|
||||
background: rgb(var(--voice-accent));
|
||||
box-shadow: 0 0 6px rgba(var(--voice-accent), 0.8);
|
||||
}
|
||||
|
||||
.voice-conversation-out-menu {
|
||||
position: absolute;
|
||||
top: 46px;
|
||||
right: 0;
|
||||
min-width: 180px;
|
||||
max-width: 260px;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 2px;
|
||||
padding: 6px;
|
||||
background: rgba(16, 22, 36, 0.96);
|
||||
border: 1px solid rgba(var(--voice-accent), 0.28);
|
||||
border-radius: 14px;
|
||||
box-shadow: 0 18px 44px rgba(0, 0, 0, 0.5);
|
||||
backdrop-filter: blur(8px);
|
||||
}
|
||||
|
||||
.voice-conversation-out-menu[hidden] {
|
||||
display: none;
|
||||
}
|
||||
|
||||
.voice-conversation-out-item {
|
||||
display: flex;
|
||||
align-items: center;
|
||||
justify-content: space-between;
|
||||
gap: 10px;
|
||||
font: inherit;
|
||||
font-size: 14px;
|
||||
font-weight: 550;
|
||||
text-align: left;
|
||||
color: rgba(231, 241, 251, 0.9);
|
||||
background: transparent;
|
||||
border: 0;
|
||||
border-radius: 9px;
|
||||
padding: 9px 12px;
|
||||
cursor: pointer;
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
white-space: nowrap;
|
||||
}
|
||||
|
||||
.voice-conversation-out-item:hover {
|
||||
background: rgba(255, 255, 255, 0.08);
|
||||
}
|
||||
|
||||
.voice-conversation-out-item:focus-visible {
|
||||
outline: 2px solid rgb(var(--voice-accent));
|
||||
outline-offset: -2px;
|
||||
}
|
||||
|
||||
.voice-conversation-out-item[aria-checked="true"] {
|
||||
color: #fff;
|
||||
background: rgba(var(--voice-accent), 0.16);
|
||||
}
|
||||
|
||||
.voice-conversation-out-item[aria-checked="true"]::after {
|
||||
content: "✓";
|
||||
font-size: 13px;
|
||||
color: rgb(var(--voice-accent));
|
||||
}
|
||||
|
||||
.voice-conversation.is-muted .voice-conversation-orb-halo,
|
||||
.voice-conversation.is-muted .voice-conversation-orb-core {
|
||||
animation: none;
|
||||
|
||||
@ -36,6 +36,18 @@
|
||||
let visualInputLevel=0;
|
||||
let streamingCapability={tts:null,stt:null,preflight:null};
|
||||
let voicePreflight=null;
|
||||
// Unified spoken-output sink. Every spoken sound — the reply's streaming PCM,
|
||||
// the thinking-cue PCM, and the WAV blob fallback — plays through this ONE
|
||||
// shared AudioContext (and, for the blob element, the same chosen sinkId), so
|
||||
// the reply and its fillers can never land on different hardware outputs
|
||||
// (the earpiece-vs-loudspeaker split). The context is created lazily on first
|
||||
// playback, persists across cues/segments/turns, and is closed only when the
|
||||
// hands-free session ends. selectedOutputSinkId is the user-chosen audio
|
||||
// output device (session-only; '' means the system default).
|
||||
let sharedPlaybackContext=null;
|
||||
let sharedPlaybackWorklet=null;
|
||||
let selectedOutputSinkId='';
|
||||
let outputDevices=[];
|
||||
// Continuous-capture state. The microphone stream and its AudioContext stay
|
||||
// hot for the whole hands-free session; each utterance is one "capture turn"
|
||||
// guarded by captureGeneration so a response-side barge (which bumps
|
||||
@ -100,10 +112,15 @@
|
||||
const TTS_SPEED_MAX=2;
|
||||
const THINKING_CUE_FIRST_MS=1900;
|
||||
const THINKING_CUE_INTERVAL_MS=6500;
|
||||
// Natural spoken fillers first (Umm/Hmm/One sec and their es/ru equivalents),
|
||||
// then the longer reassurances — one short filler per genuine >~1.9s thinking
|
||||
// gap (THINKING_CUE_FIRST_MS), the answer always preempting via
|
||||
// cancelThinkingCues, rotated deterministically per turn so they never repeat
|
||||
// back-to-back, spoken through the SAME unified sink as the reply.
|
||||
const THINKING_CUE_POOLS={
|
||||
en:[{id:'thinking',text:"I'm thinking."},{id:'let_me_think',text:'Let me think.'},{id:'still_working',text:'Still working on that.'},{id:'one_more_moment',text:'One more moment.'}],
|
||||
ru:[{id:'thinking',text:'Я думаю.'},{id:'let_me_think',text:'Дайте подумать.'},{id:'still_working',text:'Я всё ещё думаю над этим.'},{id:'one_more_moment',text:'Ещё мгновение.'}],
|
||||
es:[{id:'thinking',text:'Estoy pensando.'},{id:'let_me_think',text:'Déjame pensar.'},{id:'still_working',text:'Sigo pensando en eso.'},{id:'one_more_moment',text:'Un momento más.'}],
|
||||
en:[{id:'umm',text:'Umm.'},{id:'hmm',text:'Hmm.'},{id:'one_sec',text:'One sec.'},{id:'thinking',text:"I'm thinking."},{id:'let_me_think',text:'Let me think.'},{id:'still_working',text:'Still working on that.'},{id:'one_more_moment',text:'One more moment.'}],
|
||||
ru:[{id:'hmm',text:'Хм.'},{id:'sec',text:'Секунду.'},{id:'minute',text:'Минутку.'},{id:'thinking',text:'Я думаю.'},{id:'let_me_think',text:'Дайте подумать.'},{id:'still_working',text:'Я всё ещё думаю над этим.'},{id:'one_more_moment',text:'Ещё мгновение.'}],
|
||||
es:[{id:'mmm',text:'Mmm.'},{id:'a_ver',text:'A ver.'},{id:'un_momento',text:'Un momento.'},{id:'thinking',text:'Estoy pensando.'},{id:'let_me_think',text:'Déjame pensar.'},{id:'still_working',text:'Sigo pensando en eso.'},{id:'one_more_moment',text:'Un momento más.'}],
|
||||
};
|
||||
const STATE_LABELS={
|
||||
listening:'Listening',
|
||||
@ -145,6 +162,7 @@
|
||||
{code:'ru',label:'Русский',sublabel:'Russian'},
|
||||
];
|
||||
const GLOBE_ICON_SVG='<svg viewBox="0 0 20 20" width="18" height="18" fill="none" stroke="currentColor" stroke-width="1.5" aria-hidden="true" focusable="false"><circle cx="10" cy="10" r="7.25"/><path d="M2.9 10h14.2M10 2.75c1.9 2 2.9 4.6 2.9 7.25S11.9 15.25 10 17.25C8.1 15.25 7.1 12.65 7.1 10S8.1 4.75 10 2.75z"/></svg>';
|
||||
const SPEAKER_ICON_SVG='<svg viewBox="0 0 20 20" width="18" height="18" fill="none" stroke="currentColor" stroke-width="1.5" stroke-linejoin="round" aria-hidden="true" focusable="false"><path d="M4 7.5h2.5L10.5 4v12L6.5 12.5H4z"/><path d="M13.2 7.4a3.5 3.5 0 0 1 0 5.2M15.4 5.3a6.5 6.5 0 0 1 0 9.4"/></svg>';
|
||||
const originalAutoRead=window.autoReadLastAssistant;
|
||||
const originalApplyPreference=window._applyVoiceModePref;
|
||||
|
||||
@ -350,12 +368,15 @@
|
||||
// Escape peels one layer at a time: an open language menu closes first
|
||||
// (focus returns to its button), and only a second Escape exits the mode.
|
||||
if(conversation.langMenu&&!conversation.langMenu.hidden){closeLanguageMenu(true);return;}
|
||||
if(conversation.outMenu&&!conversation.outMenu.hidden){closeOutputMenu(true);return;}
|
||||
deactivate(true);
|
||||
return;
|
||||
}
|
||||
if(event.key==='Tab'){
|
||||
// Minimal focus trap across the overlay controls (language, mute, exit).
|
||||
const stops=[conversation.langBtn,conversation.muteBtn,conversation.exitBtn].filter(Boolean);
|
||||
// Minimal focus trap across the overlay controls (language, output, mute,
|
||||
// exit); the output button only joins the trap once it is actually shown.
|
||||
const outVisible=conversation.outWrap&&conversation.outWrap.style&&conversation.outWrap.style.display!=='none';
|
||||
const stops=[conversation.langBtn,outVisible?conversation.outBtn:null,conversation.muteBtn,conversation.exitBtn].filter(Boolean);
|
||||
const current=stops.indexOf(document.activeElement);
|
||||
const index=current<0?(event.shiftKey?0:stops.length-1):current;
|
||||
const next=stops[(index+(event.shiftKey?stops.length-1:1))%stops.length];
|
||||
@ -451,6 +472,118 @@
|
||||
return {wrap:wrap,btn:btn,menu:menu,items:items};
|
||||
}
|
||||
|
||||
// ── Output-device selector ────────────────────────────────────────────
|
||||
// A tidy corner control, styled like the language chooser, to pick which audio
|
||||
// output device the voice plays through. Feature-detected (enumerateDevices +
|
||||
// setSinkId); the whole control stays hidden when routing is unsupported or no
|
||||
// output devices are enumerable. Session-only — it never touches storage.
|
||||
function closeOutputMenu(focusButton){
|
||||
if(!conversation||!conversation.outMenu) return;
|
||||
conversation.outMenu.hidden=true;
|
||||
conversation.outMenu.setAttribute('hidden','');
|
||||
if(conversation.outBtn){
|
||||
conversation.outBtn.setAttribute('aria-expanded','false');
|
||||
if(focusButton&&conversation.outBtn.focus) conversation.outBtn.focus();
|
||||
}
|
||||
}
|
||||
|
||||
function openOutputMenu(){
|
||||
if(!conversation||!conversation.outMenu) return;
|
||||
conversation.outMenu.hidden=false;
|
||||
conversation.outMenu.removeAttribute('hidden');
|
||||
if(conversation.outBtn) conversation.outBtn.setAttribute('aria-expanded','true');
|
||||
const items=conversation.outItems||[];
|
||||
const active=items.filter(function(item){return item.getAttribute('aria-checked')==='true';})[0]||items[0];
|
||||
if(active&&active.focus) active.focus();
|
||||
}
|
||||
|
||||
function toggleOutputMenu(){
|
||||
if(!conversation||!conversation.outMenu) return;
|
||||
if(conversation.outMenu.hidden) openOutputMenu(); else closeOutputMenu(true);
|
||||
}
|
||||
|
||||
function outputDeviceLabel(device,index){
|
||||
const label=device&&device.label?String(device.label):'';
|
||||
if(label) return label;
|
||||
return 'Output '+(index+1);
|
||||
}
|
||||
|
||||
function reflectOutputSelection(){
|
||||
if(!conversation||!conversation.outItems) return;
|
||||
conversation.outItems.forEach(function(item){
|
||||
const selected=(item.getAttribute('data-device')||'')===selectedOutputSinkId;
|
||||
item.setAttribute('aria-checked',selected?'true':'false');
|
||||
});
|
||||
if(conversation.outBtn){
|
||||
const chosen=outputDevices.filter(function(device){return device.deviceId===selectedOutputSinkId;})[0];
|
||||
const label=selectedOutputSinkId&&chosen?outputDeviceLabel(chosen,0):'System default';
|
||||
conversation.outBtn.setAttribute('aria-label','Audio output: '+label);
|
||||
conversation.outBtn.setAttribute('title','Output — '+label);
|
||||
if(conversation.outBtn.classList){
|
||||
if(selectedOutputSinkId) conversation.outBtn.classList.add('is-forced');
|
||||
else conversation.outBtn.classList.remove('is-forced');
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function selectOutputDevice(deviceId){
|
||||
selectedOutputSinkId=deviceId||'';
|
||||
reflectOutputSelection();
|
||||
applyOutputSink();
|
||||
}
|
||||
|
||||
async function refreshOutputDevices(){
|
||||
if(!conversation||!conversation.outMenu||!conversation.outWrap) return;
|
||||
if(!outputRoutingSupported()){conversation.outWrap.style.display='none';return;}
|
||||
let devices=[];
|
||||
try{devices=await navigator.mediaDevices.enumerateDevices();}catch(_){devices=[];}
|
||||
// Real, routable outputs only: the empty-deviceId "default" device the
|
||||
// browser reports is represented by our own "System default" entry.
|
||||
outputDevices=(devices||[]).filter(function(device){return device&&device.kind==='audiooutput'&&device.deviceId;});
|
||||
if(!conversation||!conversation.outMenu) return;
|
||||
// Rebuild the menu: a "System default" entry plus every routable output.
|
||||
conversation.outMenu.children=[];
|
||||
const entries=[{deviceId:'',label:'System default'}].concat(
|
||||
outputDevices.map(function(device,index){return {deviceId:device.deviceId,label:outputDeviceLabel(device,index)};})
|
||||
);
|
||||
conversation.outItems=entries.map(function(entry){
|
||||
const item=conversationNode('button','voice-conversation-out-item',{
|
||||
type:'button',role:'menuitemradio','data-device':entry.deviceId,
|
||||
'aria-checked':entry.deviceId===selectedOutputSinkId?'true':'false',
|
||||
});
|
||||
item.textContent=entry.label;
|
||||
item.addEventListener('click',function(){
|
||||
selectOutputDevice(entry.deviceId);
|
||||
closeOutputMenu(true);
|
||||
});
|
||||
conversation.outMenu.appendChild(item);
|
||||
return item;
|
||||
});
|
||||
// Only the default entry means nothing is really selectable — hide it.
|
||||
conversation.outWrap.style.display=outputDevices.length?'':'none';
|
||||
reflectOutputSelection();
|
||||
}
|
||||
|
||||
function buildOutputControl(){
|
||||
const wrap=conversationNode('div','voice-conversation-out');
|
||||
// Hidden until refreshOutputDevices confirms real, routable outputs exist.
|
||||
wrap.style.display='none';
|
||||
const btn=conversationNode('button','voice-conversation-out-btn',{
|
||||
type:'button','aria-haspopup':'menu','aria-expanded':'false',
|
||||
'aria-label':'Audio output: System default',title:'Output — System default',
|
||||
});
|
||||
btn.innerHTML=SPEAKER_ICON_SVG;
|
||||
const menu=conversationNode('div','voice-conversation-out-menu',{role:'menu','aria-label':'Audio output device',hidden:''});
|
||||
menu.hidden=true;
|
||||
btn.addEventListener('click',function(event){
|
||||
if(event&&event.stopPropagation) event.stopPropagation();
|
||||
toggleOutputMenu();
|
||||
});
|
||||
wrap.appendChild(btn);
|
||||
wrap.appendChild(menu);
|
||||
return {wrap:wrap,btn:btn,menu:menu};
|
||||
}
|
||||
|
||||
function openConversationOverlay(){
|
||||
if(conversation||!conversationUsable()) return;
|
||||
try{
|
||||
@ -485,6 +618,8 @@
|
||||
controls.appendChild(exitBtn);
|
||||
const lang=buildLanguageControl();
|
||||
root.appendChild(lang.wrap);
|
||||
const out=buildOutputControl();
|
||||
root.appendChild(out.wrap);
|
||||
root.appendChild(orb);
|
||||
root.appendChild(stateEl);
|
||||
root.appendChild(captions);
|
||||
@ -494,14 +629,16 @@
|
||||
root.addEventListener('keydown',conversationKeydown);
|
||||
// A tap anywhere outside the language menu closes it (never deactivates).
|
||||
root.addEventListener('click',function(event){
|
||||
if(!conversation||!conversation.langMenu||conversation.langMenu.hidden) return;
|
||||
if(!conversation) return;
|
||||
const target=event&&event.target;
|
||||
const inWrap=target&&typeof target.closest==='function'&&target.closest('.voice-conversation-lang');
|
||||
if(!inWrap) closeLanguageMenu(false);
|
||||
const closest=target&&typeof target.closest==='function'?target.closest.bind(target):null;
|
||||
if(conversation.langMenu&&!conversation.langMenu.hidden&&!(closest&&closest('.voice-conversation-lang'))) closeLanguageMenu(false);
|
||||
if(conversation.outMenu&&!conversation.outMenu.hidden&&!(closest&&closest('.voice-conversation-out'))) closeOutputMenu(false);
|
||||
});
|
||||
document.body.appendChild(root);
|
||||
conversation={root:root,orb:orb,stateEl:stateEl,userCaption:userCaption,assistantCaption:assistantCaption,muteBtn:muteBtn,exitBtn:exitBtn,langBtn:lang.btn,langMenu:lang.menu,langItems:lang.items,muted:false};
|
||||
conversation={root:root,orb:orb,stateEl:stateEl,userCaption:userCaption,assistantCaption:assistantCaption,muteBtn:muteBtn,exitBtn:exitBtn,langBtn:lang.btn,langMenu:lang.menu,langItems:lang.items,outWrap:out.wrap,outBtn:out.btn,outMenu:out.menu,outItems:[],muted:false};
|
||||
reflectLanguageSelection();
|
||||
refreshOutputDevices();
|
||||
syncConversationOverlay(state);
|
||||
if(root.focus) root.focus();
|
||||
}catch(_){
|
||||
@ -626,7 +763,9 @@
|
||||
if(cue.controller){cue.controller.abort();cue.controller=null;}
|
||||
if(cue.audioWake){const wake=cue.audioWake;cue.audioWake=null;wake();}
|
||||
if(cue.node){try{cue.node.port.postMessage({type:'cancel'});cue.node.disconnect();}catch(_){ }cue.node=null;}
|
||||
if(cue.context){try{cue.context.close();}catch(_){ }cue.context=null;}
|
||||
// The cue shares the ONE playback context; drop the reference but never
|
||||
// close the shared sink here.
|
||||
if(cue.context){cue.context=null;}
|
||||
cue.gain=null;
|
||||
indicator.classList.remove('is-playing');
|
||||
indicator.classList.remove('is-ducked');
|
||||
@ -722,6 +861,9 @@
|
||||
cancelSpeechTurn();
|
||||
releaseMicrophone();
|
||||
stopPlayback();
|
||||
closeSharedPlayback();
|
||||
selectedOutputSinkId='';
|
||||
outputDevices=[];
|
||||
removeConversationOverlay();
|
||||
modeBtn.classList.remove('active');
|
||||
setState('error',message);
|
||||
@ -786,7 +928,9 @@
|
||||
if(playbackSession.node){
|
||||
try{playbackSession.node.port.postMessage({type:'cancel'});playbackSession.node.disconnect();}catch(_){ }
|
||||
}
|
||||
if(playbackSession.context){try{playbackSession.context.close();}catch(_){ }}
|
||||
// The context is the shared spoken-output sink; it is closed only when
|
||||
// hands-free mode ends (closeSharedPlayback), never per playback session.
|
||||
playbackSession.context=null;
|
||||
playbackSession=null;
|
||||
}
|
||||
indicator.classList.remove('is-playing');
|
||||
@ -817,6 +961,9 @@
|
||||
cancelSpeechTurn();
|
||||
releaseMicrophone();
|
||||
stopPlayback();
|
||||
closeSharedPlayback();
|
||||
selectedOutputSinkId='';
|
||||
outputDevices=[];
|
||||
modeBtn.classList.remove('active');
|
||||
setState('idle');
|
||||
if(showMessage) toast('Hands-free voice mode off');
|
||||
@ -2015,12 +2162,70 @@
|
||||
// Export the deterministic splitter as a narrow diagnostic/test seam.
|
||||
window._atlasAdaptiveChunks=function(text,final){return adaptiveChunks(text,final!==false,true).chunks;};
|
||||
|
||||
function outputRoutingSupported(){
|
||||
// The corner output-device selector needs to both ENUMERATE outputs and
|
||||
// ROUTE to one. Feature-detected so unsupported browsers hide it entirely.
|
||||
const Context=window.AudioContext||window.webkitAudioContext;
|
||||
const md=navigator.mediaDevices;
|
||||
const canRoute=(Context&&Context.prototype&&typeof Context.prototype.setSinkId==='function')||
|
||||
(window.HTMLMediaElement&&window.HTMLMediaElement.prototype&&typeof window.HTMLMediaElement.prototype.setSinkId==='function');
|
||||
return !!(md&&typeof md.enumerateDevices==='function'&&canRoute);
|
||||
}
|
||||
|
||||
async function applyContextSink(context){
|
||||
if(!context||!selectedOutputSinkId||typeof context.setSinkId!=='function') return;
|
||||
try{await context.setSinkId(selectedOutputSinkId);}catch(_){ }
|
||||
}
|
||||
|
||||
async function acquirePlaybackContext(){
|
||||
// One shared AudioContext for ALL spoken output. Its native rate is left to
|
||||
// the browser; the worklet resamples each push from Piper's rate, so cues
|
||||
// and answers (even at different sample rates) share the exact same sink.
|
||||
const Context=window.AudioContext||window.webkitAudioContext;
|
||||
if(!Context) return null;
|
||||
if(sharedPlaybackContext&&sharedPlaybackContext.state!=='closed'){
|
||||
if(sharedPlaybackWorklet){try{await sharedPlaybackWorklet;}catch(_){ }}
|
||||
return sharedPlaybackContext;
|
||||
}
|
||||
let context;
|
||||
try{context=new Context({latencyHint:'interactive'});}catch(_){context=new Context();}
|
||||
sharedPlaybackContext=context;
|
||||
sharedPlaybackWorklet=context.audioWorklet.addModule(WORKLET_URL);
|
||||
try{await sharedPlaybackWorklet;}catch(error){
|
||||
if(sharedPlaybackContext===context){sharedPlaybackContext=null;sharedPlaybackWorklet=null;}
|
||||
try{context.close();}catch(_){ }
|
||||
throw error;
|
||||
}
|
||||
await applyContextSink(context);
|
||||
return context;
|
||||
}
|
||||
|
||||
function applyOutputSink(){
|
||||
// Route the shared context and any in-flight blob element to the chosen sink.
|
||||
applyContextSink(sharedPlaybackContext);
|
||||
if(currentAudio&&typeof currentAudio.setSinkId==='function'){
|
||||
try{currentAudio.setSinkId(selectedOutputSinkId||'');}catch(_){ }
|
||||
}
|
||||
}
|
||||
|
||||
function closeSharedPlayback(){
|
||||
const context=sharedPlaybackContext;
|
||||
sharedPlaybackContext=null;
|
||||
sharedPlaybackWorklet=null;
|
||||
if(context){try{context.close();}catch(_){ }}
|
||||
}
|
||||
|
||||
function playBlob(blob,token){
|
||||
return new Promise(function(resolve,reject){
|
||||
if(!active||token!==generation){resolve();return;}
|
||||
const session=playbackSession;
|
||||
const url=URL.createObjectURL(blob);
|
||||
const audio=new Audio(url);
|
||||
// Same chosen output as the PCM path: the WAV fallback element is routed to
|
||||
// the selected sink so it can never split onto a different device.
|
||||
if(selectedOutputSinkId&&typeof audio.setSinkId==='function'){
|
||||
try{audio.setSinkId(selectedOutputSinkId);}catch(_){ }
|
||||
}
|
||||
currentAudio=audio;
|
||||
let settled=false;
|
||||
function cleanup(callback,value){
|
||||
@ -2092,7 +2297,8 @@
|
||||
}
|
||||
|
||||
function thinkingCueStillOwned(cue){
|
||||
return !!(cue&&!cue.cancelled&&thinkingCue===cue&&active&&cue.token===generation&&state==='thinking');
|
||||
// Mute-aware: a muted conversation (the user stepped away) never chatters.
|
||||
return !!(cue&&!cue.cancelled&&thinkingCue===cue&&active&&cue.token===generation&&state==='thinking'&&!(conversation&&conversation.muted));
|
||||
}
|
||||
|
||||
function scheduleNextThinkingCue(cue,delay){
|
||||
@ -2123,10 +2329,10 @@
|
||||
const sampleRate=parseInt(response.headers.get('X-Audio-Sample-Rate')||'22050',10);
|
||||
if(!Number.isFinite(sampleRate)||sampleRate<8000||sampleRate>96000) throw new Error('Cached thinking cue metadata invalid');
|
||||
if(!thinkingCueStillOwned(cue)) return;
|
||||
const Context=window.AudioContext||window.webkitAudioContext;
|
||||
try{cue.context=new Context({sampleRate:sampleRate,latencyHint:'interactive'});}catch(_){cue.context=new Context();}
|
||||
await cue.context.audioWorklet.addModule(WORKLET_URL);
|
||||
if(!thinkingCueStillOwned(cue)) return;
|
||||
// The cue plays through the ONE shared spoken-output sink, exactly like the
|
||||
// reply — never its own AudioContext (which could route to another device).
|
||||
cue.context=await acquirePlaybackContext();
|
||||
if(!cue.context||!thinkingCueStillOwned(cue)) return;
|
||||
cue.node=new AudioWorkletNode(cue.context,'atlas-pcm-playback');
|
||||
cue.gain=cue.context.createGain();
|
||||
cue.node.connect(cue.gain);
|
||||
@ -2163,7 +2369,7 @@
|
||||
}
|
||||
cue.controller=null;
|
||||
if(cue.node){try{cue.node.disconnect();}catch(_){ }cue.node=null;}
|
||||
if(cue.context){try{cue.context.close();}catch(_){ }cue.context=null;}
|
||||
if(cue.context){cue.context=null;}
|
||||
cue.gain=null;
|
||||
indicator.classList.remove('is-playing');
|
||||
if(thinkingCueStillOwned(cue)) scheduleNextThinkingCue(cue,THINKING_CUE_INTERVAL_MS);
|
||||
@ -2237,19 +2443,14 @@
|
||||
if(session.sampleRate!==asset.sampleRate) throw new Error('Streaming speech sample rate changed mid-turn');
|
||||
return session;
|
||||
}
|
||||
const Context=window.AudioContext||window.webkitAudioContext;
|
||||
let context;
|
||||
try{
|
||||
// Keep Piper at its native rate and let the browser/audio device own any
|
||||
// final hardware conversion; the worklet interpolator is a fallback.
|
||||
context=new Context({sampleRate:asset.sampleRate,latencyHint:'interactive'});
|
||||
}catch(_){
|
||||
context=new Context();
|
||||
}
|
||||
await context.audioWorklet.addModule(WORKLET_URL);
|
||||
if(!active||token!==generation){context.close();return;}
|
||||
// One shared sink for every spoken sound (reply, cues, blob fallback); the
|
||||
// worklet resamples Piper's rate to the context's native rate. The shared
|
||||
// context is never closed per-session — only when hands-free mode ends.
|
||||
const context=await acquirePlaybackContext();
|
||||
if(!context) return null;
|
||||
if(!active||token!==generation) return null;
|
||||
const node=new AudioWorkletNode(context,'atlas-pcm-playback');
|
||||
if(!session||session.cancelled){context.close();return;}
|
||||
if(!session||session.cancelled) return null;
|
||||
const gain=context.createGain();
|
||||
session.context=context;
|
||||
session.node=node;
|
||||
@ -2334,7 +2535,7 @@
|
||||
if(!session||!session.node) return;
|
||||
await drainPcmPlayback(session,token);
|
||||
if(session.node){try{session.node.disconnect();}catch(_){ }session.node=null;}
|
||||
if(session.context){try{session.context.close();}catch(_){ }session.context=null;}
|
||||
if(session.context){session.context=null;}
|
||||
session.gain=null;
|
||||
session.drained=null;
|
||||
indicator.classList.remove('is-playing');
|
||||
@ -2468,6 +2669,7 @@
|
||||
idleTimer:null,
|
||||
consumed:0,
|
||||
first:true,
|
||||
voiceResolved:false,
|
||||
queue:[],
|
||||
waiters:[],
|
||||
final:false,
|
||||
@ -2477,6 +2679,26 @@
|
||||
return speechTurn;
|
||||
}
|
||||
|
||||
function flushRetainedTail(turn){
|
||||
// An interim message we had BEGUN speaking just stopped being readable — it
|
||||
// folded into the hidden .assistant-segment-worklog-source when the model's
|
||||
// tool call began, so collectAssistantResponse() now returns '' or a
|
||||
// distinct segment. Its retained sourceText still holds an unspoken tail
|
||||
// (turn.consumed < turn.sourceText.length): speak that tail IN FULL, as
|
||||
// final-quality chunks from the text the client already holds — never by
|
||||
// re-reading the now-hidden DOM — so the whole acknowledgement is heard
|
||||
// before the turn transitions to Thinking or a new segment is chunked.
|
||||
// Marking it fully consumed makes a transient empty read that later recovers
|
||||
// the SAME message a no-op instead of a double-speak.
|
||||
if(!turn||typeof turn.sourceText!=='string') return false;
|
||||
const tail=turn.sourceText.slice(turn.consumed).trim();
|
||||
if(!tail){turn.consumed=turn.sourceText.length;return false;}
|
||||
const flushed=adaptiveChunks(tail,true,false);
|
||||
enqueueSpeech(turn,flushed.chunks.length?flushed.chunks:[tail]);
|
||||
turn.consumed=turn.sourceText.length;
|
||||
return true;
|
||||
}
|
||||
|
||||
function pumpAssistantResponse(token,isFinal){
|
||||
if(!active||token!==generation||(state!=='thinking'&&state!=='speaking')) return;
|
||||
if(isFinal){
|
||||
@ -2503,6 +2725,14 @@
|
||||
}
|
||||
const text=response.text;
|
||||
if(!text){
|
||||
// The message we had begun speaking just stopped being readable (an
|
||||
// interim acknowledgement folded into the hidden worklog source when the
|
||||
// tool call began). Speak its retained tail in full FIRST, so the
|
||||
// acknowledgement is never truncated and the queue stays non-empty — the
|
||||
// idle fallback can no longer drop to Thinking with an unspoken tail out.
|
||||
// Only for a message we have actually BEGUN speaking (consumed>0): a not-
|
||||
// yet-started message must not be flushed whole on a transient empty read.
|
||||
if(speechTurn&&speechTurn.consumed>0) flushRetainedTail(speechTurn);
|
||||
if(isFinal){
|
||||
// First-sentence-stop guard: once a reply has begun speaking, the
|
||||
// completion callback can fire a frame before the settle re-render
|
||||
@ -2538,7 +2768,7 @@
|
||||
// of the answer queue.
|
||||
cancelThinkingCues();
|
||||
const turn=ensureSpeechTurn(token);
|
||||
if(turn.first){
|
||||
if(!turn.voiceResolved){
|
||||
// Reply-voice routing, finalized when the first chunk is cut: the
|
||||
// turn's STT-detected language wins, but script evidence in the reply
|
||||
// text corrects a wrong or missing detection (a Spanish reply must
|
||||
@ -2551,14 +2781,33 @@
|
||||
const resolved=forcedLanguage||strongReplyLanguage(text)||turn.sttLanguage||detectReplyLanguage(text);
|
||||
if(resolved) turn.language=resolved;
|
||||
if(!forcedLanguage&&!turn.sttLanguage&&resolved) sessionLanguage=resolved;
|
||||
// One audio timeline, one voice: a new logical message after an interim
|
||||
// fold keeps the language already committed to the queue (see the
|
||||
// new-message reset below, which sets turn.first but never re-resolves).
|
||||
turn.voiceResolved=true;
|
||||
}
|
||||
if(text.length<turn.sourceText.length||!text.startsWith(turn.sourceText)){
|
||||
// Renderers can revise the still-unspoken tail. Already-spoken text is
|
||||
// immutable. On the completion callback, advance from the old spoken
|
||||
// offset and close the queue even when the renderer rewrote an earlier
|
||||
// span; replaying the revised prefix would be more disruptive than
|
||||
// preserving the already-heard words.
|
||||
if(turn.consumed>0){
|
||||
const spokenPrefix=turn.sourceText.slice(0,turn.consumed);
|
||||
if(turn.consumed>0&&spokenPrefix&&!text.startsWith(spokenPrefix)){
|
||||
// A DISTINCT new message replaced the one we were speaking: the interim
|
||||
// acknowledgement folded to the hidden worklog source and the final
|
||||
// answer rendered as its own segment (the already-spoken interim prefix
|
||||
// is no longer a prefix of what we read). Speak the interim's retained
|
||||
// tail in full, then fall through and process `text` as a BRAND-NEW
|
||||
// message — its chunk offset resets against its OWN text and its audio
|
||||
// queues AFTER the interim drains. Never advance the interim's consumed
|
||||
// offset into the new message's text (that resurfaced the tail out of
|
||||
// order). The voice is kept (turn.voiceResolved), not re-resolved.
|
||||
flushRetainedTail(turn);
|
||||
turn.sourceText='';
|
||||
turn.consumed=0;
|
||||
turn.first=true;
|
||||
}else if(turn.consumed>0){
|
||||
// Same message, unspoken tail revised by the renderer. Already-spoken
|
||||
// text is immutable. On the completion callback, advance from the old
|
||||
// spoken offset and close the queue even when the renderer rewrote an
|
||||
// earlier span; replaying the revised prefix would be more disruptive
|
||||
// than preserving the already-heard words.
|
||||
if(!isFinal) return;
|
||||
thinkingSession=null;
|
||||
thinkingTurnId='';
|
||||
@ -2571,8 +2820,9 @@
|
||||
}
|
||||
finishSpeechQueue(turn);
|
||||
return;
|
||||
}else{
|
||||
turn.sourceText='';
|
||||
}
|
||||
turn.sourceText='';
|
||||
}
|
||||
turn.sourceText=text;
|
||||
const remaining=text.slice(turn.consumed).trimStart();
|
||||
@ -2633,6 +2883,8 @@
|
||||
clearSttLanguage();
|
||||
sessionLanguage='';
|
||||
forcedLanguage='';
|
||||
selectedOutputSinkId='';
|
||||
outputDevices=[];
|
||||
finalizeAttempts=0;
|
||||
modeBtn.classList.add('active');
|
||||
toast('Hands-free private voice mode on');
|
||||
@ -2719,6 +2971,22 @@
|
||||
adaptiveChunks:adaptiveChunks,
|
||||
sentenceEnd:sentenceEnd,
|
||||
cleanForSpeech:cleanForSpeech,
|
||||
// Read-only, behaviour-neutral: lets a deterministic probe observe that the
|
||||
// interim fold never leaves an unspoken tail outstanding while the state
|
||||
// has fallen back to Thinking (the dropped-tail regression).
|
||||
speechTurnSnapshot:function(){
|
||||
if(!speechTurn) return null;
|
||||
return {
|
||||
state:state,
|
||||
consumed:speechTurn.consumed,
|
||||
sourceLength:(speechTurn.sourceText||'').length,
|
||||
queue:speechTurn.queue.slice(),
|
||||
queueLength:speechTurn.queue.length,
|
||||
speakingChunk:speechTurn.speakingChunk||'',
|
||||
waiterLength:speechTurn.waiters.length,
|
||||
final:!!speechTurn.final,
|
||||
};
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@ -110,6 +110,150 @@ function makeElement(id) {
|
||||
return element;
|
||||
}
|
||||
|
||||
// ── Faithful assistant-turn DOM ─────────────────────────────────────────────
|
||||
// A minimal but real querySelectorAll / closest / getAttribute / hidden /
|
||||
// recursive-textContent implementation, so the interim-acknowledgement FOLD can
|
||||
// be reproduced exactly as the live renderer performs it: the interim segment is
|
||||
// re-tagged .assistant-segment-worklog-source + hidden + aria-hidden, at which
|
||||
// point the script's real collectAssistantResponse() returns '' for it (the same
|
||||
// extraction the response probe locks) and the final answer renders as its own
|
||||
// new .assistant-segment. Supports the compound selectors the extraction uses.
|
||||
function parseDomSelector(selector) {
|
||||
return String(selector).split(',').map((group) => {
|
||||
const term = group.trim();
|
||||
const parts = [];
|
||||
const re = /([.#]?[\w-]+)|\[([\w-]+)(?:([~|^$*]?=)"?([^"\]]*)"?)?\]/g;
|
||||
let m;
|
||||
while ((m = re.exec(term))) {
|
||||
if (m[1]) {
|
||||
if (m[1][0] === '.') parts.push({ kind: 'class', value: m[1].slice(1) });
|
||||
else parts.push({ kind: 'tag', value: m[1].toLowerCase() });
|
||||
} else if (m[2]) {
|
||||
parts.push({ kind: 'attr', name: m[2], op: m[3] || null, value: m[4] });
|
||||
}
|
||||
}
|
||||
return parts;
|
||||
}).filter((parts) => parts.length);
|
||||
}
|
||||
|
||||
class DomNode {
|
||||
constructor(tag) {
|
||||
this.tag = String(tag || 'div').toLowerCase();
|
||||
this.className = '';
|
||||
this.attributes = new Map();
|
||||
this.children = [];
|
||||
this.parentNode = null;
|
||||
this.hidden = false;
|
||||
this._text = '';
|
||||
const self = this;
|
||||
this.dataset = new Proxy({}, {
|
||||
get(_t, key) {
|
||||
if (typeof key !== 'string') return undefined;
|
||||
const attr = 'data-' + key.replace(/[A-Z]/g, (c) => '-' + c.toLowerCase());
|
||||
return self.attributes.has(attr) ? self.attributes.get(attr) : undefined;
|
||||
},
|
||||
has(_t, key) {
|
||||
const attr = 'data-' + String(key).replace(/[A-Z]/g, (c) => '-' + c.toLowerCase());
|
||||
return self.attributes.has(attr);
|
||||
},
|
||||
});
|
||||
}
|
||||
get classList() {
|
||||
const el = this;
|
||||
return {
|
||||
add(name) { const s = new Set(el.className.split(/\s+/).filter(Boolean)); s.add(name); el.className = [...s].join(' '); },
|
||||
remove(name) { el.className = el.className.split(/\s+/).filter((v) => v && v !== name).join(' '); },
|
||||
contains(name) { return el.className.split(/\s+/).indexOf(name) >= 0; },
|
||||
};
|
||||
}
|
||||
setAttribute(name, value) {
|
||||
this.attributes.set(name, String(value));
|
||||
if (name === 'class') this.className = String(value);
|
||||
if (name === 'hidden') this.hidden = true;
|
||||
}
|
||||
getAttribute(name) {
|
||||
if (name === 'class') return this.className || null;
|
||||
return this.attributes.has(name) ? this.attributes.get(name) : null;
|
||||
}
|
||||
appendChild(child) { child.parentNode = this; this.children.push(child); return child; }
|
||||
set textContent(value) { this._text = String(value); this.children = []; }
|
||||
get textContent() {
|
||||
if (this.children.length) return this.children.map((c) => c.textContent).join('');
|
||||
return this._text;
|
||||
}
|
||||
_matchesTerm(parts) {
|
||||
return parts.every((p) => {
|
||||
if (p.kind === 'class') return this.classList.contains(p.value);
|
||||
if (p.kind === 'tag') return this.tag === p.value;
|
||||
if (p.kind === 'attr') {
|
||||
if (!this.attributes.has(p.name)) return false;
|
||||
return p.op ? this.attributes.get(p.name) === p.value : true;
|
||||
}
|
||||
return false;
|
||||
});
|
||||
}
|
||||
matches(selector) { return parseDomSelector(selector).some((parts) => this._matchesTerm(parts)); }
|
||||
_walk(out) { for (const c of this.children) { out.push(c); c._walk(out); } return out; }
|
||||
querySelectorAll(selector) {
|
||||
const groups = parseDomSelector(selector);
|
||||
return this._walk([]).filter((node) => groups.some((parts) => node._matchesTerm(parts)));
|
||||
}
|
||||
querySelector(selector) { const all = this.querySelectorAll(selector); return all.length ? all[0] : null; }
|
||||
closest(selector) {
|
||||
const groups = parseDomSelector(selector);
|
||||
let node = this;
|
||||
while (node) { if (groups.some((parts) => node._matchesTerm(parts))) return node; node = node.parentNode; }
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
function domEl(tag, className, attrs, text) {
|
||||
const node = new DomNode(tag);
|
||||
if (className) node.setAttribute('class', className);
|
||||
if (attrs) Object.keys(attrs).forEach((k) => node.setAttribute(k, attrs[k]));
|
||||
if (text !== undefined) node.textContent = text;
|
||||
return node;
|
||||
}
|
||||
|
||||
// A rendered assistant turn: role header, worklog chip, and an assistant-turn
|
||||
// blocks container the interim/final segments hang off of.
|
||||
function buildAssistantTurn() {
|
||||
const turn = domEl('div', 'msg-row assistant-turn', { 'data-role': 'assistant', 'data-session-id': 'session-1' });
|
||||
const role = domEl('div', 'msg-role assistant');
|
||||
role.appendChild(domEl('div', 'role-icon assistant', null, 'H'));
|
||||
role.appendChild(domEl('span', 'msg-role-name', null, 'Hermes'));
|
||||
turn.appendChild(role);
|
||||
const blocks = domEl('div', 'assistant-turn-blocks');
|
||||
turn.appendChild(blocks);
|
||||
turn.blocks = blocks;
|
||||
return turn;
|
||||
}
|
||||
|
||||
// A live (still-streaming) answer segment: no data-raw-text yet, so the script
|
||||
// reads its answer .msg-body — exactly the interim acknowledgement path.
|
||||
function addLiveAnswerSegment(turn, text) {
|
||||
const seg = domEl('div', 'assistant-segment', { 'data-live-assistant': '1' });
|
||||
seg.appendChild(domEl('div', 'msg-body', null, text));
|
||||
turn.blocks.appendChild(seg);
|
||||
return seg;
|
||||
}
|
||||
|
||||
// A settled answer segment carrying the renderer-stamped clean answer text.
|
||||
function addSettledAnswerSegment(turn, text) {
|
||||
const seg = domEl('div', 'assistant-segment', { 'data-raw-text': text });
|
||||
seg.appendChild(domEl('div', 'msg-body', null, text));
|
||||
turn.blocks.appendChild(seg);
|
||||
return seg;
|
||||
}
|
||||
|
||||
// Fold an interim segment into the hidden worklog source, exactly as the live
|
||||
// renderer does when a tool call begins: the extraction now excludes it.
|
||||
function foldSegmentIntoWorklog(seg) {
|
||||
seg.setAttribute('class', 'assistant-segment assistant-segment-worklog-source');
|
||||
seg.setAttribute('aria-hidden', 'true');
|
||||
seg.hidden = true;
|
||||
}
|
||||
|
||||
// Faithful port of StreamingTranscription's endpointing/speculation contract.
|
||||
class StubSttServer {
|
||||
constructor(words) {
|
||||
@ -243,6 +387,7 @@ function makeHarness(options = {}) {
|
||||
|
||||
const words = options.words || { 8000: 'alpha', 12000: 'bravo', 16000: 'charlie' };
|
||||
const micTracks = [];
|
||||
const sinkCalls = [];
|
||||
|
||||
const elements = {};
|
||||
['btnVoiceMode', 'voiceModeBar', 'voiceModeIndicator', 'voiceModeLabel', 'msg']
|
||||
@ -280,6 +425,7 @@ function makeHarness(options = {}) {
|
||||
createMediaStreamSource() { return { connect() {}, disconnect() {} }; }
|
||||
createGain() { return { gain: { value: 0, setTargetAtTime() {} }, connect() {}, disconnect() {} }; }
|
||||
resume() { return Promise.resolve(); }
|
||||
setSinkId(id) { sinkCalls.push({ kind: 'context', sink: id }); return Promise.resolve(); }
|
||||
close() { this.closed = true; return Promise.resolve(); }
|
||||
}
|
||||
|
||||
@ -466,6 +612,7 @@ function makeHarness(options = {}) {
|
||||
Audio: function Audio() {
|
||||
this.play = () => Promise.resolve();
|
||||
this.pause = () => {};
|
||||
this.setSinkId = (id) => { sinkCalls.push({ kind: 'audio', sink: id }); return Promise.resolve(); };
|
||||
this.onended = null;
|
||||
this.onerror = null;
|
||||
},
|
||||
@ -482,6 +629,7 @@ function makeHarness(options = {}) {
|
||||
navigator: {
|
||||
mediaDevices: {
|
||||
getSupportedConstraints: () => ({}),
|
||||
...(options.outputs ? { enumerateDevices: async () => options.outputs } : {}),
|
||||
getUserMedia: async () => {
|
||||
const track = {
|
||||
stop() {},
|
||||
@ -610,6 +758,28 @@ function makeHarness(options = {}) {
|
||||
setAssistantReply(text) { assistantRows = [{ dataset: { rawText: text } }]; },
|
||||
setAssistantError(text) { assistantRows = [{ dataset: { rawText: text, error: '1' } }]; },
|
||||
clearAssistantRows() { assistantRows = []; },
|
||||
// ── Interim-fold reproduction ─────────────────────────────────────────
|
||||
// Stream an interim acknowledgement as a live answer segment, then FOLD it
|
||||
// into the hidden worklog source (as the renderer does at the tool call) and
|
||||
// render the final answer as its own new segment.
|
||||
setInterimAck(text) {
|
||||
const turn = buildAssistantTurn();
|
||||
this._foldTurn = turn;
|
||||
this._interimSeg = addLiveAnswerSegment(turn, text);
|
||||
assistantRows = [turn];
|
||||
},
|
||||
foldInterimAck() { if (this._interimSeg) foldSegmentIntoWorklog(this._interimSeg); },
|
||||
setFinalAnswer(text) {
|
||||
if (!this._foldTurn) return;
|
||||
addSettledAnswerSegment(this._foldTurn, text);
|
||||
},
|
||||
speechTurnSnapshot() {
|
||||
const internals = context.window.__atlasVoiceInternals;
|
||||
return internals && typeof internals.speechTurnSnapshot === 'function'
|
||||
? internals.speechTurnSnapshot() : null;
|
||||
},
|
||||
ttsTexts() { return ttsCalls.map((request) => request.text); },
|
||||
sinkCalls() { return sinkCalls.slice(); },
|
||||
state() { return elements.voiceModeBar.dataset.voiceState || ''; },
|
||||
body() { return bodyElement; },
|
||||
trackStates() { return micTracks.map((track) => track.enabled); },
|
||||
@ -935,6 +1105,173 @@ scenarios.language_override_forces_stt_and_voice = async () => {
|
||||
};
|
||||
};
|
||||
|
||||
// THE FOLD REGRESSION. A quick spoken acknowledgement streams as an interim
|
||||
// answer segment while a tool call is prepared. The pump chunks the first two
|
||||
// sentences and RETAINS the still-streaming third as an unspoken tail; then the
|
||||
// renderer folds the interim into the hidden .assistant-segment-worklog-source
|
||||
// (extraction → '') and the final answer renders as its own new segment.
|
||||
//
|
||||
// Before the fix: the empty-read branch returned without flushing the retained
|
||||
// tail (the acknowledgement's last words were dropped) and the speech queue
|
||||
// drained with the state fallen back to Thinking while a tail was still
|
||||
// outstanding; the final answer then resurfaced from the interim's stale offset
|
||||
// (a garbled mid-string slice). After the fix: the whole acknowledgement is
|
||||
// spoken, in order, BEFORE the distinct final answer, and the state never falls
|
||||
// back to Thinking with an unspoken interim tail outstanding.
|
||||
scenarios.interim_ack_fold_flushes_tail_before_final = async () => {
|
||||
const harness = makeHarness();
|
||||
// The stub Audio never fires onended, so playback of the first chunk parks and
|
||||
// the queue does not drain (the same reason barge_cut_marker only inspects the
|
||||
// first synthesized chunk). The complete ordered set of chunks that WILL be
|
||||
// spoken is therefore the chunks already handed to synthesis (ttsTexts, in
|
||||
// order) followed by the chunks still queued behind the parked one, assembled
|
||||
// as spokenOrder below. Every assertion is a QUEUE/STATE fact, independent of
|
||||
// the playback stub.
|
||||
const violations = [];
|
||||
const record = () => {
|
||||
const snap = harness.speechTurnSnapshot();
|
||||
if (snap && snap.state === 'thinking' && !snap.final && snap.consumed < snap.sourceLength) {
|
||||
violations.push({ consumed: snap.consumed, sourceLength: snap.sourceLength });
|
||||
}
|
||||
};
|
||||
const settle = async (steps) => { for (let i = 0; i < steps; i += 1) { await harness.tick(100); record(); } };
|
||||
|
||||
await harness.start();
|
||||
await harness.silence(300);
|
||||
await harness.speak('alpha', 1300);
|
||||
await harness.silence(1500);
|
||||
await harness.tick(400); // dispatched; thinking; response observer running
|
||||
|
||||
// Interim acknowledgement: two complete sentences plus a third still mid-word,
|
||||
// so the pump enqueues S1+S2 and retains "One moment while I che" unspoken.
|
||||
const interim = 'Sure thing. Let me look that up. One moment while I che';
|
||||
harness.setInterimAck(interim);
|
||||
await settle(4);
|
||||
|
||||
// The tool call begins: fold the interim into the hidden worklog source.
|
||||
harness.foldInterimAck();
|
||||
await settle(6);
|
||||
const afterFold = harness.speechTurnSnapshot();
|
||||
const queuedAfterFold = (afterFold && afterFold.queue) || [];
|
||||
|
||||
// The final answer renders as its own new, distinct segment.
|
||||
const finalAnswer = 'The weather today is sunny and warm. Enjoy your afternoon out there.';
|
||||
harness.setFinalAnswer(finalAnswer);
|
||||
await settle(6);
|
||||
harness.completeResponse();
|
||||
await settle(10);
|
||||
|
||||
const finalSnap = harness.speechTurnSnapshot();
|
||||
const residualQueue = (finalSnap && finalSnap.queue) || [];
|
||||
// The full ordered list of chunks that reach the speech queue for this turn.
|
||||
const spokenOrder = harness.ttsTexts().concat(residualQueue);
|
||||
const joined = spokenOrder.join('');
|
||||
const lastInterimIdx = (() => {
|
||||
let idx = -1;
|
||||
spokenOrder.forEach((t, i) => { if (/Sure thing|look that up|One moment|while I che/.test(t)) idx = i; });
|
||||
return idx;
|
||||
})();
|
||||
const firstFinalIdx = spokenOrder.findIndex((t) => /weather|sunny|afternoon/.test(t));
|
||||
|
||||
return {
|
||||
spokenOrder,
|
||||
// After the fold, the interim's retained tail is the next thing queued, and
|
||||
// it is queued BEFORE any final-answer chunk exists.
|
||||
queuedAfterFold,
|
||||
tailQueuedAfterFold: queuedAfterFold.some((t) => /while I che/.test(t)),
|
||||
noFinalBeforeFold: !queuedAfterFold.some((t) => /weather|sunny|afternoon/.test(t)),
|
||||
// (i) every interim sentence reaches the speech queue: none dropped.
|
||||
interimS1Reached: /Sure thing/.test(joined),
|
||||
interimS2Reached: /look that up/.test(joined),
|
||||
interimTailReached: /while I che/.test(joined),
|
||||
// (iii) the distinct final answer is fully queued too.
|
||||
finalS1Reached: /weather today is sunny/.test(joined),
|
||||
finalS2Reached: /Enjoy your afternoon/.test(joined),
|
||||
// The final answer is queued from its OWN start, never resurfaced from the
|
||||
// interim's stale offset ("nny and warm" is the buggy mid-string slice).
|
||||
finalNotGarbled: !spokenOrder.some((t) => /^nny and warm/.test(t)),
|
||||
// (ii) the interim finishes before the final message's chunks are enqueued.
|
||||
interimBeforeFinal: lastInterimIdx >= 0 && firstFinalIdx >= 0 && lastInterimIdx < firstFinalIdx,
|
||||
// (iv) the state never fell back to Thinking with an unspoken interim tail.
|
||||
thinkingWithUnspokenTail: violations.length,
|
||||
state: harness.state(),
|
||||
};
|
||||
};
|
||||
|
||||
// UNIFIED OUTPUT SINK + device selector. When the browser can enumerate outputs
|
||||
// and setSinkId, a tidy corner control lists them; choosing one routes ALL spoken
|
||||
// output (the reply's blob element here, and — via the shared AudioContext — the
|
||||
// PCM reply and thinking cues) to that single device. When the APIs are missing,
|
||||
// the control hides entirely (a dead control never appears).
|
||||
scenarios.output_device_selector_routes_spoken_output = async () => {
|
||||
const harness = makeHarness({
|
||||
outputs: [
|
||||
{ deviceId: '', kind: 'audiooutput', label: 'System default' },
|
||||
{ deviceId: 'spk-1', kind: 'audiooutput', label: 'Speaker One' },
|
||||
{ deviceId: 'spk-2', kind: 'audiooutput', label: 'Headphones Two' },
|
||||
{ deviceId: 'mic-1', kind: 'audioinput', label: 'Microphone' },
|
||||
],
|
||||
});
|
||||
await harness.start();
|
||||
await harness.flush();
|
||||
await harness.flush();
|
||||
const overlay = harness.overlay();
|
||||
if (!overlay) return { overlayPresent: false };
|
||||
const outWrap = overlay.children.find(
|
||||
(child) => String(child.className).indexOf('voice-conversation-out') >= 0,
|
||||
);
|
||||
const outBtn = outWrap ? outWrap.children.find(
|
||||
(child) => String(child.className).indexOf('voice-conversation-out-btn') >= 0,
|
||||
) : null;
|
||||
const outMenu = outWrap ? outWrap.children.find(
|
||||
(child) => String(child.className).indexOf('voice-conversation-out-menu') >= 0,
|
||||
) : null;
|
||||
const shownAfterRefresh = outWrap ? outWrap.style.display !== 'none' : false;
|
||||
const items = outMenu ? outMenu.children : [];
|
||||
const itemLabels = items.map((item) => item.textContent);
|
||||
const headphones = items.find((item) => item.getAttribute('data-device') === 'spk-2');
|
||||
if (headphones) headphones.click();
|
||||
const forcedChecked = headphones ? headphones.getAttribute('aria-checked') : null;
|
||||
const btnForced = outBtn ? String(outBtn.className).indexOf('is-forced') >= 0 : false;
|
||||
// A reply now plays through the blob fallback element, which must be routed to
|
||||
// the chosen sink.
|
||||
await harness.silence(300);
|
||||
await harness.speak('alpha', 1300);
|
||||
await harness.silence(1500);
|
||||
await harness.tick(400);
|
||||
harness.setAssistantReply('A short spoken reply.');
|
||||
harness.completeResponse();
|
||||
await harness.tick(400);
|
||||
const sinks = harness.sinkCalls();
|
||||
return {
|
||||
overlayPresent: true,
|
||||
supported: true,
|
||||
shownAfterRefresh,
|
||||
itemLabels,
|
||||
forcedChecked,
|
||||
btnForced,
|
||||
audioRoutedTo: sinks.filter((c) => c.kind === 'audio').map((c) => c.sink),
|
||||
};
|
||||
};
|
||||
|
||||
// When enumerateDevices/setSinkId are unavailable the selector is present in the
|
||||
// DOM but hidden — never a dead control.
|
||||
scenarios.output_selector_hidden_when_unsupported = async () => {
|
||||
const harness = makeHarness();
|
||||
await harness.start();
|
||||
await harness.flush();
|
||||
const overlay = harness.overlay();
|
||||
if (!overlay) return { overlayPresent: false };
|
||||
const outWrap = overlay.children.find(
|
||||
(child) => String(child.className).indexOf('voice-conversation-out') >= 0,
|
||||
);
|
||||
return {
|
||||
overlayPresent: true,
|
||||
controlInDom: !!outWrap,
|
||||
hidden: outWrap ? outWrap.style.display === 'none' : null,
|
||||
};
|
||||
};
|
||||
|
||||
(async () => {
|
||||
const output = {};
|
||||
for (const name of Object.keys(scenarios)) {
|
||||
|
||||
@ -361,11 +361,12 @@ def test_chat_voice_uses_private_jetson_services_and_shared_auto_route():
|
||||
assert "Math.max(0.04,noiseFloor*2.4+0.006)" in voice_script
|
||||
assert "recorder.start();" in voice_script
|
||||
assert "recorder.start(250)" not in voice_script
|
||||
assert "function flushRetainedTail(turn)" in voice_script # folded interim tail is spoken, never dropped
|
||||
|
||||
stt_server = (ROOT / "dockerfiles" / "hermes-jetson-stt-server.py").read_text()
|
||||
assert "def _repetitive_token" in stt_server
|
||||
assert "compression_ratio_threshold=2.0" in stt_server
|
||||
assert "no_speech_threshold=0.5" in stt_server
|
||||
assert '"compression_ratio_threshold": 2.0' in stt_server
|
||||
assert '"no_speech_threshold": 0.5' in stt_server
|
||||
|
||||
|
||||
def test_worker_webui_rollouts_do_not_repeat_completed_state_migration():
|
||||
|
||||
@ -77,6 +77,138 @@ def test_normal_completion_clears_stitch_and_mic_stays_hot(probe_results):
|
||||
assert scenario["micAcquisitions"] == 1
|
||||
|
||||
|
||||
def test_interim_ack_fold_flushes_retained_tail_before_final(probe_results):
|
||||
"""The reported TTS-truncation bug: a quick spoken acknowledgement streams as
|
||||
an interim segment, the pump chunks its first sentences and RETAINS the still-
|
||||
streaming tail; then a tool call folds the interim into the hidden worklog
|
||||
source and the final answer renders as its own segment.
|
||||
|
||||
Before the fix the retained tail was dropped (the empty-read branch returned
|
||||
without flushing it) and the queue drained with the state fallen back to
|
||||
Thinking while a tail was outstanding, and the final answer resurfaced from
|
||||
the interim's stale offset. After the fix the WHOLE acknowledgement reaches
|
||||
the speech queue, in order, before the distinct final answer, and the state
|
||||
never falls back to Thinking with an unspoken interim tail outstanding.
|
||||
"""
|
||||
scenario = probe_results["interim_ack_fold_flushes_tail_before_final"]
|
||||
# (i) every interim sentence reaches the queue — the retained tail is flushed
|
||||
# the moment the segment folds, not dropped.
|
||||
assert scenario["interimS1Reached"] is True
|
||||
assert scenario["interimS2Reached"] is True
|
||||
assert scenario["interimTailReached"] is True, "the folded interim tail was dropped"
|
||||
assert scenario["tailQueuedAfterFold"] is True
|
||||
assert scenario["noFinalBeforeFold"] is True
|
||||
# (ii) the interim finishes before the final message's chunks are enqueued.
|
||||
assert scenario["interimBeforeFinal"] is True
|
||||
# (iii) the distinct final answer is also fully queued, from its own start —
|
||||
# never resurfaced from the interim's stale offset.
|
||||
assert scenario["finalS1Reached"] is True
|
||||
assert scenario["finalS2Reached"] is True
|
||||
assert scenario["finalNotGarbled"] is True
|
||||
assert scenario["spokenOrder"] == [
|
||||
"Sure thing.",
|
||||
"Let me look that up.",
|
||||
"One moment while I che",
|
||||
"The weather today is sunny and warm.",
|
||||
"Enjoy your afternoon out there.",
|
||||
]
|
||||
# (iv) the state never flipped to Thinking with an unspoken interim tail out.
|
||||
assert scenario["thinkingWithUnspokenTail"] == 0
|
||||
|
||||
|
||||
def test_interim_fold_tail_flush_source_contract():
|
||||
source = VOICE_SCRIPT.read_text(encoding="utf-8")
|
||||
|
||||
# The retained tail is flushed as final-quality chunks from the text the
|
||||
# client already holds — never by re-reading the folded/hidden DOM.
|
||||
assert "function flushRetainedTail(turn)" in source
|
||||
assert "const tail=turn.sourceText.slice(turn.consumed).trim();" in source
|
||||
assert "const flushed=adaptiveChunks(tail,true,false);" in source
|
||||
# The empty-read branch flushes the tail before anything else.
|
||||
assert "if(speechTurn&&speechTurn.consumed>0) flushRetainedTail(speechTurn);" in source
|
||||
# A distinct new message after the fold resets the offset against its OWN
|
||||
# text (never advancing the interim's consumed offset into it).
|
||||
assert "const spokenPrefix=turn.sourceText.slice(0,turn.consumed);" in source
|
||||
assert "if(turn.consumed>0&&spokenPrefix&&!text.startsWith(spokenPrefix)){" in source
|
||||
# One audio timeline, one voice: the new message keeps the resolved voice.
|
||||
assert "if(!turn.voiceResolved){" in source
|
||||
assert "turn.voiceResolved=true;" in source
|
||||
|
||||
|
||||
def test_output_device_selector_routes_all_spoken_output(probe_results):
|
||||
"""Unified sink + selector: when the browser can enumerate outputs and
|
||||
setSinkId, the corner control lists the routable outputs and choosing one
|
||||
routes the spoken output (the blob element here) to that device."""
|
||||
scenario = probe_results["output_device_selector_routes_spoken_output"]
|
||||
assert scenario["overlayPresent"] is True
|
||||
assert scenario["shownAfterRefresh"] is True
|
||||
assert scenario["itemLabels"] == ["System default", "Speaker One", "Headphones Two"]
|
||||
# Selection is reflected accessibly and marked as a non-default override.
|
||||
assert scenario["forcedChecked"] == "true"
|
||||
assert scenario["btnForced"] is True
|
||||
# The spoken reply is routed to the chosen sink.
|
||||
assert scenario["audioRoutedTo"] == ["spk-2"]
|
||||
|
||||
|
||||
def test_output_selector_hidden_when_unsupported(probe_results):
|
||||
"""No enumerateDevices/setSinkId: the control is in the DOM but hidden — it
|
||||
never appears as a dead control."""
|
||||
scenario = probe_results["output_selector_hidden_when_unsupported"]
|
||||
assert scenario["overlayPresent"] is True
|
||||
assert scenario["controlInDom"] is True
|
||||
assert scenario["hidden"] is True
|
||||
|
||||
|
||||
def test_natural_fillers_unified_sink_and_output_selector_source_contract():
|
||||
source = VOICE_SCRIPT.read_text(encoding="utf-8")
|
||||
css = (ROOT / "dockerfiles" / "hermes-webui-atlas-voice.css").read_text(encoding="utf-8")
|
||||
|
||||
# (2) Natural spoken fillers per language, with the longer reassurances kept.
|
||||
assert "{id:'umm',text:'Umm.'}" in source
|
||||
assert "{id:'hmm',text:'Hmm.'}" in source
|
||||
assert "{id:'one_sec',text:'One sec.'}" in source
|
||||
assert "{id:'mmm',text:'Mmm.'}" in source
|
||||
assert "{id:'a_ver',text:'A ver.'}" in source
|
||||
assert "{id:'hmm',text:'Хм.'}" in source
|
||||
assert "{id:'sec',text:'Секунду.'}" in source
|
||||
# Answer preempts a filler; a genuine >~1.9s gap gates the first one.
|
||||
assert "const THINKING_CUE_FIRST_MS=1900" in source
|
||||
assert "cancelThinkingCues();\n const turn=ensureSpeechTurn(token)" in source
|
||||
# Mute-aware: a muted conversation never chatters.
|
||||
assert "state==='thinking'&&!(conversation&&conversation.muted)" in source
|
||||
|
||||
# (3) ONE shared sink for every spoken sound. The only AudioContext built for
|
||||
# OUTPUT is inside acquirePlaybackContext; cues and PCM replies both use it.
|
||||
assert "async function acquirePlaybackContext()" in source
|
||||
assert "function closeSharedPlayback()" in source
|
||||
assert "cue.context=await acquirePlaybackContext();" in source
|
||||
assert "const context=await acquirePlaybackContext();" in source
|
||||
# The shared context is never closed per playback session/cue.
|
||||
assert "try{cue.context=new Context(" not in source
|
||||
assert "context=new Context({sampleRate:asset.sampleRate" not in source
|
||||
# The blob fallback element is routed to the same chosen sink.
|
||||
assert "audio.setSinkId(selectedOutputSinkId)" in source
|
||||
|
||||
# The output-device selector: feature-detected, session-only, styled.
|
||||
assert "function outputRoutingSupported()" in source
|
||||
assert "navigator.mediaDevices.enumerateDevices" in source
|
||||
assert "function buildOutputControl()" in source
|
||||
assert "function selectOutputDevice(deviceId)" in source
|
||||
assert "role:'menuitemradio','data-device':entry.deviceId" in source
|
||||
# Session-only: the chosen output never touches storage.
|
||||
out_region = source.split("function buildOutputControl()", 1)[1].split(
|
||||
"function openConversationOverlay", 1
|
||||
)[0]
|
||||
assert "localStorage" not in out_region
|
||||
for token in (
|
||||
".voice-conversation-out",
|
||||
".voice-conversation-out-btn",
|
||||
".voice-conversation-out-menu",
|
||||
".voice-conversation-out-item",
|
||||
):
|
||||
assert token in css
|
||||
|
||||
|
||||
def test_conversation_overlay_lifecycle(probe_results):
|
||||
scenario = probe_results["conversation_overlay_lifecycle"]
|
||||
assert scenario["overlayPresent"] is True
|
||||
@ -308,7 +440,7 @@ def test_conversation_language_selector_source_and_style_contract():
|
||||
assert "localStorage" not in source.split("CONVERSATION_LANGUAGES", 1)[1].split("function openConversationOverlay", 1)[0]
|
||||
# Escape peels the menu before exiting; the language button joins the trap.
|
||||
assert "if(conversation.langMenu&&!conversation.langMenu.hidden){closeLanguageMenu(true);return;}" in source
|
||||
assert "[conversation.langBtn,conversation.muteBtn,conversation.exitBtn]" in source
|
||||
assert "[conversation.langBtn,outVisible?conversation.outBtn:null,conversation.muteBtn,conversation.exitBtn]" in source
|
||||
for token in (".voice-conversation-lang-btn", ".voice-conversation-lang-menu", "menuitemradio"):
|
||||
pass # menu roles live in JS; assert CSS hooks below
|
||||
for token in (".voice-conversation-lang", ".voice-conversation-lang-btn", ".voice-conversation-lang-menu", ".voice-conversation-lang-item"):
|
||||
|
||||
Loading…
x
Reference in New Issue
Block a user