// Natural turn-taking for chat.bstein.dev using the private Jetsons. (function(){ 'use strict'; const modeBtn=document.getElementById('btnVoiceMode'); const bar=document.getElementById('voiceModeBar'); const indicator=document.getElementById('voiceModeIndicator'); const label=document.getElementById('voiceModeLabel'); const composer=document.getElementById('msg'); if(!modeBtn||!bar||!indicator||!label||!composer||!navigator.mediaDevices||!window.MediaRecorder) return; let ready=false; let active=false; let state='idle'; let generation=0; let turnSequence=0; let recorder=null; let stream=null; let audioContext=null; let captureNode=null; let vadTimer=null; let responsePollTimer=null; let currentAudio=null; let playbackSession=null; let bargeMonitor=null; let bargeCancelPromise=null; let suppressAutoRead=false; let thinkingCue=null; let streamingStt=null; let captureTurnId=''; let thinkingTurnId=''; let thinkingSession=null; let assistantBaseline=null; let speechTurn=null; let errorTimer=null; let visualInputLevel=0; let streamingCapability={tts:null,stt:null,preflight:null}; let voicePreflight=null; // Unified spoken-output sink. Every spoken sound — the reply's streaming PCM, // the thinking-cue PCM, and the WAV blob fallback — plays through this ONE // shared AudioContext (and, for the blob element, the same chosen sinkId), so // the reply and its fillers can never land on different hardware outputs // (the earpiece-vs-loudspeaker split). The context is created lazily on first // playback, persists across cues/segments/turns, and is closed only when the // hands-free session ends. selectedOutputSinkId is the user-chosen audio // output device (session-only; '' means the system default). let sharedPlaybackContext=null; let sharedPlaybackWorklet=null; let selectedOutputSinkId=''; let outputDevices=[]; // Whether the user has explicitly picked an output device this session. Until // they do, the first device enumeration auto-defaults the shared sink to the // loudspeaker (never the OS "communications"/earpiece route the browser picks // while the mic is open); an explicit choice is never overridden afterwards. let outputSinkUserChosen=false; // Continuous-capture state. The microphone stream and its AudioContext stay // hot for the whole hands-free session; each utterance is one "capture turn" // guarded by captureGeneration so a response-side barge (which bumps // `generation`) never interrupts the running microphone pipeline. let captureGeneration=0; let captureActive=false; let captureGraph=null; // Barge-in stitching: transcript whose model response was cancelled before // any visible assistant output, plus the transcript most recently sent. let pendingStitch=null; let lastSentTranscript=null; const voiceTabNonce=(function(){ try{ const bytes=new Uint8Array(16); window.crypto.getRandomValues(bytes); return Array.from(bytes,function(value){return value.toString(16).padStart(2,'0');}).join(''); }catch(_){return '';} })(); const reducedMotion=window.matchMedia?window.matchMedia('(prefers-reduced-motion: reduce)'):{matches:false}; // FIX 1: the conversation orb's centred watermark is the Hermes CHARACTER // glyph — the 512px character art, feathered to a circle and luminance-weighted // to a warm near-white, inlined as a data URI and painted by // .voice-conversation-orb-mark in atlas-voice.css cropped into the orb by a feathered circle mask (no box; normal blend) // so the face features glow over the dark orb. A low-opacity, non-animating // (reduced-motion safe) layer that scales with the orb and reads across every // state tint. No separate static asset is served for it. const ERROR_VISIBLE_MS=3200; const STREAMING_CAPABILITY_URL='/api/voice/streaming/capability'; const TTS_STREAM_URL='/api/tts/stream'; const STT_STREAM_PATH='/api/transcribe/stream'; const VOICE_PREFLIGHT_URL='/api/voice/route-preflight'; const VOICE_PREFLIGHT_DEBOUNCE_MS=200; const VOICE_PREFLIGHT_DEADLINE_MS=1100; const VOICE_PREFLIGHT_TIERS=['fast','balanced','deep','maximum']; const WORKLET_URL='static/atlas-voice-worklet.js'; const STT_MAX_QUEUED_BYTES=1048576; const STT_MAX_ARCHIVE_BYTES=2880000; const BARGE_LOOKBACK_MS=900; const BARGE_DUCK_FRAMES=2; const BARGE_TRIGGER_FRAMES=4; // A response cancelled by barge-in leaves the user mid-thought: the next // finalized utterance inside this window is sent as one stitched message. const STITCH_WINDOW_MS=20000; // Endpointing. A short utterance ("The…") is usually a sentence still being // formed, so while less than VAD_COMMITTED_SPEECH_MS of voiced audio has // been collected the endpoint silence window stretches to // VAD_EARLY_SILENCE_MS: a thinking pause right after the first word can no // longer commit a one-word fragment of the sentence. const VAD_EARLY_SILENCE_MS=1800; const VAD_COMMITTED_SPEECH_MS=1200; // The private STT server freezes a speculative end-of-speech snapshot after // 650ms of server-side silence even when this client never sent speculate. // Speech resuming after a gap that long always sends an explicit resume so // a frozen first-word snapshot can never survive into the commit. const SERVER_EOS_SILENCE_MS=650; // Never speculate on the first inter-word gap of a young utterance: a // wasted Whisper pass over one word occupies the Jetson for seconds and // stalls the real commit that follows (observed live as multi-second // commit pending waits). const SPECULATE_MIN_SPEECH_MS=700; const TTS_SPEED_DEFAULT=1.15; const TTS_SPEED_MIN=0.5; const TTS_SPEED_MAX=2; const THINKING_CUE_FIRST_MS=1900; const THINKING_CUE_INTERVAL_MS=6500; // Natural spoken fillers first (Umm/Hmm/One sec and their es/ru equivalents), // then the longer reassurances — one short filler per genuine >~1.9s thinking // gap (THINKING_CUE_FIRST_MS), the answer always preempting via // cancelThinkingCues, rotated deterministically per turn so they never repeat // back-to-back, spoken through the SAME unified sink as the reply. const THINKING_CUE_POOLS={ en:[{id:'umm',text:'Umm.'},{id:'hmm',text:'Hmm.'},{id:'one_sec',text:'One sec.'},{id:'thinking',text:"I'm thinking."},{id:'let_me_think',text:'Let me think.'},{id:'still_working',text:'Still working on that.'},{id:'one_more_moment',text:'One more moment.'}], ru:[{id:'hmm',text:'Хм.'},{id:'sec',text:'Секунду.'},{id:'minute',text:'Минутку.'},{id:'thinking',text:'Я думаю.'},{id:'let_me_think',text:'Дайте подумать.'},{id:'still_working',text:'Я всё ещё думаю над этим.'},{id:'one_more_moment',text:'Ещё мгновение.'}], es:[{id:'mmm',text:'Mmm.'},{id:'a_ver',text:'A ver.'},{id:'un_momento',text:'Un momento.'},{id:'thinking',text:'Estoy pensando.'},{id:'let_me_think',text:'Déjame pensar.'},{id:'still_working',text:'Sigo pensando en eso.'},{id:'one_more_moment',text:'Un momento más.'}], }; const STATE_LABELS={ listening:'Listening', transcribing:'Transcribing…', thinking:'Thinking…', speaking:'Speaking', error:'Voice unavailable', idle:'', }; // The only language signal this file trusts is the one the private Whisper // service returned for the audio of the turn currently being answered. It is // bound to that turn's generation token and consumed exactly once. let sttLanguage=''; let sttLanguageToken=-1; // Sticky per-hands-free-session language: set by the private Whisper // detection of the user's own speech (or a strong reply-text signal on a // detection-less turn), consumed as the NEXT streaming STT session's bias // and as the thinking-cue locale. Cleared on activate/deactivate. let sessionLanguage=''; // Forced conversation-mode language (FIX 3): when non-empty it overrides both // the streaming STT language hint and the reply-TTS voice, superseding // auto-detection until the user picks Auto again. Session-scoped; no storage. let forcedLanguage=''; // Bounded retry budget for the completion (isFinal) pump: the STREAM_DONE // callback can beat the settle re-render that stamps the answer onto // data-raw-text, so a transient empty read must never tear a live reply down. let finalizeAttempts=0; // Bounded auto-retry for a TRANSIENT provider error (a broker 5xx/502 or // "error sending request" reply — e.g. a model broker that blipped while its // pod rolled mid-conversation). Instead of silently dropping the user's // utterance and forcing them to repeat it, re-run the last user turn through // the app's own regenerate path (which truncates the errored turn, so no // duplicate user message), capped so a persistent failure still surfaces. // Reset on any real answer and on each new user utterance. let transientRetryCount=0; const MAX_TRANSIENT_RETRIES=2; const RECONNECT_CAPTION={en:'Reconnecting…',ru:'Переподключение…',es:'Reconectando…'}; const TRANSIENT_ERROR_RE=/\b(429|50[0-9])\b|error sending request|bad gateway|gateway timeout|timed? ?out|timeout|temporarily|overloaded|unavailable|connection (refused|reset|error)|reset by peer|upstream/i; const TTS_LANGUAGES=['en','ru','es']; // FIX 3: conversation-mode language chooser. Auto (default) plus the languages // the private voice map supports — kept in lockstep with // dockerfiles/hermes-jetson-tts-server.py LANGUAGE_VOICE_MAP (en → amy, // es → claude, ru → irina). Selecting one FORCES both the reply TTS voice and // the streaming STT language hint for the rest of the hands-free session (via // forcedLanguage), overriding auto-detection until Auto is chosen again. const CONVERSATION_LANGUAGES=[ {code:'',label:'Auto',sublabel:'Detect'}, {code:'en',label:'English',sublabel:'English'}, {code:'es',label:'Español',sublabel:'Spanish'}, {code:'ru',label:'Русский',sublabel:'Russian'}, ]; const GLOBE_ICON_SVG=''; const SPEAKER_ICON_SVG=''; const originalAutoRead=window.autoReadLastAssistant; const originalApplyPreference=window._applyVoiceModePref; function normalizeSttLanguage(value){ if(typeof value!=='string') return ''; const code=value.trim().toLowerCase(); return TTS_LANGUAGES.indexOf(code)>=0?code:''; } function clearSttLanguage(){ sttLanguage=''; sttLanguageToken=-1; } function rememberSttLanguage(language,token){ sttLanguage=language||''; sttLanguageToken=sttLanguage?token:-1; } function takeSttLanguage(token){ const language=sttLanguageToken===token?sttLanguage:''; clearSttLanguage(); return language; } // Inverted punctuation is exclusive to Spanish and decisive on its own. const SPANISH_UNIQUE=/[¡¿]/; // Accented vowels and ñ also occur in English loanwords and European place // names (Zürich, Málaga, Genève, jalapeño), so they signal Spanish only // alongside real Spanish stopword density — never on their own. const SPANISH_ACCENTS=/[áéíóúüñ]/gi; const SPANISH_STOPWORDS=/\b(?:el|la|los|las|un|una|es|está|qué|para|por|con|pero|como|más|sí|gracias|hola|puedo|también|muy|este|esta|todo|bien)\b/g; const CYRILLIC_CHARS=/[Ѐ-ӿ]/g; const WORD_LETTERS=/[A-Za-zÀ-ÿЀ-ӿ]/g; function strongReplyLanguage(text){ // Decisive script/orthography evidence in the REPLY text only. This is // ranked ABOVE the trusted STT detection, so it must never fire on // incidental foreign glyphs: a lone accented European place name, or a // single stray Cyrillic letter inside otherwise-English prose, keeps the // English (Amy) voice. Russian needs Cyrillic-letter density; Spanish needs // inverted punctuation, or accent/ñ density corroborated by Spanish // stopword density. (Root cause of the "English reply about Europe spoken in // a foreign voice" bug: the old rule returned 'es' for a single accented // char and 'ru' for a single Cyrillic char, overriding a correct 'en' STT.) const sample=String(text||'').slice(0,400); const letters=(sample.match(WORD_LETTERS)||[]).length; const cyrillic=(sample.match(CYRILLIC_CHARS)||[]).length; if(cyrillic>=4&&letters>0&&cyrillic/letters>=0.5) return 'ru'; if(SPANISH_UNIQUE.test(sample)) return 'es'; const accents=(sample.match(SPANISH_ACCENTS)||[]).length; if(accents>=2){ const lower=sample.toLowerCase(); const words=lower.split(/\s+/).filter(Boolean); const stops=(lower.match(SPANISH_STOPWORDS)||[]).length; if(words.length>=4&&stops>=2&&stops/words.length>=0.12) return 'es'; } return ''; } function detectReplyLanguage(text){ // Lightweight reply-language heuristic for turns without a trusted STT // detection: decisive script/orthography evidence first, then accent-free // Spanish stopword density (an accent-free Spanish sentence still routes to // the Spanish voice). Returns '' for English/unknown, which the private TTS // service resolves to its own English default voice. const strong=strongReplyLanguage(text); if(strong) return strong; const sample=String(text||'').slice(0,400).toLowerCase(); const words=sample.split(/\s+/).filter(Boolean); if(words.length>=4){ const matches=(sample.match(SPANISH_STOPWORDS)||[]).length; if(matches>=2&&matches/words.length>=0.12) return 'es'; } return ''; } function resolveReplyLanguage(text,sttLanguage,forced){ // The single source of truth for which Piper voice reads a reply. A user // force wins outright; otherwise decisive reply-text evidence corrects a // wrong or missing STT detection; otherwise the turn's trusted STT-detected // language is spoken; otherwise the accent-free stopword heuristic. Plain // English prose — including a reply full of European proper nouns — carries // no decisive evidence and matches no STT hint of its own, so it can never be // flipped off the English voice by a previous turn's sticky value: only a // real force or real reply-text evidence moves it. if(forced) return forced; const strong=strongReplyLanguage(text); if(strong) return strong; const stt=normalizeSttLanguage(sttLanguage); if(stt) return stt; return detectReplyLanguage(text)||''; } function toast(message){ if(typeof window.showToast==='function') window.showToast(message,3000); } function setState(next,customLabel){ state=next; indicator.className='voice-mode-indicator '+next; bar.dataset.voiceState=next; bar.setAttribute('aria-busy',next==='transcribing'||next==='thinking'?'true':'false'); label.textContent=customLabel||STATE_LABELS[next]||''; bar.style.display=(active&&next!=='idle')||next==='error'?'':'none'; resetInputLevel(); syncConversationOverlay(next,customLabel); } function resetInputLevel(){ visualInputLevel=0; indicator.style.setProperty('--voice-ripple-scale','1.035'); indicator.style.setProperty('--voice-ripple-opacity','0.3'); } function updateInputLevel(rms){ if(reducedMotion.matches||state!=='listening') return; const target=Math.max(0,Math.min(1,(rms-0.01)/0.18)); visualInputLevel=(visualInputLevel*0.72)+(target*0.28); indicator.style.setProperty('--voice-ripple-scale',(1.035+(visualInputLevel*0.16)).toFixed(3)); indicator.style.setProperty('--voice-ripple-opacity',(0.26+(visualInputLevel*0.48)).toFixed(3)); if(conversation) conversation.root.style.setProperty('--conversation-level',visualInputLevel.toFixed(3)); } // ── Conversation mode overlay ───────────────────────────────────────── // Hands-free is a full-screen conversation: a breathing orb that follows // microphone energy while listening and TTS playback while speaking, the // live state caption, streaming user/assistant captions, and mute/exit // controls. The overlay is created lazily on activation, removed completely // on deactivate, keeps no storage, and is silently skipped in environments // without a usable DOM (headless contract probes). let conversation=null; // FIX 4: when the model has been thinking for a beat with no reply text yet, // the overlay shows a subtle animated "working…" affordance so the dead time // reads as alive rather than frozen. Purely a client progress hint — it never // fabricates spoken acknowledgements (that is the model's job). let awaitingTimer=null; const AWAITING_AFFORDANCE_MS=2500; function conversationUsable(){ return !!(typeof document!=='undefined'&&document&&document.body&&typeof document.createElement==='function'); } function conversationWarn(message,error){ try{if(window.console&&window.console.warn) window.console.warn('[atlas-voice] '+message,error);}catch(_){ } } function clearAwaitingAffordance(){ if(awaitingTimer){window.clearTimeout(awaitingTimer);awaitingTimer=null;} if(conversation&&conversation.root.classList) conversation.root.classList.remove('is-awaiting'); } function updateAwaitingAffordance(next){ // Arm the affordance only while Thinking and only after the grace window; // any other state, or the first scrap of reply text, disarms it. clearAwaitingAffordance(); if(!conversation||next!=='thinking') return; awaitingTimer=window.setTimeout(function(){ awaitingTimer=null; if(conversation&&conversation.root.dataset.voiceState==='thinking'&&conversation.root.classList){ conversation.root.classList.add('is-awaiting'); } },AWAITING_AFFORDANCE_MS); } function conversationNode(tag,className,attributes){ const node=document.createElement(tag); if(className) node.className=className; if(attributes) Object.keys(attributes).forEach(function(name){node.setAttribute(name,attributes[name]);}); return node; } function syncConversationOverlay(next,customLabel){ if(!conversation) return; conversation.root.dataset.voiceState=next; conversation.stateEl.textContent=customLabel||STATE_LABELS[next]||''; updateAwaitingAffordance(next); } function updateCaptionRegion(element,text,limit){ // Captions are bounded scrollable regions: keep following the streaming // tail unless the user scrolled up inside this region (data-follow='0', // maintained by the scroll listener installed at overlay build time). const value=String(text||'').slice(-limit); element.textContent=value; if(!value&&element.dataset) element.dataset.follow='1'; if((!element.dataset||element.dataset.follow!=='0')&&typeof element.scrollHeight==='number'){ try{element.scrollTop=element.scrollHeight;}catch(_){ } } } function attachCaptionScroll(element){ if(!element.addEventListener) return; element.addEventListener('scroll',function(){ const gap=(element.scrollHeight||0)-(element.scrollTop||0)-(element.clientHeight||0); if(element.dataset) element.dataset.follow=gap<=24?'1':'0'; }); } function setConversationUserCaption(text){ if(conversation) updateCaptionRegion(conversation.userCaption,text,4000); } function setConversationAssistantCaption(text){ if(!conversation) return; // The first scrap of reply text means the model is no longer silently // thinking: disarm the "working…" affordance immediately (FIX 4). if(text&&String(text).trim()) clearAwaitingAffordance(); updateCaptionRegion(conversation.assistantCaption,text,9000); } function setConversationPlaying(playing){ if(!conversation) return; if(playing) conversation.root.classList.add('is-playing'); else conversation.root.classList.remove('is-playing'); } function applyConversationMute(){ // True microphone mute on the session's single retained getUserMedia // stream: disabled tracks deliver silence to the recorder, the VAD and // the streaming STT worklet alike. No second capture is ever opened. if(!conversation) return; try{ const tracks=stream&&stream.getAudioTracks?stream.getAudioTracks():[]; tracks.forEach(function(track){track.enabled=!conversation.muted;}); }catch(_){ } } function setConversationMuted(muted){ if(!conversation) return; conversation.muted=muted; applyConversationMute(); try{ conversation.muteBtn.setAttribute('aria-pressed',muted?'true':'false'); conversation.muteBtn.textContent=muted?'Unmute microphone':'Mute microphone'; if(muted) conversation.root.classList.add('is-muted'); else conversation.root.classList.remove('is-muted'); }catch(_){ } } function conversationKeydown(event){ if(!conversation) return; if(event.key==='Escape'){ if(event.preventDefault) event.preventDefault(); // Escape peels one layer at a time: an open language menu closes first // (focus returns to its button), and only a second Escape exits the mode. if(conversation.langMenu&&!conversation.langMenu.hidden){closeLanguageMenu(true);return;} if(conversation.outMenu&&!conversation.outMenu.hidden){closeOutputMenu(true);return;} deactivate(true); return; } if(event.key==='Tab'){ // Minimal focus trap across the overlay controls (language, output, mute, // exit); the output button only joins the trap once it is actually shown. const outVisible=conversation.outWrap&&conversation.outWrap.style&&conversation.outWrap.style.display!=='none'; const stops=[conversation.langBtn,outVisible?conversation.outBtn:null,conversation.muteBtn,conversation.exitBtn].filter(Boolean); const current=stops.indexOf(document.activeElement); const index=current<0?(event.shiftKey?0:stops.length-1):current; const next=stops[(index+(event.shiftKey?stops.length-1:1))%stops.length]; if(event.preventDefault) event.preventDefault(); if(next&&next.focus) next.focus(); } } function closeLanguageMenu(focusButton){ if(!conversation||!conversation.langMenu) return; conversation.langMenu.hidden=true; conversation.langMenu.setAttribute('hidden',''); if(conversation.langBtn){ conversation.langBtn.setAttribute('aria-expanded','false'); if(focusButton&&conversation.langBtn.focus) conversation.langBtn.focus(); } } function openLanguageMenu(){ if(!conversation||!conversation.langMenu) return; conversation.langMenu.hidden=false; conversation.langMenu.removeAttribute('hidden'); if(conversation.langBtn) conversation.langBtn.setAttribute('aria-expanded','true'); // Focus the currently selected option so keyboard users land on it. const items=conversation.langItems||[]; const active=items.filter(function(item){return item.getAttribute('aria-checked')==='true';})[0]||items[0]; if(active&&active.focus) active.focus(); } function toggleLanguageMenu(){ if(!conversation||!conversation.langMenu) return; if(conversation.langMenu.hidden) openLanguageMenu(); else closeLanguageMenu(true); } function reflectLanguageSelection(){ if(!conversation||!conversation.langItems) return; conversation.langItems.forEach(function(item){ const selected=(item.getAttribute('data-lang')||'')===forcedLanguage; item.setAttribute('aria-checked',selected?'true':'false'); }); if(conversation.langBtn){ const current=CONVERSATION_LANGUAGES.filter(function(entry){return entry.code===forcedLanguage;})[0]; const label=current?current.label:'Auto'; conversation.langBtn.setAttribute('aria-label','Conversation language: '+label); conversation.langBtn.setAttribute('title','Language — '+label); if(conversation.langBtn.classList){ if(forcedLanguage) conversation.langBtn.classList.add('is-forced'); else conversation.langBtn.classList.remove('is-forced'); } } } function selectConversationLanguage(code){ // Force (or, for '', release back to auto-detection) the session language. forcedLanguage=code||''; reflectLanguageSelection(); // Re-bias the open streaming STT session immediately when it is safe — i.e. // it has heard nothing yet — so the forced hint applies to the very next // utterance instead of only the one after it. Mirrors the auto-switch path. if(active&&captureActive&&streamingStt&&typeof streamingStt.latestPartial==='function'&&!streamingStt.latestPartial()){ startListening(generation,{preserveDisplay:true}); } } // Never throws: on any failure it returns an inert, hidden control object whose // .wrap is still a safe, appendable element, so the overlay can attach without // this enhancement. langBtn/langMenu stay null and every menu helper no-ops. function inertControl(className){ let wrap; try{wrap=conversationNode('div',className);wrap.style.display='none';} catch(_){wrap=null;} return {wrap:wrap,btn:null,menu:null,items:[]}; } function buildLanguageControl(){ try{ const wrap=conversationNode('div','voice-conversation-lang'); const btn=conversationNode('button','voice-conversation-lang-btn',{ type:'button','aria-haspopup':'menu','aria-expanded':'false', 'aria-label':'Conversation language: Auto',title:'Language — Auto', }); btn.innerHTML=GLOBE_ICON_SVG; const menu=conversationNode('div','voice-conversation-lang-menu',{role:'menu','aria-label':'Conversation language',hidden:''}); menu.hidden=true; const items=CONVERSATION_LANGUAGES.map(function(entry){ const item=conversationNode('button','voice-conversation-lang-item',{ type:'button',role:'menuitemradio','data-lang':entry.code, 'aria-checked':entry.code===forcedLanguage?'true':'false', }); item.textContent=entry.label; item.addEventListener('click',function(){ selectConversationLanguage(entry.code); closeLanguageMenu(true); }); menu.appendChild(item); return item; }); btn.addEventListener('click',function(event){ if(event&&event.stopPropagation) event.stopPropagation(); toggleLanguageMenu(); }); wrap.appendChild(btn); wrap.appendChild(menu); return {wrap:wrap,btn:btn,menu:menu,items:items}; }catch(error){ conversationWarn('language control unavailable',error); return inertControl('voice-conversation-lang'); } } // ── Output-device selector ──────────────────────────────────────────── // A tidy corner control, styled like the language chooser, to pick which audio // output device the voice plays through. Feature-detected (enumerateDevices + // setSinkId); the whole control stays hidden when routing is unsupported or no // output devices are enumerable. Session-only — it never touches storage. function closeOutputMenu(focusButton){ if(!conversation||!conversation.outMenu) return; conversation.outMenu.hidden=true; conversation.outMenu.setAttribute('hidden',''); if(conversation.outBtn){ conversation.outBtn.setAttribute('aria-expanded','false'); if(focusButton&&conversation.outBtn.focus) conversation.outBtn.focus(); } } function openOutputMenu(){ if(!conversation||!conversation.outMenu) return; conversation.outMenu.hidden=false; conversation.outMenu.removeAttribute('hidden'); if(conversation.outBtn) conversation.outBtn.setAttribute('aria-expanded','true'); const items=conversation.outItems||[]; const active=items.filter(function(item){return item.getAttribute('aria-checked')==='true';})[0]||items[0]; if(active&&active.focus) active.focus(); } function toggleOutputMenu(){ if(!conversation||!conversation.outMenu) return; if(conversation.outMenu.hidden) openOutputMenu(); else closeOutputMenu(true); } function outputDeviceLabel(device,index){ const label=device&&device.label?String(device.label):''; if(label) return label; return 'Output '+(index+1); } function reflectOutputSelection(){ if(!conversation||!conversation.outItems) return; conversation.outItems.forEach(function(item){ const selected=(item.getAttribute('data-device')||'')===selectedOutputSinkId; item.setAttribute('aria-checked',selected?'true':'false'); }); if(conversation.outBtn){ const chosen=outputDevices.filter(function(device){return device.deviceId===selectedOutputSinkId;})[0]; const label=selectedOutputSinkId&&chosen?outputDeviceLabel(chosen,0):'System default'; conversation.outBtn.setAttribute('aria-label','Audio output: '+label); conversation.outBtn.setAttribute('title','Output — '+label); if(conversation.outBtn.classList){ if(selectedOutputSinkId) conversation.outBtn.classList.add('is-forced'); else conversation.outBtn.classList.remove('is-forced'); } } } function selectOutputDevice(deviceId){ // An explicit user choice: it is honoured for the rest of the session and is // never overridden by the auto-default-to-loudspeaker on later refreshes. outputSinkUserChosen=true; selectedOutputSinkId=deviceId||''; reflectOutputSelection(); applyOutputSink(); } function pickLoudspeakerSink(devices){ // FIX 2(a): choose the actual LOUDSPEAKER so hands-free playback does not // follow the OS "communication" route — while getUserMedia holds the mic // open the browser/OS puts audio in communication mode and the system // default endpoint becomes the EARPIECE. We therefore bias to an explicit // speaker deviceId rather than leaving the sink on the system/communications // default. Never target the "communications" pseudo-endpoint (that IS the // earpiece route). Prefer a labelled speaker; else the first concrete, // non-earpiece output; else '' (fall back to the system default). const outs=(devices||[]).filter(function(device){ return device&&device.kind==='audiooutput'&&device.deviceId&&device.deviceId!=='communications'; }); const speaker=/(speaker|speakerphone|loud)/i; const earpiece=/(earpiece|receiver|handset|headset|headphone|earbud|bluetooth|communication)/i; const named=outs.filter(function(device){ const label=device.label||'';return speaker.test(label)&&!earpiece.test(label); })[0]; if(named) return named.deviceId; const concrete=outs.filter(function(device){ return device.deviceId!=='default'&&!earpiece.test(device.label||''); })[0]; return concrete?concrete.deviceId:''; } async function refreshOutputDevices(){ if(!conversation||!conversation.outMenu||!conversation.outWrap) return; if(!outputRoutingSupported()){conversation.outWrap.style.display='none';return;} let devices=[]; try{devices=await navigator.mediaDevices.enumerateDevices();}catch(_){devices=[];} // Real, routable outputs only: the empty-deviceId "default" device the // browser reports is represented by our own "System default" entry. outputDevices=(devices||[]).filter(function(device){return device&&device.kind==='audiooutput'&&device.deviceId;}); if(!conversation||!conversation.outMenu) return; // Auto-default the shared sink to the loudspeaker on first enumeration so // hands-free speech never plays out the earpiece. Only when the user has not // made an explicit choice and nothing is selected yet — a user choice wins. if(!outputSinkUserChosen&&!selectedOutputSinkId){ const speaker=pickLoudspeakerSink(devices); if(speaker){selectedOutputSinkId=speaker;applyOutputSink();} } // Rebuild the menu: a "System default" entry plus every routable output. conversation.outMenu.children=[]; const entries=[{deviceId:'',label:'System default'}].concat( outputDevices.map(function(device,index){return {deviceId:device.deviceId,label:outputDeviceLabel(device,index)};}) ); conversation.outItems=entries.map(function(entry){ const item=conversationNode('button','voice-conversation-out-item',{ type:'button',role:'menuitemradio','data-device':entry.deviceId, 'aria-checked':entry.deviceId===selectedOutputSinkId?'true':'false', }); item.textContent=entry.label; item.addEventListener('click',function(){ selectOutputDevice(entry.deviceId); closeOutputMenu(true); }); conversation.outMenu.appendChild(item); return item; }); // Only the default entry means nothing is really selectable — hide it. conversation.outWrap.style.display=outputDevices.length?'':'none'; reflectOutputSelection(); } // Never throws (see buildLanguageControl): device access is done later, and // asynchronously, in refreshOutputDevices — buildOutputControl only assembles // the inert corner button, hidden until routable outputs are confirmed. function buildOutputControl(){ try{ const wrap=conversationNode('div','voice-conversation-out'); // Hidden until refreshOutputDevices confirms real, routable outputs exist. wrap.style.display='none'; const btn=conversationNode('button','voice-conversation-out-btn',{ type:'button','aria-haspopup':'menu','aria-expanded':'false', 'aria-label':'Audio output: System default',title:'Output — System default', }); btn.innerHTML=SPEAKER_ICON_SVG; const menu=conversationNode('div','voice-conversation-out-menu',{role:'menu','aria-label':'Audio output device',hidden:''}); menu.hidden=true; btn.addEventListener('click',function(event){ if(event&&event.stopPropagation) event.stopPropagation(); toggleOutputMenu(); }); wrap.appendChild(btn); wrap.appendChild(menu); return {wrap:wrap,btn:btn,menu:menu}; }catch(error){ conversationWarn('output control unavailable',error); return inertControl('voice-conversation-out'); } } function openConversationOverlay(){ if(conversation||!conversationUsable()) return; let root=null; // ── Phase 1: the ESSENTIAL overlay (orb + captions + mute/exit). This is the // full-screen visualization the user relies on; it MUST attach whenever // conversation mode activates. The corner language/output selectors are // enhancements added in phase 2 — a failure there can never keep this from // rendering (which would drop us back to the inline voice bar). Only a // failure to build this core is fatal, and is the sole reason for fallback. try{ root=conversationNode('div','voice-conversation',{ role:'dialog','aria-modal':'true','aria-label':'Voice conversation',tabindex:'-1', }); const orb=conversationNode('div','voice-conversation-orb',{'aria-hidden':'true'}); orb.appendChild(conversationNode('span','voice-conversation-orb-halo')); orb.appendChild(conversationNode('span','voice-conversation-orb-core')); orb.appendChild(conversationNode('span','voice-conversation-orb-ring')); // FIX 1: the Hermes character glyph watermark, centred in the orb beneath // the energy layers. It is a presentation-only span; atlas-voice.css paints // the inlined, circle-feathered character data URI into it (screen blend). const orbMark=conversationNode('span','voice-conversation-orb-mark',{'aria-hidden':'true'}); orb.appendChild(orbMark); const stateEl=conversationNode('div','voice-conversation-state',{'aria-hidden':'true'}); const captions=conversationNode('div','voice-conversation-captions',{'aria-live':'polite'}); // Each caption is its own bounded, independently scrollable region // (keyboard focusable, touch scrollable); streaming keeps following the // tail until the user scrolls up inside that region. const userCaption=conversationNode('p','voice-conversation-caption-user',{tabindex:'0'}); const assistantCaption=conversationNode('p','voice-conversation-caption-assistant',{tabindex:'0'}); attachCaptionScroll(userCaption); attachCaptionScroll(assistantCaption); captions.appendChild(userCaption); captions.appendChild(assistantCaption); const controls=conversationNode('div','voice-conversation-controls'); const muteBtn=conversationNode('button','voice-conversation-mute',{type:'button','aria-pressed':'false'}); muteBtn.textContent='Mute microphone'; const exitBtn=conversationNode('button','voice-conversation-exit',{type:'button'}); exitBtn.textContent='Exit voice mode'; controls.appendChild(muteBtn); controls.appendChild(exitBtn); root.appendChild(orb); root.appendChild(stateEl); root.appendChild(captions); root.appendChild(controls); muteBtn.addEventListener('click',function(){if(conversation) setConversationMuted(!conversation.muted);}); exitBtn.addEventListener('click',function(){deactivate(true);}); root.addEventListener('keydown',conversationKeydown); // A tap anywhere outside a menu closes it (never deactivates). root.addEventListener('click',function(event){ if(!conversation) return; const target=event&&event.target; const closest=target&&typeof target.closest==='function'?target.closest.bind(target):null; if(conversation.langMenu&&!conversation.langMenu.hidden&&!(closest&&closest('.voice-conversation-lang'))) closeLanguageMenu(false); if(conversation.outMenu&&!conversation.outMenu.hidden&&!(closest&&closest('.voice-conversation-out'))) closeOutputMenu(false); }); document.body.appendChild(root); // Selector fields start null; phase 2 fills them in if their build succeeds. conversation={root:root,orb:orb,stateEl:stateEl,userCaption:userCaption,assistantCaption:assistantCaption,muteBtn:muteBtn,exitBtn:exitBtn,langBtn:null,langMenu:null,langItems:[],outWrap:null,outBtn:null,outMenu:null,outItems:[],muted:false}; }catch(error){ // The essential overlay could not even be built/attached: fall back to the // compact voice bar as a last resort, leaving nothing half-attached. conversationWarn('conversation overlay unavailable',error); if(root&&root.parentNode){try{root.parentNode.removeChild(root);}catch(_e){ }} conversation=null; return; } // ── Phase 2: OPTIONAL corner selectors. Each is built and wired under its own // guard so a synchronous failure (e.g. a mobile browser where mediaDevices // access throws) omits just that one control — the overlay stays up. Device // enumeration itself is async (refreshOutputDevices), run after attachment. try{ const lang=buildLanguageControl(); if(lang.wrap) root.appendChild(lang.wrap); conversation.langBtn=lang.btn;conversation.langMenu=lang.menu;conversation.langItems=lang.items||[]; reflectLanguageSelection(); }catch(error){conversationWarn('language selector omitted',error);} try{ const out=buildOutputControl(); if(out.wrap) root.appendChild(out.wrap); conversation.outWrap=out.wrap;conversation.outBtn=out.btn;conversation.outMenu=out.menu; // Async device enumeration; tolerate rejection without touching the overlay. refreshOutputDevices().catch(function(error){conversationWarn('output devices unavailable',error);}); }catch(error){conversationWarn('output selector omitted',error);} try{ syncConversationOverlay(state); if(root.focus) root.focus(); }catch(error){conversationWarn('overlay finalize warning',error);} } function removeConversationOverlay(){ clearAwaitingAffordance(); const overlay=conversation; conversation=null; if(!overlay) return; try{ if(overlay.root.removeEventListener) overlay.root.removeEventListener('keydown',conversationKeydown); if(overlay.root.parentNode) overlay.root.parentNode.removeChild(overlay.root); }catch(_){ } // The session's tracks are always left enabled on the way out; a real // deactivation stops them entirely via releaseMicrophone(). try{ const tracks=stream&&stream.getAudioTracks?stream.getAudioTracks():[]; tracks.forEach(function(track){track.enabled=true;}); }catch(_){ } } function clearErrorTimer(){ if(errorTimer){window.clearTimeout(errorTimer);errorTimer=null;} } function errorMessage(error,fallback){ return String((error&&error.message)||fallback||'Voice unavailable').trim(); } function createAbortController(){ const Controller=window.AbortController||(typeof AbortController!=='undefined'?AbortController:null); return Controller?new Controller():{signal:undefined,abort:function(){ }}; } function cancelledError(message){ const error=new Error(message||'Voice turn cancelled'); error.name='AbortError'; return error; } function cancelSpeechTurn(){ if(!speechTurn) return; cancelSpeakingIdleFallback(speechTurn); speechTurn.cancelled=true; speechTurn.final=true; while(speechTurn.waiters.length) speechTurn.waiters.shift()(null); speechTurn=null; } function cancelStreamingStt(){ if(!streamingStt) return; streamingStt.cancel(); streamingStt=null; } function cancelVoicePreflight(turnId){ const preflight=voicePreflight; if(!preflight||(turnId&&preflight.turnId!==turnId)) return; voicePreflight=null; if(preflight.timer){window.clearTimeout(preflight.timer);preflight.timer=null;} if(preflight.deadline){window.clearTimeout(preflight.deadline);preflight.deadline=null;} if(preflight.controller){preflight.controller.abort();preflight.controller=null;} } function scheduleVoicePreflight(turnId,revision,transcript){ const capability=streamingCapability.preflight; const stable=String(transcript||'').replace(/\s+/g,' ').trim(); if(!capability||!voiceTabNonce||!active||captureTurnId!==turnId||state!=='listening') return; if(!Number.isInteger(revision)||revision<1||stable.length<12||stable.length>512) return; if(voicePreflight&&voicePreflight.turnId===turnId&&voicePreflight.transcript===stable) return; cancelVoicePreflight(); const preflight={turnId:turnId,revision:revision,transcript:stable,timer:null,deadline:null,controller:null}; voicePreflight=preflight; preflight.timer=window.setTimeout(async function(){ preflight.timer=null; if(voicePreflight!==preflight||!active||captureTurnId!==turnId||state!=='listening') return; const controller=createAbortController(); preflight.controller=controller; preflight.deadline=window.setTimeout(function(){controller.abort();},VOICE_PREFLIGHT_DEADLINE_MS); try{ const response=await fetch(VOICE_PREFLIGHT_URL,{ method:'POST', credentials:'same-origin', cache:'no-store', headers:{'Content-Type':'application/json'}, body:JSON.stringify({turn_id:turnId,revision:revision,transcript:stable}), signal:controller.signal, }); if(!response.ok) return; const advisory=await response.json(); const tier=advisory&&advisory.tier; if( voicePreflight!==preflight||!active||captureTurnId!==turnId||state!=='listening'|| advisory.turn_id!==turnId||advisory.revision!==revision||advisory.advisory!==true|| VOICE_PREFLIGHT_TIERS.indexOf(tier)<0||advisory.target!=='atlas/auto/'+tier ) return; // This is intentionally feedback-only. The completed transcript still // takes the normal Switchyard path and remains the sole routing input. const preview=stable.length>60?stable.slice(0,57)+'…':stable; label.textContent='Listening · '+preview+' · '+tier; }catch(_){ // A partial can be superseded at any time; advisory failure is silent. }finally{ if(preflight.deadline){window.clearTimeout(preflight.deadline);preflight.deadline=null;} preflight.controller=null; } },VOICE_PREFLIGHT_DEBOUNCE_MS); } function cancelThinkingCues(){ const cue=thinkingCue; thinkingCue=null; if(!cue) return; cue.cancelled=true; if(cue.timer){window.clearTimeout(cue.timer);cue.timer=null;} if(cue.controller){cue.controller.abort();cue.controller=null;} if(cue.audioWake){const wake=cue.audioWake;cue.audioWake=null;wake();} if(cue.node){try{cue.node.port.postMessage({type:'cancel'});cue.node.disconnect();}catch(_){ }cue.node=null;} // The cue shares the ONE playback context; drop the reference but never // close the shared sink here. if(cue.context){cue.context=null;} cue.gain=null; indicator.classList.remove('is-playing'); indicator.classList.remove('is-ducked'); } function clearBargeCancellation(){ const cancellation=bargeCancelPromise; bargeCancelPromise=null; if(cancellation&&cancellation.controller) cancellation.controller.abort(); } function microphoneConstraints(){ const constraints={echoCancellation:true,noiseSuppression:true,autoGainControl:true}; const supported=navigator.mediaDevices.getSupportedConstraints?navigator.mediaDevices.getSupportedConstraints():{}; if(supported.voiceIsolation) constraints.voiceIsolation=true; return constraints; } function acquireMicrophone(){ return navigator.mediaDevices.getUserMedia({audio:microphoneConstraints()}); } function captureAecIsUsable(capture){ try{ const tracks=capture&&capture.getAudioTracks?capture.getAudioTracks():[]; const settings=tracks.length&&tracks[0].getSettings?tracks[0].getSettings():{}; return settings.echoCancellation!==false; }catch(_){ return true; } } window._atlasCaptureAecIsUsable=captureAecIsUsable; function setPlaybackDucked(ducked){ const session=playbackSession; if(session&&session.gain&&session.context){ const gain=ducked?0.16:1; try{session.gain.gain.setTargetAtTime(gain,session.context.currentTime,0.018);}catch(_){session.gain.gain.value=gain;} } if(currentAudio) currentAudio.volume=ducked?0.16:1; const cue=thinkingCue; if(cue&&cue.gain&&cue.context){ const gain=ducked?0.16:1; try{cue.gain.gain.setTargetAtTime(gain,cue.context.currentTime,0.018);}catch(_){cue.gain.gain.value=gain;} } if(ducked) indicator.classList.add('is-ducked'); else indicator.classList.remove('is-ducked'); } function playbackAudible(){ // True only while audio is actually sounding. A streaming PCM session // keeps its worklet node alive across sentence chunks and between the // speech segments of one turn; the buffered frames the worklet reports // decide whether it is audible right now. if(currentAudio) return true; if(thinkingCue&&thinkingCue.node) return true; const session=playbackSession; if(session&&session.node) return session.bufferedFrames>0&&!session.playbackEnded; return indicator.classList.contains('is-playing'); } function disposeBargeResources(monitor,keepCapture){ if(!monitor) return; monitor.cancelled=true; if(monitor.timer){window.clearInterval(monitor.timer);monitor.timer=null;} try{if(monitor.node) monitor.node.disconnect();}catch(_){ } try{if(monitor.source) monitor.source.disconnect();}catch(_){ } try{if(monitor.silentGain) monitor.silentGain.disconnect();}catch(_){ } if(!keepCapture){ if(monitor.stream) monitor.stream.getTracks().forEach(function(track){track.stop();}); if(monitor.context){try{monitor.context.close();}catch(_){ }} } if(!keepCapture) setPlaybackDucked(false); } function stopBargeMonitor(){ const monitor=bargeMonitor; bargeMonitor=null; disposeBargeResources(monitor,false); } function showUnavailable(message){ generation+=1; active=false; thinkingSession=null; thinkingTurnId=''; suppressAutoRead=false; pendingStitch=null; lastSentTranscript=null; clearBargeCancellation(); stopResponseObserver(); cancelThinkingCues(); cancelSpeechTurn(); releaseMicrophone(); stopPlayback(); closeSharedPlayback(); selectedOutputSinkId=''; outputSinkUserChosen=false; outputDevices=[]; removeConversationOverlay(); modeBtn.classList.remove('active'); setState('error',message); clearErrorTimer(); errorTimer=window.setTimeout(function(){ errorTimer=null; if(!active&&state==='error'&&indicator.classList.contains('error')) setState('idle'); },ERROR_VISIBLE_MS); } function stopCapture(){ // Tear down the per-utterance capture pipeline (recorder, VAD, streaming // STT, WebAudio graph) but deliberately KEEP the microphone stream and its // AudioContext, so the next capture turn starts instantly and speech // during transcribing/thinking is never lost. Only releaseMicrophone() // actually ends the session's single getUserMedia lease. captureGeneration+=1; captureActive=false; cancelVoicePreflight(captureTurnId); stopBargeMonitor(); if(vadTimer){clearInterval(vadTimer);vadTimer=null;} if(recorder&&recorder.state!=='inactive'){ try{recorder.stop();}catch(_){ } } recorder=null; if(captureNode){try{captureNode.disconnect();}catch(_){ }} captureNode=null; if(captureGraph){ // The AudioContext persists across turns, so every per-turn node must be // detached here or the shared graph would grow with each utterance. try{if(captureGraph.mediaSource) captureGraph.mediaSource.disconnect();}catch(_){ } try{if(captureGraph.highpass) captureGraph.highpass.disconnect();}catch(_){ } try{if(captureGraph.analyser) captureGraph.analyser.disconnect();}catch(_){ } try{if(captureGraph.silentGain) captureGraph.silentGain.disconnect();}catch(_){ } captureGraph=null; } cancelStreamingStt(); } function releaseMicrophone(){ // Full microphone teardown. Used only by deactivate(), showUnavailable() // and fatal capture errors; every other path retains the hot microphone. stopCapture(); if(stream){stream.getTracks().forEach(function(track){track.stop();});stream=null;} if(audioContext){try{audioContext.close();}catch(_){ }audioContext=null;} } function stopPlayback(){ if(currentAudio){ try{currentAudio.pause();currentAudio.currentTime=0;}catch(_){ } currentAudio=null; } if(playbackSession){ playbackSession.cancelled=true; if(playbackSession.blobWake){playbackSession.blobWake();playbackSession.blobWake=null;} playbackSession.controllers.forEach(function(controller){controller.abort();}); if(playbackSession.lowWaterWake){ playbackSession.lowWaterWake(); playbackSession.lowWaterWake=null; } if(playbackSession.drainWake){playbackSession.drainWake();playbackSession.drainWake=null;} if(playbackSession.node){ try{playbackSession.node.port.postMessage({type:'cancel'});playbackSession.node.disconnect();}catch(_){ } } // The context is the shared spoken-output sink; it is closed only when // hands-free mode ends (closeSharedPlayback), never per playback session. playbackSession.context=null; playbackSession=null; } indicator.classList.remove('is-playing'); indicator.classList.remove('is-ducked'); } function stopResponseObserver(){ if(responsePollTimer){clearInterval(responsePollTimer);responsePollTimer=null;} } function deactivate(showMessage){ generation+=1; active=false; thinkingSession=null; thinkingTurnId=''; clearErrorTimer(); removeConversationOverlay(); clearSttLanguage(); sessionLanguage=''; forcedLanguage=''; finalizeAttempts=0; suppressAutoRead=false; pendingStitch=null; lastSentTranscript=null; clearBargeCancellation(); stopResponseObserver(); cancelThinkingCues(); cancelSpeechTurn(); releaseMicrophone(); stopPlayback(); closeSharedPlayback(); selectedOutputSinkId=''; outputSinkUserChosen=false; outputDevices=[]; modeBtn.classList.remove('active'); setState('idle'); if(showMessage) toast('Conversation mode off'); } function restartSoon(token,delay){ window.setTimeout(function(){ if(!active||token!==generation) return; // Continuous capture: while the microphone pipeline is already running, // finishing a response only needs the display returned to Listening. // Rebuilding capture here would drop speech already being collected. if(captureActive){setState('listening');return;} startListening(token); },delay||500); } function resyncCapture(token,statusLabel){ // Full capture-turn resync: cancel any streaming STT session, drop the // lookback/pending buffers and start a fresh capture turn on the hot // microphone. Used after an errored turn and by barge-in edge cases // where the previous stream state is no longer trustworthy. if(!active||token!==generation) return; stopCapture(); startListening(token); setState('listening',statusLabel); } function errorTurnText(turn){ // The human-visible error text of an errored assistant turn: the provider // details block the renderer appends for a provider failure, else the body. if(!turn||typeof turn.querySelector!=='function') return ''; let txt=''; try{ const details=turn.querySelector('.provider-error-details'); if(details&&typeof details.textContent==='string') txt=details.textContent; if(!txt){ const body=turn.querySelector('.msg-body'); if(body&&typeof body.textContent==='string') txt=body.textContent; } }catch(_){ } return txt||''; } function errorTurnIsTransient(turn){ // A broker/provider blip (5xx/502, "error sending request", timeout, // overloaded) is recoverable by re-running the same turn; a content or // policy error is not, and must never be retried into an identical failure. return TRANSIENT_ERROR_RE.test(errorTurnText(turn)); } function findRegenerateButton(turn){ // The app renders a single "regenerate" action on the last assistant turn // (onclick="regenerateResponse(this)"): clicking it truncates the errored // turn and re-runs the last user message with no duplicate user turn. Match // the action buttons by class and filter by the onclick target so this stays // correct without relying on a descendant/substring attribute selector. if(!turn||typeof turn.querySelectorAll!=='function') return null; let buttons; try{buttons=turn.querySelectorAll('.msg-action-btn');}catch(_){return null;} if(!buttons) return null; for(let i=0;i=0) return button; } return null; } function retryTransientResponse(token,turn){ // Re-run the errored user turn through the app's own regenerate path and // keep the conversation in Thinking — its cues cover the reconnect gap — so // the retried answer streams straight into TTS. Bounded by // MAX_TRANSIENT_RETRIES; a busy session or a missing regenerate control // declines the retry so the caller drops cleanly. Returns true iff launched. if(typeof S==='undefined'||!S||!S.session||S.busy) return false; if(transientRetryCount>=MAX_TRANSIENT_RETRIES) return false; const button=findRegenerateButton(turn); if(!button) return false; transientRetryCount+=1; // Tear down only the response-side speech/observer state; do NOT resync // capture (that would drop the turn and re-open the microphone). thinkingSession=null; thinkingTurnId=''; finalizeAttempts=0; stopResponseObserver(); cancelThinkingCues(); cancelSpeechTurn(); stopPlayback(); pendingStitch=null; const localized=normalizeSttLanguage(sessionLanguage)||'en'; setConversationAssistantCaption(RECONNECT_CAPTION[localized]||RECONNECT_CAPTION.en); setState('thinking'); // Reset the answer baseline to the errored turn so the regenerated answer // (which replaces it) is detected as fresh output, then arm the observer // and the thinking cues before triggering the app's regenerate. rememberAssistantBaseline(); thinkingTurnId=String(token)+'-retry-'+String(transientRetryCount); startResponseObserver(token); scheduleThinkingCues(token,sessionLanguage,thinkingTurnId); try{button.click();}catch(_){return false;} return true; } function handleAssistantResponseError(token){ // An error/system envelope (cancellation notice, provider failure) is a // transcript artifact, not a reply: never feed it to TTS or the reply // caption. A TRANSIENT provider blip (a broker 5xx/502) is auto-retried in // place — up to MAX_TRANSIENT_RETRIES — so a momentary backend hiccup does // not silently discard the user's utterance and force them to repeat it. A // non-transient error, or an exhausted retry budget, shows a brief non- // spoken state and resynchronizes capture for the next utterance. const rows=assistantRows(); const turn=rows.length?assistantTurnOf(rows[rows.length-1]):null; if(turn&&errorTurnIsTransient(turn)&&retryTransientResponse(token,turn)) return; transientRetryCount=0; thinkingSession=null; thinkingTurnId=''; finalizeAttempts=0; clearSttLanguage(); stopResponseObserver(); cancelThinkingCues(); cancelSpeechTurn(); stopPlayback(); pendingStitch=null; lastSentTranscript=null; clearBargeCancellation(); setConversationAssistantCaption(''); resyncCapture(token,'Let’s try that again — listening'); } function assistantRows(){ return document.querySelectorAll('.msg-row[data-role="assistant"], .assistant-segment[data-raw-text]'); } function assistantTurnOf(row){ // Resolve a matched node to its whole turn container so multi-segment // turns (interim messages around tool calls) are read as one response. if(row&&typeof row.closest==='function'){ const turn=row.closest('.msg-row[data-role="assistant"]'); if(turn) return turn; } return row; } function segmentIsHidden(segment){ if(!segment) return true; if(segment.hidden===true) return true; return !!(typeof segment.getAttribute==='function'&&segment.getAttribute('aria-hidden')==='true'); } function segmentIsError(segment){ // Error/system envelopes: the renderer stamps data-error="1" on segments // matching its error patterns, and provider errors and cancellation // notices carry a .provider-error-details block inside the body. if(!segment) return false; if(segment.dataset&&segment.dataset.error==='1') return true; return !!(typeof segment.querySelector==='function'&&segment.querySelector('.provider-error-details')); } // Ground-truth DOM contract (verified against the live build-24 bundle, // ui.js renderMessages / messages.js ensureAssistantRow): // //
//
//
H
← avatar letter // Hermes ← author name //
//
//
… "Processed 13s" …
← worklog chip // ← folded interim //
//
← reasoning (optional) //
ANSWER
← the spoken answer //
//
//
// // The settled renderer stamps the CLEAN answer text (no avatar, no author, no // "Processed Ns" chip, no reasoning) onto every answer segment's data-raw-text // — the exact field the app's own autoReadLastAssistant() reads (ui.js:~8891). // Reasoning uses .thinking-card-body, tool activity uses .tool-call-group-body // and the worklog "Processed Ns" chip lives in the worklog group beside the // segments — so .msg-body is answer-only, and the concatenated data-raw-text of // the visible answer segments is the whole reply. The prior round's fallback to // turn.textContent scraped the avatar "H" + "Hermes" + "Processed 13s" whenever // a settle-frame race left no readable answer segment; that fallback is gone. const NON_ANSWER_SEGMENT_CLASSES=['assistant-segment-worklog-source','assistant-segment-anchor']; // A .msg-body is the spoken answer ONLY when it is not nested inside reasoning, // tool, worklog, run-status, error or role chrome. The live renderer uses // distinct classes for each, but this guard keeps chrome out of the caption and // TTS even if a future build nests a body differently. const CHROME_CONTAINER_SELECTOR='.thinking-card,.agent-activity-thinking,.tool-call-group,.tool-worklog-group,.tool-group,.tool-card,.wl-reason,.process-wakeup-card,.provider-error-details,.msg-role,.assistant-run-status,.assistant-segment-worklog-source'; function segmentHasClass(segment,name){ if(!segment) return false; if(segment.classList&&typeof segment.classList.contains==='function') return segment.classList.contains(name); return typeof segment.className==='string'&&segment.className.split(/\s+/).indexOf(name)>=0; } function segmentIsAnswer(segment){ // A visible answer segment: not folded into the worklog, not an anchor // scaffold, not hidden, not an error/system envelope. if(!segment) return false; if(segmentIsHidden(segment)) return false; if(segmentIsError(segment)) return false; for(let i=0;i120?tail.slice(tail.length-120):tail; return '\n[voice interruption: you were cut off after "'+bounded+'"]'; } async function sendTranscript(transcript,token,language,turnId){ if(!active||token!==generation) return; cancelVoicePreflight(turnId||captureTurnId); const text=String(transcript||'').trim(); if(!text){clearSttLanguage();restartSoon(token,350);return;} // A fresh user utterance starts its own transient-retry budget. transientRetryCount=0; if(!bargeCancelPromise&&typeof S!=='undefined'&&(S.busy||S.activeStreamId)){ // A new utterance finished while the previous model turn was still in // flight (continuous capture makes this a normal interruption): remember // the interrupted transcript for stitching, then cancel the stale turn. recordPendingStitch(); suppressAutoRead=true; bargeCancelPromise=cancelActiveModelTurn(); } // A turn whose response was barge-cancelled before any assistant output is // restated as one stitched message, so the model answers the full thought. const stitch=(pendingStitch&&pendingStitch.text&&(Date.now()-pendingStitch.at)<=STITCH_WINDOW_MS)?pendingStitch:null; pendingStitch=null; composer.value=stitch?stitch.text+' '+text+voiceCutMarker(stitch.cut):text; if(typeof window.autoResize==='function') window.autoResize(); setConversationUserCaption(composer.value); setConversationAssistantCaption(''); setState('thinking'); // The live VAD/STT capture is the barge-in detector while it runs; the // energy-only monitor remains only as a fallback when capture is down. if(!captureActive) startBargeMonitor(token); const cancellationSettled=await settleBargeCancellation(token); if(!active||token!==generation) return; if(!cancellationSettled){ pendingStitch=stitch; toast('The previous response did not stop. Please repeat your interruption.'); // The stream state is now suspect: rebuild the capture turn instead of // resuming a session whose server-side epoch may be stale. resyncCapture(token); return; } thinkingSession=(typeof S!=='undefined'&&S.session)?S.session.session_id:null; thinkingTurnId=turnId||captureTurnId||String(token)+'-'+String(++turnSequence); rememberAssistantBaseline(); rememberSttLanguage(language,token); // The user's detected speech language is the sticky hands-free session // language: it biases the next streaming STT session and localizes the // thinking cues until the user audibly switches again. if(language){ const switched=language!==sessionLanguage; sessionLanguage=language; // Continuous capture already opened the next session (with the // previous bias) before this turn's detection arrived: restart it with // the switched language, but only while it has heard nothing yet. if(switched&&captureActive&&streamingStt&&typeof streamingStt.latestPartial==='function'&&!streamingStt.latestPartial()){ startListening(generation,{preserveDisplay:true}); } } if(typeof window.send==='function'){ lastSentTranscript={text:composer.value,token:token}; window.send(); suppressAutoRead=false; startResponseObserver(token); scheduleThinkingCues(token,language||sessionLanguage,thinkingTurnId); } } function audioExtension(mimeType){ const normalized=String(mimeType||'').toLowerCase(); if(normalized.indexOf('wav')>=0) return 'wav'; if(normalized.indexOf('ogg')>=0) return 'ogg'; if(normalized.indexOf('mp4')>=0) return 'mp4'; return 'webm'; } async function transcribe(blob,token,turnId){ if(!active||token!==generation) return; setState('transcribing'); const ext=audioExtension(blob.type); const form=new FormData(); form.append('file',new File([blob],'voice-input.'+ext,{type:blob.type||'audio/'+ext})); try{ const response=await fetch('/api/transcribe',{method:'POST',body:form}); const payload=await response.json().catch(function(){return {};}); if(!response.ok) throw new Error(payload.error||('Whisper request failed: '+response.status)); sendTranscript(payload.transcript,token,normalizeSttLanguage(payload.language),turnId); }catch(error){ if(!active||token!==generation) return; const message=errorMessage(error,'Private Whisper is unavailable'); showUnavailable(message); toast(message); // If the browser supplies its own recognizer, hand control back to the // upstream voice implementation until the Jetson becomes healthy again. if(window.SpeechRecognition||window.webkitSpeechRecognition){ modeBtn.removeEventListener('click',onVoiceClick,true); window.setTimeout(function(){modeBtn.click();},50); } } } function websocketUrl(path){ const protocol=window.location&&window.location.protocol==='https:'?'wss:':'ws:'; return protocol+'//'+window.location.host+path; } function encodePcm16(samples){ const bytes=new ArrayBuffer(samples.length*2); const view=new DataView(bytes); for(let index=0;index=STT_MAX_ARCHIVE_BYTES) return; const retained=Math.min(bytes.byteLength,STT_MAX_ARCHIVE_BYTES-archiveBytes); const even=retained-(retained%2); if(!even) return; archive.push(bytes.slice(0,even)); archiveBytes+=even; } function reject(error){ if(settled) return; settled=true; cancelVoicePreflight(turnId); clearQueue(); rejectFinal(error instanceof Error?error:new Error(String(error||'Streaming transcription failed'))); } function sendJson(payload){ if(socket&&socket.readyState===1){socket.send(JSON.stringify(payload));return true;} return false; } function scheduleFlush(){ if(flushTimer||settled||!queue.length||!socket||socket.readyState!==1) return; flushTimer=window.setTimeout(function(){flushTimer=null;flush();},20); } function flush(){ clearFlushTimer(); if(settled){clearQueue();return;} if(!socket||socket.readyState!==1) return; while(queue.length&&socket.bufferedAmount<524288){ const bytes=queue.shift(); queuedBytes=Math.max(0,queuedBytes-bytes.byteLength); socket.send(bytes); } if(queue.length) scheduleFlush(); } try{ const csrf=String((window.__HERMES_CONFIG__&&window.__HERMES_CONFIG__.csrfToken)||''); const protocols=['hermes-voice-v1']; if(csrf) protocols.push('hermes-csrf.'+csrf); socket=new WebSocket(websocketUrl(STT_STREAM_PATH),protocols); socket.binaryType='arraybuffer'; }catch(error){ reject(error); return null; } socket.onopen=function(){ // Bias recognition toward the session's sticky language the moment the // user has audibly switched; 'auto' remains the cold-start default. sendJson({type:'start',turn_id:turnId,format:'pcm_s16le',sample_rate:16000,language:forcedLanguage||sessionLanguage||'auto'}); flush(); }; socket.onmessage=function(event){ if(typeof event.data!=='string') return; let payload; try{payload=JSON.parse(event.data);}catch(_){return;} if(payload.turn_id!==turnId) return; if(payload.type==='partial'&&payload.rolling===true){ const revision=Number(payload.revision); if(Number.isFinite(revision)&&revision>partialRevision){ partialRevision=revision; const stable=String(payload.stable_transcript||'').trim(); const provisional=String(payload.transcript||'').trim(); const visible=stable||provisional; // The newest rolling partial also feeds the dynamic endpoint below, // so it is tracked regardless of the display-only gating that keeps // the label and captions quiet outside the listening state. if(visible) lastPartialText=visible; if(visible&&active&&captureTurnId===turnId&&state==='listening'){ const preview=visible.length>72?visible.slice(0,69)+'…':visible; label.textContent='Listening · '+preview+(stable?'':' · provisional'); setConversationUserCaption(visible); if(stable!==lastPreflightText){ if(stable){ lastPreflightText=stable; scheduleVoicePreflight(turnId,revision,stable); } } } } }else if(payload.type==='final'&&!settled){ settled=true; cancelVoicePreflight(turnId); clearQueue(); clearArchive(); resolveFinal({transcript:payload.transcript||'',language:normalizeSttLanguage(payload.language)}); }else if(payload.type==='error'){ reject(new Error(payload.error||'Streaming transcription failed')); } }; socket.onerror=function(){reject(new Error('Streaming transcription connection failed'));}; socket.onclose=function(){if(!settled) reject(new Error('Streaming transcription closed before final result'));}; return { finalPromise:finalPromise, latestPartial:function(){return lastPartialText;}, setWorkletReady:function(value){workletReady=value;}, push:function(samples){ if(cancelled||committed||!workletReady) return; const pcm=resample(samples); if(!pcm.length) return; const bytes=encodePcm16(pcm); archivePcm(bytes); if(settled) return; if(queuedBytes+bytes.byteLength>STT_MAX_QUEUED_BYTES){ reject(new Error('Streaming transcription backpressure limit exceeded')); if(socket&&socket.readyState<2) socket.close(1008,'client backpressure'); return; } queue.push(bytes); queuedBytes+=bytes.byteLength; flush(); }, takeFallbackBlob:function(){ if(!archiveBytes) return null; const blob=pcm16WavBlob(archive,16000); clearArchive(); return blob; }, speculate:function(){ if(settled||committed||speculative) return; speculative=true; sendJson({type:'speculate',turn_id:turnId}); }, resume:function(){ // Deliberately not gated on a prior speculate() from this client: the // server also freezes an EOS snapshot on its own silence detector, so // resume must always reach the wire when speech restarts. The server // treats a redundant resume as a harmless epoch bump. if(settled||committed) return; speculative=false; sendJson({type:'resume',turn_id:turnId}); }, commit:function(){ if(settled||committed) return finalPromise; committed=true; flush(); const commitDeadline=Date.now()+30000; const sendCommit=function(){ if(settled) return; if(!socket||socket.readyState>1||Date.now()>=commitDeadline){ reject(new Error('Streaming transcription closed before commit')); return; } if(socket.readyState!==1||queue.length){flush();window.setTimeout(sendCommit,20);return;} if(!sendJson({type:'commit',turn_id:turnId})) reject(new Error('Streaming transcription commit failed')); }; sendCommit(); return finalPromise; }, cancel:function(){ cancelled=true; cancelVoicePreflight(turnId); clearQueue(); clearArchive(); if(!settled){sendJson({type:'cancel',turn_id:turnId});reject(new Error('Streaming transcription cancelled'));} if(socket&&socket.readyState<2) socket.close(1000,'cancelled'); }, }; } async function installCaptureWorklet(context,source,session){ if(!session) return false; if(!context.audioWorklet||!window.AudioWorkletNode){ session.cancel(); if(streamingStt===session) streamingStt=null; return false; } try{ await context.audioWorklet.addModule(WORKLET_URL); if(!streamingStt||streamingStt!==session) return false; const node=new AudioWorkletNode(context,'atlas-pcm-capture'); const silentGain=context.createGain(); silentGain.gain.value=0; let flushResolve=null; node.port.onmessage=function(event){ if(!event.data) return; if(event.data.type==='flushed'&&flushResolve){flushResolve();flushResolve=null;return;} if(!streamingStt||streamingStt!==session||event.data.type!=='pcm') return; session.push(new Float32Array(event.data.samples)); }; node._atlasFlush=function(){ return new Promise(function(resolve){flushResolve=resolve;node.port.postMessage({type:'flush'});}); }; source.connect(node); node.connect(silentGain); silentGain.connect(context.destination); captureNode=node; if(captureGraph) captureGraph.silentGain=silentGain; session.setWorkletReady(true); return true; }catch(_){ session.cancel(); if(streamingStt===session) streamingStt=null; return false; } } function cancelActiveModelTurn(){ const sessionId=(typeof S!=='undefined'&&S.session)?String(S.session.session_id||''):''; const streamId=typeof S!=='undefined'?String(S.activeStreamId||''):''; if(!streamId) return {sessionId:sessionId,streamId:'',controller:null,promise:Promise.resolve(false)}; const controller=createAbortController(); const timer=window.setTimeout(function(){controller.abort();},1800); const promise=(async function(){ try{ const url=new URL('api/chat/cancel?stream_id='+encodeURIComponent(streamId),document.baseURI||location.href).href; const response=await fetch(url,{credentials:'include',signal:controller.signal}); let payload=null; try{payload=await response.json();}catch(_){ } const currentSession=(typeof S!=='undefined'&&S.session)?String(S.session.session_id||''):''; if(response.ok&&payload&&payload.cancelled===false&¤tSession===sessionId&&String(S.activeStreamId||'')===streamId){ S.activeStreamId=null; if(S.session) S.session.active_stream_id=null; if(typeof setBusy==='function') setBusy(false); else S.busy=false; } return !!response.ok; }catch(_){ return false; }finally{ window.clearTimeout(timer); } })(); return {sessionId:sessionId,streamId:streamId,controller:controller,promise:promise}; } function trimBargeLookback(monitor){ const maximum=Math.max(1,Math.round(monitor.context.sampleRate*BARGE_LOOKBACK_MS/1000)); while(monitor.lookbackFrames>maximum&&monitor.lookback.length){ const overflow=monitor.lookbackFrames-maximum; const first=monitor.lookback[0]; if(first.length<=overflow){monitor.lookback.shift();monitor.lookbackFrames-=first.length;continue;} monitor.lookback[0]=first.slice(overflow); monitor.lookbackFrames-=overflow; } } async function triggerBargeIn(monitor){ if(!bargeMonitor||bargeMonitor!==monitor||monitor.cancelled||!active||monitor.token!==generation) return; bargeMonitor=null; if(monitor.timer){window.clearInterval(monitor.timer);monitor.timer=null;} // Keep the old capture worklet alive until the new STT worklet is attached. // Its lookback becomes an untrimmed handoff buffer so no syllables disappear // during AudioWorklet/session setup. monitor.handoff=true; const oldToken=monitor.token; recordPendingStitch(); generation+=1; const token=generation; thinkingSession=null; thinkingTurnId=''; suppressAutoRead=true; clearSttLanguage(); stopResponseObserver(); cancelThinkingCues(); cancelSpeechTurn(); stopPlayback(); if(typeof window.stopTTS==='function') window.stopTTS(); clearBargeCancellation(); bargeCancelPromise=cancelActiveModelTurn(); setState('listening','Listening — interrupted'); if(oldToken===token) return; startListening(token,{ stream:monitor.stream, context:monitor.context, lookback:monitor.lookback, handoffMonitor:monitor, heardSpeech:true, requireStreamingLookback:true, }); } function bargeFromLiveCapture(){ // Live-capture barge-in: the always-on VAD/STT listener detected a real // speech onset while Hermes was mid-response. Cancel the RESPONSE side // only — cues, playback, the model stream — and bump `generation` so every // stale response-side continuation dies. The already-running capture is // guarded by captureGeneration and keeps collecting the interrupting // utterance without losing a syllable. if(!active||(state!=='thinking'&&state!=='speaking')) return; recordPendingStitch(); generation+=1; thinkingSession=null; thinkingTurnId=''; suppressAutoRead=true; clearSttLanguage(); stopResponseObserver(); cancelThinkingCues(); cancelSpeechTurn(); stopPlayback(); if(typeof window.stopTTS==='function') window.stopTTS(); clearBargeCancellation(); bargeCancelPromise=cancelActiveModelTurn(); setState('listening','Listening — interrupted'); } async function startBargeMonitor(token){ if(!active||token!==generation||(state!=='thinking'&&state!=='speaking')||bargeMonitor) return; const monitor={token:token,cancelled:false,handoff:false,aecUsable:true,timer:null,stream:null,context:null,source:null,node:null,silentGain:null,lookback:[],lookbackFrames:0}; bargeMonitor=monitor; try{ monitor.stream=await acquireMicrophone(); if(!active||token!==generation||bargeMonitor!==monitor){disposeBargeResources(monitor,false);return;} monitor.aecUsable=captureAecIsUsable(monitor.stream); const Context=window.AudioContext||window.webkitAudioContext; try{monitor.context=new Context({latencyHint:'interactive'});}catch(_){monitor.context=new Context();} const analyser=monitor.context.createAnalyser(); analyser.fftSize=1024; monitor.source=monitor.context.createMediaStreamSource(monitor.stream); monitor.source.connect(analyser); if(monitor.context.audioWorklet&&window.AudioWorkletNode){ await monitor.context.audioWorklet.addModule(WORKLET_URL); if(bargeMonitor!==monitor||monitor.cancelled){disposeBargeResources(monitor,false);return;} monitor.node=new AudioWorkletNode(monitor.context,'atlas-pcm-capture'); monitor.silentGain=monitor.context.createGain(); monitor.silentGain.gain.value=0; monitor.node.port.onmessage=function(event){ if(monitor.cancelled||(!monitor.handoff&&bargeMonitor!==monitor)||!event.data||event.data.type!=='pcm') return; const samples=new Float32Array(event.data.samples); monitor.lookback.push(samples); monitor.lookbackFrames+=samples.length; if(!monitor.handoff) trimBargeLookback(monitor); }; monitor.source.connect(monitor.node); monitor.node.connect(monitor.silentGain); monitor.silentGain.connect(monitor.context.destination); } await monitor.context.resume(); const samples=new Uint8Array(analyser.fftSize); let noiseFloor=0.008; let voiceFrames=0; let ducked=false; let speechArmAt=Date.now()+100; let playbackWasActive=false; monitor.timer=window.setInterval(function(){ if(!active||token!==generation||bargeMonitor!==monitor||(state!=='thinking'&&state!=='speaking')){stopBargeMonitor();return;} analyser.getByteTimeDomainData(samples); let energy=0; for(let index=0;index=speechArmAt&&rms>threshold; if(!voiceNow) noiseFloor=(noiseFloor*0.975)+(Math.min(rms,threshold)*0.025); voiceFrames=voiceNow?Math.min(voiceFrames+1,BARGE_TRIGGER_FRAMES):Math.max(voiceFrames-1,0); if(voiceFrames>=BARGE_DUCK_FRAMES&&!ducked){ducked=true;setPlaybackDucked(true);} if(!voiceFrames&&ducked){ducked=false;setPlaybackDucked(false);} if(voiceFrames>=BARGE_TRIGGER_FRAMES) triggerBargeIn(monitor); },50); }catch(_){ if(bargeMonitor===monitor) bargeMonitor=null; disposeBargeResources(monitor,false); // Barge-in is an optional full-duplex enhancement. The normal finalized // microphone turn remains available if the browser rejects concurrent // capture or AudioWorklet initialization. } } async function transcribeStreamingOrFallback(blob,token,session,allowContainerFallback,turnId){ if(!active||token!==generation) return; setState('transcribing'); let pcmFallback=null; if(session){ try{ const result=await session.commit(); if(!String(result.transcript||'').trim()) throw new Error('Streaming transcription returned no final text'); if(active&&token===generation){ if(streamingStt===session) streamingStt=null; sendTranscript(result.transcript,token,result.language,turnId); return; } }catch(_){ // The finalized browser container below is the quality-preserving path // whenever rolling PCM transport or speculative inference fails. if(typeof session.takeFallbackBlob==='function') pcmFallback=session.takeFallbackBlob(); } if(streamingStt===session) streamingStt=null; } if(pcmFallback){ transcribe(pcmFallback,token,turnId); return; } if(allowContainerFallback===false){ toast('I missed the start of that interruption. Please repeat it.'); restartSoon(token,250); return; } transcribe(blob,token,turnId); } async function startListening(token,reusedCapture){ if(!active||token!==generation) return; stopCapture(); // A continuous-capture restart (reusedCapture.preserveDisplay) happens the // instant an utterance finalizes, while the display legitimately shows // Hermes's own activity (transcribing/thinking/speaking). Capture is // therefore tracked by captureActive, never by the visible state, and the // response-side teardown below is skipped so that turn is not disturbed. const preserveDisplay=!!(reusedCapture&&reusedCapture.preserveDisplay); if(!preserveDisplay){ stopPlayback(); cancelSpeechTurn(); clearSttLanguage(); } captureTurnId=(voiceTabNonce?voiceTabNonce+'-':'')+String(token)+'-'+String(++turnSequence); // Capture turns carry their own epoch: a response-side barge bumps // `generation` but never this counter, so capture survives it seamlessly. const captureToken=++captureGeneration; captureActive=true; if(!preserveDisplay) setState('listening'); try{ // A barge-monitor handoff supplies its own stream/context; otherwise the // retained session microphone is reused. Only a fresh hands-free session // actually asks for the microphone again, keeping acquireMicrophone() // the single getUserMedia call site. if(reusedCapture&&reusedCapture.stream&&stream&&reusedCapture.stream!==stream){ stream.getTracks().forEach(function(track){track.stop();}); stream=null; } if(reusedCapture&&reusedCapture.context&&audioContext&&reusedCapture.context!==audioContext){ try{audioContext.close();}catch(_){ } audioContext=null; } const capture=(reusedCapture&&reusedCapture.stream)||stream||await acquireMicrophone(); if(!active||captureToken!==captureGeneration){ if(reusedCapture&&reusedCapture.handoffMonitor) disposeBargeResources(reusedCapture.handoffMonitor,false); else if(capture!==stream) capture.getTracks().forEach(function(track){track.stop();}); return; } stream=capture; applyConversationMute(); const captureAec=captureAecIsUsable(stream); const Context=window.AudioContext||window.webkitAudioContext; if(reusedCapture&&reusedCapture.context){ audioContext=reusedCapture.context; }else if(!audioContext){ try{ // Let the browser's native resampler produce Whisper's 16 kHz input. audioContext=new Context({sampleRate:16000,latencyHint:'interactive'}); }catch(_){ audioContext=new Context(); } } const analyser=audioContext.createAnalyser(); analyser.fftSize=2048; const highpass=audioContext.createBiquadFilter(); highpass.type='highpass'; highpass.frequency.value=140; highpass.Q.value=0.7; const mediaSource=audioContext.createMediaStreamSource(stream); mediaSource.connect(highpass); highpass.connect(analyser); captureGraph={mediaSource:mediaSource,highpass:highpass,analyser:analyser,silentGain:null}; streamingStt=createStreamingSttSession(captureTurnId,audioContext); // Whisper receives the browser's full-band processed microphone signal; // the 140 Hz high-pass remains a VAD-only aid so low voices are not // needlessly altered before final recognition. if(streamingStt) await installCaptureWorklet(audioContext,mediaSource,streamingStt); if(streamingStt&&reusedCapture&&Array.isArray(reusedCapture.lookback)){ reusedCapture.lookback.forEach(function(samples){streamingStt.push(samples);}); } if(reusedCapture&&reusedCapture.handoffMonitor){ disposeBargeResources(reusedCapture.handoffMonitor,true); reusedCapture.handoffMonitor=null; } if(reusedCapture&&reusedCapture.requireStreamingLookback&&!streamingStt){ toast('Streaming transcription is unavailable. Please repeat your interruption.'); stopCapture(); restartSoon(token,250); return; } const samples=new Uint8Array(analyser.fftSize); const mimeTypes=['audio/webm;codecs=opus','audio/ogg;codecs=opus','audio/mp4;codecs=mp4a.40.2','audio/mp4','audio/webm']; const mime=mimeTypes.find(function(value){return MediaRecorder.isTypeSupported(value);})||''; const chunks=[]; let heardSpeech=!!(reusedCapture&&reusedCapture.heardSpeech); let voiceFrames=heardSpeech?3:0; // Voiced milliseconds collected this utterance. A barge handoff already // carries confirmed speech, so it starts past the young-utterance hold. let speechMs=heardSpeech?VAD_COMMITTED_SPEECH_MS:0; let noiseFloor=0.008; let lastSpeech=Date.now(); let speculative=false; let ducked=false; let bargeArmAt=0; let playbackWasLive=false; const started=Date.now(); recorder=new MediaRecorder(stream,mime?{mimeType:mime}:undefined); let recordedMime=recorder.mimeType||mime||''; recorder.ondataavailable=function(event){ if(!event.data||!event.data.size) return; if(event.data.type) recordedMime=event.data.type; chunks.push(event.data); }; recorder.onstop=async function(){ if(vadTimer){clearInterval(vadTimer);vadTimer=null;} const finalCaptureNode=captureNode; if(finalCaptureNode&&typeof finalCaptureNode._atlasFlush==='function'){ await Promise.race([ finalCaptureNode._atlasFlush(), new Promise(function(resolve){window.setTimeout(resolve,100);}), ]); } // A stale onstop (teardown by stopCapture/releaseMicrophone, or a // newer capture turn already running) must not touch the globals: by // now they can belong to the NEXT turn, and its session was already // cancelled by stopCapture. Real recorders fire onstop asynchronously, // so this guard has to come before anything else is read. if(!active||captureToken!==captureGeneration) return; // Continuous capture: the microphone stream and AudioContext stay hot. // Detach only this utterance's session, then re-enter capture below so // speech during transcribing/thinking becomes the next utterance. const utteranceTurnId=captureTurnId; const session=streamingStt; streamingStt=null; recorder=null; if(!heardSpeech||!chunks.length){ if(session) session.cancel(); // Nothing worth transcribing: recycle the recorder on the hot mic. startListening(generation,{preserveDisplay:true}); return; } // The utterance is dispatched under the CURRENT response epoch: when a // live-capture barge just cancelled a model turn, this transcript is // the interruption that replaces it. transcribeStreamingOrFallback( new Blob(chunks,{type:recordedMime||'audio/webm'}), generation, session, !(reusedCapture&&reusedCapture.requireStreamingLookback), utteranceTurnId ); // Re-enter capture immediately — never wait for transcription or the // response. The display may show transcribing/thinking while the next // capture turn is already live underneath. startListening(generation,{preserveDisplay:true}); }; // Ask the browser for one finalized container at stop. Android Chromium // can emit timeslice fragments without a reusable EBML initialization // header; concatenating those fragments made otherwise valid recordings // intermittently unreadable by ffmpeg. A bounded 90-second Opus capture // is small enough to retain as one browser-owned recording. recorder.start(); const silenceMs=Math.max(900,parseInt(localStorage.getItem('hermes-voice-silence-ms')||'1100',10)||1100); const speculateMs=Math.min(silenceMs-250,Math.max(450,Math.round(silenceMs*0.55))); vadTimer=window.setInterval(function(){ if(!active||captureToken!==captureGeneration||!recorder||recorder.state==='inactive') return; analyser.getByteTimeDomainData(samples); let energy=0; for(let index=0;indexspeechThreshold&&(!playbackLive||(captureAec&&now>=bargeArmAt)); // Adapt the floor only toward energies at or below the current // threshold (like the fallback energy monitor): echo residue during // continuous capture must never ratchet the threshold above real // speech and deafen onset detection for the next utterance. if(!heardSpeech&&!voiceNow){noiseFloor=(noiseFloor*0.94)+(Math.min(rms,speechThreshold)*0.06);} voiceFrames=voiceNow?Math.min(voiceFrames+1,5):Math.max(voiceFrames-1,0); if(voiceNow) speechMs+=100; if(playbackLive){ if(voiceFrames>=BARGE_DUCK_FRAMES&&!ducked){ducked=true;setPlaybackDucked(true);} if(!voiceFrames&&ducked){ducked=false;setPlaybackDucked(false);} }else if(ducked){ ducked=false; setPlaybackDucked(false); } // Speech onset. While playback runs an echo-safe barge needs the same // sustained evidence (BARGE_TRIGGER_FRAMES) as the old energy monitor. const onsetFrames=playbackLive?BARGE_TRIGGER_FRAMES:3; if(!heardSpeech&&voiceFrames>=onsetFrames){ heardSpeech=true; lastSpeech=now; // The user started a new utterance while Hermes was mid-response: // cancel the response side only; this capture keeps running. if(state==='thinking'||state==='speaking') bargeFromLiveCapture(); }else if(heardSpeech&&voiceNow){ const speechGapMs=now-lastSpeech; lastSpeech=now; // Resume after our own speculate AND after any gap long enough for // the server's independent 650ms end-of-speech detector: either // party may have frozen a partial-utterance snapshot by now. if((speculative||speechGapMs>=SERVER_EOS_SILENCE_MS)&&streamingStt){streamingStt.resume();speculative=false;} } if(heardSpeech&&!voiceNow&&!speculative&&speechMs>=SPECULATE_MIN_SPEECH_MS&&(now-lastSpeech)>=speculateMs&&streamingStt){ streamingStt.speculate(); speculative=true; } // A young utterance holds its endpoint longer: pausing to think right // after the first word must not send a one-word fragment. When the // streaming partial already reads as a plausibly complete utterance // (>=3 words, or terminal punctuation), endpoint at the base window // instead so short commands are not delayed by the clipping guard — // the long hold remains only for 1-2 word partials. const partialText=streamingStt&&streamingStt.latestPartial?streamingStt.latestPartial():''; const partialWords=partialText?partialText.split(/\s+/).filter(Boolean).length:0; const partialComplete=partialWords>=3||(partialWords>0&&/[.!?…]["')\]}]*$/.test(partialText)); const endpointSilenceMs=(speechMs=endpointSilenceMs; const timedOut=now-started>=90000; const idle=(!heardSpeech)&&(now-started)>=20000; if(finished||timedOut||idle){ if(vadTimer){clearInterval(vadTimer);vadTimer=null;} try{recorder.stop();}catch(_){ } } },100); }catch(error){ if(reusedCapture&&reusedCapture.handoffMonitor){ disposeBargeResources(reusedCapture.handoffMonitor,true); reusedCapture.handoffMonitor=null; } if(!active||captureToken!==captureGeneration) return; // This capture turn is dead. A fatal microphone error ends the session // and releases the retained microphone via showUnavailable(). captureActive=false; const message=errorMessage(error,'Microphone permission is required'); showUnavailable(message); toast(message); } } function stripHttpUrlsForSpeech(text){ return String(text||'').replace(/(^|\s+[([{]|\s+|[([{"'])https?:\/\/[^\s<>"']+/gi,function(match,prefix){ let address=match.slice(prefix.length); let suffix=''; while(address){ const last=address.slice(-1); let trailing=/[.,!?;:…,。!?;:]/.test(last); if(last===')') trailing=(address.match(/\)/g)||[]).length>(address.match(/\(/g)||[]).length; if(last===']') trailing=(address.match(/\]/g)||[]).length>(address.match(/\[/g)||[]).length; if(last==='}') trailing=(address.match(/\}/g)||[]).length>(address.match(/\{/g)||[]).length; if(!trailing) break; suffix=last+suffix; address=address.slice(0,-1); } const pairs={'(':')','[':']','{':'}'}; const opening=prefix.slice(-1); if(pairs[opening]&&suffix.startsWith(pairs[opening])){ prefix=prefix.slice(0,-1); suffix=suffix.slice(1); } if(/^\s+$/.test(prefix)&&/^[.,!?;:…,。!?;:]/.test(suffix)) prefix=''; return prefix+suffix; }); } window._atlasStripHttpUrlsForSpeech=stripHttpUrlsForSpeech; function cleanForSpeech(text){ const cleaned=typeof window._stripForTTS==='function'?window._stripForTTS(text):String(text||'').replace(/```[\s\S]*?```/g,' code block '); return stripHttpUrlsForSpeech(cleaned).replace(/\s+/g,' ').trim(); } function sentenceEnd(text){ const source=String(text||''); const boundary=/[.!?…]+["')\]}]*(?:\s|$)/g; const abbreviation=/(?:\b(?:mr|mrs|ms|dr|prof|sr|jr|st|vs|etc|e\.g|i\.e)|\b[A-Z])\.$/i; let match; while((match=boundary.exec(source))){ const end=match.index+match[0].trimEnd().length; const prefix=source.slice(0,end).replace(/["')\]}]+$/,''); if(prefix.endsWith('.')&&abbreviation.test(prefix)) continue; return end; } return -1; } function preferredCut(text,min,target,max){ const bounded=text.slice(0,max); const candidates=[]; const punctuation=/[,;:—–.!?…]+["')\]}]*(?:\s|$)/g; let match; while((match=punctuation.exec(bounded))){ const end=match.index+match[0].trimEnd().length; if(end>=min) candidates.push({end:end,weight:Math.abs(end-target)}); } if(candidates.length){ candidates.sort(function(left,right){return left.weight-right.weight;}); return candidates[0].end; } const spaces=[]; const whitespace=/\s+/g; while((match=whitespace.exec(bounded))){if(match.index>=min) spaces.push(match.index);} if(spaces.length){ spaces.sort(function(left,right){return Math.abs(left-target)-Math.abs(right-target);}); return spaces[0]; } return Math.min(max,text.length); } function adaptiveChunks(text,final,firstChunk){ const source=String(text||'').trim(); if(!source) return {chunks:[],consumed:0}; const chunks=[]; let offset=0; if(firstChunk!==false){ const firstSentence=sentenceEnd(source); if(firstSentence<0&&!final) return {chunks:[],consumed:0}; const firstAvailable=firstSentence>=0?firstSentence:source.length; const firstSlice=source.slice(0,firstAvailable); const firstCut=firstSlice.length<=60?firstSlice.length:preferredCut(firstSlice,40,52,60); chunks.push(source.slice(0,firstCut).trim()); offset=firstCut; } while(offset>>0; return length?hash%length:0; } function thinkingCueStillOwned(cue){ // Mute-aware: a muted conversation (the user stepped away) never chatters. return !!(cue&&!cue.cancelled&&thinkingCue===cue&&active&&cue.token===generation&&state==='thinking'&&!(conversation&&conversation.muted)); } function scheduleNextThinkingCue(cue,delay){ if(!thinkingCueStillOwned(cue)||cue.issued>=cue.pool.length) return; cue.timer=window.setTimeout(function(){ cue.timer=null; issueThinkingCue(cue); },delay); } async function issueThinkingCue(cue){ if(!thinkingCueStillOwned(cue)||cue.issued>=cue.pool.length) return; const entry=cue.pool[(cue.offset+cue.issued)%cue.pool.length]; const cueNumber=cue.issued+1; cue.issued+=1; const controller=createAbortController(); cue.controller=controller; try{ const request=ttsRequest(entry.text,cue.language,cue.turnId+':thinking-cue:'+cueNumber); request.cue_id=entry.id; const response=await fetch(TTS_STREAM_URL,{ method:'POST', headers:{'Content-Type':'application/json','Accept':'audio/pcm'}, body:JSON.stringify(request), signal:controller.signal, }); if(!response.ok||!response.body) throw new Error('Cached thinking cue unavailable'); const sampleRate=parseInt(response.headers.get('X-Audio-Sample-Rate')||'22050',10); if(!Number.isFinite(sampleRate)||sampleRate<8000||sampleRate>96000) throw new Error('Cached thinking cue metadata invalid'); if(!thinkingCueStillOwned(cue)) return; // The cue plays through the ONE shared spoken-output sink, exactly like the // reply — never its own AudioContext (which could route to another device). cue.context=await acquirePlaybackContext(); if(!cue.context||!thinkingCueStillOwned(cue)) return; cue.node=new AudioWorkletNode(cue.context,'atlas-pcm-playback'); cue.gain=cue.context.createGain(); cue.node.connect(cue.gain); cue.gain.connect(cue.context.destination); const drained=new Promise(function(resolve){ cue.audioWake=resolve; cue.node.port.onmessage=function(event){ if((event.data||{}).type==='drained'){ const wake=cue.audioWake; cue.audioWake=null; if(wake) wake(); } }; }); await cue.context.resume(); indicator.classList.add('is-playing'); const reader=response.body.getReader(); let carry=null; while(thinkingCueStillOwned(cue)){ const result=await reader.read(); if(result.done) break; let bytes=result.value; if(carry!==null){const joined=new Uint8Array(bytes.length+1);joined[0]=carry;joined.set(bytes,1);bytes=joined;carry=null;} if(bytes.length%2){carry=bytes[bytes.length-1];bytes=bytes.slice(0,-1);} if(!bytes.length) continue; const copy=bytes.buffer.slice(bytes.byteOffset,bytes.byteOffset+bytes.byteLength); cue.node.port.postMessage({type:'push',samples:copy,sampleRate:sampleRate},[copy]); } if(thinkingCueStillOwned(cue)){ cue.node.port.postMessage({type:'end'}); await drained; }else{ try{await reader.cancel();}catch(_){ } } cue.controller=null; if(cue.node){try{cue.node.disconnect();}catch(_){ }cue.node=null;} if(cue.context){cue.context=null;} cue.gain=null; indicator.classList.remove('is-playing'); if(thinkingCueStillOwned(cue)) scheduleNextThinkingCue(cue,THINKING_CUE_INTERVAL_MS); }catch(error){ cue.controller=null; if(cue.cancelled||(error&&error.name==='AbortError')) return; // Cues are optional. A playback or synthesis failure must not disturb the // answer, retry noisily, or disable the primary hands-free conversation. if(thinkingCue===cue) cancelThinkingCues(); } } function scheduleThinkingCues(token,language,turnId){ cancelThinkingCues(); if(!active||token!==generation||state!=='thinking') return; // Missing/unknown is the same fail-safe route as answer TTS: English/Amy. // Known STT languages remain bound exactly to their turn voice. const localized=normalizeSttLanguage(language)||'en'; const pool=THINKING_CUE_POOLS[localized]; const cue={ token:token, turnId:String(turnId||token), language:localized, pool:pool, offset:cuePoolOffset(turnId,pool.length), issued:0, timer:null, controller:null, audioWake:null, context:null, node:null, gain:null, cancelled:false, }; thinkingCue=cue; scheduleNextThinkingCue(cue,THINKING_CUE_FIRST_MS); } async function prepareSpeech(chunk,language,turnId,token){ if(streamingCapability.tts&&window.ReadableStream&&window.AudioWorkletNode){ const controller=createAbortController(); if(playbackSession) playbackSession.controllers.add(controller); try{ const response=await fetch(TTS_STREAM_URL,{ method:'POST', headers:{'Content-Type':'application/json','Accept':'audio/pcm'}, body:JSON.stringify(ttsRequest(chunk,language,turnId)), signal:controller.signal, }); if(!response.ok||!response.body) throw new Error('Streaming speech unavailable'); const sampleRate=parseInt(response.headers.get('X-Audio-Sample-Rate')||String(streamingCapability.tts.sample_rate||22050),10); const channels=parseInt(response.headers.get('X-Audio-Channels')||'1',10); if(!Number.isFinite(sampleRate)||sampleRate<8000||sampleRate>96000||channels!==1) throw new Error('Unsupported streaming speech format'); return {kind:'pcm',response:response,sampleRate:sampleRate,controller:controller,token:token,chunk:chunk,language:language,turnId:turnId,started:false}; }catch(error){ if(playbackSession) playbackSession.controllers.delete(controller); if(error&&error.name==='AbortError') throw error; // Disable only this optional transport for the browser session. The // complete WAV endpoint preserves voice quality and availability. streamingCapability.tts=null; } } return {kind:'blob',chunk:chunk,blob:await fetchSpeech(chunk,language,turnId,token)}; } async function ensurePcmPlayback(asset,token){ if(!active||token!==generation) return null; const session=playbackSession; if(!session||session.cancelled) return null; if(session.node){ if(session.sampleRate!==asset.sampleRate) throw new Error('Streaming speech sample rate changed mid-turn'); return session; } // One shared sink for every spoken sound (reply, cues, blob fallback); the // worklet resamples Piper's rate to the context's native rate. The shared // context is never closed per-session — only when hands-free mode ends. const context=await acquirePlaybackContext(); if(!context) return null; if(!active||token!==generation) return null; const node=new AudioWorkletNode(context,'atlas-pcm-playback'); if(!session||session.cancelled) return null; const gain=context.createGain(); session.context=context; session.node=node; session.gain=gain; session.sampleRate=asset.sampleRate; session.bufferedFrames=0; session.playbackEnded=false; node.connect(gain); gain.connect(context.destination); await context.resume(); session.drained=new Promise(function(resolve,reject){ session.drainWake=resolve; session.drainReject=reject; }); indicator.classList.add('is-playing'); node.port.onmessage=function(event){ const data=event.data||{}; if(data.type==='buffer'){ session.bufferedFrames=data.frames||0; if(session.bufferedFramesasset.sampleRate*2){ await new Promise(function(resolve){ let completed=false; const wake=function(){if(completed)return;completed=true;session.lowWaterWake=null;resolve();}; session.lowWaterWake=wake; window.setTimeout(wake,3000); }); } const result=await reader.read(); if(result.done) break; let bytes=result.value; if(carry!==null){ const joined=new Uint8Array(bytes.length+1); joined[0]=carry;joined.set(bytes,1);bytes=joined;carry=null; } if(bytes.length%2){carry=bytes[bytes.length-1];bytes=bytes.slice(0,-1);} if(!bytes.length) continue; const copy=bytes.buffer.slice(bytes.byteOffset,bytes.byteOffset+bytes.byteLength); session.bufferedFrames+=bytes.byteLength/2; asset.started=true; session.node.port.postMessage({type:'push',samples:copy,sampleRate:asset.sampleRate},[copy]); } }finally{ if((!active||token!==generation||session.cancelled)){try{await reader.cancel();}catch(_){ }} session.controllers.delete(asset.controller); } } async function drainPcmPlayback(session,token){ if(!session||!session.node||session.cancelled||!active||token!==generation) return; if(session.playbackEnded) return; session.node.port.postMessage({type:'end'}); await session.drained; } async function closePcmBeforeBlob(session,token){ if(!session||!session.node) return; await drainPcmPlayback(session,token); if(session.node){try{session.node.disconnect();}catch(_){ }session.node=null;} if(session.context){session.context=null;} session.gain=null; session.drained=null; indicator.classList.remove('is-playing'); } async function playPrepared(asset,token){ if(asset.kind!=='pcm'){ if(playbackSession&&playbackSession.node) await closePcmBeforeBlob(playbackSession,token); return playBlob(asset.blob,token); } try{ return await playPcm(asset,token); }catch(error){ if(asset.started) throw error; streamingCapability.tts=null; if(playbackSession&&playbackSession.node) await closePcmBeforeBlob(playbackSession,token); return playBlob(await fetchSpeech(asset.chunk,asset.language,asset.turnId,token),token); } } function nextSpeechChunk(turn){ if(turn.cancelled) return Promise.resolve(null); if(turn.queue.length) return Promise.resolve(turn.queue.shift()); if(turn.final) return Promise.resolve(null); return new Promise(function(resolve){turn.waiters.push(resolve);}); } function enqueueSpeech(turn,chunks){ chunks.forEach(function(chunk){ if(!chunk) return; if(turn.waiters.length) turn.waiters.shift()(chunk); else turn.queue.push(chunk); }); } function finishSpeechQueue(turn){ turn.final=true; while(turn.waiters.length&&!turn.queue.length) turn.waiters.shift()(null); } function scheduleSpeakingIdleFallback(turn,session){ // Interim-message turns speak in cycles (speak → tools → speak): once no // further chunk is queued and the turn is not final, fall back to // Thinking after the buffered audio has played out. The observer flips // the state back to Speaking when the next segment yields a chunk. cancelSpeakingIdleFallback(turn); const bufferedMs=session&&session.sampleRate?Math.ceil(((session.bufferedFrames||0)/session.sampleRate)*1000):0; turn.idleTimer=window.setTimeout(function(){ turn.idleTimer=null; if(!active||turn.token!==generation||turn.cancelled||speechTurn!==turn) return; if(turn.queue.length||turn.final) return; if(state==='speaking') setState('thinking'); },bufferedMs+250); } function cancelSpeakingIdleFallback(turn){ if(turn&&turn.idleTimer){window.clearTimeout(turn.idleTimer);turn.idleTimer=null;} } async function runSpeechQueue(turn){ if(turn.running) return; turn.running=true; playbackSession={turnId:turn.turnId,controllers:new Set(),context:null,node:null,gain:null,sampleRate:0,bufferedFrames:0,drained:null,drainReject:null,playbackEnded:false,cancelled:false,lowWaterWake:null,drainWake:null,blobWake:null}; const session=playbackSession; try{ let chunk=await nextSpeechChunk(turn); let current=chunk?prepareSpeech(chunk,turn.language,turn.turnId,turn.token):null; while(current){ const asset=await current; // At most one later synthesis request exists while this asset plays. // If streaming has not produced it yet, the waiter starts it the // instant a punctuation-safe chunk arrives. const next=nextSpeechChunk(turn); const nextPrepared=next.then(function(value){ return value?prepareSpeech(value,turn.language,turn.turnId,turn.token):null; }); // Barge-in can abort both the playing request and its one-ahead request. // Observe the latter even when the cancelled current turn returns first. nextPrepared.catch(function(){ }); turn.speakingChunk=asset.chunk||''; await playPrepared(asset,turn.token); turn.lastSpokenChunk=asset.chunk||turn.lastSpokenChunk; turn.speakingChunk=''; if(!active||turn.token!==generation||turn.cancelled||session.cancelled) return; if(!turn.queue.length&&!turn.final) scheduleSpeakingIdleFallback(turn,session); current=await nextPrepared; cancelSpeakingIdleFallback(turn); if(current) setState('speaking'); } await drainPcmPlayback(session,turn.token); if(turn.final&&active&&turn.token===generation&&!turn.cancelled) restartSoon(turn.token,300); }catch(error){ if(active&&turn.token===generation&&!turn.cancelled){ const message=errorMessage(error,'Local speech is unavailable'); setState('error',message); toast(message); restartSoon(turn.token,500); } }finally{ if(playbackSession===session) stopPlayback(); if(speechTurn===turn) speechTurn=null; } } function collectAssistantResponse(){ const rows=assistantRows(); if(!rows.length) return {text:'',error:false}; const turn=assistantTurnOf(rows[rows.length-1]); const response=readAssistantTurn(turn); if(assistantBaseline){ if(rows.length