// Deterministic continuous-capture probe for dockerfiles/hermes-webui-atlas-voice.js. // // Drives the real script through complete streaming hands-free turns against a // stub STT WebSocket whose VAD, speculative EOS-freeze, resume and commit // semantics faithfully mirror dockerfiles/hermes-jetson-stt-server.py. Words // are encoded as distinct constant PCM amplitudes so the "transcript" of any // audio snapshot is exactly the ordered set of words whose samples survived // endpointing — a first-word-only regression is directly observable as a // truncated transcript in the sent composer text. // // node hermes_voice_capture_probe.js // // Prints one JSON object per scenario to stdout; the python wrapper asserts. 'use strict'; const fs = require('fs'); const vm = require('vm'); const SCRIPT_PATH = process.argv[2]; if (!SCRIPT_PATH) { throw new Error('usage: hermes_voice_capture_probe.js '); } const SOURCE = fs.readFileSync(SCRIPT_PATH, 'utf8'); class FakeURL extends URL {} FakeURL.createObjectURL = () => 'blob:capture-probe'; FakeURL.revokeObjectURL = () => {}; const SAMPLE_RATE = 16000; const TICK_MS = 100; const TICK_SAMPLES = (SAMPLE_RATE * TICK_MS) / 1000; // Server constants mirrored from hermes-jetson-stt-server.py. const SERVER_END_SILENCE_SAMPLES = (SAMPLE_RATE * 650) / 1000; const SERVER_TAIL_SAMPLES = (SAMPLE_RATE * 220) / 1000; function flush() { return new Promise((resolve) => { let hops = 0; (function hop() { hops += 1; if (hops > 16) { resolve(); return; } setImmediate(hop); })(); }); } function makeElement(id) { const element = { id, value: '', textContent: '', className: '', dataset: {}, attributes: {}, children: [], style: { values: new Map(), display: '', setProperty(name, value) { this.values.set(name, String(value)); }, getPropertyValue(name) { return this.values.get(name) || ''; }, removeProperty(name) { this.values.delete(name); }, }, listeners: [], setAttribute(name, value) { this.attributes[name] = String(value); }, getAttribute(name) { return Object.prototype.hasOwnProperty.call(this.attributes, name) ? this.attributes[name] : null; }, addEventListener(type, handler) { this.listeners.push({ type, handler }); }, removeEventListener(type, handler) { element.listeners = element.listeners.filter((entry) => entry.handler !== handler); }, appendChild(child) { element.children.push(child); child.parentNode = element; return child; }, removeChild(child) { element.children = element.children.filter((entry) => entry !== child); child.parentNode = null; return child; }, contains() { return false; }, focus() {}, querySelector() { return null; }, querySelectorAll() { return []; }, insertBefore() {}, click() { const event = { preventDefault() {}, stopImmediatePropagation() {} }; element.listeners .filter((entry) => entry.type === 'click') .forEach((entry) => entry.handler(event)); }, }; element.classList = { add(name) { const names = new Set(element.className.split(/\s+/).filter(Boolean)); names.add(name); element.className = [...names].join(' '); }, remove(name) { element.className = element.className .split(/\s+/).filter((value) => value && value !== name).join(' '); }, contains(name) { return element.className.split(/\s+/).includes(name); }, toggle(name, force) { const present = this.contains(name); const next = force === undefined ? !present : !!force; if (next) this.add(name); else this.remove(name); return next; }, }; return element; } // Faithful port of StreamingTranscription's endpointing/speculation contract. class StubSttServer { constructor(words) { // words: {amplitude(int16) -> word string} this.words = words; this.chunks = []; // Int16Array chunks in arrival order this.totalSamples = 0; this.noiseFloor = 0.004; this.heardSpeech = false; this.atEos = false; this.silenceSamples = 0; this.lastSpeechSample = 0; this.clientActive = false; this.epoch = 0; this.frozenSnapshotSamples = -1; this.committed = false; this.events = []; } snapshotSamples() { if (this.atEos && this.frozenSnapshotSamples >= 0) return this.frozenSnapshotSamples; let cutoff = this.totalSamples; if (this.heardSpeech && this.lastSpeechSample) { cutoff = Math.min(cutoff, this.lastSpeechSample + SERVER_TAIL_SAMPLES); } return cutoff; } append(int16) { if (this.committed) throw new Error('server: audio after commit'); this.chunks.push(int16); this.totalSamples += int16.length; let energy = 0; for (let index = 0; index < int16.length; index += 1) { const value = int16[index] / 32768; energy += value * value; } const rms = int16.length ? Math.sqrt(energy / int16.length) : 0; const threshold = Math.max(0.012, this.noiseFloor * 2.5 + 0.004); const speech = this.clientActive || rms >= threshold; if (!this.heardSpeech && !speech) { this.noiseFloor = this.noiseFloor * 0.96 + rms * 0.04; } if (speech) { if (this.atEos) { this.epoch += 1; this.frozenSnapshotSamples = -1; this.events.push('recover-by-rms'); } this.heardSpeech = true; this.atEos = false; this.silenceSamples = 0; this.lastSpeechSample = this.totalSamples; } else if (this.heardSpeech) { this.silenceSamples += int16.length; if (this.silenceSamples >= SERVER_END_SILENCE_SAMPLES && !this.atEos) { this.atEos = true; this.frozenSnapshotSamples = this.snapshotSamples(); this.events.push('auto-eos-freeze'); } } } speculate() { this.clientActive = false; this.atEos = true; this.frozenSnapshotSamples = this.snapshotSamples(); this.events.push('speculate-freeze'); } resume() { this.atEos = false; this.silenceSamples = 0; this.epoch += 1; this.clientActive = true; this.frozenSnapshotSamples = -1; this.events.push('resume-clear'); } transcript() { const cutoff = this.snapshotSamples(); const decoded = []; let position = 0; let currentWord = null; let runLength = 0; const flushRun = () => { // Ignore sub-40ms runs: resampler transition samples, not words. if (currentWord && runLength >= SAMPLE_RATE * 0.04) decoded.push(currentWord); currentWord = null; runLength = 0; }; for (const chunk of this.chunks) { for (let index = 0; index < chunk.length; index += 1) { if (position >= cutoff) { flushRun(); return decoded.join(' '); } position += 1; const magnitude = Math.abs(chunk[index]); let word = null; for (const [amplitude, name] of Object.entries(this.words)) { if (Math.abs(magnitude - Number(amplitude)) <= 200) { word = name; break; } } if (word === currentWord) { runLength += 1; } else { flushRun(); currentWord = word; runLength = word ? 1 : 0; } } } flushRun(); return decoded.join(' '); } } function makeHarness(options = {}) { const clock = { now: 1000000 }; const timeouts = []; const intervals = new Map(); let timerId = 1; const sends = []; const toasts = []; const transcribeUploads = []; const ttsCalls = []; const servers = []; const captureNodes = []; const analysers = []; const recorders = []; let assistantRows = []; let streamCounter = 0; const words = options.words || { 8000: 'alpha', 12000: 'bravo', 16000: 'charlie' }; const micTracks = []; const elements = {}; ['btnVoiceMode', 'voiceModeBar', 'voiceModeIndicator', 'voiceModeLabel', 'msg'] .forEach((id) => { elements[id] = makeElement(id); }); const bodyElement = makeElement('body'); // ── Microphone model ──────────────────────────────────────────────────── // Per tick the mic has one (analyser level, pcm amplitude) pair. The level // feeds the script's VAD analyser; the amplitude becomes the PCM the // capture worklet delivers to the streaming session. const mic = { level: 0, amplitude: 0 }; class FakeAnalyser { constructor() { this.fftSize = 0; analysers.push(this); } getByteTimeDomainData(samples) { for (let index = 0; index < samples.length; index += 1) { const offset = Math.round(mic.level * 127); samples[index] = 128 + (index % 2 ? offset : -offset); } } disconnect() {} } class FakeAudioContext { constructor() { this.sampleRate = SAMPLE_RATE; this.destination = {}; this.audioWorklet = { addModule: async () => {} }; this.closed = false; } createAnalyser() { return new FakeAnalyser(); } createBiquadFilter() { return { type: '', frequency: { value: 0 }, Q: { value: 0 }, connect() {}, disconnect() {} }; } createMediaStreamSource() { return { connect() {}, disconnect() {} }; } createGain() { return { gain: { value: 0, setTargetAtTime() {} }, connect() {}, disconnect() {} }; } resume() { return Promise.resolve(); } close() { this.closed = true; return Promise.resolve(); } } class FakeAudioWorkletNode { constructor(_context, name) { this.name = name; this.connected = true; const node = this; this.port = { onmessage: null, postMessage(message) { if ((message || {}).type === 'flush') { queueMicrotask(() => { if (node.port.onmessage) node.port.onmessage({ data: { type: 'flushed' } }); }); } }, }; if (name === 'atlas-pcm-capture') captureNodes.push(this); } connect() {} disconnect() { this.connected = false; } } class FakeMediaRecorder { static isTypeSupported() { return true; } constructor(stream, recorderOptions) { this.stream = stream; this.state = 'inactive'; this.mimeType = (recorderOptions && recorderOptions.mimeType) || 'audio/webm;codecs=opus'; this.ondataavailable = null; this.onstop = null; recorders.push(this); } start() { this.state = 'recording'; } stop() { if (this.state === 'inactive') return; this.state = 'inactive'; if (this.ondataavailable) { this.ondataavailable({ data: { size: 512, type: 'audio/webm;codecs=opus' } }); } const recorder = this; // Real recorders finalize asynchronously. queueMicrotask(() => { if (recorder.onstop) recorder.onstop(); }); } } class FakeWebSocket { constructor(url, protocols) { this.url = url; this.protocols = protocols; this.readyState = 0; this.bufferedAmount = 0; this.binaryType = ''; this.onopen = null; this.onmessage = null; this.onerror = null; this.onclose = null; this.server = new StubSttServer(words); this.turnId = ''; servers.push(this); const socket = this; queueMicrotask(() => { if (socket.readyState !== 0) return; socket.readyState = 1; if (socket.onopen) socket.onopen(); }); } deliver(payload) { const socket = this; queueMicrotask(() => { if (socket.readyState !== 1) return; if (socket.onmessage) socket.onmessage({ data: JSON.stringify(payload) }); }); } send(data) { if (this.readyState !== 1) throw new Error('socket not open'); if (typeof data === 'string') { const message = JSON.parse(data); if (message.type === 'start') { this.turnId = message.turn_id; this.startLanguage = message.language; return; } if (message.type === 'speculate') { this.server.speculate(); return; } if (message.type === 'resume') { this.server.resume(); return; } if (message.type === 'cancel') { this.server.events.push('cancel'); return; } if (message.type === 'commit') { this.server.committed = true; this.deliver({ type: 'final', turn_id: this.turnId, transcript: this.server.transcript(), language: 'en', }); } return; } this.server.append(new Int16Array(data)); // Mirror the server's rolling-partial stream: one stable partial per // decoded-transcript change, so the client's dynamic endpoint sees the // same signal the Jetson emits. const partialText = this.server.transcript(); if (partialText && partialText !== this.lastPartialText) { this.lastPartialText = partialText; this.partialRevision = (this.partialRevision || 0) + 1; this.deliver({ type: 'partial', rolling: true, turn_id: this.turnId, revision: this.partialRevision, transcript: partialText, stable_transcript: partialText, }); } } close(code) { this.readyState = 3; this.closeCode = code; } } async function fetchStub(url, init) { if (url === '/api/transcribe/capability') { return { ok: true, status: 200, json: async () => ({ available: true, provider: 'local_command' }) }; } if (url === '/api/voice/streaming/capability') { return { ok: true, status: 200, json: async () => ({ tts: { available: false }, stt: { available: true, transport: 'websocket', format: 'pcm_s16le', sample_rate: 16000 }, preflight: { available: false }, }), }; } if (url === '/api/transcribe') { transcribeUploads.push({ body: init && init.body }); return { ok: true, status: 200, json: async () => ({ transcript: 'CONTAINER-FALLBACK', language: 'en' }) }; } if (url === '/api/tts') { ttsCalls.push(JSON.parse((init && init.body) || '{}')); return { ok: true, status: 200, blob: async () => ({ synthetic: true }), json: async () => ({}) }; } if (String(url).indexOf('api/chat/cancel') >= 0) { context.S.busy = false; context.S.activeStreamId = null; if (context.S.session) context.S.session.active_stream_id = null; return { ok: true, status: 200, json: async () => ({ cancelled: true }) }; } throw new Error(`unexpected fetch: ${url}`); } const storage = new Map(); const context = { AbortController, console, Uint8Array, Int16Array, Float32Array, ArrayBuffer, DataView, Promise, Math, JSON, String, Number, Error, parseInt, parseFloat, isNaN, Set, Map, Array, Object, URL: FakeURL, queueMicrotask, Date: { now: () => clock.now }, Blob: function Blob(parts, blobOptions) { this.parts = parts; this.type = (blobOptions || {}).type || ''; this.size = 512; }, File: function File(parts, name, fileOptions) { this.parts = parts; this.name = name; this.type = (fileOptions || {}).type || ''; }, FormData: function FormData() { this.entries = []; this.append = (key, value) => this.entries.push([key, value]); }, Audio: function Audio() { this.play = () => Promise.resolve(); this.pause = () => {}; this.onended = null; this.onerror = null; }, WebSocket: FakeWebSocket, AudioWorkletNode: FakeAudioWorkletNode, MediaRecorder: FakeMediaRecorder, AudioContext: FakeAudioContext, fetch: fetchStub, localStorage: { getItem: (key) => (storage.has(key) ? storage.get(key) : null), setItem: (key, value) => { storage.set(key, String(value)); }, removeItem: (key) => { storage.delete(key); }, }, navigator: { mediaDevices: { getSupportedConstraints: () => ({}), getUserMedia: async () => { const track = { stop() {}, enabled: true, getSettings: () => ({ echoCancellation: true }), }; micTracks.push(track); return { getTracks: () => [track], getAudioTracks: () => [track] }; }, }, }, document: { baseURI: 'https://chat.test/', body: bodyElement, getElementById: (id) => elements[id] || null, querySelectorAll: () => assistantRows, createElement: (tag) => makeElement(`created-${tag}`), addEventListener() {}, removeEventListener() {}, }, location: { protocol: 'https:', host: 'chat.test', href: 'https://chat.test/' }, crypto: { getRandomValues(bytes) { for (let i = 0; i < bytes.length; i += 1) bytes[i] = i + 1; } }, S: { session: { session_id: 'session-1' }, busy: false, activeStreamId: null }, setTimeout: (fn, delay) => { const id = timerId; timerId += 1; timeouts.push({ id, fn, at: clock.now + (delay || 0) }); return id; }, clearTimeout: (id) => { const index = timeouts.findIndex((entry) => entry.id === id); if (index >= 0) timeouts.splice(index, 1); }, setInterval: (fn, delay) => { const id = timerId; timerId += 1; intervals.set(id, { fn, delay: delay || 0 }); return id; }, clearInterval: (id) => { intervals.delete(id); }, }; context.window = context; context.window.location = context.location; context.window.crypto = context.crypto; context.showToast = (message) => { toasts.push(message); }; context.send = () => { streamCounter += 1; sends.push(elements.msg.value); context.S.busy = true; context.S.activeStreamId = `stream-${streamCounter}`; if (context.S.session) context.S.session.active_stream_id = context.S.activeStreamId; }; context.autoResize = () => {}; context.stopTTS = () => {}; context._stripForTTS = (text) => text; vm.createContext(context); vm.runInContext(SOURCE, context, { filename: 'atlas-voice.js' }); function runDueTimeouts() { const due = timeouts.filter((entry) => entry.at <= clock.now); due.forEach((entry) => { const index = timeouts.indexOf(entry); if (index >= 0) timeouts.splice(index, 1); entry.fn(); }); } function currentCaptureNode() { for (let index = captureNodes.length - 1; index >= 0; index -= 1) { if (captureNodes[index].connected) return captureNodes[index]; } return null; } function deliverMicFrame() { const node = currentCaptureNode(); if (!node || !node.port.onmessage) return; const samples = new Float32Array(TICK_SAMPLES); const value = mic.amplitude / 32768; for (let index = 0; index < samples.length; index += 1) { samples[index] = index % 2 ? value : -value; } node.port.onmessage({ data: { type: 'pcm', samples: samples.buffer } }); } async function tick(ms) { const steps = Math.max(1, Math.round(ms / TICK_MS)); for (let step = 0; step < steps; step += 1) { clock.now += TICK_MS; deliverMicFrame(); Array.from(intervals.values()).forEach((entry) => entry.fn()); runDueTimeouts(); // eslint-disable-next-line no-await-in-loop await flush(); } } async function speak(word, ms) { const amplitude = Number(Object.keys(words).find((key) => words[key] === word)); if (!amplitude) throw new Error(`unknown word: ${word}`); mic.level = Math.min(0.9, amplitude / 32768 + 0.25); mic.amplitude = amplitude; await tick(ms); } async function silence(ms) { mic.level = 0; mic.amplitude = 0; await tick(ms); } return { context, elements, sends, toasts, transcribeUploads, ttsCalls, servers, recorders, clock, tick, speak, silence, flush, runDueTimeouts, setAssistantReply(text) { assistantRows = [{ dataset: { rawText: text } }]; }, setAssistantError(text) { assistantRows = [{ dataset: { rawText: text, error: '1' } }]; }, clearAssistantRows() { assistantRows = []; }, state() { return elements.voiceModeBar.dataset.voiceState || ''; }, body() { return bodyElement; }, trackStates() { return micTracks.map((track) => track.enabled); }, overlay() { return bodyElement.children.find( (child) => String(child.className).indexOf('voice-conversation') >= 0, ) || null; }, completeResponse() { context.S.busy = false; context.S.activeStreamId = null; if (context.S.session) context.S.session.active_stream_id = null; context.autoReadLastAssistant(); }, async start() { await flush(); elements.btnVoiceMode.click(); await flush(); await flush(); }, }; } const scenarios = {}; // The reported live regression: a normal multi-word utterance with a natural // inter-word pause long enough for the client to speculate (>=605ms at the // default 1100ms endpoint) must still transcribe completely. scenarios.multiword_utterance_with_interword_pause = async () => { const harness = makeHarness(); await harness.start(); await harness.silence(300); await harness.speak('alpha', 400); await harness.silence(700); // client speculates at ~605ms await harness.speak('bravo', 600); await harness.silence(1400); // past the 1100ms endpoint await harness.tick(400); // commit retries + final delivery return { sends: harness.sends, state: harness.state(), serverEvents: harness.servers.map((socket) => socket.server.events), toasts: harness.toasts, }; }; // Continuous capture: the mic stays hot after an utterance is dispatched, and // the next utterance (spoken while the first response is still in flight) // barges in and is transcribed completely, including words after a pause. scenarios.second_utterance_during_response_is_complete = async () => { const harness = makeHarness(); await harness.start(); await harness.silence(300); await harness.speak('alpha', 500); await harness.silence(1400); await harness.tick(400); // first send dispatched const firstSends = harness.sends.slice(); // The model is now answering (S.busy). Assistant text starts streaming. harness.setAssistantReply('Partial answer already visible.'); await harness.tick(200); // User talks over the response with a two-word interruption + pause. await harness.speak('bravo', 500); await harness.silence(700); await harness.speak('charlie', 500); await harness.silence(1400); await harness.tick(400); return { firstSends, sends: harness.sends, state: harness.state(), serverEvents: harness.servers.map((socket) => socket.server.events), toasts: harness.toasts, }; }; // The mic never re-acquires getUserMedia across utterances and a second // utterance right after a normally completed response is sent alone: a // normal completion clears any stitch context. scenarios.back_to_back_utterances_stay_hot = async () => { const harness = makeHarness(); await harness.start(); await harness.silence(300); await harness.speak('alpha', 1300); await harness.silence(1500); await harness.tick(400); // first send dispatched (past the endpoint) const sendsAfterFirst = harness.sends.slice(); harness.completeResponse(); await harness.tick(400); await harness.speak('bravo', 1300); await harness.silence(1500); await harness.tick(400); return { sendsAfterFirst, sends: harness.sends, micAcquisitions: harness.trackStates().length, recorderCount: harness.recorders.length, serverCount: harness.servers.length, toasts: harness.toasts, }; }; // The exact live regression: "The … rest of the sentence" // used to endpoint on the first inter-word gap (>=1100ms) and send the first // word alone. A young utterance now holds its endpoint to 1800ms, so the // pause stays inside one utterance and one complete message is sent. scenarios.thinking_pause_after_first_word_does_not_split = async () => { const harness = makeHarness(); await harness.start(); await harness.silence(300); await harness.speak('alpha', 400); await harness.silence(1300); // longer than the old 1100ms endpoint await harness.speak('bravo', 700); await harness.silence(2000); await harness.tick(400); return { sends: harness.sends, serverEvents: harness.servers.map((socket) => socket.server.events), toasts: harness.toasts, }; }; // Conversation mode: activation builds the full-screen overlay, captions // follow the turn, mute drives the real capture track, and both the exit // control and Escape tear the overlay down completely. scenarios.conversation_overlay_lifecycle = async () => { const harness = makeHarness(); await harness.start(); const overlay = harness.overlay(); if (!overlay) return { overlayPresent: false }; const find = (root, needle) => root.children.find( (child) => String(child.className).indexOf(needle) >= 0, ); const role = overlay.getAttribute('role'); const ariaModal = overlay.getAttribute('aria-modal'); await harness.silence(300); const listeningState = overlay.dataset.voiceState; await harness.speak('alpha', 1300); await harness.silence(1500); await harness.tick(400); const captions = find(overlay, 'voice-conversation-captions'); const userCaption = captions ? captions.children[0].textContent : ''; const captionsLive = captions ? captions.getAttribute('aria-live') : ''; harness.setAssistantReply('A visible answer.'); await harness.tick(300); const assistantCaption = captions ? captions.children[1].textContent : ''; const thinkingState = overlay.dataset.voiceState; const controls = find(overlay, 'voice-conversation-controls'); const muteBtn = controls.children[0]; const exitBtn = controls.children[1]; muteBtn.click(); const mutedPressed = muteBtn.getAttribute('aria-pressed'); const mutedTracks = harness.trackStates(); muteBtn.click(); const unmutedTracks = harness.trackStates(); exitBtn.click(); await harness.flush(); const removedOnExit = !harness.overlay(); const inactiveAfterExit = !harness.context._voiceModeActive(); // Re-activate and leave through Escape instead. harness.elements.btnVoiceMode.click(); await harness.flush(); const overlayAgain = harness.overlay(); const keydown = overlayAgain.listeners.find((entry) => entry.type === 'keydown'); keydown.handler({ key: 'Escape', preventDefault() {} }); await harness.flush(); return { overlayPresent: true, role, ariaModal, listeningState, thinkingState, userCaption, assistantCaption, captionsLive, mutedPressed, mutedTracks, unmutedTracks, removedOnExit, inactiveAfterExit, removedOnEscape: !harness.overlay(), sends: harness.sends, }; }; // Dynamic endpointing: a >=3-word partial reads as a plausibly complete // utterance and endpoints at the base silence window; 1-2 word partials keep // the young-utterance hold so thinking pauses still never clip. scenarios.three_word_partial_endpoints_at_base_silence = async () => { const harness = makeHarness(); await harness.start(); await harness.silence(300); await harness.speak('alpha', 300); await harness.speak('bravo', 300); await harness.speak('charlie', 400); await harness.silence(1200); // past the base 1100ms endpoint, well under the 1800ms hold await harness.tick(300); return { sends: harness.sends, state: harness.state() }; }; scenarios.two_word_young_utterance_keeps_the_hold = async () => { const harness = makeHarness(); await harness.start(); await harness.silence(300); await harness.speak('alpha', 300); await harness.speak('bravo', 300); await harness.silence(1300); // longer than base, shorter than the hold const sendsEarly = harness.sends.slice(); await harness.silence(700); await harness.tick(300); return { sendsEarly, sends: harness.sends }; }; // An errored turn (error/system envelope in the transcript) is never spoken, // never captioned as a reply, and fully resynchronizes capture. scenarios.errored_turn_is_not_spoken_and_capture_resyncs = async () => { const harness = makeHarness(); await harness.start(); await harness.silence(300); await harness.speak('alpha', 1300); await harness.silence(1500); await harness.tick(400); // dispatched; thinking const serversBefore = harness.servers.length; const ttsBefore = harness.ttsCalls.length; harness.setAssistantError('**Task cancelled:** Task cancelled.'); await harness.tick(300); // the response observer sees the envelope const stateAfterError = harness.state(); const labelAfterError = harness.elements.voiceModeLabel.textContent; await harness.speak('bravo', 1300); await harness.silence(1500); await harness.tick(400); return { stateAfterError, labelAfterError, ttsDuringError: harness.ttsCalls.length - ttsBefore, freshSessions: harness.servers.length - serversBefore, sends: harness.sends, }; }; // Voice barge-in appends a single-line cut marker naming the sentence that // was playing, so the model knows where its reply was cut off. scenarios.barge_cut_marker_records_spoken_tail = async () => { const harness = makeHarness(); await harness.start(); await harness.silence(300); await harness.speak('alpha', 1300); await harness.silence(1500); await harness.tick(400); // dispatched; thinking harness.setAssistantReply('The first point is ready. The second point needs many more words before it becomes another speakable chunk.'); await harness.tick(300); // first sentence reaches TTS and starts playing await harness.speak('bravo', 900); // talk over the reply await harness.silence(1900); await harness.tick(400); return { sends: harness.sends, ttsTexts: harness.ttsCalls.map((request) => request.text) }; }; // The detected utterance language re-biases the following streaming STT // session (sticky per hands-free session). scenarios.sticky_language_biases_next_stt_session = async () => { const harness = makeHarness(); await harness.start(); await harness.silence(300); await harness.speak('alpha', 1300); await harness.silence(1500); await harness.tick(400); harness.completeResponse(); await harness.tick(400); await harness.speak('bravo', 1300); await harness.silence(1500); await harness.tick(400); return { startLanguages: harness.servers.map((socket) => socket.startLanguage) }; }; // FIX 3: choosing a language in the conversation overlay FORCES both the // streaming STT hint and the reply TTS voice for the whole session, overriding // auto-detection, until Auto is chosen again. scenarios.language_override_forces_stt_and_voice = async () => { const harness = makeHarness(); await harness.start(); const overlay = harness.overlay(); if (!overlay) return { overlayPresent: false }; const langWrap = overlay.children.find( (child) => String(child.className).indexOf('voice-conversation-lang') >= 0, ); const menu = langWrap ? langWrap.children.find( (child) => String(child.className).indexOf('voice-conversation-lang-menu') >= 0, ) : null; const items = menu ? menu.children : []; const ru = items.find((item) => item.getAttribute('data-lang') === 'ru'); const auto = items.find((item) => item.getAttribute('data-lang') === ''); const langBtn = langWrap ? langWrap.children.find( (child) => String(child.className).indexOf('voice-conversation-lang-btn') >= 0, ) : null; if (!ru) return { overlayPresent: true, ruItemPresent: false }; // Force Russian. ru.click(); const forcedChecked = ru.getAttribute('aria-checked'); const autoChecked = auto ? auto.getAttribute('aria-checked') : null; const btnForced = langBtn ? String(langBtn.className).indexOf('is-forced') >= 0 : false; // Run a turn with an English reply: STT hint and TTS voice must both be 'ru'. await harness.silence(300); await harness.speak('alpha', 1300); await harness.silence(1500); await harness.tick(400); harness.setAssistantReply('A short English reply. It has two sentences.'); harness.completeResponse(); await harness.tick(600); const startLanguages = harness.servers.map((socket) => socket.startLanguage); const ttsLanguages = harness.ttsCalls.map((call) => call.language); // Release back to Auto. if (auto) auto.click(); const releasedChecked = auto ? auto.getAttribute('aria-checked') : null; const btnForcedAfterAuto = langBtn ? String(langBtn.className).indexOf('is-forced') >= 0 : false; return { overlayPresent: true, ruItemPresent: true, forcedChecked, autoChecked, btnForced, startLanguages, ttsLanguages, releasedChecked, btnForcedAfterAuto, }; }; (async () => { const output = {}; for (const name of Object.keys(scenarios)) { // eslint-disable-next-line no-await-in-loop output[name] = await scenarios[name](); } process.stdout.write(JSON.stringify(output, null, 1)); })().catch((error) => { process.stderr.write(String((error && error.stack) || error)); process.exit(1); });