Diagnosed from a live voice session (e1b9fccb90ef): three turns failed with
raw '**Error:** HTTP 502 ... hermes-{claude,codex}-broker' because the agent
pod hosting the model brokers rolled mid-conversation. The voice client
correctly refused to speak the error envelope, but it then dropped the user's
utterance and forced them to repeat it three times.
Voice: on a TRANSIENT provider error (5xx/502/'error sending request'/timeout),
conversation mode now re-runs the errored turn in place through the app's own
regenerate action (which truncates the errored turn — no duplicate user
message) and stays in Thinking so its cues cover the reconnect gap. Bounded to
MAX_TRANSIENT_RETRIES (2); a non-transient error or an exhausted budget still
drops cleanly to 'let's try that again — listening'. The raw error is never
spoken. New probe scenarios cover retry-then-recover and the bounded-then-drop
path; source contract updated.
STT: the same session mis-transcribed 'CUI' as 'cue'. Prime the default
initial_prompt with the domain acronyms the user uses (CUI, FOUO, DoD, NIST,
CMMC, FIPS, RMF, POA&M, ATO, SBU) so they bias to uppercase forms.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01BvMSXH8VH2tMWXanb8SJdf
1529 lines
58 KiB
JavaScript
1529 lines
58 KiB
JavaScript
// Deterministic continuous-capture probe for dockerfiles/hermes-webui-atlas-voice.js.
|
|
//
|
|
// Drives the real script through complete streaming hands-free turns against a
|
|
// stub STT WebSocket whose VAD, speculative EOS-freeze, resume and commit
|
|
// semantics faithfully mirror dockerfiles/hermes-jetson-stt-server.py. Words
|
|
// are encoded as distinct constant PCM amplitudes so the "transcript" of any
|
|
// audio snapshot is exactly the ordered set of words whose samples survived
|
|
// endpointing — a first-word-only regression is directly observable as a
|
|
// truncated transcript in the sent composer text.
|
|
//
|
|
// node hermes_voice_capture_probe.js <path-to-atlas-voice.js>
|
|
//
|
|
// Prints one JSON object per scenario to stdout; the python wrapper asserts.
|
|
'use strict';
|
|
|
|
const fs = require('fs');
|
|
const vm = require('vm');
|
|
|
|
const SCRIPT_PATH = process.argv[2];
|
|
if (!SCRIPT_PATH) {
|
|
throw new Error('usage: hermes_voice_capture_probe.js <atlas-voice.js>');
|
|
}
|
|
const SOURCE = fs.readFileSync(SCRIPT_PATH, 'utf8');
|
|
|
|
class FakeURL extends URL {}
|
|
FakeURL.createObjectURL = () => 'blob:capture-probe';
|
|
FakeURL.revokeObjectURL = () => {};
|
|
|
|
const SAMPLE_RATE = 16000;
|
|
const TICK_MS = 100;
|
|
const TICK_SAMPLES = (SAMPLE_RATE * TICK_MS) / 1000;
|
|
|
|
// Server constants mirrored from hermes-jetson-stt-server.py.
|
|
const SERVER_END_SILENCE_SAMPLES = (SAMPLE_RATE * 650) / 1000;
|
|
const SERVER_TAIL_SAMPLES = (SAMPLE_RATE * 220) / 1000;
|
|
|
|
function flush() {
|
|
return new Promise((resolve) => {
|
|
let hops = 0;
|
|
(function hop() {
|
|
hops += 1;
|
|
if (hops > 16) { resolve(); return; }
|
|
setImmediate(hop);
|
|
})();
|
|
});
|
|
}
|
|
|
|
function makeElement(id) {
|
|
const element = {
|
|
id,
|
|
value: '',
|
|
textContent: '',
|
|
className: '',
|
|
dataset: {},
|
|
attributes: {},
|
|
children: [],
|
|
style: {
|
|
values: new Map(),
|
|
display: '',
|
|
setProperty(name, value) { this.values.set(name, String(value)); },
|
|
getPropertyValue(name) { return this.values.get(name) || ''; },
|
|
removeProperty(name) { this.values.delete(name); },
|
|
},
|
|
listeners: [],
|
|
setAttribute(name, value) { this.attributes[name] = String(value); },
|
|
getAttribute(name) {
|
|
return Object.prototype.hasOwnProperty.call(this.attributes, name)
|
|
? this.attributes[name] : null;
|
|
},
|
|
addEventListener(type, handler) { this.listeners.push({ type, handler }); },
|
|
removeEventListener(type, handler) {
|
|
element.listeners = element.listeners.filter((entry) => entry.handler !== handler);
|
|
},
|
|
appendChild(child) { element.children.push(child); child.parentNode = element; return child; },
|
|
removeChild(child) {
|
|
element.children = element.children.filter((entry) => entry !== child);
|
|
child.parentNode = null;
|
|
return child;
|
|
},
|
|
contains() { return false; },
|
|
focus() {},
|
|
querySelector() { return null; },
|
|
querySelectorAll() { return []; },
|
|
insertBefore() {},
|
|
click() {
|
|
const event = { preventDefault() {}, stopImmediatePropagation() {} };
|
|
element.listeners
|
|
.filter((entry) => entry.type === 'click')
|
|
.forEach((entry) => entry.handler(event));
|
|
},
|
|
};
|
|
element.classList = {
|
|
add(name) {
|
|
const names = new Set(element.className.split(/\s+/).filter(Boolean));
|
|
names.add(name);
|
|
element.className = [...names].join(' ');
|
|
},
|
|
remove(name) {
|
|
element.className = element.className
|
|
.split(/\s+/).filter((value) => value && value !== name).join(' ');
|
|
},
|
|
contains(name) { return element.className.split(/\s+/).includes(name); },
|
|
toggle(name, force) {
|
|
const present = this.contains(name);
|
|
const next = force === undefined ? !present : !!force;
|
|
if (next) this.add(name); else this.remove(name);
|
|
return next;
|
|
},
|
|
};
|
|
return element;
|
|
}
|
|
|
|
// ── Faithful assistant-turn DOM ─────────────────────────────────────────────
|
|
// A minimal but real querySelectorAll / closest / getAttribute / hidden /
|
|
// recursive-textContent implementation, so the interim-acknowledgement FOLD can
|
|
// be reproduced exactly as the live renderer performs it: the interim segment is
|
|
// re-tagged .assistant-segment-worklog-source + hidden + aria-hidden, at which
|
|
// point the script's real collectAssistantResponse() returns '' for it (the same
|
|
// extraction the response probe locks) and the final answer renders as its own
|
|
// new .assistant-segment. Supports the compound selectors the extraction uses.
|
|
function parseDomSelector(selector) {
|
|
return String(selector).split(',').map((group) => {
|
|
const term = group.trim();
|
|
const parts = [];
|
|
const re = /([.#]?[\w-]+)|\[([\w-]+)(?:([~|^$*]?=)"?([^"\]]*)"?)?\]/g;
|
|
let m;
|
|
while ((m = re.exec(term))) {
|
|
if (m[1]) {
|
|
if (m[1][0] === '.') parts.push({ kind: 'class', value: m[1].slice(1) });
|
|
else parts.push({ kind: 'tag', value: m[1].toLowerCase() });
|
|
} else if (m[2]) {
|
|
parts.push({ kind: 'attr', name: m[2], op: m[3] || null, value: m[4] });
|
|
}
|
|
}
|
|
return parts;
|
|
}).filter((parts) => parts.length);
|
|
}
|
|
|
|
class DomNode {
|
|
constructor(tag) {
|
|
this.tag = String(tag || 'div').toLowerCase();
|
|
this.className = '';
|
|
this.attributes = new Map();
|
|
this.children = [];
|
|
this.parentNode = null;
|
|
this.hidden = false;
|
|
this._text = '';
|
|
const self = this;
|
|
this.dataset = new Proxy({}, {
|
|
get(_t, key) {
|
|
if (typeof key !== 'string') return undefined;
|
|
const attr = 'data-' + key.replace(/[A-Z]/g, (c) => '-' + c.toLowerCase());
|
|
return self.attributes.has(attr) ? self.attributes.get(attr) : undefined;
|
|
},
|
|
has(_t, key) {
|
|
const attr = 'data-' + String(key).replace(/[A-Z]/g, (c) => '-' + c.toLowerCase());
|
|
return self.attributes.has(attr);
|
|
},
|
|
});
|
|
}
|
|
get classList() {
|
|
const el = this;
|
|
return {
|
|
add(name) { const s = new Set(el.className.split(/\s+/).filter(Boolean)); s.add(name); el.className = [...s].join(' '); },
|
|
remove(name) { el.className = el.className.split(/\s+/).filter((v) => v && v !== name).join(' '); },
|
|
contains(name) { return el.className.split(/\s+/).indexOf(name) >= 0; },
|
|
};
|
|
}
|
|
setAttribute(name, value) {
|
|
this.attributes.set(name, String(value));
|
|
if (name === 'class') this.className = String(value);
|
|
if (name === 'hidden') this.hidden = true;
|
|
}
|
|
getAttribute(name) {
|
|
if (name === 'class') return this.className || null;
|
|
return this.attributes.has(name) ? this.attributes.get(name) : null;
|
|
}
|
|
appendChild(child) { child.parentNode = this; this.children.push(child); return child; }
|
|
set textContent(value) { this._text = String(value); this.children = []; }
|
|
get textContent() {
|
|
if (this.children.length) return this.children.map((c) => c.textContent).join('');
|
|
return this._text;
|
|
}
|
|
_matchesTerm(parts) {
|
|
return parts.every((p) => {
|
|
if (p.kind === 'class') return this.classList.contains(p.value);
|
|
if (p.kind === 'tag') return this.tag === p.value;
|
|
if (p.kind === 'attr') {
|
|
if (!this.attributes.has(p.name)) return false;
|
|
return p.op ? this.attributes.get(p.name) === p.value : true;
|
|
}
|
|
return false;
|
|
});
|
|
}
|
|
matches(selector) { return parseDomSelector(selector).some((parts) => this._matchesTerm(parts)); }
|
|
_walk(out) { for (const c of this.children) { out.push(c); c._walk(out); } return out; }
|
|
querySelectorAll(selector) {
|
|
const groups = parseDomSelector(selector);
|
|
return this._walk([]).filter((node) => groups.some((parts) => node._matchesTerm(parts)));
|
|
}
|
|
querySelector(selector) { const all = this.querySelectorAll(selector); return all.length ? all[0] : null; }
|
|
closest(selector) {
|
|
const groups = parseDomSelector(selector);
|
|
let node = this;
|
|
while (node) { if (groups.some((parts) => node._matchesTerm(parts))) return node; node = node.parentNode; }
|
|
return null;
|
|
}
|
|
}
|
|
|
|
function domEl(tag, className, attrs, text) {
|
|
const node = new DomNode(tag);
|
|
if (className) node.setAttribute('class', className);
|
|
if (attrs) Object.keys(attrs).forEach((k) => node.setAttribute(k, attrs[k]));
|
|
if (text !== undefined) node.textContent = text;
|
|
return node;
|
|
}
|
|
|
|
// A rendered assistant turn: role header, worklog chip, and an assistant-turn
|
|
// blocks container the interim/final segments hang off of.
|
|
function buildAssistantTurn() {
|
|
const turn = domEl('div', 'msg-row assistant-turn', { 'data-role': 'assistant', 'data-session-id': 'session-1' });
|
|
const role = domEl('div', 'msg-role assistant');
|
|
role.appendChild(domEl('div', 'role-icon assistant', null, 'H'));
|
|
role.appendChild(domEl('span', 'msg-role-name', null, 'Hermes'));
|
|
turn.appendChild(role);
|
|
const blocks = domEl('div', 'assistant-turn-blocks');
|
|
turn.appendChild(blocks);
|
|
turn.blocks = blocks;
|
|
return turn;
|
|
}
|
|
|
|
// A live (still-streaming) answer segment: no data-raw-text yet, so the script
|
|
// reads its answer .msg-body — exactly the interim acknowledgement path.
|
|
function addLiveAnswerSegment(turn, text) {
|
|
const seg = domEl('div', 'assistant-segment', { 'data-live-assistant': '1' });
|
|
seg.appendChild(domEl('div', 'msg-body', null, text));
|
|
turn.blocks.appendChild(seg);
|
|
return seg;
|
|
}
|
|
|
|
// A settled answer segment carrying the renderer-stamped clean answer text.
|
|
function addSettledAnswerSegment(turn, text) {
|
|
const seg = domEl('div', 'assistant-segment', { 'data-raw-text': text });
|
|
seg.appendChild(domEl('div', 'msg-body', null, text));
|
|
turn.blocks.appendChild(seg);
|
|
return seg;
|
|
}
|
|
|
|
// Fold an interim segment into the hidden worklog source, exactly as the live
|
|
// renderer does when a tool call begins: the extraction now excludes it.
|
|
function foldSegmentIntoWorklog(seg) {
|
|
seg.setAttribute('class', 'assistant-segment assistant-segment-worklog-source');
|
|
seg.setAttribute('aria-hidden', 'true');
|
|
seg.hidden = true;
|
|
}
|
|
|
|
// Faithful port of StreamingTranscription's endpointing/speculation contract.
|
|
class StubSttServer {
|
|
constructor(words) {
|
|
// words: {amplitude(int16) -> word string}
|
|
this.words = words;
|
|
this.chunks = []; // Int16Array chunks in arrival order
|
|
this.totalSamples = 0;
|
|
this.noiseFloor = 0.004;
|
|
this.heardSpeech = false;
|
|
this.atEos = false;
|
|
this.silenceSamples = 0;
|
|
this.lastSpeechSample = 0;
|
|
this.clientActive = false;
|
|
this.epoch = 0;
|
|
this.frozenSnapshotSamples = -1;
|
|
this.committed = false;
|
|
this.events = [];
|
|
}
|
|
|
|
snapshotSamples() {
|
|
if (this.atEos && this.frozenSnapshotSamples >= 0) return this.frozenSnapshotSamples;
|
|
let cutoff = this.totalSamples;
|
|
if (this.heardSpeech && this.lastSpeechSample) {
|
|
cutoff = Math.min(cutoff, this.lastSpeechSample + SERVER_TAIL_SAMPLES);
|
|
}
|
|
return cutoff;
|
|
}
|
|
|
|
append(int16) {
|
|
if (this.committed) throw new Error('server: audio after commit');
|
|
this.chunks.push(int16);
|
|
this.totalSamples += int16.length;
|
|
let energy = 0;
|
|
for (let index = 0; index < int16.length; index += 1) {
|
|
const value = int16[index] / 32768;
|
|
energy += value * value;
|
|
}
|
|
const rms = int16.length ? Math.sqrt(energy / int16.length) : 0;
|
|
const threshold = Math.max(0.012, this.noiseFloor * 2.5 + 0.004);
|
|
const speech = this.clientActive || rms >= threshold;
|
|
if (!this.heardSpeech && !speech) {
|
|
this.noiseFloor = this.noiseFloor * 0.96 + rms * 0.04;
|
|
}
|
|
if (speech) {
|
|
if (this.atEos) {
|
|
this.epoch += 1;
|
|
this.frozenSnapshotSamples = -1;
|
|
this.events.push('recover-by-rms');
|
|
}
|
|
this.heardSpeech = true;
|
|
this.atEos = false;
|
|
this.silenceSamples = 0;
|
|
this.lastSpeechSample = this.totalSamples;
|
|
} else if (this.heardSpeech) {
|
|
this.silenceSamples += int16.length;
|
|
if (this.silenceSamples >= SERVER_END_SILENCE_SAMPLES && !this.atEos) {
|
|
this.atEos = true;
|
|
this.frozenSnapshotSamples = this.snapshotSamples();
|
|
this.events.push('auto-eos-freeze');
|
|
}
|
|
}
|
|
}
|
|
|
|
speculate() {
|
|
this.clientActive = false;
|
|
this.atEos = true;
|
|
this.frozenSnapshotSamples = this.snapshotSamples();
|
|
this.events.push('speculate-freeze');
|
|
}
|
|
|
|
resume() {
|
|
this.atEos = false;
|
|
this.silenceSamples = 0;
|
|
this.epoch += 1;
|
|
this.clientActive = true;
|
|
this.frozenSnapshotSamples = -1;
|
|
this.events.push('resume-clear');
|
|
}
|
|
|
|
transcript() {
|
|
const cutoff = this.snapshotSamples();
|
|
const decoded = [];
|
|
let position = 0;
|
|
let currentWord = null;
|
|
let runLength = 0;
|
|
const flushRun = () => {
|
|
// Ignore sub-40ms runs: resampler transition samples, not words.
|
|
if (currentWord && runLength >= SAMPLE_RATE * 0.04) decoded.push(currentWord);
|
|
currentWord = null;
|
|
runLength = 0;
|
|
};
|
|
for (const chunk of this.chunks) {
|
|
for (let index = 0; index < chunk.length; index += 1) {
|
|
if (position >= cutoff) { flushRun(); return decoded.join(' '); }
|
|
position += 1;
|
|
const magnitude = Math.abs(chunk[index]);
|
|
let word = null;
|
|
for (const [amplitude, name] of Object.entries(this.words)) {
|
|
if (Math.abs(magnitude - Number(amplitude)) <= 200) { word = name; break; }
|
|
}
|
|
if (word === currentWord) {
|
|
runLength += 1;
|
|
} else {
|
|
flushRun();
|
|
currentWord = word;
|
|
runLength = word ? 1 : 0;
|
|
}
|
|
}
|
|
}
|
|
flushRun();
|
|
return decoded.join(' ');
|
|
}
|
|
}
|
|
|
|
function makeHarness(options = {}) {
|
|
const clock = { now: 1000000 };
|
|
const timeouts = [];
|
|
const intervals = new Map();
|
|
let timerId = 1;
|
|
|
|
const sends = [];
|
|
const toasts = [];
|
|
const transcribeUploads = [];
|
|
const ttsCalls = [];
|
|
const servers = [];
|
|
const captureNodes = [];
|
|
const analysers = [];
|
|
const recorders = [];
|
|
let assistantRows = [];
|
|
let streamCounter = 0;
|
|
|
|
const words = options.words || { 8000: 'alpha', 12000: 'bravo', 16000: 'charlie' };
|
|
const micTracks = [];
|
|
const sinkCalls = [];
|
|
|
|
const elements = {};
|
|
['btnVoiceMode', 'voiceModeBar', 'voiceModeIndicator', 'voiceModeLabel', 'msg']
|
|
.forEach((id) => { elements[id] = makeElement(id); });
|
|
const bodyElement = makeElement('body');
|
|
|
|
// ── Microphone model ────────────────────────────────────────────────────
|
|
// Per tick the mic has one (analyser level, pcm amplitude) pair. The level
|
|
// feeds the script's VAD analyser; the amplitude becomes the PCM the
|
|
// capture worklet delivers to the streaming session.
|
|
const mic = { level: 0, amplitude: 0 };
|
|
|
|
class FakeAnalyser {
|
|
constructor() { this.fftSize = 0; analysers.push(this); }
|
|
getByteTimeDomainData(samples) {
|
|
for (let index = 0; index < samples.length; index += 1) {
|
|
const offset = Math.round(mic.level * 127);
|
|
samples[index] = 128 + (index % 2 ? offset : -offset);
|
|
}
|
|
}
|
|
disconnect() {}
|
|
}
|
|
|
|
class FakeAudioContext {
|
|
constructor() {
|
|
this.sampleRate = SAMPLE_RATE;
|
|
this.destination = {};
|
|
this.audioWorklet = { addModule: async () => {} };
|
|
this.closed = false;
|
|
}
|
|
createAnalyser() { return new FakeAnalyser(); }
|
|
createBiquadFilter() {
|
|
return { type: '', frequency: { value: 0 }, Q: { value: 0 }, connect() {}, disconnect() {} };
|
|
}
|
|
createMediaStreamSource() { return { connect() {}, disconnect() {} }; }
|
|
createGain() { return { gain: { value: 0, setTargetAtTime() {} }, connect() {}, disconnect() {} }; }
|
|
resume() { return Promise.resolve(); }
|
|
setSinkId(id) { sinkCalls.push({ kind: 'context', sink: id }); return Promise.resolve(); }
|
|
close() { this.closed = true; return Promise.resolve(); }
|
|
}
|
|
|
|
class FakeAudioWorkletNode {
|
|
constructor(_context, name) {
|
|
this.name = name;
|
|
this.connected = true;
|
|
const node = this;
|
|
this.port = {
|
|
onmessage: null,
|
|
postMessage(message) {
|
|
if ((message || {}).type === 'flush') {
|
|
queueMicrotask(() => {
|
|
if (node.port.onmessage) node.port.onmessage({ data: { type: 'flushed' } });
|
|
});
|
|
}
|
|
},
|
|
};
|
|
if (name === 'atlas-pcm-capture') captureNodes.push(this);
|
|
}
|
|
connect() {}
|
|
disconnect() { this.connected = false; }
|
|
}
|
|
|
|
class FakeMediaRecorder {
|
|
static isTypeSupported() { return true; }
|
|
constructor(stream, recorderOptions) {
|
|
this.stream = stream;
|
|
this.state = 'inactive';
|
|
this.mimeType = (recorderOptions && recorderOptions.mimeType) || 'audio/webm;codecs=opus';
|
|
this.ondataavailable = null;
|
|
this.onstop = null;
|
|
recorders.push(this);
|
|
}
|
|
start() { this.state = 'recording'; }
|
|
stop() {
|
|
if (this.state === 'inactive') return;
|
|
this.state = 'inactive';
|
|
if (this.ondataavailable) {
|
|
this.ondataavailable({ data: { size: 512, type: 'audio/webm;codecs=opus' } });
|
|
}
|
|
const recorder = this;
|
|
// Real recorders finalize asynchronously.
|
|
queueMicrotask(() => { if (recorder.onstop) recorder.onstop(); });
|
|
}
|
|
}
|
|
|
|
class FakeWebSocket {
|
|
constructor(url, protocols) {
|
|
this.url = url;
|
|
this.protocols = protocols;
|
|
this.readyState = 0;
|
|
this.bufferedAmount = 0;
|
|
this.binaryType = '';
|
|
this.onopen = null;
|
|
this.onmessage = null;
|
|
this.onerror = null;
|
|
this.onclose = null;
|
|
this.server = new StubSttServer(words);
|
|
this.turnId = '';
|
|
servers.push(this);
|
|
const socket = this;
|
|
queueMicrotask(() => {
|
|
if (socket.readyState !== 0) return;
|
|
socket.readyState = 1;
|
|
if (socket.onopen) socket.onopen();
|
|
});
|
|
}
|
|
deliver(payload) {
|
|
const socket = this;
|
|
queueMicrotask(() => {
|
|
if (socket.readyState !== 1) return;
|
|
if (socket.onmessage) socket.onmessage({ data: JSON.stringify(payload) });
|
|
});
|
|
}
|
|
send(data) {
|
|
if (this.readyState !== 1) throw new Error('socket not open');
|
|
if (typeof data === 'string') {
|
|
const message = JSON.parse(data);
|
|
if (message.type === 'start') { this.turnId = message.turn_id; this.startLanguage = message.language; return; }
|
|
if (message.type === 'speculate') { this.server.speculate(); return; }
|
|
if (message.type === 'resume') { this.server.resume(); return; }
|
|
if (message.type === 'cancel') { this.server.events.push('cancel'); return; }
|
|
if (message.type === 'commit') {
|
|
this.server.committed = true;
|
|
this.deliver({
|
|
type: 'final',
|
|
turn_id: this.turnId,
|
|
transcript: this.server.transcript(),
|
|
language: 'en',
|
|
});
|
|
}
|
|
return;
|
|
}
|
|
this.server.append(new Int16Array(data));
|
|
// Mirror the server's rolling-partial stream: one stable partial per
|
|
// decoded-transcript change, so the client's dynamic endpoint sees the
|
|
// same signal the Jetson emits.
|
|
const partialText = this.server.transcript();
|
|
if (partialText && partialText !== this.lastPartialText) {
|
|
this.lastPartialText = partialText;
|
|
this.partialRevision = (this.partialRevision || 0) + 1;
|
|
this.deliver({
|
|
type: 'partial',
|
|
rolling: true,
|
|
turn_id: this.turnId,
|
|
revision: this.partialRevision,
|
|
transcript: partialText,
|
|
stable_transcript: partialText,
|
|
});
|
|
}
|
|
}
|
|
close(code) { this.readyState = 3; this.closeCode = code; }
|
|
}
|
|
|
|
async function fetchStub(url, init) {
|
|
if (url === '/api/transcribe/capability') {
|
|
return { ok: true, status: 200, json: async () => ({ available: true, provider: 'local_command' }) };
|
|
}
|
|
if (url === '/api/voice/streaming/capability') {
|
|
return {
|
|
ok: true,
|
|
status: 200,
|
|
json: async () => ({
|
|
tts: { available: false },
|
|
stt: { available: true, transport: 'websocket', format: 'pcm_s16le', sample_rate: 16000 },
|
|
preflight: { available: false },
|
|
}),
|
|
};
|
|
}
|
|
if (url === '/api/transcribe') {
|
|
transcribeUploads.push({ body: init && init.body });
|
|
return { ok: true, status: 200, json: async () => ({ transcript: 'CONTAINER-FALLBACK', language: 'en' }) };
|
|
}
|
|
if (url === '/api/tts') {
|
|
ttsCalls.push(JSON.parse((init && init.body) || '{}'));
|
|
return { ok: true, status: 200, blob: async () => ({ synthetic: true }), json: async () => ({}) };
|
|
}
|
|
if (String(url).indexOf('api/chat/cancel') >= 0) {
|
|
context.S.busy = false;
|
|
context.S.activeStreamId = null;
|
|
if (context.S.session) context.S.session.active_stream_id = null;
|
|
return { ok: true, status: 200, json: async () => ({ cancelled: true }) };
|
|
}
|
|
throw new Error(`unexpected fetch: ${url}`);
|
|
}
|
|
|
|
const storage = new Map();
|
|
const context = {
|
|
AbortController,
|
|
console,
|
|
Uint8Array,
|
|
Int16Array,
|
|
Float32Array,
|
|
ArrayBuffer,
|
|
DataView,
|
|
Promise,
|
|
Math,
|
|
JSON,
|
|
String,
|
|
Number,
|
|
Error,
|
|
parseInt,
|
|
parseFloat,
|
|
isNaN,
|
|
Set,
|
|
Map,
|
|
Array,
|
|
Object,
|
|
URL: FakeURL,
|
|
queueMicrotask,
|
|
Date: { now: () => clock.now },
|
|
Blob: function Blob(parts, blobOptions) {
|
|
this.parts = parts;
|
|
this.type = (blobOptions || {}).type || '';
|
|
this.size = 512;
|
|
},
|
|
File: function File(parts, name, fileOptions) {
|
|
this.parts = parts; this.name = name; this.type = (fileOptions || {}).type || '';
|
|
},
|
|
FormData: function FormData() {
|
|
this.entries = []; this.append = (key, value) => this.entries.push([key, value]);
|
|
},
|
|
Audio: function Audio() {
|
|
this.play = () => Promise.resolve();
|
|
this.pause = () => {};
|
|
this.setSinkId = (id) => { sinkCalls.push({ kind: 'audio', sink: id }); return Promise.resolve(); };
|
|
this.onended = null;
|
|
this.onerror = null;
|
|
},
|
|
WebSocket: FakeWebSocket,
|
|
AudioWorkletNode: FakeAudioWorkletNode,
|
|
MediaRecorder: FakeMediaRecorder,
|
|
AudioContext: FakeAudioContext,
|
|
fetch: fetchStub,
|
|
localStorage: {
|
|
getItem: (key) => (storage.has(key) ? storage.get(key) : null),
|
|
setItem: (key, value) => { storage.set(key, String(value)); },
|
|
removeItem: (key) => { storage.delete(key); },
|
|
},
|
|
navigator: {
|
|
mediaDevices: {
|
|
getSupportedConstraints: () => ({}),
|
|
...(options.outputs ? { enumerateDevices: async () => options.outputs } : {}),
|
|
getUserMedia: async () => {
|
|
const track = {
|
|
stop() {},
|
|
enabled: true,
|
|
getSettings: () => ({ echoCancellation: true }),
|
|
};
|
|
micTracks.push(track);
|
|
return { getTracks: () => [track], getAudioTracks: () => [track] };
|
|
},
|
|
},
|
|
},
|
|
document: {
|
|
baseURI: 'https://chat.test/',
|
|
body: bodyElement,
|
|
getElementById: (id) => elements[id] || null,
|
|
querySelectorAll: () => assistantRows,
|
|
createElement: (tag) => makeElement(`created-${tag}`),
|
|
addEventListener() {},
|
|
removeEventListener() {},
|
|
},
|
|
location: { protocol: 'https:', host: 'chat.test', href: 'https://chat.test/' },
|
|
crypto: { getRandomValues(bytes) { for (let i = 0; i < bytes.length; i += 1) bytes[i] = i + 1; } },
|
|
S: { session: { session_id: 'session-1' }, busy: false, activeStreamId: null },
|
|
setTimeout: (fn, delay) => {
|
|
const id = timerId; timerId += 1;
|
|
timeouts.push({ id, fn, at: clock.now + (delay || 0) });
|
|
return id;
|
|
},
|
|
clearTimeout: (id) => {
|
|
const index = timeouts.findIndex((entry) => entry.id === id);
|
|
if (index >= 0) timeouts.splice(index, 1);
|
|
},
|
|
setInterval: (fn, delay) => {
|
|
const id = timerId; timerId += 1;
|
|
intervals.set(id, { fn, delay: delay || 0 });
|
|
return id;
|
|
},
|
|
clearInterval: (id) => { intervals.delete(id); },
|
|
};
|
|
context.window = context;
|
|
context.window.location = context.location;
|
|
context.window.crypto = context.crypto;
|
|
context.showToast = (message) => { toasts.push(message); };
|
|
context.send = () => {
|
|
streamCounter += 1;
|
|
sends.push(elements.msg.value);
|
|
context.S.busy = true;
|
|
context.S.activeStreamId = `stream-${streamCounter}`;
|
|
if (context.S.session) context.S.session.active_stream_id = context.S.activeStreamId;
|
|
};
|
|
context.autoResize = () => {};
|
|
context.stopTTS = () => {};
|
|
context._stripForTTS = (text) => text;
|
|
|
|
vm.createContext(context);
|
|
vm.runInContext(SOURCE, context, { filename: 'atlas-voice.js' });
|
|
|
|
function runDueTimeouts() {
|
|
const due = timeouts.filter((entry) => entry.at <= clock.now);
|
|
due.forEach((entry) => {
|
|
const index = timeouts.indexOf(entry);
|
|
if (index >= 0) timeouts.splice(index, 1);
|
|
entry.fn();
|
|
});
|
|
}
|
|
|
|
function currentCaptureNode() {
|
|
for (let index = captureNodes.length - 1; index >= 0; index -= 1) {
|
|
if (captureNodes[index].connected) return captureNodes[index];
|
|
}
|
|
return null;
|
|
}
|
|
|
|
function deliverMicFrame() {
|
|
const node = currentCaptureNode();
|
|
if (!node || !node.port.onmessage) return;
|
|
const samples = new Float32Array(TICK_SAMPLES);
|
|
const value = mic.amplitude / 32768;
|
|
for (let index = 0; index < samples.length; index += 1) {
|
|
samples[index] = index % 2 ? value : -value;
|
|
}
|
|
node.port.onmessage({ data: { type: 'pcm', samples: samples.buffer } });
|
|
}
|
|
|
|
async function tick(ms) {
|
|
const steps = Math.max(1, Math.round(ms / TICK_MS));
|
|
for (let step = 0; step < steps; step += 1) {
|
|
clock.now += TICK_MS;
|
|
deliverMicFrame();
|
|
Array.from(intervals.values()).forEach((entry) => entry.fn());
|
|
runDueTimeouts();
|
|
// eslint-disable-next-line no-await-in-loop
|
|
await flush();
|
|
}
|
|
}
|
|
|
|
async function speak(word, ms) {
|
|
const amplitude = Number(Object.keys(words).find((key) => words[key] === word));
|
|
if (!amplitude) throw new Error(`unknown word: ${word}`);
|
|
mic.level = Math.min(0.9, amplitude / 32768 + 0.25);
|
|
mic.amplitude = amplitude;
|
|
await tick(ms);
|
|
}
|
|
|
|
async function silence(ms) {
|
|
mic.level = 0;
|
|
mic.amplitude = 0;
|
|
await tick(ms);
|
|
}
|
|
|
|
return {
|
|
context,
|
|
elements,
|
|
sends,
|
|
toasts,
|
|
transcribeUploads,
|
|
ttsCalls,
|
|
servers,
|
|
recorders,
|
|
clock,
|
|
tick,
|
|
speak,
|
|
silence,
|
|
flush,
|
|
runDueTimeouts,
|
|
setAssistantReply(text) { assistantRows = [{ dataset: { rawText: text } }]; },
|
|
setAssistantError(text) { assistantRows = [{ dataset: { rawText: text, error: '1' } }]; },
|
|
// A TRANSIENT provider error turn: an error envelope carrying a
|
|
// .provider-error-details block (so the extraction reports error), plus the
|
|
// last-assistant "regenerate" action button the auto-retry clicks. The
|
|
// button records each click so the scenario can assert the bounded retry.
|
|
setAssistantTransientError(text) {
|
|
// An errored turn is a COMPLETED turn: its stream ended, so the session is
|
|
// no longer busy (the same precondition the app's regenerate action needs).
|
|
context.S.busy = false;
|
|
context.S.activeStreamId = null;
|
|
if (context.S.session) context.S.session.active_stream_id = null;
|
|
const turn = buildAssistantTurn();
|
|
const seg = domEl('div', 'assistant-segment');
|
|
seg.appendChild(domEl('div', 'msg-body', null, text));
|
|
seg.appendChild(domEl('details', 'provider-error-details', null, text));
|
|
turn.blocks.appendChild(seg);
|
|
const foot = domEl('div', 'msg-foot');
|
|
const actions = domEl('span', 'msg-actions');
|
|
const regen = domEl('button', 'msg-action-btn', { onclick: 'regenerateResponse(this)', title: 'regenerate' });
|
|
regen.click = () => {
|
|
// Mirror the app's regenerate: truncate the errored turn and start a
|
|
// fresh stream. The scenario injects the next state (another error, or a
|
|
// recovered answer) to model whether the retried turn succeeds.
|
|
this._regenerateClicks = (this._regenerateClicks || 0) + 1;
|
|
assistantRows = [];
|
|
context.S.busy = true;
|
|
streamCounter += 1;
|
|
context.S.activeStreamId = `stream-${streamCounter}`;
|
|
};
|
|
actions.appendChild(regen);
|
|
foot.appendChild(actions);
|
|
turn.appendChild(foot);
|
|
assistantRows = [turn];
|
|
},
|
|
regenerateClicks() { return this._regenerateClicks || 0; },
|
|
endStream() { context.S.busy = false; context.S.activeStreamId = null; if (context.S.session) context.S.session.active_stream_id = null; },
|
|
clearAssistantRows() { assistantRows = []; },
|
|
// ── Interim-fold reproduction ─────────────────────────────────────────
|
|
// Stream an interim acknowledgement as a live answer segment, then FOLD it
|
|
// into the hidden worklog source (as the renderer does at the tool call) and
|
|
// render the final answer as its own new segment.
|
|
setInterimAck(text) {
|
|
const turn = buildAssistantTurn();
|
|
this._foldTurn = turn;
|
|
this._interimSeg = addLiveAnswerSegment(turn, text);
|
|
assistantRows = [turn];
|
|
},
|
|
foldInterimAck() { if (this._interimSeg) foldSegmentIntoWorklog(this._interimSeg); },
|
|
setFinalAnswer(text) {
|
|
if (!this._foldTurn) return;
|
|
addSettledAnswerSegment(this._foldTurn, text);
|
|
},
|
|
speechTurnSnapshot() {
|
|
const internals = context.window.__atlasVoiceInternals;
|
|
return internals && typeof internals.speechTurnSnapshot === 'function'
|
|
? internals.speechTurnSnapshot() : null;
|
|
},
|
|
ttsTexts() { return ttsCalls.map((request) => request.text); },
|
|
sinkCalls() { return sinkCalls.slice(); },
|
|
state() { return elements.voiceModeBar.dataset.voiceState || ''; },
|
|
body() { return bodyElement; },
|
|
trackStates() { return micTracks.map((track) => track.enabled); },
|
|
overlay() {
|
|
return bodyElement.children.find(
|
|
(child) => String(child.className).indexOf('voice-conversation') >= 0,
|
|
) || null;
|
|
},
|
|
completeResponse() {
|
|
context.S.busy = false;
|
|
context.S.activeStreamId = null;
|
|
if (context.S.session) context.S.session.active_stream_id = null;
|
|
context.autoReadLastAssistant();
|
|
},
|
|
async start() {
|
|
await flush();
|
|
elements.btnVoiceMode.click();
|
|
await flush();
|
|
await flush();
|
|
},
|
|
};
|
|
}
|
|
|
|
const scenarios = {};
|
|
|
|
// The reported live regression: a normal multi-word utterance with a natural
|
|
// inter-word pause long enough for the client to speculate (>=605ms at the
|
|
// default 1100ms endpoint) must still transcribe completely.
|
|
scenarios.multiword_utterance_with_interword_pause = async () => {
|
|
const harness = makeHarness();
|
|
await harness.start();
|
|
await harness.silence(300);
|
|
await harness.speak('alpha', 400);
|
|
await harness.silence(700); // client speculates at ~605ms
|
|
await harness.speak('bravo', 600);
|
|
await harness.silence(1400); // past the 1100ms endpoint
|
|
await harness.tick(400); // commit retries + final delivery
|
|
return {
|
|
sends: harness.sends,
|
|
state: harness.state(),
|
|
serverEvents: harness.servers.map((socket) => socket.server.events),
|
|
toasts: harness.toasts,
|
|
};
|
|
};
|
|
|
|
// Continuous capture: the mic stays hot after an utterance is dispatched, and
|
|
// the next utterance (spoken while the first response is still in flight)
|
|
// barges in and is transcribed completely, including words after a pause.
|
|
scenarios.second_utterance_during_response_is_complete = async () => {
|
|
const harness = makeHarness();
|
|
await harness.start();
|
|
await harness.silence(300);
|
|
await harness.speak('alpha', 500);
|
|
await harness.silence(1400);
|
|
await harness.tick(400); // first send dispatched
|
|
const firstSends = harness.sends.slice();
|
|
// The model is now answering (S.busy). Assistant text starts streaming.
|
|
harness.setAssistantReply('Partial answer already visible.');
|
|
await harness.tick(200);
|
|
// User talks over the response with a two-word interruption + pause.
|
|
await harness.speak('bravo', 500);
|
|
await harness.silence(700);
|
|
await harness.speak('charlie', 500);
|
|
await harness.silence(1400);
|
|
await harness.tick(400);
|
|
return {
|
|
firstSends,
|
|
sends: harness.sends,
|
|
state: harness.state(),
|
|
serverEvents: harness.servers.map((socket) => socket.server.events),
|
|
toasts: harness.toasts,
|
|
};
|
|
};
|
|
|
|
// The mic never re-acquires getUserMedia across utterances and a second
|
|
// utterance right after a normally completed response is sent alone: a
|
|
// normal completion clears any stitch context.
|
|
scenarios.back_to_back_utterances_stay_hot = async () => {
|
|
const harness = makeHarness();
|
|
await harness.start();
|
|
await harness.silence(300);
|
|
await harness.speak('alpha', 1300);
|
|
await harness.silence(1500);
|
|
await harness.tick(400); // first send dispatched (past the endpoint)
|
|
const sendsAfterFirst = harness.sends.slice();
|
|
harness.completeResponse();
|
|
await harness.tick(400);
|
|
await harness.speak('bravo', 1300);
|
|
await harness.silence(1500);
|
|
await harness.tick(400);
|
|
return {
|
|
sendsAfterFirst,
|
|
sends: harness.sends,
|
|
micAcquisitions: harness.trackStates().length,
|
|
recorderCount: harness.recorders.length,
|
|
serverCount: harness.servers.length,
|
|
toasts: harness.toasts,
|
|
};
|
|
};
|
|
|
|
// The exact live regression: "The … <thinking pause> rest of the sentence"
|
|
// used to endpoint on the first inter-word gap (>=1100ms) and send the first
|
|
// word alone. A young utterance now holds its endpoint to 1800ms, so the
|
|
// pause stays inside one utterance and one complete message is sent.
|
|
scenarios.thinking_pause_after_first_word_does_not_split = async () => {
|
|
const harness = makeHarness();
|
|
await harness.start();
|
|
await harness.silence(300);
|
|
await harness.speak('alpha', 400);
|
|
await harness.silence(1300); // longer than the old 1100ms endpoint
|
|
await harness.speak('bravo', 700);
|
|
await harness.silence(2000);
|
|
await harness.tick(400);
|
|
return {
|
|
sends: harness.sends,
|
|
serverEvents: harness.servers.map((socket) => socket.server.events),
|
|
toasts: harness.toasts,
|
|
};
|
|
};
|
|
|
|
// Conversation mode: activation builds the full-screen overlay, captions
|
|
// follow the turn, mute drives the real capture track, and both the exit
|
|
// control and Escape tear the overlay down completely.
|
|
scenarios.conversation_overlay_lifecycle = async () => {
|
|
const harness = makeHarness();
|
|
await harness.start();
|
|
const overlay = harness.overlay();
|
|
if (!overlay) return { overlayPresent: false };
|
|
const find = (root, needle) => root.children.find(
|
|
(child) => String(child.className).indexOf(needle) >= 0,
|
|
);
|
|
const role = overlay.getAttribute('role');
|
|
const ariaModal = overlay.getAttribute('aria-modal');
|
|
await harness.silence(300);
|
|
const listeningState = overlay.dataset.voiceState;
|
|
await harness.speak('alpha', 1300);
|
|
await harness.silence(1500);
|
|
await harness.tick(400);
|
|
const captions = find(overlay, 'voice-conversation-captions');
|
|
const userCaption = captions ? captions.children[0].textContent : '';
|
|
const captionsLive = captions ? captions.getAttribute('aria-live') : '';
|
|
harness.setAssistantReply('A visible answer.');
|
|
await harness.tick(300);
|
|
const assistantCaption = captions ? captions.children[1].textContent : '';
|
|
const thinkingState = overlay.dataset.voiceState;
|
|
const controls = find(overlay, 'voice-conversation-controls');
|
|
const muteBtn = controls.children[0];
|
|
const exitBtn = controls.children[1];
|
|
muteBtn.click();
|
|
const mutedPressed = muteBtn.getAttribute('aria-pressed');
|
|
const mutedTracks = harness.trackStates();
|
|
muteBtn.click();
|
|
const unmutedTracks = harness.trackStates();
|
|
exitBtn.click();
|
|
await harness.flush();
|
|
const removedOnExit = !harness.overlay();
|
|
const inactiveAfterExit = !harness.context._voiceModeActive();
|
|
// Re-activate and leave through Escape instead.
|
|
harness.elements.btnVoiceMode.click();
|
|
await harness.flush();
|
|
const overlayAgain = harness.overlay();
|
|
const keydown = overlayAgain.listeners.find((entry) => entry.type === 'keydown');
|
|
keydown.handler({ key: 'Escape', preventDefault() {} });
|
|
await harness.flush();
|
|
return {
|
|
overlayPresent: true,
|
|
role,
|
|
ariaModal,
|
|
listeningState,
|
|
thinkingState,
|
|
userCaption,
|
|
assistantCaption,
|
|
captionsLive,
|
|
mutedPressed,
|
|
mutedTracks,
|
|
unmutedTracks,
|
|
removedOnExit,
|
|
inactiveAfterExit,
|
|
removedOnEscape: !harness.overlay(),
|
|
sends: harness.sends,
|
|
};
|
|
};
|
|
|
|
// Dynamic endpointing: a >=3-word partial reads as a plausibly complete
|
|
// utterance and endpoints at the base silence window; 1-2 word partials keep
|
|
// the young-utterance hold so thinking pauses still never clip.
|
|
scenarios.three_word_partial_endpoints_at_base_silence = async () => {
|
|
const harness = makeHarness();
|
|
await harness.start();
|
|
await harness.silence(300);
|
|
await harness.speak('alpha', 300);
|
|
await harness.speak('bravo', 300);
|
|
await harness.speak('charlie', 400);
|
|
await harness.silence(1200); // past the base 1100ms endpoint, well under the 1800ms hold
|
|
await harness.tick(300);
|
|
return { sends: harness.sends, state: harness.state() };
|
|
};
|
|
|
|
scenarios.two_word_young_utterance_keeps_the_hold = async () => {
|
|
const harness = makeHarness();
|
|
await harness.start();
|
|
await harness.silence(300);
|
|
await harness.speak('alpha', 300);
|
|
await harness.speak('bravo', 300);
|
|
await harness.silence(1300); // longer than base, shorter than the hold
|
|
const sendsEarly = harness.sends.slice();
|
|
await harness.silence(700);
|
|
await harness.tick(300);
|
|
return { sendsEarly, sends: harness.sends };
|
|
};
|
|
|
|
// An errored turn (error/system envelope in the transcript) is never spoken,
|
|
// never captioned as a reply, and fully resynchronizes capture.
|
|
scenarios.errored_turn_is_not_spoken_and_capture_resyncs = async () => {
|
|
const harness = makeHarness();
|
|
await harness.start();
|
|
await harness.silence(300);
|
|
await harness.speak('alpha', 1300);
|
|
await harness.silence(1500);
|
|
await harness.tick(400); // dispatched; thinking
|
|
const serversBefore = harness.servers.length;
|
|
const ttsBefore = harness.ttsCalls.length;
|
|
harness.setAssistantError('**Task cancelled:** Task cancelled.');
|
|
await harness.tick(300); // the response observer sees the envelope
|
|
const stateAfterError = harness.state();
|
|
const labelAfterError = harness.elements.voiceModeLabel.textContent;
|
|
await harness.speak('bravo', 1300);
|
|
await harness.silence(1500);
|
|
await harness.tick(400);
|
|
return {
|
|
stateAfterError,
|
|
labelAfterError,
|
|
ttsDuringError: harness.ttsCalls.length - ttsBefore,
|
|
freshSessions: harness.servers.length - serversBefore,
|
|
sends: harness.sends,
|
|
};
|
|
};
|
|
|
|
// A TRANSIENT provider error (a broker 5xx/502 that blipped mid-conversation)
|
|
// is auto-retried in place through the app's regenerate action instead of
|
|
// dropping the user's utterance: the error is never spoken, the overlay stays
|
|
// in Thinking (its cues cover the reconnect gap), the regenerate button is
|
|
// clicked once, and when the retried answer arrives it is spoken normally.
|
|
scenarios.transient_error_auto_retries_then_speaks = async () => {
|
|
const harness = makeHarness();
|
|
await harness.start();
|
|
await harness.silence(300);
|
|
await harness.speak('alpha', 1300);
|
|
await harness.silence(1500);
|
|
await harness.tick(400); // dispatched; thinking
|
|
const ttsBefore = harness.ttsCalls.length;
|
|
harness.setAssistantTransientError('**Error:** HTTP 502: error sending request for url (http://hermes-codex-broker:9003/v1/responses)');
|
|
await harness.tick(120); // observer sees the transient error -> one auto-retry
|
|
const stateAfterError = harness.state();
|
|
const labelAfterError = harness.elements.voiceModeLabel.textContent;
|
|
const clicksAfterError = harness.regenerateClicks();
|
|
const ttsDuringError = harness.ttsCalls.slice(ttsBefore).map((r) => r.text);
|
|
// The regenerated turn recovers with a real answer.
|
|
harness.setAssistantReply('Controlled Unclassified Information.');
|
|
harness.endStream();
|
|
await harness.tick(400);
|
|
return {
|
|
stateAfterError,
|
|
labelAfterError,
|
|
clicksAfterError,
|
|
ttsDuringError,
|
|
ttsTexts: harness.ttsCalls.map((r) => r.text),
|
|
state: harness.state(),
|
|
};
|
|
};
|
|
|
|
// The auto-retry is bounded: a provider that keeps returning a transient error
|
|
// is retried at most MAX_TRANSIENT_RETRIES (2) times, then the turn is dropped
|
|
// and capture resynchronizes so the failure surfaces instead of looping.
|
|
scenarios.transient_error_retry_is_bounded = async () => {
|
|
const harness = makeHarness();
|
|
await harness.start();
|
|
await harness.silence(300);
|
|
await harness.speak('alpha', 1300);
|
|
await harness.silence(1500);
|
|
await harness.tick(400);
|
|
harness.setAssistantTransientError('**Error:** HTTP 503: service unavailable');
|
|
await harness.tick(150); // retry 1 (click truncates + starts a fresh stream)
|
|
harness.setAssistantTransientError('**Error:** HTTP 503: service unavailable');
|
|
await harness.tick(150); // retry 2
|
|
harness.setAssistantTransientError('**Error:** HTTP 503: service unavailable');
|
|
await harness.tick(150); // budget exhausted -> drop + resync
|
|
return {
|
|
clicks: harness.regenerateClicks(),
|
|
state: harness.state(),
|
|
label: harness.elements.voiceModeLabel.textContent,
|
|
};
|
|
};
|
|
|
|
// Voice barge-in appends a single-line cut marker naming the sentence that
|
|
// was playing, so the model knows where its reply was cut off.
|
|
scenarios.barge_cut_marker_records_spoken_tail = async () => {
|
|
const harness = makeHarness();
|
|
await harness.start();
|
|
await harness.silence(300);
|
|
await harness.speak('alpha', 1300);
|
|
await harness.silence(1500);
|
|
await harness.tick(400); // dispatched; thinking
|
|
harness.setAssistantReply('The first point is ready. The second point needs many more words before it becomes another speakable chunk.');
|
|
await harness.tick(300); // first sentence reaches TTS and starts playing
|
|
await harness.speak('bravo', 900); // talk over the reply
|
|
await harness.silence(1900);
|
|
await harness.tick(400);
|
|
return { sends: harness.sends, ttsTexts: harness.ttsCalls.map((request) => request.text) };
|
|
};
|
|
|
|
// The detected utterance language re-biases the following streaming STT
|
|
// session (sticky per hands-free session).
|
|
scenarios.sticky_language_biases_next_stt_session = async () => {
|
|
const harness = makeHarness();
|
|
await harness.start();
|
|
await harness.silence(300);
|
|
await harness.speak('alpha', 1300);
|
|
await harness.silence(1500);
|
|
await harness.tick(400);
|
|
harness.completeResponse();
|
|
await harness.tick(400);
|
|
await harness.speak('bravo', 1300);
|
|
await harness.silence(1500);
|
|
await harness.tick(400);
|
|
return { startLanguages: harness.servers.map((socket) => socket.startLanguage) };
|
|
};
|
|
|
|
// FIX 3: choosing a language in the conversation overlay FORCES both the
|
|
// streaming STT hint and the reply TTS voice for the whole session, overriding
|
|
// auto-detection, until Auto is chosen again.
|
|
scenarios.language_override_forces_stt_and_voice = async () => {
|
|
const harness = makeHarness();
|
|
await harness.start();
|
|
const overlay = harness.overlay();
|
|
if (!overlay) return { overlayPresent: false };
|
|
const langWrap = overlay.children.find(
|
|
(child) => String(child.className).indexOf('voice-conversation-lang') >= 0,
|
|
);
|
|
const menu = langWrap ? langWrap.children.find(
|
|
(child) => String(child.className).indexOf('voice-conversation-lang-menu') >= 0,
|
|
) : null;
|
|
const items = menu ? menu.children : [];
|
|
const ru = items.find((item) => item.getAttribute('data-lang') === 'ru');
|
|
const auto = items.find((item) => item.getAttribute('data-lang') === '');
|
|
const langBtn = langWrap ? langWrap.children.find(
|
|
(child) => String(child.className).indexOf('voice-conversation-lang-btn') >= 0,
|
|
) : null;
|
|
if (!ru) return { overlayPresent: true, ruItemPresent: false };
|
|
// Force Russian.
|
|
ru.click();
|
|
const forcedChecked = ru.getAttribute('aria-checked');
|
|
const autoChecked = auto ? auto.getAttribute('aria-checked') : null;
|
|
const btnForced = langBtn ? String(langBtn.className).indexOf('is-forced') >= 0 : false;
|
|
// Run a turn with an English reply: STT hint and TTS voice must both be 'ru'.
|
|
await harness.silence(300);
|
|
await harness.speak('alpha', 1300);
|
|
await harness.silence(1500);
|
|
await harness.tick(400);
|
|
harness.setAssistantReply('A short English reply. It has two sentences.');
|
|
harness.completeResponse();
|
|
await harness.tick(600);
|
|
const startLanguages = harness.servers.map((socket) => socket.startLanguage);
|
|
const ttsLanguages = harness.ttsCalls.map((call) => call.language);
|
|
// Release back to Auto.
|
|
if (auto) auto.click();
|
|
const releasedChecked = auto ? auto.getAttribute('aria-checked') : null;
|
|
const btnForcedAfterAuto = langBtn ? String(langBtn.className).indexOf('is-forced') >= 0 : false;
|
|
return {
|
|
overlayPresent: true,
|
|
ruItemPresent: true,
|
|
forcedChecked,
|
|
autoChecked,
|
|
btnForced,
|
|
startLanguages,
|
|
ttsLanguages,
|
|
releasedChecked,
|
|
btnForcedAfterAuto,
|
|
};
|
|
};
|
|
|
|
// THE FOLD REGRESSION. A quick spoken acknowledgement streams as an interim
|
|
// answer segment while a tool call is prepared. The pump chunks the first two
|
|
// sentences and RETAINS the still-streaming third as an unspoken tail; then the
|
|
// renderer folds the interim into the hidden .assistant-segment-worklog-source
|
|
// (extraction → '') and the final answer renders as its own new segment.
|
|
//
|
|
// Before the fix: the empty-read branch returned without flushing the retained
|
|
// tail (the acknowledgement's last words were dropped) and the speech queue
|
|
// drained with the state fallen back to Thinking while a tail was still
|
|
// outstanding; the final answer then resurfaced from the interim's stale offset
|
|
// (a garbled mid-string slice). After the fix: the whole acknowledgement is
|
|
// spoken, in order, BEFORE the distinct final answer, and the state never falls
|
|
// back to Thinking with an unspoken interim tail outstanding.
|
|
scenarios.interim_ack_fold_flushes_tail_before_final = async () => {
|
|
const harness = makeHarness();
|
|
// The stub Audio never fires onended, so playback of the first chunk parks and
|
|
// the queue does not drain (the same reason barge_cut_marker only inspects the
|
|
// first synthesized chunk). The complete ordered set of chunks that WILL be
|
|
// spoken is therefore the chunks already handed to synthesis (ttsTexts, in
|
|
// order) followed by the chunks still queued behind the parked one, assembled
|
|
// as spokenOrder below. Every assertion is a QUEUE/STATE fact, independent of
|
|
// the playback stub.
|
|
const violations = [];
|
|
const record = () => {
|
|
const snap = harness.speechTurnSnapshot();
|
|
if (snap && snap.state === 'thinking' && !snap.final && snap.consumed < snap.sourceLength) {
|
|
violations.push({ consumed: snap.consumed, sourceLength: snap.sourceLength });
|
|
}
|
|
};
|
|
const settle = async (steps) => { for (let i = 0; i < steps; i += 1) { await harness.tick(100); record(); } };
|
|
|
|
await harness.start();
|
|
await harness.silence(300);
|
|
await harness.speak('alpha', 1300);
|
|
await harness.silence(1500);
|
|
await harness.tick(400); // dispatched; thinking; response observer running
|
|
|
|
// Interim acknowledgement: two complete sentences plus a third still mid-word,
|
|
// so the pump enqueues S1+S2 and retains "One moment while I che" unspoken.
|
|
const interim = 'Sure thing. Let me look that up. One moment while I che';
|
|
harness.setInterimAck(interim);
|
|
await settle(4);
|
|
|
|
// The tool call begins: fold the interim into the hidden worklog source.
|
|
harness.foldInterimAck();
|
|
await settle(6);
|
|
const afterFold = harness.speechTurnSnapshot();
|
|
const queuedAfterFold = (afterFold && afterFold.queue) || [];
|
|
|
|
// The final answer renders as its own new, distinct segment.
|
|
const finalAnswer = 'The weather today is sunny and warm. Enjoy your afternoon out there.';
|
|
harness.setFinalAnswer(finalAnswer);
|
|
await settle(6);
|
|
harness.completeResponse();
|
|
await settle(10);
|
|
|
|
const finalSnap = harness.speechTurnSnapshot();
|
|
const residualQueue = (finalSnap && finalSnap.queue) || [];
|
|
// The full ordered list of chunks that reach the speech queue for this turn.
|
|
const spokenOrder = harness.ttsTexts().concat(residualQueue);
|
|
const joined = spokenOrder.join('');
|
|
const lastInterimIdx = (() => {
|
|
let idx = -1;
|
|
spokenOrder.forEach((t, i) => { if (/Sure thing|look that up|One moment|while I che/.test(t)) idx = i; });
|
|
return idx;
|
|
})();
|
|
const firstFinalIdx = spokenOrder.findIndex((t) => /weather|sunny|afternoon/.test(t));
|
|
|
|
return {
|
|
spokenOrder,
|
|
// After the fold, the interim's retained tail is the next thing queued, and
|
|
// it is queued BEFORE any final-answer chunk exists.
|
|
queuedAfterFold,
|
|
tailQueuedAfterFold: queuedAfterFold.some((t) => /while I che/.test(t)),
|
|
noFinalBeforeFold: !queuedAfterFold.some((t) => /weather|sunny|afternoon/.test(t)),
|
|
// (i) every interim sentence reaches the speech queue: none dropped.
|
|
interimS1Reached: /Sure thing/.test(joined),
|
|
interimS2Reached: /look that up/.test(joined),
|
|
interimTailReached: /while I che/.test(joined),
|
|
// (iii) the distinct final answer is fully queued too.
|
|
finalS1Reached: /weather today is sunny/.test(joined),
|
|
finalS2Reached: /Enjoy your afternoon/.test(joined),
|
|
// The final answer is queued from its OWN start, never resurfaced from the
|
|
// interim's stale offset ("nny and warm" is the buggy mid-string slice).
|
|
finalNotGarbled: !spokenOrder.some((t) => /^nny and warm/.test(t)),
|
|
// (ii) the interim finishes before the final message's chunks are enqueued.
|
|
interimBeforeFinal: lastInterimIdx >= 0 && firstFinalIdx >= 0 && lastInterimIdx < firstFinalIdx,
|
|
// (iv) the state never fell back to Thinking with an unspoken interim tail.
|
|
thinkingWithUnspokenTail: violations.length,
|
|
state: harness.state(),
|
|
};
|
|
};
|
|
|
|
// UNIFIED OUTPUT SINK + device selector. When the browser can enumerate outputs
|
|
// and setSinkId, a tidy corner control lists them; choosing one routes ALL spoken
|
|
// output (the reply's blob element here, and — via the shared AudioContext — the
|
|
// PCM reply and thinking cues) to that single device. When the APIs are missing,
|
|
// the control hides entirely (a dead control never appears).
|
|
scenarios.output_device_selector_routes_spoken_output = async () => {
|
|
const harness = makeHarness({
|
|
outputs: [
|
|
{ deviceId: '', kind: 'audiooutput', label: 'System default' },
|
|
{ deviceId: 'spk-1', kind: 'audiooutput', label: 'Speaker One' },
|
|
{ deviceId: 'spk-2', kind: 'audiooutput', label: 'Headphones Two' },
|
|
{ deviceId: 'mic-1', kind: 'audioinput', label: 'Microphone' },
|
|
],
|
|
});
|
|
await harness.start();
|
|
await harness.flush();
|
|
await harness.flush();
|
|
const overlay = harness.overlay();
|
|
if (!overlay) return { overlayPresent: false };
|
|
const outWrap = overlay.children.find(
|
|
(child) => String(child.className).indexOf('voice-conversation-out') >= 0,
|
|
);
|
|
const outBtn = outWrap ? outWrap.children.find(
|
|
(child) => String(child.className).indexOf('voice-conversation-out-btn') >= 0,
|
|
) : null;
|
|
const outMenu = outWrap ? outWrap.children.find(
|
|
(child) => String(child.className).indexOf('voice-conversation-out-menu') >= 0,
|
|
) : null;
|
|
const shownAfterRefresh = outWrap ? outWrap.style.display !== 'none' : false;
|
|
const items = outMenu ? outMenu.children : [];
|
|
const itemLabels = items.map((item) => item.textContent);
|
|
const headphones = items.find((item) => item.getAttribute('data-device') === 'spk-2');
|
|
if (headphones) headphones.click();
|
|
const forcedChecked = headphones ? headphones.getAttribute('aria-checked') : null;
|
|
const btnForced = outBtn ? String(outBtn.className).indexOf('is-forced') >= 0 : false;
|
|
// A reply now plays through the blob fallback element, which must be routed to
|
|
// the chosen sink.
|
|
await harness.silence(300);
|
|
await harness.speak('alpha', 1300);
|
|
await harness.silence(1500);
|
|
await harness.tick(400);
|
|
harness.setAssistantReply('A short spoken reply.');
|
|
harness.completeResponse();
|
|
await harness.tick(400);
|
|
const sinks = harness.sinkCalls();
|
|
return {
|
|
overlayPresent: true,
|
|
supported: true,
|
|
shownAfterRefresh,
|
|
itemLabels,
|
|
forcedChecked,
|
|
btnForced,
|
|
audioRoutedTo: sinks.filter((c) => c.kind === 'audio').map((c) => c.sink),
|
|
};
|
|
};
|
|
|
|
// When enumerateDevices/setSinkId are unavailable the selector is present in the
|
|
// DOM but hidden — never a dead control.
|
|
scenarios.output_selector_hidden_when_unsupported = async () => {
|
|
const harness = makeHarness();
|
|
await harness.start();
|
|
await harness.flush();
|
|
const overlay = harness.overlay();
|
|
if (!overlay) return { overlayPresent: false };
|
|
const outWrap = overlay.children.find(
|
|
(child) => String(child.className).indexOf('voice-conversation-out') >= 0,
|
|
);
|
|
return {
|
|
overlayPresent: true,
|
|
controlInDom: !!outWrap,
|
|
hidden: outWrap ? outWrap.style.display === 'none' : null,
|
|
};
|
|
};
|
|
|
|
// Shape of the attached full-screen overlay: the essential orb + captions +
|
|
// mute/exit that MUST render whenever conversation mode opens, independent of the
|
|
// language/output corner selectors.
|
|
function overlayShape(harness) {
|
|
const overlay = harness.overlay();
|
|
if (!overlay) return { overlayPresent: false };
|
|
const kids = overlay.children;
|
|
const find = (sub) => kids.find((c) => String(c.className).indexOf(sub) >= 0) || null;
|
|
const orb = find('voice-conversation-orb');
|
|
const orbMark = orb
|
|
? orb.children.find((c) => String(c.className).indexOf('voice-conversation-orb-mark') >= 0)
|
|
: null;
|
|
const captions = find('voice-conversation-captions');
|
|
const controls = find('voice-conversation-controls');
|
|
const mute = controls
|
|
? controls.children.find((c) => String(c.className).indexOf('voice-conversation-mute') >= 0)
|
|
: null;
|
|
const exit = controls
|
|
? controls.children.find((c) => String(c.className).indexOf('voice-conversation-exit') >= 0)
|
|
: null;
|
|
return {
|
|
overlayPresent: true,
|
|
isDialog: overlay.getAttribute('role') === 'dialog',
|
|
hasOrb: !!orb,
|
|
hasOrbMark: !!orbMark,
|
|
hasCaptions: !!captions,
|
|
captionCount: captions ? captions.children.length : 0,
|
|
hasMute: !!mute,
|
|
hasExit: !!exit,
|
|
};
|
|
}
|
|
|
|
// REGRESSION LOCK: the full-screen overlay must ALWAYS attach when conversation
|
|
// mode opens, even when the audio-output APIs are hostile — mediaDevices absent,
|
|
// or enumerateDevices rejecting. A failure here previously discarded the whole
|
|
// overlay and dropped back to the inline voice bar.
|
|
scenarios.overlay_opens_despite_hostile_output_apis = async () => {
|
|
const results = {};
|
|
|
|
// (A) navigator.mediaDevices is undefined when the overlay is built.
|
|
{
|
|
const harness = makeHarness();
|
|
await harness.flush();
|
|
const internals = harness.context.window.__atlasVoiceInternals;
|
|
harness.context.navigator.mediaDevices = undefined;
|
|
internals.openConversationOverlay();
|
|
results.mediaDevicesUndefined = overlayShape(harness);
|
|
const overlay = harness.overlay();
|
|
const outWrap = overlay
|
|
? overlay.children.find((c) => String(c.className).indexOf('voice-conversation-out') >= 0)
|
|
: null;
|
|
results.mediaDevicesUndefined.outputHidden = outWrap ? outWrap.style.display === 'none' : null;
|
|
internals.removeConversationOverlay();
|
|
}
|
|
|
|
// (B) enumerateDevices is present (routing LOOKS supported) but REJECTS.
|
|
{
|
|
const harness = makeHarness();
|
|
await harness.flush();
|
|
const internals = harness.context.window.__atlasVoiceInternals;
|
|
harness.context.navigator.mediaDevices.enumerateDevices = async () => {
|
|
throw new Error('enumerate blocked');
|
|
};
|
|
internals.openConversationOverlay();
|
|
// The overlay must be attached synchronously, before the async rejection.
|
|
const immediate = overlayShape(harness);
|
|
await harness.flush();
|
|
await harness.flush();
|
|
results.enumerateRejects = overlayShape(harness);
|
|
results.enumerateRejects.attachedBeforeReject = immediate.overlayPresent;
|
|
internals.removeConversationOverlay();
|
|
}
|
|
return results;
|
|
};
|
|
|
|
// FIX 2(a): with the mic open the OS routes the system default to the earpiece;
|
|
// the shared sink must auto-default to the LOUDSPEAKER (never "communications"),
|
|
// with no user interaction — and that default must reach real playback.
|
|
scenarios.output_defaults_to_loudspeaker = async () => {
|
|
const harness = makeHarness({
|
|
outputs: [
|
|
{ deviceId: '', kind: 'audiooutput', label: 'System default' },
|
|
{ deviceId: 'communications', kind: 'audiooutput', label: 'Communications' },
|
|
{ deviceId: 'ear-1', kind: 'audiooutput', label: 'Earpiece' },
|
|
{ deviceId: 'spk-1', kind: 'audiooutput', label: 'Speakerphone' },
|
|
{ deviceId: 'mic-1', kind: 'audioinput', label: 'Microphone' },
|
|
],
|
|
});
|
|
await harness.start();
|
|
await harness.flush();
|
|
await harness.flush();
|
|
const overlay = harness.overlay();
|
|
if (!overlay) return { overlayPresent: false };
|
|
const outWrap = overlay.children.find((c) => String(c.className).indexOf('voice-conversation-out') >= 0);
|
|
const outBtn = outWrap ? outWrap.children.find((c) => String(c.className).indexOf('voice-conversation-out-btn') >= 0) : null;
|
|
const outMenu = outWrap ? outWrap.children.find((c) => String(c.className).indexOf('voice-conversation-out-menu') >= 0) : null;
|
|
const items = outMenu ? outMenu.children : [];
|
|
const defaultChecked = items
|
|
.filter((i) => i.getAttribute('aria-checked') === 'true')
|
|
.map((i) => i.getAttribute('data-device'));
|
|
const btnForced = outBtn ? String(outBtn.className).indexOf('is-forced') >= 0 : false;
|
|
const btnLabel = outBtn ? outBtn.getAttribute('aria-label') : null;
|
|
// The default must reach real playback with NO user interaction: a reply now
|
|
// plays through the blob fallback and must route to the auto-selected speaker.
|
|
await harness.silence(300);
|
|
await harness.speak('alpha', 1300);
|
|
await harness.silence(1500);
|
|
await harness.tick(400);
|
|
harness.setAssistantReply('A short spoken reply.');
|
|
harness.completeResponse();
|
|
await harness.tick(400);
|
|
const audioRoutedTo = harness.sinkCalls().filter((c) => c.kind === 'audio').map((c) => c.sink);
|
|
return {
|
|
overlayPresent: true,
|
|
defaultChecked,
|
|
btnForced,
|
|
btnLabel,
|
|
audioRoutedTo,
|
|
};
|
|
};
|
|
|
|
// FIX 2(a) pure logic: the loudspeaker chooser prefers a labelled speaker, never
|
|
// the "communications" endpoint, falls back to the first concrete non-earpiece
|
|
// output, and returns '' (system default) when only an earpiece/comms exists.
|
|
scenarios.loudspeaker_selection_logic = async () => {
|
|
const harness = makeHarness();
|
|
await harness.flush();
|
|
const pick = harness.context.window.__atlasVoiceInternals.pickLoudspeakerSink;
|
|
return {
|
|
labelledSpeaker: pick([
|
|
{ deviceId: 'ear', kind: 'audiooutput', label: 'Earpiece' },
|
|
{ deviceId: 'spk', kind: 'audiooutput', label: 'Speakerphone' },
|
|
]),
|
|
skipsCommunications: pick([
|
|
{ deviceId: 'communications', kind: 'audiooutput', label: 'Communications' },
|
|
{ deviceId: 'spk', kind: 'audiooutput', label: 'Speaker' },
|
|
]),
|
|
concreteFallback: pick([
|
|
{ deviceId: 'default', kind: 'audiooutput', label: '' },
|
|
{ deviceId: 'dev-9', kind: 'audiooutput', label: '' },
|
|
]),
|
|
onlyEarpiece: pick([
|
|
{ deviceId: 'communications', kind: 'audiooutput', label: 'Communications' },
|
|
{ deviceId: 'ear', kind: 'audiooutput', label: 'Earpiece' },
|
|
]),
|
|
ignoresInputs: pick([
|
|
{ deviceId: 'mic', kind: 'audioinput', label: 'Speaker Mic' },
|
|
{ deviceId: 'spk', kind: 'audiooutput', label: 'Loudspeaker' },
|
|
]),
|
|
};
|
|
};
|
|
|
|
(async () => {
|
|
const output = {};
|
|
for (const name of Object.keys(scenarios)) {
|
|
// eslint-disable-next-line no-await-in-loop
|
|
output[name] = await scenarios[name]();
|
|
}
|
|
process.stdout.write(JSON.stringify(output, null, 1));
|
|
})().catch((error) => {
|
|
process.stderr.write(String((error && error.stack) || error));
|
|
process.exit(1);
|
|
});
|