Files
openchamber/packages/web/server/lib/dictation/stream-manager.test.js
T
Bohdan Triapitsyn de1b85ac56 feat(voice): first-class voice input and local TTS across web, desktop, and mobile (#2018)
Complete rebuild of voice input on a server-authoritative streaming
architecture, replacing the legacy Web Speech / whole-blob / WASM engines
and the dead voice-agent layer (~4k lines removed).

Speech-to-text (dictation):
- Client streams 16 kHz mono PCM16 chunks over /api/dictation/ws with
  seq/ack ordering; buffered audio is retained and replayed on reconnect
- Server transcribes and streams live partial transcripts back;
  segments auto-commit every ~15s with silence suppression and adaptive
  finalization timeouts
- Local provider (default, zero config): sherpa-onnx models in a forked
  worker process — auto-download with progress, staged extraction with
  verification, corrupt-model auto-recovery, idle shutdown after 5 min
- Model catalog with settings picker (accuracy/speed ratings, sizes,
  download/delete): Parakeet TDT v2 (English) and v3 (25 European
  languages, auto-detected), Whisper base and tiny (multilingual, light)
- OpenAI-compatible provider for any Whisper endpoint
- Composer overlay with live transcript, volume meter, timer, and
  cancel / insert / insert-and-send actions; failed transcriptions keep
  their audio for retry or accepting the partial text as-is
- Configurable keyboard shortcut (default mod+alt+v) toggles dictation;
  Enter confirms and Escape cancels while recording
- Overlay is pixel-aligned with the composer (measured footer height,
  matching paddings/typography/gaps) — no layout shift when toggling

Text-to-speech:
- Local Kokoro provider (English, 11 voices) synthesized in the same
  worker via /api/dictation/tts/speak, managed by the shared model
  pipeline; sentence-pipelined playback keeps time-to-first-audio at
  ~1 sentence regardless of message length, and stop cancels in-flight
  synthesis
- Sanitizer keeps inline-code content (strips backticks only), reads
  interword slashes aloud, and removes only absolute file paths

Settings:
- Voice page unified: a single read-aloud toggle owns all playback
  options (the confusing "Enable Voice Mode" is gone); a new "Enable
  voice input" toggle (default on, persisted to settings.json) hides
  the composer mic entirely when disabled

Mobile and transport:
- iOS/Android microphone permissions added (dictation was previously
  impossible on mobile)
- Fixed Android WebSocket upgrades: the Capacitor WebView origin
  (https://localhost) was missing from the packaged-client allowlist,
  403-ing every WS connection — root cause of the old mobile SSE lock,
  which is now removed for all transports

Security and conventions:
- All HTTP routes sit behind the global /api auth gate; the WS upgrade
  explicitly validates the UI session and origin, with oc_url_token
  narrowly allowlisted and covered by tests; the dictation socket mints
  a fresh URL token before connecting
- Routes register before the generic OpenCode proxy; the client goes
  through runtimeFetch/getRuntimeUrlResolver, and runtime switches
  reset the dictation socket
- VS Code deliberately reports dictation as unavailable (no server
  process in that runtime)

CI: workflow Node bumped 20 -> 22 to match the repo engines and fix
better-sqlite3 installs broken by node-gyp@latest on Node 20.

New dependency: sherpa-onnx-node (prebuilt N-API; macOS/Linux x64+arm64,
Windows x64 — Windows-on-ARM falls back to the OpenAI-compatible provider)
2026-07-04 02:48:07 +03:00

195 lines
6.1 KiB
JavaScript

import { describe, it, expect } from 'bun:test';
import { EventEmitter } from 'events';
import { DictationStreamManager } from './stream-manager.js';
const FORMAT = 'audio/pcm;rate=16000;bits=16';
class FakeSttSession extends EventEmitter {
constructor({ transcriptBySegment = () => 'hello world' } = {}) {
super();
this.requiredSampleRate = 16000;
this.appended = [];
this.commits = 0;
this.clears = 0;
this.closed = false;
this.segmentCounter = 0;
this.transcriptBySegment = transcriptBySegment;
}
async connect() {}
appendPcm16(buf) {
this.appended.push(buf);
}
commit() {
this.commits += 1;
const segmentId = `seg-${this.segmentCounter}`;
this.segmentCounter += 1;
this.emit('committed', { segmentId, previousSegmentId: null });
setTimeout(() => {
this.emit('transcript', {
segmentId,
transcript: this.transcriptBySegment(segmentId),
isFinal: true,
});
}, 0);
}
clear() {
this.clears += 1;
}
close() {
this.closed = true;
}
}
function loudChunkBase64(samples = 1600, amplitude = 8000) {
const arr = new Int16Array(samples);
for (let i = 0; i < samples; i += 1) {
arr[i] = i % 2 === 0 ? amplitude : -amplitude;
}
return Buffer.from(arr.buffer).toString('base64');
}
function silentChunkBase64(samples = 1600) {
return Buffer.from(new Int16Array(samples).buffer).toString('base64');
}
function createManager(session) {
const messages = [];
const manager = new DictationStreamManager({
emit: (msg) => messages.push(msg),
createSttSession: async () => ({ session }),
});
return { manager, messages };
}
function waitFor(predicate, timeoutMs = 1000) {
return new Promise((resolve, reject) => {
const startedAt = Date.now();
const tick = () => {
if (predicate()) {
resolve(undefined);
return;
}
if (Date.now() - startedAt > timeoutMs) {
reject(new Error('waitFor timed out'));
return;
}
setTimeout(tick, 5);
};
tick();
});
}
describe('DictationStreamManager', () => {
it('transcribes ordered chunks and emits final text', async () => {
const session = new FakeSttSession();
const { manager, messages } = createManager(session);
await manager.handleStart('d1', FORMAT, {});
manager.handleChunk({ dictationId: 'd1', seq: 0, audioBase64: loudChunkBase64() });
manager.handleChunk({ dictationId: 'd1', seq: 1, audioBase64: loudChunkBase64() });
manager.handleFinish('d1', 1);
await waitFor(() => messages.some((m) => m.type === 'final'));
const final = messages.find((m) => m.type === 'final');
expect(final.payload.text).toBe('hello world');
expect(session.commits).toBe(1);
expect(session.closed).toBe(true);
const acks = messages.filter((m) => m.type === 'ack');
expect(acks[acks.length - 1].payload.ackSeq).toBe(1);
});
it('reorders out-of-order chunks before appending', async () => {
const session = new FakeSttSession();
const { manager, messages } = createManager(session);
await manager.handleStart('d1', FORMAT, {});
manager.handleChunk({ dictationId: 'd1', seq: 1, audioBase64: loudChunkBase64() });
expect(session.appended.length).toBe(0);
manager.handleChunk({ dictationId: 'd1', seq: 0, audioBase64: loudChunkBase64() });
expect(session.appended.length).toBe(2);
manager.handleFinish('d1', 1);
await waitFor(() => messages.some((m) => m.type === 'final'));
});
it('clears silence-only tails instead of committing', async () => {
const session = new FakeSttSession();
const { manager, messages } = createManager(session);
await manager.handleStart('d1', FORMAT, {});
manager.handleChunk({ dictationId: 'd1', seq: 0, audioBase64: silentChunkBase64() });
manager.handleFinish('d1', 0);
await waitFor(() => messages.some((m) => m.type === 'final'));
const final = messages.find((m) => m.type === 'final');
expect(final.payload.text).toBe('');
expect(session.commits).toBe(0);
expect(session.clears).toBe(1);
});
it('fails fast when finish arrives with no chunks', async () => {
const session = new FakeSttSession();
const { manager, messages } = createManager(session);
await manager.handleStart('d1', FORMAT, {});
manager.handleFinish('d1', 3);
const error = messages.find((m) => m.type === 'error');
expect(error).toBeDefined();
expect(error.payload.retryable).toBe(true);
expect(session.closed).toBe(true);
});
it('reports provider readiness errors from createSttSession', async () => {
const messages = [];
const manager = new DictationStreamManager({
emit: (msg) => messages.push(msg),
createSttSession: async () => ({
error: 'Dictation model is downloading',
retryable: true,
reasonCode: 'model_download_in_progress',
}),
});
await manager.handleStart('d1', FORMAT, {});
const error = messages.find((m) => m.type === 'error');
expect(error.payload.reasonCode).toBe('model_download_in_progress');
expect(error.payload.retryable).toBe(true);
});
it('emits partials as segment transcripts arrive', async () => {
let segment = 0;
const session = new FakeSttSession({
transcriptBySegment: () => {
segment += 1;
return segment === 1 ? 'first part' : 'second part';
},
});
const { manager, messages } = createManager(session);
// Force auto-commit after ~0.05s of audio so two segments form.
manager.autoCommitSeconds = 0.05;
await manager.handleStart('d1', FORMAT, {});
manager.handleChunk({ dictationId: 'd1', seq: 0, audioBase64: loudChunkBase64(1600) });
await waitFor(() => session.commits >= 1);
manager.handleChunk({ dictationId: 'd1', seq: 1, audioBase64: loudChunkBase64(1600) });
manager.handleFinish('d1', 1);
await waitFor(() => messages.some((m) => m.type === 'final'));
const final = messages.find((m) => m.type === 'final');
expect(final.payload.text).toBe('first part second part');
const partials = messages.filter((m) => m.type === 'partial');
expect(partials.length).toBeGreaterThan(0);
});
});