Complete rebuild of voice input on a server-authoritative streaming architecture, replacing the legacy Web Speech / whole-blob / WASM engines and the dead voice-agent layer (~4k lines removed). Speech-to-text (dictation): - Client streams 16 kHz mono PCM16 chunks over /api/dictation/ws with seq/ack ordering; buffered audio is retained and replayed on reconnect - Server transcribes and streams live partial transcripts back; segments auto-commit every ~15s with silence suppression and adaptive finalization timeouts - Local provider (default, zero config): sherpa-onnx models in a forked worker process — auto-download with progress, staged extraction with verification, corrupt-model auto-recovery, idle shutdown after 5 min - Model catalog with settings picker (accuracy/speed ratings, sizes, download/delete): Parakeet TDT v2 (English) and v3 (25 European languages, auto-detected), Whisper base and tiny (multilingual, light) - OpenAI-compatible provider for any Whisper endpoint - Composer overlay with live transcript, volume meter, timer, and cancel / insert / insert-and-send actions; failed transcriptions keep their audio for retry or accepting the partial text as-is - Configurable keyboard shortcut (default mod+alt+v) toggles dictation; Enter confirms and Escape cancels while recording - Overlay is pixel-aligned with the composer (measured footer height, matching paddings/typography/gaps) — no layout shift when toggling Text-to-speech: - Local Kokoro provider (English, 11 voices) synthesized in the same worker via /api/dictation/tts/speak, managed by the shared model pipeline; sentence-pipelined playback keeps time-to-first-audio at ~1 sentence regardless of message length, and stop cancels in-flight synthesis - Sanitizer keeps inline-code content (strips backticks only), reads interword slashes aloud, and removes only absolute file paths Settings: - Voice page unified: a single read-aloud toggle owns all playback options (the confusing "Enable Voice Mode" is gone); a new "Enable voice input" toggle (default on, persisted to settings.json) hides the composer mic entirely when disabled Mobile and transport: - iOS/Android microphone permissions added (dictation was previously impossible on mobile) - Fixed Android WebSocket upgrades: the Capacitor WebView origin (https://localhost) was missing from the packaged-client allowlist, 403-ing every WS connection — root cause of the old mobile SSE lock, which is now removed for all transports Security and conventions: - All HTTP routes sit behind the global /api auth gate; the WS upgrade explicitly validates the UI session and origin, with oc_url_token narrowly allowlisted and covered by tests; the dictation socket mints a fresh URL token before connecting - Routes register before the generic OpenCode proxy; the client goes through runtimeFetch/getRuntimeUrlResolver, and runtime switches reset the dictation socket - VS Code deliberately reports dictation as unavailable (no server process in that runtime) CI: workflow Node bumped 20 -> 22 to match the repo engines and fix better-sqlite3 installs broken by node-gyp@latest on Node 20. New dependency: sherpa-onnx-node (prebuilt N-API; macOS/Linux x64+arm64, Windows x64 — Windows-on-ARM falls back to the OpenAI-compatible provider)
195 lines
6.1 KiB
JavaScript
195 lines
6.1 KiB
JavaScript
import { describe, it, expect } from 'bun:test';
|
|
import { EventEmitter } from 'events';
|
|
|
|
import { DictationStreamManager } from './stream-manager.js';
|
|
|
|
const FORMAT = 'audio/pcm;rate=16000;bits=16';
|
|
|
|
class FakeSttSession extends EventEmitter {
|
|
constructor({ transcriptBySegment = () => 'hello world' } = {}) {
|
|
super();
|
|
this.requiredSampleRate = 16000;
|
|
this.appended = [];
|
|
this.commits = 0;
|
|
this.clears = 0;
|
|
this.closed = false;
|
|
this.segmentCounter = 0;
|
|
this.transcriptBySegment = transcriptBySegment;
|
|
}
|
|
|
|
async connect() {}
|
|
|
|
appendPcm16(buf) {
|
|
this.appended.push(buf);
|
|
}
|
|
|
|
commit() {
|
|
this.commits += 1;
|
|
const segmentId = `seg-${this.segmentCounter}`;
|
|
this.segmentCounter += 1;
|
|
this.emit('committed', { segmentId, previousSegmentId: null });
|
|
setTimeout(() => {
|
|
this.emit('transcript', {
|
|
segmentId,
|
|
transcript: this.transcriptBySegment(segmentId),
|
|
isFinal: true,
|
|
});
|
|
}, 0);
|
|
}
|
|
|
|
clear() {
|
|
this.clears += 1;
|
|
}
|
|
|
|
close() {
|
|
this.closed = true;
|
|
}
|
|
}
|
|
|
|
function loudChunkBase64(samples = 1600, amplitude = 8000) {
|
|
const arr = new Int16Array(samples);
|
|
for (let i = 0; i < samples; i += 1) {
|
|
arr[i] = i % 2 === 0 ? amplitude : -amplitude;
|
|
}
|
|
return Buffer.from(arr.buffer).toString('base64');
|
|
}
|
|
|
|
function silentChunkBase64(samples = 1600) {
|
|
return Buffer.from(new Int16Array(samples).buffer).toString('base64');
|
|
}
|
|
|
|
function createManager(session) {
|
|
const messages = [];
|
|
const manager = new DictationStreamManager({
|
|
emit: (msg) => messages.push(msg),
|
|
createSttSession: async () => ({ session }),
|
|
});
|
|
return { manager, messages };
|
|
}
|
|
|
|
function waitFor(predicate, timeoutMs = 1000) {
|
|
return new Promise((resolve, reject) => {
|
|
const startedAt = Date.now();
|
|
const tick = () => {
|
|
if (predicate()) {
|
|
resolve(undefined);
|
|
return;
|
|
}
|
|
if (Date.now() - startedAt > timeoutMs) {
|
|
reject(new Error('waitFor timed out'));
|
|
return;
|
|
}
|
|
setTimeout(tick, 5);
|
|
};
|
|
tick();
|
|
});
|
|
}
|
|
|
|
describe('DictationStreamManager', () => {
|
|
it('transcribes ordered chunks and emits final text', async () => {
|
|
const session = new FakeSttSession();
|
|
const { manager, messages } = createManager(session);
|
|
|
|
await manager.handleStart('d1', FORMAT, {});
|
|
manager.handleChunk({ dictationId: 'd1', seq: 0, audioBase64: loudChunkBase64() });
|
|
manager.handleChunk({ dictationId: 'd1', seq: 1, audioBase64: loudChunkBase64() });
|
|
manager.handleFinish('d1', 1);
|
|
|
|
await waitFor(() => messages.some((m) => m.type === 'final'));
|
|
|
|
const final = messages.find((m) => m.type === 'final');
|
|
expect(final.payload.text).toBe('hello world');
|
|
expect(session.commits).toBe(1);
|
|
expect(session.closed).toBe(true);
|
|
|
|
const acks = messages.filter((m) => m.type === 'ack');
|
|
expect(acks[acks.length - 1].payload.ackSeq).toBe(1);
|
|
});
|
|
|
|
it('reorders out-of-order chunks before appending', async () => {
|
|
const session = new FakeSttSession();
|
|
const { manager, messages } = createManager(session);
|
|
|
|
await manager.handleStart('d1', FORMAT, {});
|
|
manager.handleChunk({ dictationId: 'd1', seq: 1, audioBase64: loudChunkBase64() });
|
|
expect(session.appended.length).toBe(0);
|
|
manager.handleChunk({ dictationId: 'd1', seq: 0, audioBase64: loudChunkBase64() });
|
|
expect(session.appended.length).toBe(2);
|
|
manager.handleFinish('d1', 1);
|
|
|
|
await waitFor(() => messages.some((m) => m.type === 'final'));
|
|
});
|
|
|
|
it('clears silence-only tails instead of committing', async () => {
|
|
const session = new FakeSttSession();
|
|
const { manager, messages } = createManager(session);
|
|
|
|
await manager.handleStart('d1', FORMAT, {});
|
|
manager.handleChunk({ dictationId: 'd1', seq: 0, audioBase64: silentChunkBase64() });
|
|
manager.handleFinish('d1', 0);
|
|
|
|
await waitFor(() => messages.some((m) => m.type === 'final'));
|
|
|
|
const final = messages.find((m) => m.type === 'final');
|
|
expect(final.payload.text).toBe('');
|
|
expect(session.commits).toBe(0);
|
|
expect(session.clears).toBe(1);
|
|
});
|
|
|
|
it('fails fast when finish arrives with no chunks', async () => {
|
|
const session = new FakeSttSession();
|
|
const { manager, messages } = createManager(session);
|
|
|
|
await manager.handleStart('d1', FORMAT, {});
|
|
manager.handleFinish('d1', 3);
|
|
|
|
const error = messages.find((m) => m.type === 'error');
|
|
expect(error).toBeDefined();
|
|
expect(error.payload.retryable).toBe(true);
|
|
expect(session.closed).toBe(true);
|
|
});
|
|
|
|
it('reports provider readiness errors from createSttSession', async () => {
|
|
const messages = [];
|
|
const manager = new DictationStreamManager({
|
|
emit: (msg) => messages.push(msg),
|
|
createSttSession: async () => ({
|
|
error: 'Dictation model is downloading',
|
|
retryable: true,
|
|
reasonCode: 'model_download_in_progress',
|
|
}),
|
|
});
|
|
|
|
await manager.handleStart('d1', FORMAT, {});
|
|
const error = messages.find((m) => m.type === 'error');
|
|
expect(error.payload.reasonCode).toBe('model_download_in_progress');
|
|
expect(error.payload.retryable).toBe(true);
|
|
});
|
|
|
|
it('emits partials as segment transcripts arrive', async () => {
|
|
let segment = 0;
|
|
const session = new FakeSttSession({
|
|
transcriptBySegment: () => {
|
|
segment += 1;
|
|
return segment === 1 ? 'first part' : 'second part';
|
|
},
|
|
});
|
|
const { manager, messages } = createManager(session);
|
|
// Force auto-commit after ~0.05s of audio so two segments form.
|
|
manager.autoCommitSeconds = 0.05;
|
|
|
|
await manager.handleStart('d1', FORMAT, {});
|
|
manager.handleChunk({ dictationId: 'd1', seq: 0, audioBase64: loudChunkBase64(1600) });
|
|
await waitFor(() => session.commits >= 1);
|
|
manager.handleChunk({ dictationId: 'd1', seq: 1, audioBase64: loudChunkBase64(1600) });
|
|
manager.handleFinish('d1', 1);
|
|
|
|
await waitFor(() => messages.some((m) => m.type === 'final'));
|
|
|
|
const final = messages.find((m) => m.type === 'final');
|
|
expect(final.payload.text).toBe('first part second part');
|
|
const partials = messages.filter((m) => m.type === 'partial');
|
|
expect(partials.length).toBeGreaterThan(0);
|
|
});
|
|
});
|