feat(voice): first-class voice input and local TTS across web, desktop, and mobile (#2018)
Complete rebuild of voice input on a server-authoritative streaming architecture, replacing the legacy Web Speech / whole-blob / WASM engines and the dead voice-agent layer (~4k lines removed). Speech-to-text (dictation): - Client streams 16 kHz mono PCM16 chunks over /api/dictation/ws with seq/ack ordering; buffered audio is retained and replayed on reconnect - Server transcribes and streams live partial transcripts back; segments auto-commit every ~15s with silence suppression and adaptive finalization timeouts - Local provider (default, zero config): sherpa-onnx models in a forked worker process — auto-download with progress, staged extraction with verification, corrupt-model auto-recovery, idle shutdown after 5 min - Model catalog with settings picker (accuracy/speed ratings, sizes, download/delete): Parakeet TDT v2 (English) and v3 (25 European languages, auto-detected), Whisper base and tiny (multilingual, light) - OpenAI-compatible provider for any Whisper endpoint - Composer overlay with live transcript, volume meter, timer, and cancel / insert / insert-and-send actions; failed transcriptions keep their audio for retry or accepting the partial text as-is - Configurable keyboard shortcut (default mod+alt+v) toggles dictation; Enter confirms and Escape cancels while recording - Overlay is pixel-aligned with the composer (measured footer height, matching paddings/typography/gaps) — no layout shift when toggling Text-to-speech: - Local Kokoro provider (English, 11 voices) synthesized in the same worker via /api/dictation/tts/speak, managed by the shared model pipeline; sentence-pipelined playback keeps time-to-first-audio at ~1 sentence regardless of message length, and stop cancels in-flight synthesis - Sanitizer keeps inline-code content (strips backticks only), reads interword slashes aloud, and removes only absolute file paths Settings: - Voice page unified: a single read-aloud toggle owns all playback options (the confusing "Enable Voice Mode" is gone); a new "Enable voice input" toggle (default on, persisted to settings.json) hides the composer mic entirely when disabled Mobile and transport: - iOS/Android microphone permissions added (dictation was previously impossible on mobile) - Fixed Android WebSocket upgrades: the Capacitor WebView origin (https://localhost) was missing from the packaged-client allowlist, 403-ing every WS connection — root cause of the old mobile SSE lock, which is now removed for all transports Security and conventions: - All HTTP routes sit behind the global /api auth gate; the WS upgrade explicitly validates the UI session and origin, with oc_url_token narrowly allowlisted and covered by tests; the dictation socket mints a fresh URL token before connecting - Routes register before the generic OpenCode proxy; the client goes through runtimeFetch/getRuntimeUrlResolver, and runtime switches reset the dictation socket - VS Code deliberately reports dictation as unavailable (no server process in that runtime) CI: workflow Node bumped 20 -> 22 to match the repo engines and fix better-sqlite3 installs broken by node-gyp@latest on Node 20. New dependency: sherpa-onnx-node (prebuilt N-API; macOS/Linux x64+arm64, Windows x64 — Windows-on-ARM falls back to the OpenAI-compatible provider)
This commit is contained in:
committed by
GitHub
parent
3f5151d424
commit
de1b85ac56
@@ -0,0 +1,194 @@
|
||||
import { describe, it, expect } from 'bun:test';
|
||||
import { EventEmitter } from 'events';
|
||||
|
||||
import { DictationStreamManager } from './stream-manager.js';
|
||||
|
||||
const FORMAT = 'audio/pcm;rate=16000;bits=16';
|
||||
|
||||
class FakeSttSession extends EventEmitter {
|
||||
constructor({ transcriptBySegment = () => 'hello world' } = {}) {
|
||||
super();
|
||||
this.requiredSampleRate = 16000;
|
||||
this.appended = [];
|
||||
this.commits = 0;
|
||||
this.clears = 0;
|
||||
this.closed = false;
|
||||
this.segmentCounter = 0;
|
||||
this.transcriptBySegment = transcriptBySegment;
|
||||
}
|
||||
|
||||
async connect() {}
|
||||
|
||||
appendPcm16(buf) {
|
||||
this.appended.push(buf);
|
||||
}
|
||||
|
||||
commit() {
|
||||
this.commits += 1;
|
||||
const segmentId = `seg-${this.segmentCounter}`;
|
||||
this.segmentCounter += 1;
|
||||
this.emit('committed', { segmentId, previousSegmentId: null });
|
||||
setTimeout(() => {
|
||||
this.emit('transcript', {
|
||||
segmentId,
|
||||
transcript: this.transcriptBySegment(segmentId),
|
||||
isFinal: true,
|
||||
});
|
||||
}, 0);
|
||||
}
|
||||
|
||||
clear() {
|
||||
this.clears += 1;
|
||||
}
|
||||
|
||||
close() {
|
||||
this.closed = true;
|
||||
}
|
||||
}
|
||||
|
||||
function loudChunkBase64(samples = 1600, amplitude = 8000) {
|
||||
const arr = new Int16Array(samples);
|
||||
for (let i = 0; i < samples; i += 1) {
|
||||
arr[i] = i % 2 === 0 ? amplitude : -amplitude;
|
||||
}
|
||||
return Buffer.from(arr.buffer).toString('base64');
|
||||
}
|
||||
|
||||
function silentChunkBase64(samples = 1600) {
|
||||
return Buffer.from(new Int16Array(samples).buffer).toString('base64');
|
||||
}
|
||||
|
||||
function createManager(session) {
|
||||
const messages = [];
|
||||
const manager = new DictationStreamManager({
|
||||
emit: (msg) => messages.push(msg),
|
||||
createSttSession: async () => ({ session }),
|
||||
});
|
||||
return { manager, messages };
|
||||
}
|
||||
|
||||
function waitFor(predicate, timeoutMs = 1000) {
|
||||
return new Promise((resolve, reject) => {
|
||||
const startedAt = Date.now();
|
||||
const tick = () => {
|
||||
if (predicate()) {
|
||||
resolve(undefined);
|
||||
return;
|
||||
}
|
||||
if (Date.now() - startedAt > timeoutMs) {
|
||||
reject(new Error('waitFor timed out'));
|
||||
return;
|
||||
}
|
||||
setTimeout(tick, 5);
|
||||
};
|
||||
tick();
|
||||
});
|
||||
}
|
||||
|
||||
describe('DictationStreamManager', () => {
|
||||
it('transcribes ordered chunks and emits final text', async () => {
|
||||
const session = new FakeSttSession();
|
||||
const { manager, messages } = createManager(session);
|
||||
|
||||
await manager.handleStart('d1', FORMAT, {});
|
||||
manager.handleChunk({ dictationId: 'd1', seq: 0, audioBase64: loudChunkBase64() });
|
||||
manager.handleChunk({ dictationId: 'd1', seq: 1, audioBase64: loudChunkBase64() });
|
||||
manager.handleFinish('d1', 1);
|
||||
|
||||
await waitFor(() => messages.some((m) => m.type === 'final'));
|
||||
|
||||
const final = messages.find((m) => m.type === 'final');
|
||||
expect(final.payload.text).toBe('hello world');
|
||||
expect(session.commits).toBe(1);
|
||||
expect(session.closed).toBe(true);
|
||||
|
||||
const acks = messages.filter((m) => m.type === 'ack');
|
||||
expect(acks[acks.length - 1].payload.ackSeq).toBe(1);
|
||||
});
|
||||
|
||||
it('reorders out-of-order chunks before appending', async () => {
|
||||
const session = new FakeSttSession();
|
||||
const { manager, messages } = createManager(session);
|
||||
|
||||
await manager.handleStart('d1', FORMAT, {});
|
||||
manager.handleChunk({ dictationId: 'd1', seq: 1, audioBase64: loudChunkBase64() });
|
||||
expect(session.appended.length).toBe(0);
|
||||
manager.handleChunk({ dictationId: 'd1', seq: 0, audioBase64: loudChunkBase64() });
|
||||
expect(session.appended.length).toBe(2);
|
||||
manager.handleFinish('d1', 1);
|
||||
|
||||
await waitFor(() => messages.some((m) => m.type === 'final'));
|
||||
});
|
||||
|
||||
it('clears silence-only tails instead of committing', async () => {
|
||||
const session = new FakeSttSession();
|
||||
const { manager, messages } = createManager(session);
|
||||
|
||||
await manager.handleStart('d1', FORMAT, {});
|
||||
manager.handleChunk({ dictationId: 'd1', seq: 0, audioBase64: silentChunkBase64() });
|
||||
manager.handleFinish('d1', 0);
|
||||
|
||||
await waitFor(() => messages.some((m) => m.type === 'final'));
|
||||
|
||||
const final = messages.find((m) => m.type === 'final');
|
||||
expect(final.payload.text).toBe('');
|
||||
expect(session.commits).toBe(0);
|
||||
expect(session.clears).toBe(1);
|
||||
});
|
||||
|
||||
it('fails fast when finish arrives with no chunks', async () => {
|
||||
const session = new FakeSttSession();
|
||||
const { manager, messages } = createManager(session);
|
||||
|
||||
await manager.handleStart('d1', FORMAT, {});
|
||||
manager.handleFinish('d1', 3);
|
||||
|
||||
const error = messages.find((m) => m.type === 'error');
|
||||
expect(error).toBeDefined();
|
||||
expect(error.payload.retryable).toBe(true);
|
||||
expect(session.closed).toBe(true);
|
||||
});
|
||||
|
||||
it('reports provider readiness errors from createSttSession', async () => {
|
||||
const messages = [];
|
||||
const manager = new DictationStreamManager({
|
||||
emit: (msg) => messages.push(msg),
|
||||
createSttSession: async () => ({
|
||||
error: 'Dictation model is downloading',
|
||||
retryable: true,
|
||||
reasonCode: 'model_download_in_progress',
|
||||
}),
|
||||
});
|
||||
|
||||
await manager.handleStart('d1', FORMAT, {});
|
||||
const error = messages.find((m) => m.type === 'error');
|
||||
expect(error.payload.reasonCode).toBe('model_download_in_progress');
|
||||
expect(error.payload.retryable).toBe(true);
|
||||
});
|
||||
|
||||
it('emits partials as segment transcripts arrive', async () => {
|
||||
let segment = 0;
|
||||
const session = new FakeSttSession({
|
||||
transcriptBySegment: () => {
|
||||
segment += 1;
|
||||
return segment === 1 ? 'first part' : 'second part';
|
||||
},
|
||||
});
|
||||
const { manager, messages } = createManager(session);
|
||||
// Force auto-commit after ~0.05s of audio so two segments form.
|
||||
manager.autoCommitSeconds = 0.05;
|
||||
|
||||
await manager.handleStart('d1', FORMAT, {});
|
||||
manager.handleChunk({ dictationId: 'd1', seq: 0, audioBase64: loudChunkBase64(1600) });
|
||||
await waitFor(() => session.commits >= 1);
|
||||
manager.handleChunk({ dictationId: 'd1', seq: 1, audioBase64: loudChunkBase64(1600) });
|
||||
manager.handleFinish('d1', 1);
|
||||
|
||||
await waitFor(() => messages.some((m) => m.type === 'final'));
|
||||
|
||||
const final = messages.find((m) => m.type === 'final');
|
||||
expect(final.payload.text).toBe('first part second part');
|
||||
const partials = messages.filter((m) => m.type === 'partial');
|
||||
expect(partials.length).toBeGreaterThan(0);
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user