feat(voice): first-class voice input and local TTS across web, desktop, and mobile (#2018)
Complete rebuild of voice input on a server-authoritative streaming architecture, replacing the legacy Web Speech / whole-blob / WASM engines and the dead voice-agent layer (~4k lines removed). Speech-to-text (dictation): - Client streams 16 kHz mono PCM16 chunks over /api/dictation/ws with seq/ack ordering; buffered audio is retained and replayed on reconnect - Server transcribes and streams live partial transcripts back; segments auto-commit every ~15s with silence suppression and adaptive finalization timeouts - Local provider (default, zero config): sherpa-onnx models in a forked worker process — auto-download with progress, staged extraction with verification, corrupt-model auto-recovery, idle shutdown after 5 min - Model catalog with settings picker (accuracy/speed ratings, sizes, download/delete): Parakeet TDT v2 (English) and v3 (25 European languages, auto-detected), Whisper base and tiny (multilingual, light) - OpenAI-compatible provider for any Whisper endpoint - Composer overlay with live transcript, volume meter, timer, and cancel / insert / insert-and-send actions; failed transcriptions keep their audio for retry or accepting the partial text as-is - Configurable keyboard shortcut (default mod+alt+v) toggles dictation; Enter confirms and Escape cancels while recording - Overlay is pixel-aligned with the composer (measured footer height, matching paddings/typography/gaps) — no layout shift when toggling Text-to-speech: - Local Kokoro provider (English, 11 voices) synthesized in the same worker via /api/dictation/tts/speak, managed by the shared model pipeline; sentence-pipelined playback keeps time-to-first-audio at ~1 sentence regardless of message length, and stop cancels in-flight synthesis - Sanitizer keeps inline-code content (strips backticks only), reads interword slashes aloud, and removes only absolute file paths Settings: - Voice page unified: a single read-aloud toggle owns all playback options (the confusing "Enable Voice Mode" is gone); a new "Enable voice input" toggle (default on, persisted to settings.json) hides the composer mic entirely when disabled Mobile and transport: - iOS/Android microphone permissions added (dictation was previously impossible on mobile) - Fixed Android WebSocket upgrades: the Capacitor WebView origin (https://localhost) was missing from the packaged-client allowlist, 403-ing every WS connection — root cause of the old mobile SSE lock, which is now removed for all transports Security and conventions: - All HTTP routes sit behind the global /api auth gate; the WS upgrade explicitly validates the UI session and origin, with oc_url_token narrowly allowlisted and covered by tests; the dictation socket mints a fresh URL token before connecting - Routes register before the generic OpenCode proxy; the client goes through runtimeFetch/getRuntimeUrlResolver, and runtime switches reset the dictation socket - VS Code deliberately reports dictation as unavailable (no server process in that runtime) CI: workflow Node bumped 20 -> 22 to match the repo engines and fix better-sqlite3 installs broken by node-gyp@latest on Node 20. New dependency: sherpa-onnx-node (prebuilt N-API; macOS/Linux x64+arm64, Windows x64 — Windows-on-ARM falls back to the OpenAI-compatible provider)
This commit is contained in:
committed by
GitHub
parent
3f5151d424
commit
de1b85ac56
@@ -0,0 +1,352 @@
|
||||
/**
|
||||
* Client for the dictation local-speech worker process.
|
||||
*
|
||||
* Lazily forks the worker on first use, correlates request/response messages
|
||||
* by requestId, routes session events to per-session EventEmitters, and
|
||||
* shuts the worker down after an idle TTL so the ONNX runtime does not sit
|
||||
* in memory while dictation is unused.
|
||||
*/
|
||||
|
||||
import { fork } from 'child_process';
|
||||
import { randomUUID } from 'crypto';
|
||||
import { EventEmitter } from 'events';
|
||||
import { fileURLToPath } from 'url';
|
||||
|
||||
import { applySherpaLoaderEnv } from './sherpa-loader.js';
|
||||
|
||||
const DEFAULT_REQUEST_TIMEOUT_MS = 30000;
|
||||
const DEFAULT_IDLE_TTL_MS = 5 * 60 * 1000;
|
||||
const DEFAULT_LOCAL_SAMPLE_RATE = 16000;
|
||||
const STDERR_TAIL_MAX_CHARS = 2000;
|
||||
|
||||
function forkDictationWorker() {
|
||||
const env = { ...process.env };
|
||||
applySherpaLoaderEnv(env);
|
||||
return fork(fileURLToPath(new URL('./worker-process.js', import.meta.url)), [], {
|
||||
env,
|
||||
serialization: 'advanced',
|
||||
stdio: ['ignore', 'ignore', 'pipe', 'ipc'],
|
||||
windowsHide: true,
|
||||
});
|
||||
}
|
||||
|
||||
export class DictationWorkerClient {
|
||||
/**
|
||||
* @param {{ requestTimeoutMs?: number, idleTtlMs?: number }} [options]
|
||||
*/
|
||||
constructor(options = {}) {
|
||||
this.requestTimeoutMs = options.requestTimeoutMs ?? DEFAULT_REQUEST_TIMEOUT_MS;
|
||||
this.idleTtlMs = options.idleTtlMs ?? DEFAULT_IDLE_TTL_MS;
|
||||
this.pendingRequests = new Map();
|
||||
this.sessionEmitters = new Map();
|
||||
this.worker = null;
|
||||
this.stderrTail = '';
|
||||
this.inFlightRequests = 0;
|
||||
this.idleTimer = null;
|
||||
this.intentionalCloses = new WeakSet();
|
||||
}
|
||||
|
||||
/**
|
||||
* Synthesize speech in the worker. Returns WAV bytes.
|
||||
* @param {{ modelsDir: string, modelId: string, text: string, speakerId?: number, speed?: number }} params
|
||||
* @returns {Promise<{ audio: Buffer, format: string }>}
|
||||
*/
|
||||
async synthesizeSpeech(params) {
|
||||
// Long texts on slow hardware can exceed the default request timeout.
|
||||
const result = await this.sendRequest(
|
||||
{ type: 'tts.synthesize', ...params },
|
||||
{ timeoutMs: 120000 },
|
||||
);
|
||||
return {
|
||||
audio: Buffer.isBuffer(result.audio) ? result.audio : Buffer.from(result.audio),
|
||||
format: result.format || 'audio/wav',
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Create a streaming STT session in the worker.
|
||||
* @param {{ modelsDir: string, modelId: string }} params
|
||||
* @param {EventEmitter} emitter receives 'committed' | 'transcript' | 'error'
|
||||
* @returns {Promise<{ sessionId: string, requiredSampleRate: number }>}
|
||||
*/
|
||||
async createSession({ modelsDir, modelId }, emitter) {
|
||||
const sessionId = randomUUID();
|
||||
this.sessionEmitters.set(sessionId, emitter);
|
||||
try {
|
||||
const result = await this.sendRequest({
|
||||
type: 'session.create',
|
||||
sessionId,
|
||||
modelsDir,
|
||||
modelId,
|
||||
});
|
||||
return { sessionId, requiredSampleRate: result?.requiredSampleRate ?? DEFAULT_LOCAL_SAMPLE_RATE };
|
||||
} catch (err) {
|
||||
this.sessionEmitters.delete(sessionId);
|
||||
this.scheduleIdleShutdownIfReady();
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
appendSessionAudio(sessionId, audio) {
|
||||
void this.sendRequest({ type: 'session.append', sessionId, audio }).catch((err) => {
|
||||
this.emitSessionError(sessionId, err);
|
||||
});
|
||||
}
|
||||
|
||||
commitSession(sessionId) {
|
||||
void this.sendRequest({ type: 'session.commit', sessionId }).catch((err) => {
|
||||
this.emitSessionError(sessionId, err);
|
||||
});
|
||||
}
|
||||
|
||||
clearSession(sessionId) {
|
||||
void this.sendRequest({ type: 'session.clear', sessionId }).catch((err) => {
|
||||
this.emitSessionError(sessionId, err);
|
||||
});
|
||||
}
|
||||
|
||||
closeSession(sessionId) {
|
||||
this.sessionEmitters.delete(sessionId);
|
||||
void this.sendRequest({ type: 'session.close', sessionId }).catch(() => {
|
||||
// Closing is best-effort; the parent already dropped the session.
|
||||
});
|
||||
this.scheduleIdleShutdownIfReady();
|
||||
}
|
||||
|
||||
shutdown() {
|
||||
this.clearIdleTimer();
|
||||
this.rejectAllPending(new Error('Dictation worker shut down'));
|
||||
this.sessionEmitters.clear();
|
||||
const worker = this.worker;
|
||||
this.worker = null;
|
||||
if (worker && !worker.killed) {
|
||||
this.intentionalCloses.add(worker);
|
||||
try {
|
||||
worker.disconnect();
|
||||
} catch {
|
||||
// ignore
|
||||
}
|
||||
try {
|
||||
worker.kill();
|
||||
} catch {
|
||||
// ignore
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
sendRequest(input, options = {}) {
|
||||
const worker = this.ensureWorker();
|
||||
const requestId = randomUUID();
|
||||
const message = { ...input, requestId };
|
||||
this.inFlightRequests += 1;
|
||||
this.clearIdleTimer();
|
||||
|
||||
return new Promise((resolve, reject) => {
|
||||
const timeout = setTimeout(() => {
|
||||
this.pendingRequests.delete(requestId);
|
||||
this.inFlightRequests = Math.max(0, this.inFlightRequests - 1);
|
||||
this.scheduleIdleShutdownIfReady();
|
||||
reject(new Error(`Dictation worker request timed out: ${input.type}`));
|
||||
}, options.timeoutMs ?? this.requestTimeoutMs);
|
||||
|
||||
this.pendingRequests.set(requestId, { resolve, reject, timeout });
|
||||
|
||||
worker.send(message, (error) => {
|
||||
if (!error) {
|
||||
return;
|
||||
}
|
||||
const pending = this.pendingRequests.get(requestId);
|
||||
if (!pending) {
|
||||
return;
|
||||
}
|
||||
clearTimeout(pending.timeout);
|
||||
this.pendingRequests.delete(requestId);
|
||||
this.inFlightRequests = Math.max(0, this.inFlightRequests - 1);
|
||||
this.scheduleIdleShutdownIfReady();
|
||||
pending.reject(error);
|
||||
});
|
||||
});
|
||||
}
|
||||
|
||||
ensureWorker() {
|
||||
if (this.worker && !this.worker.killed && this.worker.connected) {
|
||||
return this.worker;
|
||||
}
|
||||
const worker = forkDictationWorker();
|
||||
this.worker = worker;
|
||||
this.stderrTail = '';
|
||||
worker.stderr?.on('data', (chunk) => {
|
||||
const text = Buffer.isBuffer(chunk) ? chunk.toString('utf8') : String(chunk);
|
||||
this.stderrTail = (this.stderrTail + text).slice(-STDERR_TAIL_MAX_CHARS);
|
||||
});
|
||||
worker.on('message', (message) => this.handleWorkerMessage(message));
|
||||
worker.on('close', (code, signal) => this.handleWorkerExit(worker, code, signal));
|
||||
return worker;
|
||||
}
|
||||
|
||||
handleWorkerMessage(message) {
|
||||
if (message?.type === 'response') {
|
||||
const pending = this.pendingRequests.get(message.requestId);
|
||||
if (!pending) {
|
||||
return;
|
||||
}
|
||||
clearTimeout(pending.timeout);
|
||||
this.pendingRequests.delete(message.requestId);
|
||||
this.inFlightRequests = Math.max(0, this.inFlightRequests - 1);
|
||||
this.scheduleIdleShutdownIfReady();
|
||||
if (message.ok) {
|
||||
pending.resolve(message.result);
|
||||
} else {
|
||||
pending.reject(new Error(message.error || 'Dictation worker request failed'));
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
const emitter = this.sessionEmitters.get(message?.sessionId);
|
||||
if (!emitter) {
|
||||
return;
|
||||
}
|
||||
switch (message.type) {
|
||||
case 'session.committed':
|
||||
emitter.emit('committed', message.payload);
|
||||
return;
|
||||
case 'session.transcript':
|
||||
emitter.emit('transcript', message.payload);
|
||||
return;
|
||||
case 'session.error':
|
||||
emitter.emit('error', new Error(message.error));
|
||||
return;
|
||||
default:
|
||||
}
|
||||
}
|
||||
|
||||
handleWorkerExit(worker, code, signal) {
|
||||
const wasCurrentWorker = this.worker === worker;
|
||||
const wasIntentional = this.intentionalCloses.has(worker);
|
||||
this.intentionalCloses.delete(worker);
|
||||
if (!wasCurrentWorker || wasIntentional) {
|
||||
if (wasCurrentWorker) {
|
||||
this.worker = null;
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
const stderr = this.stderrTail.trim();
|
||||
const error = new Error(
|
||||
`Dictation worker exited (code ${code ?? 'null'}${signal ? `, signal ${signal}` : ''}).` +
|
||||
(stderr ? ` Last stderr: ${stderr.slice(-500)}` : ''),
|
||||
);
|
||||
|
||||
this.worker = null;
|
||||
this.clearIdleTimer();
|
||||
this.rejectAllPending(error);
|
||||
for (const emitter of this.sessionEmitters.values()) {
|
||||
if (emitter.listenerCount('error') > 0) {
|
||||
emitter.emit('error', error);
|
||||
}
|
||||
}
|
||||
this.sessionEmitters.clear();
|
||||
this.inFlightRequests = 0;
|
||||
}
|
||||
|
||||
rejectAllPending(error) {
|
||||
for (const [requestId, pending] of this.pendingRequests) {
|
||||
clearTimeout(pending.timeout);
|
||||
pending.reject(error);
|
||||
this.pendingRequests.delete(requestId);
|
||||
}
|
||||
}
|
||||
|
||||
emitSessionError(sessionId, error) {
|
||||
const emitter = this.sessionEmitters.get(sessionId);
|
||||
if (emitter && emitter.listenerCount('error') > 0) {
|
||||
emitter.emit('error', error instanceof Error ? error : new Error(String(error)));
|
||||
}
|
||||
}
|
||||
|
||||
scheduleIdleShutdownIfReady() {
|
||||
if (!this.worker || this.inFlightRequests > 0 || this.sessionEmitters.size > 0) {
|
||||
return;
|
||||
}
|
||||
this.clearIdleTimer();
|
||||
this.idleTimer = setTimeout(() => {
|
||||
if (this.inFlightRequests === 0 && this.sessionEmitters.size === 0) {
|
||||
this.shutdown();
|
||||
}
|
||||
}, this.idleTtlMs);
|
||||
}
|
||||
|
||||
clearIdleTimer() {
|
||||
if (this.idleTimer) {
|
||||
clearTimeout(this.idleTimer);
|
||||
this.idleTimer = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* StreamingTranscriptionSession backed by the worker process.
|
||||
* Matches the session contract consumed by DictationStreamManager.
|
||||
*/
|
||||
export class WorkerBackedTranscriptionSession extends EventEmitter {
|
||||
/**
|
||||
* @param {DictationWorkerClient} client
|
||||
* @param {{ modelsDir: string, modelId: string }} modelConfig
|
||||
*/
|
||||
constructor(client, modelConfig) {
|
||||
super();
|
||||
this.client = client;
|
||||
this.modelConfig = modelConfig;
|
||||
this.requiredSampleRate = DEFAULT_LOCAL_SAMPLE_RATE;
|
||||
this.connectedSessionId = null;
|
||||
this.connecting = null;
|
||||
}
|
||||
|
||||
async connect() {
|
||||
if (this.connectedSessionId) {
|
||||
return;
|
||||
}
|
||||
if (!this.connecting) {
|
||||
this.connecting = (async () => {
|
||||
try {
|
||||
const result = await this.client.createSession(this.modelConfig, this);
|
||||
this.connectedSessionId = result.sessionId;
|
||||
this.requiredSampleRate = result.requiredSampleRate;
|
||||
} finally {
|
||||
this.connecting = null;
|
||||
}
|
||||
})();
|
||||
}
|
||||
await this.connecting;
|
||||
}
|
||||
|
||||
appendPcm16(pcm16le) {
|
||||
if (!this.connectedSessionId) {
|
||||
this.emit('error', new Error('Local STT session not connected'));
|
||||
return;
|
||||
}
|
||||
this.client.appendSessionAudio(this.connectedSessionId, pcm16le);
|
||||
}
|
||||
|
||||
commit() {
|
||||
if (!this.connectedSessionId) {
|
||||
this.emit('error', new Error('Local STT session not connected'));
|
||||
return;
|
||||
}
|
||||
this.client.commitSession(this.connectedSessionId);
|
||||
}
|
||||
|
||||
clear() {
|
||||
if (this.connectedSessionId) {
|
||||
this.client.clearSession(this.connectedSessionId);
|
||||
}
|
||||
}
|
||||
|
||||
close() {
|
||||
const sessionId = this.connectedSessionId;
|
||||
this.connectedSessionId = null;
|
||||
if (sessionId) {
|
||||
this.client.closeSession(sessionId);
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user