feat(voice): first-class voice input and local TTS across web, desktop, and mobile (#2018)

Complete rebuild of voice input on a server-authoritative streaming
architecture, replacing the legacy Web Speech / whole-blob / WASM engines
and the dead voice-agent layer (~4k lines removed).

Speech-to-text (dictation):
- Client streams 16 kHz mono PCM16 chunks over /api/dictation/ws with
  seq/ack ordering; buffered audio is retained and replayed on reconnect
- Server transcribes and streams live partial transcripts back;
  segments auto-commit every ~15s with silence suppression and adaptive
  finalization timeouts
- Local provider (default, zero config): sherpa-onnx models in a forked
  worker process — auto-download with progress, staged extraction with
  verification, corrupt-model auto-recovery, idle shutdown after 5 min
- Model catalog with settings picker (accuracy/speed ratings, sizes,
  download/delete): Parakeet TDT v2 (English) and v3 (25 European
  languages, auto-detected), Whisper base and tiny (multilingual, light)
- OpenAI-compatible provider for any Whisper endpoint
- Composer overlay with live transcript, volume meter, timer, and
  cancel / insert / insert-and-send actions; failed transcriptions keep
  their audio for retry or accepting the partial text as-is
- Configurable keyboard shortcut (default mod+alt+v) toggles dictation;
  Enter confirms and Escape cancels while recording
- Overlay is pixel-aligned with the composer (measured footer height,
  matching paddings/typography/gaps) — no layout shift when toggling

Text-to-speech:
- Local Kokoro provider (English, 11 voices) synthesized in the same
  worker via /api/dictation/tts/speak, managed by the shared model
  pipeline; sentence-pipelined playback keeps time-to-first-audio at
  ~1 sentence regardless of message length, and stop cancels in-flight
  synthesis
- Sanitizer keeps inline-code content (strips backticks only), reads
  interword slashes aloud, and removes only absolute file paths

Settings:
- Voice page unified: a single read-aloud toggle owns all playback
  options (the confusing "Enable Voice Mode" is gone); a new "Enable
  voice input" toggle (default on, persisted to settings.json) hides
  the composer mic entirely when disabled

Mobile and transport:
- iOS/Android microphone permissions added (dictation was previously
  impossible on mobile)
- Fixed Android WebSocket upgrades: the Capacitor WebView origin
  (https://localhost) was missing from the packaged-client allowlist,
  403-ing every WS connection — root cause of the old mobile SSE lock,
  which is now removed for all transports

Security and conventions:
- All HTTP routes sit behind the global /api auth gate; the WS upgrade
  explicitly validates the UI session and origin, with oc_url_token
  narrowly allowlisted and covered by tests; the dictation socket mints
  a fresh URL token before connecting
- Routes register before the generic OpenCode proxy; the client goes
  through runtimeFetch/getRuntimeUrlResolver, and runtime switches
  reset the dictation socket
- VS Code deliberately reports dictation as unavailable (no server
  process in that runtime)

CI: workflow Node bumped 20 -> 22 to match the repo engines and fix
better-sqlite3 installs broken by node-gyp@latest on Node 20.

New dependency: sherpa-onnx-node (prebuilt N-API; macOS/Linux x64+arm64,
Windows x64 — Windows-on-ARM falls back to the OpenAI-compatible provider)
This commit is contained in:
Bohdan Triapitsyn
2026-07-04 02:48:07 +03:00
committed by GitHub
parent 3f5151d424
commit de1b85ac56
89 changed files with 8740 additions and 6061 deletions
@@ -1,397 +0,0 @@
/**
* Audio Stream Service
*
* Captures microphone audio using MediaRecorder, detects utterance boundaries
* via an AnalyserNode-based silence detector (VAD), then POSTs each utterance
* as a raw audio blob to the OpenChamber server's /api/stt/transcribe endpoint.
*
* Mimics the BrowserVoiceService.startListening interface so useBrowserVoice
* can swap providers without changing its internal logic.
*
* @example
* ```typescript
* audioStreamService.configure({ baseURL: 'http://localhost:8001/v1', model: 'whisper-1' });
* audioStreamService.startListening('en', (text, isFinal) => {
* if (isFinal) console.log('transcript:', text);
* });
* audioStreamService.stopListening();
* ```
*/
import { runtimeFetch } from '@/lib/runtime-fetch';
type SpeechResultCallback = (text: string, isFinal: boolean) => void;
type ErrorCallback = (error: string) => void;
interface AudioStreamConfig {
/** Base URL of the OpenAI-compatible STT server (e.g. http://localhost:8001/v1) */
baseURL: string;
/** Whisper-compatible model name */
model: string;
/** Optional BCP-47 language hint (e.g. 'en'). Empty string = auto-detect. */
language?: string;
/**
* Silence threshold in dB below which audio is considered silence.
* Lower (more negative) = only very quiet audio counts as silence.
* Default: -45
*/
silenceThresholdDb?: number;
/**
* How long continuous silence must last (ms) before the utterance is finalised.
* Default: 1500
*/
silenceHoldMs?: number;
/** Optional API key for the STT server. */
apiKey?: string;
}
// How often (ms) the VAD samples the analyser
const VAD_POLL_MS = 80;
// Minimum audio duration (ms) to bother uploading (avoids blank clips)
const MIN_UTTERANCE_MS = 300;
class AudioStreamService {
private stream: MediaStream | null = null;
private mediaRecorder: MediaRecorder | null = null;
private audioContext: AudioContext | null = null;
private analyser: AnalyserNode | null = null;
private vadTimer: ReturnType<typeof setInterval> | null = null;
private chunks: Blob[] = [];
private recordingStartMs = 0;
private isActive = false;
private isSpeaking = false;
private silenceSince: number | null = null;
private onResult: SpeechResultCallback | null = null;
private onError: ErrorCallback | null = null;
private finishResolver: (() => void) | null = null;
private lang = 'en';
// Configurable parameters
private cfg: Required<AudioStreamConfig> = {
baseURL: '',
model: 'deepdml/faster-whisper-large-v3-turbo-ct2',
language: '',
silenceThresholdDb: -45,
silenceHoldMs: 1500,
apiKey: '',
};
/** Update service configuration. Can be called before or after startListening. */
configure(config: AudioStreamConfig): void {
this.cfg = {
silenceThresholdDb: -45,
silenceHoldMs: 1500,
language: '',
apiKey: '',
...config,
};
this.cfg.apiKey = config.apiKey ?? '';
}
/** Whether the browser supports the required APIs. */
isSupported(): boolean {
return (
typeof window !== 'undefined' &&
typeof navigator !== 'undefined' &&
typeof navigator.mediaDevices?.getUserMedia === 'function' &&
typeof window.MediaRecorder !== 'undefined' &&
typeof window.AudioContext !== 'undefined'
);
}
/**
* Start listening. Requests microphone access if not already held.
* Calls onResult(text, true) for each completed utterance.
*/
async startListening(
lang: string,
onResult: SpeechResultCallback,
onError?: ErrorCallback
): Promise<void> {
if (this.isActive) {
this.stopListening();
}
this.lang = lang;
this.onResult = onResult;
this.onError = onError ?? null;
this.isActive = true;
try {
this.stream = await navigator.mediaDevices.getUserMedia({ audio: true, video: false });
} catch (err) {
this.isActive = false;
const msg = err instanceof Error ? err.message : 'Microphone access denied';
onError?.(msg);
return;
}
this._setupAudioContext();
this._startRecorder();
this._startVAD();
}
/** Stop listening and clean up all resources. */
stopListening(): void {
this._stopVAD();
this._stopRecorder();
this._cleanupAfterStop(true);
}
async finishListening(): Promise<void> {
if (!this.isActive) return;
this._stopVAD();
this.isSpeaking = false;
this.silenceSince = null;
if (!this.mediaRecorder || this.mediaRecorder.state === 'inactive') {
this._cleanupAfterStop(true);
return;
}
await new Promise<void>((resolve) => {
this.finishResolver = resolve;
this._finaliseUtterance(false);
});
this._cleanupAfterStop(true);
}
/** Whether currently listening. */
getIsListening(): boolean {
return this.isActive;
}
// ── Private helpers ──────────────────────────────────────────────────────
private _setupAudioContext(): void {
if (!this.stream) return;
const AudioContextClass = window.AudioContext ?? (window as unknown as { webkitAudioContext: typeof AudioContext }).webkitAudioContext;
this.audioContext = new AudioContextClass();
const source = this.audioContext.createMediaStreamSource(this.stream);
this.analyser = this.audioContext.createAnalyser();
this.analyser.fftSize = 512;
source.connect(this.analyser);
}
private _teardownAudioContext(): void {
try {
this.audioContext?.close();
} catch {
// ignore
}
this.audioContext = null;
this.analyser = null;
}
private _startRecorder(): void {
if (!this.stream) return;
const mimeType = this._pickMimeType();
const options: MediaRecorderOptions = {};
if (mimeType && MediaRecorder.isTypeSupported(mimeType)) {
options.mimeType = mimeType;
}
this.mediaRecorder = new MediaRecorder(this.stream, options);
this.chunks = [];
this.recordingStartMs = Date.now();
this.mediaRecorder.ondataavailable = (e) => {
if (e.data && e.data.size > 0) {
this.chunks.push(e.data);
}
};
this.mediaRecorder.onstop = () => {
const blobs = this.chunks.splice(0);
const durationMs = Date.now() - this.recordingStartMs;
if (blobs.length === 0 || durationMs < MIN_UTTERANCE_MS) {
this.finishResolver?.();
this.finishResolver = null;
return;
}
const mType = blobs[0].type || mimeType || 'audio/webm';
const blob = new Blob(blobs, { type: mType });
void this._upload(blob, mType).finally(() => {
this.finishResolver?.();
this.finishResolver = null;
});
};
// Collect data every 250 ms so we don't lose the tail on stop()
this.mediaRecorder.start(250);
}
private _stopRecorder(): void {
if (this.mediaRecorder && this.mediaRecorder.state !== 'inactive') {
try {
this.mediaRecorder.stop();
} catch {
this.finishResolver?.();
this.finishResolver = null;
// ignore
}
}
this.mediaRecorder = null;
}
private _releaseStream(): void {
if (this.stream) {
this.stream.getTracks().forEach((t) => t.stop());
this.stream = null;
}
}
private _startVAD(): void {
this._stopVAD();
this.silenceSince = null;
this.isSpeaking = false;
this.vadTimer = setInterval(() => {
if (!this.isActive || !this.analyser) return;
const db = this._getRmsDb();
const isSilent = db < this.cfg.silenceThresholdDb;
if (!isSilent) {
// Audio detected
this.silenceSince = null;
if (!this.isSpeaking) {
this.isSpeaking = true;
// Restart recorder to capture from the start of speech
if (this.mediaRecorder?.state === 'recording') {
this.recordingStartMs = Date.now();
}
}
} else {
// Silence detected
if (this.isSpeaking) {
if (this.silenceSince === null) {
this.silenceSince = Date.now();
} else if (Date.now() - this.silenceSince >= this.cfg.silenceHoldMs) {
// End of utterance — stop recorder (triggers onstop → upload)
this.isSpeaking = false;
this.silenceSince = null;
this._finaliseUtterance(true);
}
}
}
}, VAD_POLL_MS);
}
private _stopVAD(): void {
if (this.vadTimer !== null) {
clearInterval(this.vadTimer);
this.vadTimer = null;
}
}
private _cleanupAfterStop(clearChunks: boolean): void {
const pendingResolver = this.finishResolver;
this.isActive = false;
this.finishResolver = null;
this.mediaRecorder = null;
this._teardownAudioContext();
this._releaseStream();
if (clearChunks) {
this.chunks = [];
}
this.isSpeaking = false;
this.silenceSince = null;
this.onResult = null;
this.onError = null;
pendingResolver?.();
}
/** Stop the current recorder to flush the utterance, optionally restarting for the next one. */
private _finaliseUtterance(restart: boolean): void {
if (!this.isActive) return;
if (this.mediaRecorder && this.mediaRecorder.state === 'recording') {
this.mediaRecorder.stop();
}
if (!restart) return;
// Restart recorder for the next utterance after a short delay
// (MediaRecorder.onstop fires asynchronously; we wait for it to complete)
setTimeout(() => {
if (this.isActive && this.stream) {
this._startRecorder();
}
}, 100);
}
/** Compute RMS of current analyser frame in dBFS. */
private _getRmsDb(): number {
if (!this.analyser) return -Infinity;
const buf = new Float32Array(this.analyser.fftSize);
this.analyser.getFloatTimeDomainData(buf);
let sumSq = 0;
for (const s of buf) sumSq += s * s;
const rms = Math.sqrt(sumSq / buf.length);
return rms === 0 ? -Infinity : 20 * Math.log10(rms);
}
/** POST utterance blob to server, call onResult with transcript. */
private async _upload(blob: Blob, mimeType: string): Promise<void> {
if (!this.onResult) return;
try {
const headers: Record<string, string> = {
'Content-Type': mimeType,
'X-Base-URL': this.cfg.baseURL,
'X-Model': this.cfg.model,
};
if (this.cfg.apiKey) {
headers['Authorization'] = `Bearer ${this.cfg.apiKey}`;
}
if (this.cfg.language) {
headers['X-Language'] = this.cfg.language;
} else if (this.lang && this.lang !== 'auto') {
// Use BCP-47 base language code (e.g. 'en' from 'en-US')
const baseLang = this.lang.split('-')[0];
headers['X-Language'] = baseLang;
}
const response = await runtimeFetch('/api/stt/transcribe', {
method: 'POST',
headers,
body: blob,
});
if (!response.ok) {
const errData = await response.json().catch(() => ({ error: 'Unknown error' }));
throw new Error(errData.error ?? `HTTP ${response.status}`);
}
const data = await response.json();
const transcript: string = (data.transcript ?? '').trim();
if (transcript) {
this.onResult(transcript, true);
}
} catch (err) {
if (!this.isActive) return; // Stopped — ignore
const msg = err instanceof Error ? err.message : 'Transcription upload failed';
console.error('[AudioStreamService] Upload error:', msg);
this.onError?.(msg);
}
}
/** Pick the best supported MIME type for MediaRecorder. */
private _pickMimeType(): string {
const candidates = [
'audio/webm;codecs=opus',
'audio/webm',
'audio/ogg;codecs=opus',
'audio/ogg',
'audio/mp4',
];
if (typeof MediaRecorder !== 'undefined' && MediaRecorder.isTypeSupported) {
return candidates.find((t) => MediaRecorder.isTypeSupported(t)) ?? '';
}
return '';
}
}
export const audioStreamService = new AudioStreamService();
@@ -1,131 +0,0 @@
/**
* Context formatters for voice-native output
* Formats session events (messages, permissions, ready events) into natural language
* for the ElevenLabs voice agent to speak aloud.
*
* @example
* ```typescript
* import { formatMessage, formatPermissionRequest } from '@/lib/voice';
*
* const voiceText = formatMessage({ role: 'assistant', content: 'Hello!' });
* // Returns: "Claude Code: Hello!"
* ```
*/
import { VOICE_CONFIG } from "./voiceConfig";
/** Message type for voice formatting */
export interface VoiceMessage {
role: string;
content: string;
}
/**
* Format a single message for voice output
* - Assistant messages: Code blocks replaced with "[code block]", prefixed with "Claude Code: "
* - User messages: Prefixed with "User: "
* - Other roles: Returns null (not spoken)
*
* @param message - The message to format
* @returns Formatted text for voice, or null if should not be spoken
*/
function formatMessage(message: VoiceMessage): string | null {
// Handle edge cases
if (!message || typeof message.content !== "string") {
return null;
}
const content = message.content.trim();
if (!content) {
return null;
}
if (message.role === "assistant") {
// Replace code blocks with description (don't read code aloud)
const textOnly = content.replace(/```[\s\S]*?```/g, "[code block]");
return `Claude Code: ${textOnly}`;
}
if (message.role === "user") {
return `User: ${content}`;
}
// Skip system, tool, and other roles for voice
return null;
}
/**
* Format multiple new messages for voice output
* - Maps messages through formatMessage
* - Filters out nulls (unspoken roles)
* - Joins with newlines
*
* @param sessionId - The session ID (for future use/debugging)
* @param messages - Array of messages to format
* @returns Formatted text for voice, or null if no speakable messages
*/
export function formatNewMessages(
sessionId: string,
messages: VoiceMessage[]
): string | null {
// Handle edge cases
if (!Array.isArray(messages) || messages.length === 0) {
return null;
}
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
console.log(`[Voice] Formatting ${messages.length} messages for session ${sessionId}`);
}
// Format each message and filter out nulls
const formattedMessages = messages
.map(formatMessage)
.filter((msg): msg is string => msg !== null);
if (formattedMessages.length === 0) {
return null;
}
return formattedMessages.join("\n");
}
/**
* Format a permission request for voice announcement
* - Per CONTEXT.md: Only tool name, not arguments (LIMITED_TOOL_CALLS)
* - Prompts user to say "allow" or "deny"
*
* @param sessionId - The session ID
* @param requestId - The permission request ID
* @param toolName - Name of the tool requesting permission
* @param toolArgs - Tool arguments (not included in voice output per config)
* @returns Formatted permission request for voice
*/
export function formatPermissionRequest(
sessionId: string,
requestId: string,
toolName: string,
// eslint-disable-next-line @typescript-eslint/no-unused-vars
toolArgs: unknown
): string {
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
console.log(`[Voice] Formatting permission request ${requestId} for session ${sessionId}`);
}
// Per VOICE_CONFIG.LIMITED_TOOL_CALLS, we don't include toolArgs in voice output
return `Claude Code is requesting permission to use ${toolName}. Say "allow" or "deny".`;
}
/**
* Format a ready event for voice announcement
* - Indicates the AI has finished working and is ready for next instruction
*
* @param sessionId - The session ID
* @returns Formatted ready event for voice
*/
export function formatReadyEvent(sessionId: string): string {
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
console.log(`[Voice] Formatting ready event for session ${sessionId}`);
}
return "Claude Code finished working. Ready for next instruction.";
}
-17
View File
@@ -1,17 +0,0 @@
/**
* Voice module barrel export
* Provides a clean import path for voice session hooks.
*
* @example
* ```typescript
* import { voiceHooks } from '@/lib/voice';
* ```
*/
// Voice session registry (from voiceSession.ts)
export {
isVoiceSessionStarted,
} from "./voiceSession";
// Voice hooks for session-to-voice event routing (from voiceHooks.ts)
export { voiceHooks } from "./voiceHooks";
+10 -4
View File
@@ -6,15 +6,21 @@
export function sanitizeForTTS(text: string): string {
if (!text) return '';
return text
// Remove code blocks
// Remove fenced code blocks entirely (multi-line code is unreadable),
// but keep inline-code CONTENT and only strip the backticks: agents
// routinely inline meaningful words ("You are on `main`").
.replace(/```[\s\S]*?```/g, '')
.replace(/`[^`]*`/g, '')
.replace(/`([^`\n]*)`/g, '$1')
// Remove markdown formatting
.replace(/[*_~#]/g, '')
// Remove URLs
.replace(/https?:\/\/[^\s]+/g, '')
// Remove file paths
.replace(/\/[\w\-./]+/g, '')
// Remove absolute file paths (leading slash, one or more segments).
// Deliberately NOT matching interword slashes: "iOS/Android" and
// "origin/main" are speech, not paths.
.replace(/(^|\s)\/(?:[\w.-]+\/)*[\w.-]+/g, '$1')
// Read remaining interword slashes out loud ("iOS slash Android").
.replace(/([\w.])\/([\w.])/g, '$1 slash $2')
// Remove shell-like patterns
.replace(/^\s*[$#>]\s*/gm, '')
// Remove brackets and special chars
-32
View File
@@ -1,32 +0,0 @@
/**
* Static voice context configuration
* Controls voice behavior and feature flags for the ElevenLabs voice agent
*/
export const VOICE_CONFIG = {
/** Disable all tool call information from being sent to voice context */
DISABLE_TOOL_CALLS: false,
/** Send only tool names and descriptions, exclude arguments */
LIMITED_TOOL_CALLS: true,
/** Disable permission request forwarding */
DISABLE_PERMISSION_REQUESTS: false,
/** Disable session online/offline notifications */
DISABLE_SESSION_STATUS: true,
/** Disable message forwarding */
DISABLE_MESSAGES: false,
/** Disable session focus notifications */
DISABLE_SESSION_FOCUS: false,
/** Disable ready event notifications */
DISABLE_READY_EVENTS: false,
/** Maximum number of messages to include in session history */
MAX_HISTORY_MESSAGES: 50,
/** Enable debug logging for voice context updates */
ENABLE_DEBUG_LOGGING: true,
} as const;
-125
View File
@@ -1,125 +0,0 @@
/**
* Voice hooks for session-to-voice event routing
* Routes session events (messages, permissions, ready events) to the ElevenLabs
* voice agent via contextual updates.
*
* This module provides hooks that can be called when session events occur,
* using the voice session registry from voiceSession.ts.
*
* @example
* ```typescript
* import { voiceHooks } from '@/lib/voice';
*
* // Route session messages to voice
* voiceHooks.onMessages(sessionId, messages);
* ```
*/
import { VOICE_CONFIG } from "./voiceConfig";
import {
formatNewMessages,
formatPermissionRequest,
formatReadyEvent,
type VoiceMessage,
} from "./contextFormatters";
import { getVoiceSession, isVoiceSessionStarted } from "./voiceSession";
/**
* Report a contextual update to the voice session
* Internal helper that checks preconditions and handles errors
*
* @param update - The text update to send, or null/undefined to skip
*/
function reportContextualUpdate(update: string | null | undefined): void {
// Skip empty/null/undefined updates
if (!update || update.trim().length === 0) {
return;
}
// Skip if no voice session or not started
const voiceSession = getVoiceSession();
if (!voiceSession || !isVoiceSessionStarted()) {
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
console.log("[Voice] Skipping contextual update - no active session");
}
return;
}
try {
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
console.log("[Voice] Sending contextual update:", update.substring(0, 100));
}
voiceSession.sendContextualUpdate(update);
} catch (error) {
// Log error but don't throw - voice updates shouldn't break the app
console.error("[Voice] Failed to send contextual update:", error);
}
}
/**
* Voice hooks - exported functions to route session events to voice
*
* These hooks should be called when corresponding session events occur.
* They respect VOICE_CONFIG feature flags to enable/disable specific
* event types.
*/
export const voiceHooks = {
/**
* Called when new messages arrive in the session
* Formats and sends messages to voice agent (if not disabled)
*
* @param sessionId - The session ID
* @param messages - Array of messages to format and send
*/
onMessages(sessionId: string, messages: VoiceMessage[]): void {
if (VOICE_CONFIG.DISABLE_MESSAGES) {
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
console.log("[Voice] Message forwarding disabled");
}
return;
}
reportContextualUpdate(formatNewMessages(sessionId, messages));
},
/**
* Called when a permission request is made
* Announces the permission request to the voice agent (if not disabled)
*
* @param sessionId - The session ID
* @param requestId - The permission request ID
* @param toolName - Name of the tool requesting permission
* @param toolArgs - Arguments for the tool (not sent to voice per LIMITED_TOOL_CALLS)
*/
onPermissionRequested(
sessionId: string,
requestId: string,
toolName: string,
toolArgs: unknown
): void {
if (VOICE_CONFIG.DISABLE_PERMISSION_REQUESTS) {
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
console.log("[Voice] Permission request forwarding disabled");
}
return;
}
reportContextualUpdate(
formatPermissionRequest(sessionId, requestId, toolName, toolArgs)
);
},
/**
* Called when the AI is ready for the next instruction
* Announces ready state to the voice agent (if not disabled)
*
* @param sessionId - The session ID
*/
onReady(sessionId: string): void {
if (VOICE_CONFIG.DISABLE_READY_EVENTS) {
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
console.log("[Voice] Ready event forwarding disabled");
}
return;
}
reportContextualUpdate(formatReadyEvent(sessionId));
},
};
-28
View File
@@ -1,28 +0,0 @@
/**
* Voice session interface
* Used for type safety without importing ReturnType from SDK
*/
interface VoiceSession {
sendContextualUpdate: (text: string) => void;
}
/**
* Global storage for the active voice session.
* Used by voiceHooks to send contextual updates to the voice agent.
*/
const activeVoiceSession: VoiceSession | null = null;
/**
* Get the currently registered voice session.
* Used by voiceHooks to send contextual updates.
*/
export function getVoiceSession(): VoiceSession | null {
return activeVoiceSession;
}
/**
* Check if a voice session is currently active.
*/
export function isVoiceSessionStarted(): boolean {
return activeVoiceSession !== null;
}
-555
View File
@@ -1,555 +0,0 @@
/**
* WASM Speech-to-Text Service
*
* Local Whisper transcription via Transformers.js (ONNX Runtime Web).
* Captures microphone audio, detects utterance boundaries via silence-based
* VAD, then transcribes each utterance locally — no cloud API required.
*
* Works in Electron and all modern browsers that support Web Audio API.
* First use downloads a Whisper model (~40–166 MB, cached).
*/
export type WasmModelStatus =
| { state: 'unloaded' }
| { state: 'downloading'; progress: number }
| { state: 'loading' }
| { state: 'ready' }
| { state: 'error'; error: string };
export interface WasmModelInfo {
id: string;
name: string;
size: string;
languages: string;
description: string;
}
export const WASM_MODELS: WasmModelInfo[] = [
{
id: 'Xenova/whisper-tiny.en',
name: 'Whisper Tiny (EN)',
size: '~39 MB',
languages: 'English',
description: 'Fastest, lowest accuracy. Good for quick dictation.',
},
{
id: 'Xenova/whisper-base.en',
name: 'Whisper Base (EN)',
size: '~73 MB',
languages: 'English',
description: 'Balanced speed and accuracy. Default for English.',
},
{
id: 'Xenova/whisper-small.en',
name: 'Whisper Small (EN)',
size: '~166 MB',
languages: 'English',
description: 'Higher accuracy, slower. Best for noisy environments.',
},
];
type SpeechResultCallback = (text: string, isFinal: boolean) => void;
type ErrorCallback = (error: string) => void;
const VAD_POLL_MS = 80;
const MIN_UTTERANCE_MS = 300;
const WHISPER_SAMPLE_RATE = 16000;
interface WasmSttConfig {
silenceThresholdDb?: number;
silenceHoldMs?: number;
}
class WasmSttService {
private transcriber: unknown = null;
private worker: Worker | null = null;
private modelStatus: WasmModelStatus = { state: 'unloaded' };
private currentModelId: string | null = null;
private stream: MediaStream | null = null;
private mediaRecorder: MediaRecorder | null = null;
private audioContext: AudioContext | null = null;
private analyser: AnalyserNode | null = null;
private vadTimer: ReturnType<typeof setInterval> | null = null;
private chunks: Blob[] = [];
private recordingStartMs = 0;
private isActive = false;
private isSpeaking = false;
private silenceSince: number | null = null;
private onResult: SpeechResultCallback | null = null;
private onError: ErrorCallback | null = null;
private finishResolver: (() => void) | null = null;
private lang = 'en';
private cfg: Required<WasmSttConfig> = {
silenceThresholdDb: -45,
silenceHoldMs: 1500,
};
public onModelStatusChange: ((status: WasmModelStatus) => void) | null = null;
configure(config: WasmSttConfig): void {
this.cfg = { ...this.cfg, ...config };
}
isSupported(): boolean {
return (
typeof window !== 'undefined' &&
typeof navigator !== 'undefined' &&
typeof navigator.mediaDevices?.getUserMedia === 'function' &&
typeof window.MediaRecorder !== 'undefined' &&
typeof window.AudioContext !== 'undefined'
);
}
getModelStatus(): WasmModelStatus {
return this.modelStatus;
}
getCurrentModelId(): string | null {
return this.currentModelId;
}
private setModelStatus(status: WasmModelStatus): void {
this.modelStatus = status;
this.onModelStatusChange?.(status);
}
async loadModel(modelId: string): Promise<void> {
if (this.currentModelId === modelId && this.modelStatus.state === 'ready') {
return;
}
if (this.modelStatus.state === 'downloading' || this.modelStatus.state === 'loading') {
return;
}
this._terminateWorker();
this.transcriber = null;
this.setModelStatus({ state: 'downloading', progress: 0 });
this.currentModelId = modelId;
// Try Web Worker first — inference off main thread = no UI freeze.
try {
const WasmWorkerMod = await import('./wasmSttWorker?worker');
const WasmWorker = WasmWorkerMod.default as new () => Worker;
this.worker = new WasmWorker();
await new Promise<void>((resolve, reject) => {
const timer = setTimeout(() => reject(new Error('Worker init timed out')), 10000);
this.worker!.onmessage = (e: MessageEvent) => {
const data = e.data as { type: string; progress?: number; error?: string; text?: string };
if (data.type === 'progress') {
this.setModelStatus({ state: 'downloading', progress: data.progress ?? 0 });
} else if (data.type === 'loaded') {
clearTimeout(timer);
resolve();
} else if (data.type === 'error') {
clearTimeout(timer);
reject(new Error(data.error ?? 'Worker load failed'));
}
};
this.worker!.onerror = (err) => {
clearTimeout(timer);
reject(new Error(err.message || 'Worker error'));
};
this.worker!.postMessage({ type: 'load', modelId });
});
this.setModelStatus({ state: 'ready' });
return;
} catch (err) {
console.warn('[WasmStt] Worker failed, using main-thread:', err instanceof Error ? err.message : err);
this._terminateWorker();
}
// Fallback: main-thread pipeline (causes brief UI freeze during inference).
try {
const { pipeline, env } = await import('@xenova/transformers');
env.backends.onnx.wasm.numThreads = 1;
env.allowLocalModels = false;
const fileDoneBytes = new Map<string, number>();
let totalDone = 0;
let totalEstimate = 0;
this.transcriber = await pipeline('automatic-speech-recognition', modelId, {
progress_callback: (info: { status?: string; file?: string; loaded?: number; total?: number }) => {
if (info.status === 'progress' && info.file) {
const prevDone = fileDoneBytes.get(info.file) ?? 0;
const currentDone = info.loaded ?? 0;
const delta = Math.max(0, currentDone - prevDone);
fileDoneBytes.set(info.file, currentDone);
totalDone += delta;
if (info.total && info.total > totalEstimate) totalEstimate = info.total;
const effectiveTotal = Math.max(totalEstimate, totalDone);
const pct = effectiveTotal > 0 ? Math.min(100, Math.round((totalDone / effectiveTotal) * 100)) : 0;
this.setModelStatus({ state: 'downloading', progress: pct });
}
},
});
this.setModelStatus({ state: 'ready' });
} catch (err) {
const msg = err instanceof Error ? err.message : 'Unknown error loading model';
this.setModelStatus({ state: 'error', error: msg });
this.transcriber = null;
this.currentModelId = null;
throw err;
}
}
private _terminateWorker(): void {
if (this.worker) {
this.worker.terminate();
this.worker = null;
}
}
async unloadModel(): Promise<void> {
this._terminateWorker();
this.transcriber = null;
this.currentModelId = null;
this.setModelStatus({ state: 'unloaded' });
}
async startListening(
lang: string,
onResult: SpeechResultCallback,
onError?: ErrorCallback,
): Promise<void> {
if (this.isActive) {
this.stopListening();
}
if (!this.transcriber && !this.worker) {
onError?.('Whisper model not loaded. Select a model in Voice Settings first.');
return;
}
this.lang = lang;
this.onResult = onResult;
this.onError = onError ?? null;
this.isActive = true;
try {
this.stream = await navigator.mediaDevices.getUserMedia({ audio: true, video: false });
} catch (err) {
this.isActive = false;
const msg = err instanceof Error ? err.message : 'Microphone access denied';
onError?.(msg);
return;
}
this._setupAudioContext();
this._startRecorder();
this._startVAD();
}
stopListening(): void {
this._stopVAD();
if (this.mediaRecorder && this.mediaRecorder.state === 'recording') {
try { this.mediaRecorder.stop(); } catch { /* ignore */ }
}
this._cleanupAfterStop(true);
}
async finishListening(): Promise<void> {
if (!this.isActive) return;
this._stopVAD();
this.isSpeaking = false;
this.silenceSince = null;
if (!this.mediaRecorder || this.mediaRecorder.state === 'inactive') {
this._cleanupAfterStop(true);
return;
}
await new Promise<void>((resolve) => {
this.finishResolver = resolve;
this._finaliseUtterance(false);
});
this._cleanupAfterStop(true);
}
getIsListening(): boolean {
return this.isActive;
}
// ── Audio capture ────────────────────────────────────────────────────
private _setupAudioContext(): void {
if (!this.stream) return;
const AudioContextClass = window.AudioContext ?? (window as unknown as { webkitAudioContext: typeof AudioContext }).webkitAudioContext;
this.audioContext = new AudioContextClass();
const source = this.audioContext.createMediaStreamSource(this.stream);
this.analyser = this.audioContext.createAnalyser();
this.analyser.fftSize = 512;
source.connect(this.analyser);
}
private _teardownAudioContext(): void {
try { this.audioContext?.close(); } catch { /* ignore */ }
this.audioContext = null;
this.analyser = null;
}
private _startRecorder(): void {
if (!this.stream) return;
const mimeType = this._pickMimeType();
const options: MediaRecorderOptions = {};
if (mimeType && MediaRecorder.isTypeSupported(mimeType)) {
options.mimeType = mimeType;
}
this.mediaRecorder = new MediaRecorder(this.stream, options);
this.chunks = [];
this.recordingStartMs = Date.now();
this.mediaRecorder.ondataavailable = (e) => {
if (e.data && e.data.size > 0) {
this.chunks.push(e.data);
}
};
this.mediaRecorder.onstop = () => {
const blobs = this.chunks.splice(0);
const durationMs = Date.now() - this.recordingStartMs;
if (blobs.length === 0 || durationMs < MIN_UTTERANCE_MS) {
this.finishResolver?.();
this.finishResolver = null;
return;
}
const mType = blobs[0].type || mimeType || 'audio/webm';
const blob = new Blob(blobs, { type: mType });
void this._transcribe(blob).finally(() => {
this.finishResolver?.();
this.finishResolver = null;
});
};
this.mediaRecorder.start(250);
}
private _releaseStream(): void {
if (this.stream) {
this.stream.getTracks().forEach((t) => t.stop());
this.stream = null;
}
}
// ── VAD ──────────────────────────────────────────────────────────────
private _startVAD(): void {
this._stopVAD();
this.silenceSince = null;
this.isSpeaking = false;
this.vadTimer = setInterval(() => {
if (!this.isActive || !this.analyser) return;
const db = this._getRmsDb();
const isSilent = db < this.cfg.silenceThresholdDb;
if (!isSilent) {
this.silenceSince = null;
if (!this.isSpeaking) {
this.isSpeaking = true;
if (this.mediaRecorder?.state === 'recording') {
this.recordingStartMs = Date.now();
}
}
} else {
if (this.isSpeaking) {
if (this.silenceSince === null) {
this.silenceSince = Date.now();
} else if (Date.now() - this.silenceSince >= this.cfg.silenceHoldMs) {
this.isSpeaking = false;
this.silenceSince = null;
this._finaliseUtterance(true);
}
}
}
}, VAD_POLL_MS);
}
private _stopVAD(): void {
if (this.vadTimer !== null) {
clearInterval(this.vadTimer);
this.vadTimer = null;
}
}
private _cleanupAfterStop(clearChunks: boolean): void {
const pendingResolver = this.finishResolver;
this.isActive = false;
this.finishResolver = null;
this.mediaRecorder = null;
this._teardownAudioContext();
this._releaseStream();
if (clearChunks) this.chunks = [];
this.isSpeaking = false;
this.silenceSince = null;
this.onResult = null;
this.onError = null;
pendingResolver?.();
}
private _finaliseUtterance(restart: boolean): void {
if (!this.isActive) return;
if (this.mediaRecorder && this.mediaRecorder.state === 'recording') {
this.mediaRecorder.stop();
}
if (!restart) return;
setTimeout(() => {
if (this.isActive && this.stream) {
this._startRecorder();
}
}, 100);
}
private _getRmsDb(): number {
if (!this.analyser) return -Infinity;
const buf = new Float32Array(this.analyser.fftSize);
this.analyser.getFloatTimeDomainData(buf);
let sumSq = 0;
for (const s of buf) sumSq += s * s;
const rms = Math.sqrt(sumSq / buf.length);
return rms === 0 ? -Infinity : 20 * Math.log10(rms);
}
// ── Transcription ────────────────────────────────────────────────────
private async _transcribe(blob: Blob): Promise<void> {
if (!this.onResult) return;
if (!this.transcriber && !this.worker) {
this.onError?.('Model not loaded');
return;
}
try {
const audioData = await this._decodeToFloat32(blob);
if (!audioData || audioData.length === 0) {
this.onError?.(`Failed to decode audio (${blob.size} bytes)`);
return;
}
const langHint = this._resolveLanguageHint();
// Prefer worker (non-blocking); fall back to main-thread pipeline.
const transcript = this.worker
? await this._transcribeViaWorker(audioData, langHint)
: await this._transcribeMainThread(audioData, langHint);
if (transcript) {
this.onResult(transcript, true);
}
} catch (err) {
if (!this.isActive) return;
const msg = err instanceof Error ? err.message : 'Local transcription failed';
this.onError?.(msg);
}
}
private _transcribeViaWorker(audioData: Float32Array, langHint: string | undefined): Promise<string> {
return new Promise((resolve, reject) => {
if (!this.worker) return reject(new Error('Worker gone'));
const onMessage = (e: MessageEvent) => {
const data = e.data as { type: string; error?: string; transcript?: string; text?: string };
if (data.type === 'result') {
this.worker!.removeEventListener('message', onMessage);
resolve(data.transcript ?? '');
} else if (data.type === 'log') {
console.log('[WasmStt Worker]', data.text);
} else if (data.type === 'error') {
this.worker!.removeEventListener('message', onMessage);
reject(new Error(data.error ?? 'Transcription failed'));
}
};
this.worker.addEventListener('message', onMessage);
this.worker.postMessage(
{ type: 'transcribe', audio: audioData.buffer, language: langHint },
[audioData.buffer],
);
setTimeout(() => {
this.worker?.removeEventListener('message', onMessage);
reject(new Error('Transcription timed out'));
}, 30000);
});
}
private async _transcribeMainThread(audioData: Float32Array, langHint: string | undefined): Promise<string> {
const pipelineFn = this.transcriber as (
input: Float32Array,
options?: Record<string, unknown>,
) => Promise<{ text: string }>;
const result = await pipelineFn(audioData, {
task: 'transcribe',
...(langHint ? { language: langHint } : {}),
});
return (result?.text ?? '').trim();
}
private async _decodeToFloat32(blob: Blob): Promise<Float32Array | null> {
if (!this.audioContext) return null;
const arrayBuffer = await blob.arrayBuffer();
let audioBuffer: AudioBuffer;
try {
audioBuffer = await this.audioContext.decodeAudioData(arrayBuffer);
} catch {
return null;
}
const origRate = audioBuffer.sampleRate;
const origData = audioBuffer.getChannelData(0);
const targetRate = WHISPER_SAMPLE_RATE;
if (origRate === targetRate) {
return new Float32Array(origData);
}
const ratio = origRate / targetRate;
const newLength = Math.ceil(origData.length / ratio);
const result = new Float32Array(newLength);
for (let i = 0; i < newLength; i++) {
const origIdx = i * ratio;
const idx0 = Math.floor(origIdx);
const idx1 = Math.min(idx0 + 1, origData.length - 1);
const frac = origIdx - idx0;
result[i] = origData[idx0] * (1 - frac) + origData[idx1] * frac;
}
return result;
}
private _resolveLanguageHint(): string | undefined {
if (this.lang && this.lang !== 'auto') {
return this.lang.split('-')[0];
}
return undefined;
}
private _pickMimeType(): string {
const candidates = [
'audio/webm;codecs=opus',
'audio/webm',
'audio/ogg;codecs=opus',
'audio/ogg',
'audio/mp4',
];
if (typeof MediaRecorder !== 'undefined' && MediaRecorder.isTypeSupported) {
return candidates.find((t) => MediaRecorder.isTypeSupported(t)) ?? '';
}
return '';
}
}
export const wasmSttService = new WasmSttService();
@@ -1,92 +0,0 @@
/**
* Web Worker for off-main-thread Whisper transcription.
*
* Receives `{ type: 'load', modelId }` to load a model, then
* `{ type: 'transcribe', audio: Float32Array (transferred buffer), language? }`
* to run inference. Posts progress, results, and errors back.
*/
import { pipeline, env } from '@xenova/transformers';
let transcriber: unknown = null;
self.onmessage = async (e: MessageEvent) => {
const { type } = e.data as { type: string };
if (type === 'load') {
const { modelId } = e.data as { modelId: string };
try {
env.backends.onnx.wasm.numThreads = 1;
const fileDoneBytes = new Map<string, number>();
let totalDone = 0;
let totalEstimate = 0;
transcriber = await pipeline('automatic-speech-recognition', modelId, {
progress_callback: (info: { status?: string; file?: string; loaded?: number; total?: number }) => {
if (info.status === 'progress' && info.file) {
const prevDone = fileDoneBytes.get(info.file) ?? 0;
const currentDone = info.loaded ?? 0;
const delta = Math.max(0, currentDone - prevDone);
fileDoneBytes.set(info.file, currentDone);
totalDone += delta;
if (info.total && info.total > totalEstimate) {
totalEstimate = info.total;
}
const effectiveTotal = Math.max(totalEstimate, totalDone);
const pct = effectiveTotal > 0 ? Math.min(100, Math.round((totalDone / effectiveTotal) * 100)) : 0;
self.postMessage({ type: 'progress', progress: pct });
}
},
});
self.postMessage({ type: 'loaded' });
} catch (err) {
self.postMessage({
type: 'error',
error: err instanceof Error ? err.message : 'Failed to load model',
});
}
} else if (type === 'transcribe') {
if (!transcriber) {
self.postMessage({ type: 'error', error: 'Model not loaded', seq: (e.data as { seq?: number }).seq });
return;
}
const { audio, language, seq } = e.data as { audio: ArrayBuffer; language?: string; seq?: number };
try {
const samples = new Float32Array(audio);
if (samples.length === 0) {
self.postMessage({ type: 'error', error: 'Empty audio received', seq });
return;
}
self.postMessage({ type: 'log', text: `Transcribing ${samples.length} samples (${(samples.length / 16000).toFixed(1)}s)` });
const pipelineFn = transcriber as (
input: Float32Array,
options?: Record<string, unknown>,
) => Promise<{ text: string }>;
const result = await pipelineFn(samples, {
task: 'transcribe',
...(language ? { language } : {}),
});
self.postMessage({
type: 'result',
transcript: (result?.text ?? '').trim(),
seq,
});
} catch (err) {
self.postMessage({
type: 'error',
error: err instanceof Error ? err.message : 'Transcription failed',
seq,
});
}
}
};