feat(voice): first-class voice input and local TTS across web, desktop, and mobile (#2018)
Complete rebuild of voice input on a server-authoritative streaming architecture, replacing the legacy Web Speech / whole-blob / WASM engines and the dead voice-agent layer (~4k lines removed). Speech-to-text (dictation): - Client streams 16 kHz mono PCM16 chunks over /api/dictation/ws with seq/ack ordering; buffered audio is retained and replayed on reconnect - Server transcribes and streams live partial transcripts back; segments auto-commit every ~15s with silence suppression and adaptive finalization timeouts - Local provider (default, zero config): sherpa-onnx models in a forked worker process — auto-download with progress, staged extraction with verification, corrupt-model auto-recovery, idle shutdown after 5 min - Model catalog with settings picker (accuracy/speed ratings, sizes, download/delete): Parakeet TDT v2 (English) and v3 (25 European languages, auto-detected), Whisper base and tiny (multilingual, light) - OpenAI-compatible provider for any Whisper endpoint - Composer overlay with live transcript, volume meter, timer, and cancel / insert / insert-and-send actions; failed transcriptions keep their audio for retry or accepting the partial text as-is - Configurable keyboard shortcut (default mod+alt+v) toggles dictation; Enter confirms and Escape cancels while recording - Overlay is pixel-aligned with the composer (measured footer height, matching paddings/typography/gaps) — no layout shift when toggling Text-to-speech: - Local Kokoro provider (English, 11 voices) synthesized in the same worker via /api/dictation/tts/speak, managed by the shared model pipeline; sentence-pipelined playback keeps time-to-first-audio at ~1 sentence regardless of message length, and stop cancels in-flight synthesis - Sanitizer keeps inline-code content (strips backticks only), reads interword slashes aloud, and removes only absolute file paths Settings: - Voice page unified: a single read-aloud toggle owns all playback options (the confusing "Enable Voice Mode" is gone); a new "Enable voice input" toggle (default on, persisted to settings.json) hides the composer mic entirely when disabled Mobile and transport: - iOS/Android microphone permissions added (dictation was previously impossible on mobile) - Fixed Android WebSocket upgrades: the Capacitor WebView origin (https://localhost) was missing from the packaged-client allowlist, 403-ing every WS connection — root cause of the old mobile SSE lock, which is now removed for all transports Security and conventions: - All HTTP routes sit behind the global /api auth gate; the WS upgrade explicitly validates the UI session and origin, with oc_url_token narrowly allowlisted and covered by tests; the dictation socket mints a fresh URL token before connecting - Routes register before the generic OpenCode proxy; the client goes through runtimeFetch/getRuntimeUrlResolver, and runtime switches reset the dictation socket - VS Code deliberately reports dictation as unavailable (no server process in that runtime) CI: workflow Node bumped 20 -> 22 to match the repo engines and fix better-sqlite3 installs broken by node-gyp@latest on Node 20. New dependency: sherpa-onnx-node (prebuilt N-API; macOS/Linux x64+arm64, Windows x64 — Windows-on-ARM falls back to the OpenAI-compatible provider)
This commit is contained in:
committed by
GitHub
parent
3f5151d424
commit
de1b85ac56
@@ -1,397 +0,0 @@
|
||||
/**
|
||||
* Audio Stream Service
|
||||
*
|
||||
* Captures microphone audio using MediaRecorder, detects utterance boundaries
|
||||
* via an AnalyserNode-based silence detector (VAD), then POSTs each utterance
|
||||
* as a raw audio blob to the OpenChamber server's /api/stt/transcribe endpoint.
|
||||
*
|
||||
* Mimics the BrowserVoiceService.startListening interface so useBrowserVoice
|
||||
* can swap providers without changing its internal logic.
|
||||
*
|
||||
* @example
|
||||
* ```typescript
|
||||
* audioStreamService.configure({ baseURL: 'http://localhost:8001/v1', model: 'whisper-1' });
|
||||
* audioStreamService.startListening('en', (text, isFinal) => {
|
||||
* if (isFinal) console.log('transcript:', text);
|
||||
* });
|
||||
* audioStreamService.stopListening();
|
||||
* ```
|
||||
*/
|
||||
|
||||
import { runtimeFetch } from '@/lib/runtime-fetch';
|
||||
|
||||
type SpeechResultCallback = (text: string, isFinal: boolean) => void;
|
||||
type ErrorCallback = (error: string) => void;
|
||||
|
||||
interface AudioStreamConfig {
|
||||
/** Base URL of the OpenAI-compatible STT server (e.g. http://localhost:8001/v1) */
|
||||
baseURL: string;
|
||||
/** Whisper-compatible model name */
|
||||
model: string;
|
||||
/** Optional BCP-47 language hint (e.g. 'en'). Empty string = auto-detect. */
|
||||
language?: string;
|
||||
/**
|
||||
* Silence threshold in dB below which audio is considered silence.
|
||||
* Lower (more negative) = only very quiet audio counts as silence.
|
||||
* Default: -45
|
||||
*/
|
||||
silenceThresholdDb?: number;
|
||||
/**
|
||||
* How long continuous silence must last (ms) before the utterance is finalised.
|
||||
* Default: 1500
|
||||
*/
|
||||
silenceHoldMs?: number;
|
||||
/** Optional API key for the STT server. */
|
||||
apiKey?: string;
|
||||
}
|
||||
|
||||
// How often (ms) the VAD samples the analyser
|
||||
const VAD_POLL_MS = 80;
|
||||
// Minimum audio duration (ms) to bother uploading (avoids blank clips)
|
||||
const MIN_UTTERANCE_MS = 300;
|
||||
|
||||
class AudioStreamService {
|
||||
private stream: MediaStream | null = null;
|
||||
private mediaRecorder: MediaRecorder | null = null;
|
||||
private audioContext: AudioContext | null = null;
|
||||
private analyser: AnalyserNode | null = null;
|
||||
private vadTimer: ReturnType<typeof setInterval> | null = null;
|
||||
private chunks: Blob[] = [];
|
||||
private recordingStartMs = 0;
|
||||
private isActive = false;
|
||||
private isSpeaking = false;
|
||||
private silenceSince: number | null = null;
|
||||
private onResult: SpeechResultCallback | null = null;
|
||||
private onError: ErrorCallback | null = null;
|
||||
private finishResolver: (() => void) | null = null;
|
||||
private lang = 'en';
|
||||
|
||||
// Configurable parameters
|
||||
private cfg: Required<AudioStreamConfig> = {
|
||||
baseURL: '',
|
||||
model: 'deepdml/faster-whisper-large-v3-turbo-ct2',
|
||||
language: '',
|
||||
silenceThresholdDb: -45,
|
||||
silenceHoldMs: 1500,
|
||||
apiKey: '',
|
||||
};
|
||||
|
||||
/** Update service configuration. Can be called before or after startListening. */
|
||||
configure(config: AudioStreamConfig): void {
|
||||
this.cfg = {
|
||||
silenceThresholdDb: -45,
|
||||
silenceHoldMs: 1500,
|
||||
language: '',
|
||||
apiKey: '',
|
||||
...config,
|
||||
};
|
||||
this.cfg.apiKey = config.apiKey ?? '';
|
||||
}
|
||||
|
||||
/** Whether the browser supports the required APIs. */
|
||||
isSupported(): boolean {
|
||||
return (
|
||||
typeof window !== 'undefined' &&
|
||||
typeof navigator !== 'undefined' &&
|
||||
typeof navigator.mediaDevices?.getUserMedia === 'function' &&
|
||||
typeof window.MediaRecorder !== 'undefined' &&
|
||||
typeof window.AudioContext !== 'undefined'
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Start listening. Requests microphone access if not already held.
|
||||
* Calls onResult(text, true) for each completed utterance.
|
||||
*/
|
||||
async startListening(
|
||||
lang: string,
|
||||
onResult: SpeechResultCallback,
|
||||
onError?: ErrorCallback
|
||||
): Promise<void> {
|
||||
if (this.isActive) {
|
||||
this.stopListening();
|
||||
}
|
||||
|
||||
this.lang = lang;
|
||||
this.onResult = onResult;
|
||||
this.onError = onError ?? null;
|
||||
this.isActive = true;
|
||||
|
||||
try {
|
||||
this.stream = await navigator.mediaDevices.getUserMedia({ audio: true, video: false });
|
||||
} catch (err) {
|
||||
this.isActive = false;
|
||||
const msg = err instanceof Error ? err.message : 'Microphone access denied';
|
||||
onError?.(msg);
|
||||
return;
|
||||
}
|
||||
|
||||
this._setupAudioContext();
|
||||
this._startRecorder();
|
||||
this._startVAD();
|
||||
}
|
||||
|
||||
/** Stop listening and clean up all resources. */
|
||||
stopListening(): void {
|
||||
this._stopVAD();
|
||||
this._stopRecorder();
|
||||
this._cleanupAfterStop(true);
|
||||
}
|
||||
|
||||
async finishListening(): Promise<void> {
|
||||
if (!this.isActive) return;
|
||||
|
||||
this._stopVAD();
|
||||
this.isSpeaking = false;
|
||||
this.silenceSince = null;
|
||||
|
||||
if (!this.mediaRecorder || this.mediaRecorder.state === 'inactive') {
|
||||
this._cleanupAfterStop(true);
|
||||
return;
|
||||
}
|
||||
|
||||
await new Promise<void>((resolve) => {
|
||||
this.finishResolver = resolve;
|
||||
this._finaliseUtterance(false);
|
||||
});
|
||||
|
||||
this._cleanupAfterStop(true);
|
||||
}
|
||||
|
||||
/** Whether currently listening. */
|
||||
getIsListening(): boolean {
|
||||
return this.isActive;
|
||||
}
|
||||
|
||||
// ── Private helpers ──────────────────────────────────────────────────────
|
||||
|
||||
private _setupAudioContext(): void {
|
||||
if (!this.stream) return;
|
||||
const AudioContextClass = window.AudioContext ?? (window as unknown as { webkitAudioContext: typeof AudioContext }).webkitAudioContext;
|
||||
this.audioContext = new AudioContextClass();
|
||||
const source = this.audioContext.createMediaStreamSource(this.stream);
|
||||
this.analyser = this.audioContext.createAnalyser();
|
||||
this.analyser.fftSize = 512;
|
||||
source.connect(this.analyser);
|
||||
}
|
||||
|
||||
private _teardownAudioContext(): void {
|
||||
try {
|
||||
this.audioContext?.close();
|
||||
} catch {
|
||||
// ignore
|
||||
}
|
||||
this.audioContext = null;
|
||||
this.analyser = null;
|
||||
}
|
||||
|
||||
private _startRecorder(): void {
|
||||
if (!this.stream) return;
|
||||
|
||||
const mimeType = this._pickMimeType();
|
||||
const options: MediaRecorderOptions = {};
|
||||
if (mimeType && MediaRecorder.isTypeSupported(mimeType)) {
|
||||
options.mimeType = mimeType;
|
||||
}
|
||||
|
||||
this.mediaRecorder = new MediaRecorder(this.stream, options);
|
||||
this.chunks = [];
|
||||
this.recordingStartMs = Date.now();
|
||||
|
||||
this.mediaRecorder.ondataavailable = (e) => {
|
||||
if (e.data && e.data.size > 0) {
|
||||
this.chunks.push(e.data);
|
||||
}
|
||||
};
|
||||
|
||||
this.mediaRecorder.onstop = () => {
|
||||
const blobs = this.chunks.splice(0);
|
||||
const durationMs = Date.now() - this.recordingStartMs;
|
||||
if (blobs.length === 0 || durationMs < MIN_UTTERANCE_MS) {
|
||||
this.finishResolver?.();
|
||||
this.finishResolver = null;
|
||||
return;
|
||||
}
|
||||
|
||||
const mType = blobs[0].type || mimeType || 'audio/webm';
|
||||
const blob = new Blob(blobs, { type: mType });
|
||||
void this._upload(blob, mType).finally(() => {
|
||||
this.finishResolver?.();
|
||||
this.finishResolver = null;
|
||||
});
|
||||
};
|
||||
|
||||
// Collect data every 250 ms so we don't lose the tail on stop()
|
||||
this.mediaRecorder.start(250);
|
||||
}
|
||||
|
||||
private _stopRecorder(): void {
|
||||
if (this.mediaRecorder && this.mediaRecorder.state !== 'inactive') {
|
||||
try {
|
||||
this.mediaRecorder.stop();
|
||||
} catch {
|
||||
this.finishResolver?.();
|
||||
this.finishResolver = null;
|
||||
// ignore
|
||||
}
|
||||
}
|
||||
this.mediaRecorder = null;
|
||||
}
|
||||
|
||||
private _releaseStream(): void {
|
||||
if (this.stream) {
|
||||
this.stream.getTracks().forEach((t) => t.stop());
|
||||
this.stream = null;
|
||||
}
|
||||
}
|
||||
|
||||
private _startVAD(): void {
|
||||
this._stopVAD();
|
||||
this.silenceSince = null;
|
||||
this.isSpeaking = false;
|
||||
|
||||
this.vadTimer = setInterval(() => {
|
||||
if (!this.isActive || !this.analyser) return;
|
||||
const db = this._getRmsDb();
|
||||
const isSilent = db < this.cfg.silenceThresholdDb;
|
||||
|
||||
if (!isSilent) {
|
||||
// Audio detected
|
||||
this.silenceSince = null;
|
||||
if (!this.isSpeaking) {
|
||||
this.isSpeaking = true;
|
||||
// Restart recorder to capture from the start of speech
|
||||
if (this.mediaRecorder?.state === 'recording') {
|
||||
this.recordingStartMs = Date.now();
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// Silence detected
|
||||
if (this.isSpeaking) {
|
||||
if (this.silenceSince === null) {
|
||||
this.silenceSince = Date.now();
|
||||
} else if (Date.now() - this.silenceSince >= this.cfg.silenceHoldMs) {
|
||||
// End of utterance — stop recorder (triggers onstop → upload)
|
||||
this.isSpeaking = false;
|
||||
this.silenceSince = null;
|
||||
this._finaliseUtterance(true);
|
||||
}
|
||||
}
|
||||
}
|
||||
}, VAD_POLL_MS);
|
||||
}
|
||||
|
||||
private _stopVAD(): void {
|
||||
if (this.vadTimer !== null) {
|
||||
clearInterval(this.vadTimer);
|
||||
this.vadTimer = null;
|
||||
}
|
||||
}
|
||||
|
||||
private _cleanupAfterStop(clearChunks: boolean): void {
|
||||
const pendingResolver = this.finishResolver;
|
||||
this.isActive = false;
|
||||
this.finishResolver = null;
|
||||
this.mediaRecorder = null;
|
||||
this._teardownAudioContext();
|
||||
this._releaseStream();
|
||||
if (clearChunks) {
|
||||
this.chunks = [];
|
||||
}
|
||||
this.isSpeaking = false;
|
||||
this.silenceSince = null;
|
||||
this.onResult = null;
|
||||
this.onError = null;
|
||||
pendingResolver?.();
|
||||
}
|
||||
|
||||
/** Stop the current recorder to flush the utterance, optionally restarting for the next one. */
|
||||
private _finaliseUtterance(restart: boolean): void {
|
||||
if (!this.isActive) return;
|
||||
if (this.mediaRecorder && this.mediaRecorder.state === 'recording') {
|
||||
this.mediaRecorder.stop();
|
||||
}
|
||||
if (!restart) return;
|
||||
|
||||
// Restart recorder for the next utterance after a short delay
|
||||
// (MediaRecorder.onstop fires asynchronously; we wait for it to complete)
|
||||
setTimeout(() => {
|
||||
if (this.isActive && this.stream) {
|
||||
this._startRecorder();
|
||||
}
|
||||
}, 100);
|
||||
}
|
||||
|
||||
/** Compute RMS of current analyser frame in dBFS. */
|
||||
private _getRmsDb(): number {
|
||||
if (!this.analyser) return -Infinity;
|
||||
const buf = new Float32Array(this.analyser.fftSize);
|
||||
this.analyser.getFloatTimeDomainData(buf);
|
||||
let sumSq = 0;
|
||||
for (const s of buf) sumSq += s * s;
|
||||
const rms = Math.sqrt(sumSq / buf.length);
|
||||
return rms === 0 ? -Infinity : 20 * Math.log10(rms);
|
||||
}
|
||||
|
||||
/** POST utterance blob to server, call onResult with transcript. */
|
||||
private async _upload(blob: Blob, mimeType: string): Promise<void> {
|
||||
if (!this.onResult) return;
|
||||
|
||||
try {
|
||||
const headers: Record<string, string> = {
|
||||
'Content-Type': mimeType,
|
||||
'X-Base-URL': this.cfg.baseURL,
|
||||
'X-Model': this.cfg.model,
|
||||
};
|
||||
if (this.cfg.apiKey) {
|
||||
headers['Authorization'] = `Bearer ${this.cfg.apiKey}`;
|
||||
}
|
||||
if (this.cfg.language) {
|
||||
headers['X-Language'] = this.cfg.language;
|
||||
} else if (this.lang && this.lang !== 'auto') {
|
||||
// Use BCP-47 base language code (e.g. 'en' from 'en-US')
|
||||
const baseLang = this.lang.split('-')[0];
|
||||
headers['X-Language'] = baseLang;
|
||||
}
|
||||
|
||||
const response = await runtimeFetch('/api/stt/transcribe', {
|
||||
method: 'POST',
|
||||
headers,
|
||||
body: blob,
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const errData = await response.json().catch(() => ({ error: 'Unknown error' }));
|
||||
throw new Error(errData.error ?? `HTTP ${response.status}`);
|
||||
}
|
||||
|
||||
const data = await response.json();
|
||||
const transcript: string = (data.transcript ?? '').trim();
|
||||
if (transcript) {
|
||||
this.onResult(transcript, true);
|
||||
}
|
||||
} catch (err) {
|
||||
if (!this.isActive) return; // Stopped — ignore
|
||||
const msg = err instanceof Error ? err.message : 'Transcription upload failed';
|
||||
console.error('[AudioStreamService] Upload error:', msg);
|
||||
this.onError?.(msg);
|
||||
}
|
||||
}
|
||||
|
||||
/** Pick the best supported MIME type for MediaRecorder. */
|
||||
private _pickMimeType(): string {
|
||||
const candidates = [
|
||||
'audio/webm;codecs=opus',
|
||||
'audio/webm',
|
||||
'audio/ogg;codecs=opus',
|
||||
'audio/ogg',
|
||||
'audio/mp4',
|
||||
];
|
||||
if (typeof MediaRecorder !== 'undefined' && MediaRecorder.isTypeSupported) {
|
||||
return candidates.find((t) => MediaRecorder.isTypeSupported(t)) ?? '';
|
||||
}
|
||||
return '';
|
||||
}
|
||||
}
|
||||
|
||||
export const audioStreamService = new AudioStreamService();
|
||||
@@ -1,131 +0,0 @@
|
||||
/**
|
||||
* Context formatters for voice-native output
|
||||
* Formats session events (messages, permissions, ready events) into natural language
|
||||
* for the ElevenLabs voice agent to speak aloud.
|
||||
*
|
||||
* @example
|
||||
* ```typescript
|
||||
* import { formatMessage, formatPermissionRequest } from '@/lib/voice';
|
||||
*
|
||||
* const voiceText = formatMessage({ role: 'assistant', content: 'Hello!' });
|
||||
* // Returns: "Claude Code: Hello!"
|
||||
* ```
|
||||
*/
|
||||
|
||||
import { VOICE_CONFIG } from "./voiceConfig";
|
||||
|
||||
/** Message type for voice formatting */
|
||||
export interface VoiceMessage {
|
||||
role: string;
|
||||
content: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Format a single message for voice output
|
||||
* - Assistant messages: Code blocks replaced with "[code block]", prefixed with "Claude Code: "
|
||||
* - User messages: Prefixed with "User: "
|
||||
* - Other roles: Returns null (not spoken)
|
||||
*
|
||||
* @param message - The message to format
|
||||
* @returns Formatted text for voice, or null if should not be spoken
|
||||
*/
|
||||
function formatMessage(message: VoiceMessage): string | null {
|
||||
// Handle edge cases
|
||||
if (!message || typeof message.content !== "string") {
|
||||
return null;
|
||||
}
|
||||
|
||||
const content = message.content.trim();
|
||||
if (!content) {
|
||||
return null;
|
||||
}
|
||||
|
||||
if (message.role === "assistant") {
|
||||
// Replace code blocks with description (don't read code aloud)
|
||||
const textOnly = content.replace(/```[\s\S]*?```/g, "[code block]");
|
||||
return `Claude Code: ${textOnly}`;
|
||||
}
|
||||
|
||||
if (message.role === "user") {
|
||||
return `User: ${content}`;
|
||||
}
|
||||
|
||||
// Skip system, tool, and other roles for voice
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Format multiple new messages for voice output
|
||||
* - Maps messages through formatMessage
|
||||
* - Filters out nulls (unspoken roles)
|
||||
* - Joins with newlines
|
||||
*
|
||||
* @param sessionId - The session ID (for future use/debugging)
|
||||
* @param messages - Array of messages to format
|
||||
* @returns Formatted text for voice, or null if no speakable messages
|
||||
*/
|
||||
export function formatNewMessages(
|
||||
sessionId: string,
|
||||
messages: VoiceMessage[]
|
||||
): string | null {
|
||||
// Handle edge cases
|
||||
if (!Array.isArray(messages) || messages.length === 0) {
|
||||
return null;
|
||||
}
|
||||
|
||||
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
|
||||
console.log(`[Voice] Formatting ${messages.length} messages for session ${sessionId}`);
|
||||
}
|
||||
|
||||
// Format each message and filter out nulls
|
||||
const formattedMessages = messages
|
||||
.map(formatMessage)
|
||||
.filter((msg): msg is string => msg !== null);
|
||||
|
||||
if (formattedMessages.length === 0) {
|
||||
return null;
|
||||
}
|
||||
|
||||
return formattedMessages.join("\n");
|
||||
}
|
||||
|
||||
/**
|
||||
* Format a permission request for voice announcement
|
||||
* - Per CONTEXT.md: Only tool name, not arguments (LIMITED_TOOL_CALLS)
|
||||
* - Prompts user to say "allow" or "deny"
|
||||
*
|
||||
* @param sessionId - The session ID
|
||||
* @param requestId - The permission request ID
|
||||
* @param toolName - Name of the tool requesting permission
|
||||
* @param toolArgs - Tool arguments (not included in voice output per config)
|
||||
* @returns Formatted permission request for voice
|
||||
*/
|
||||
export function formatPermissionRequest(
|
||||
sessionId: string,
|
||||
requestId: string,
|
||||
toolName: string,
|
||||
// eslint-disable-next-line @typescript-eslint/no-unused-vars
|
||||
toolArgs: unknown
|
||||
): string {
|
||||
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
|
||||
console.log(`[Voice] Formatting permission request ${requestId} for session ${sessionId}`);
|
||||
}
|
||||
|
||||
// Per VOICE_CONFIG.LIMITED_TOOL_CALLS, we don't include toolArgs in voice output
|
||||
return `Claude Code is requesting permission to use ${toolName}. Say "allow" or "deny".`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Format a ready event for voice announcement
|
||||
* - Indicates the AI has finished working and is ready for next instruction
|
||||
*
|
||||
* @param sessionId - The session ID
|
||||
* @returns Formatted ready event for voice
|
||||
*/
|
||||
export function formatReadyEvent(sessionId: string): string {
|
||||
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
|
||||
console.log(`[Voice] Formatting ready event for session ${sessionId}`);
|
||||
}
|
||||
|
||||
return "Claude Code finished working. Ready for next instruction.";
|
||||
}
|
||||
@@ -1,17 +0,0 @@
|
||||
/**
|
||||
* Voice module barrel export
|
||||
* Provides a clean import path for voice session hooks.
|
||||
*
|
||||
* @example
|
||||
* ```typescript
|
||||
* import { voiceHooks } from '@/lib/voice';
|
||||
* ```
|
||||
*/
|
||||
|
||||
// Voice session registry (from voiceSession.ts)
|
||||
export {
|
||||
isVoiceSessionStarted,
|
||||
} from "./voiceSession";
|
||||
|
||||
// Voice hooks for session-to-voice event routing (from voiceHooks.ts)
|
||||
export { voiceHooks } from "./voiceHooks";
|
||||
@@ -6,15 +6,21 @@
|
||||
export function sanitizeForTTS(text: string): string {
|
||||
if (!text) return '';
|
||||
return text
|
||||
// Remove code blocks
|
||||
// Remove fenced code blocks entirely (multi-line code is unreadable),
|
||||
// but keep inline-code CONTENT and only strip the backticks: agents
|
||||
// routinely inline meaningful words ("You are on `main`").
|
||||
.replace(/```[\s\S]*?```/g, '')
|
||||
.replace(/`[^`]*`/g, '')
|
||||
.replace(/`([^`\n]*)`/g, '$1')
|
||||
// Remove markdown formatting
|
||||
.replace(/[*_~#]/g, '')
|
||||
// Remove URLs
|
||||
.replace(/https?:\/\/[^\s]+/g, '')
|
||||
// Remove file paths
|
||||
.replace(/\/[\w\-./]+/g, '')
|
||||
// Remove absolute file paths (leading slash, one or more segments).
|
||||
// Deliberately NOT matching interword slashes: "iOS/Android" and
|
||||
// "origin/main" are speech, not paths.
|
||||
.replace(/(^|\s)\/(?:[\w.-]+\/)*[\w.-]+/g, '$1')
|
||||
// Read remaining interword slashes out loud ("iOS slash Android").
|
||||
.replace(/([\w.])\/([\w.])/g, '$1 slash $2')
|
||||
// Remove shell-like patterns
|
||||
.replace(/^\s*[$#>]\s*/gm, '')
|
||||
// Remove brackets and special chars
|
||||
|
||||
@@ -1,32 +0,0 @@
|
||||
/**
|
||||
* Static voice context configuration
|
||||
* Controls voice behavior and feature flags for the ElevenLabs voice agent
|
||||
*/
|
||||
export const VOICE_CONFIG = {
|
||||
/** Disable all tool call information from being sent to voice context */
|
||||
DISABLE_TOOL_CALLS: false,
|
||||
|
||||
/** Send only tool names and descriptions, exclude arguments */
|
||||
LIMITED_TOOL_CALLS: true,
|
||||
|
||||
/** Disable permission request forwarding */
|
||||
DISABLE_PERMISSION_REQUESTS: false,
|
||||
|
||||
/** Disable session online/offline notifications */
|
||||
DISABLE_SESSION_STATUS: true,
|
||||
|
||||
/** Disable message forwarding */
|
||||
DISABLE_MESSAGES: false,
|
||||
|
||||
/** Disable session focus notifications */
|
||||
DISABLE_SESSION_FOCUS: false,
|
||||
|
||||
/** Disable ready event notifications */
|
||||
DISABLE_READY_EVENTS: false,
|
||||
|
||||
/** Maximum number of messages to include in session history */
|
||||
MAX_HISTORY_MESSAGES: 50,
|
||||
|
||||
/** Enable debug logging for voice context updates */
|
||||
ENABLE_DEBUG_LOGGING: true,
|
||||
} as const;
|
||||
@@ -1,125 +0,0 @@
|
||||
/**
|
||||
* Voice hooks for session-to-voice event routing
|
||||
* Routes session events (messages, permissions, ready events) to the ElevenLabs
|
||||
* voice agent via contextual updates.
|
||||
*
|
||||
* This module provides hooks that can be called when session events occur,
|
||||
* using the voice session registry from voiceSession.ts.
|
||||
*
|
||||
* @example
|
||||
* ```typescript
|
||||
* import { voiceHooks } from '@/lib/voice';
|
||||
*
|
||||
* // Route session messages to voice
|
||||
* voiceHooks.onMessages(sessionId, messages);
|
||||
* ```
|
||||
*/
|
||||
|
||||
import { VOICE_CONFIG } from "./voiceConfig";
|
||||
import {
|
||||
formatNewMessages,
|
||||
formatPermissionRequest,
|
||||
formatReadyEvent,
|
||||
type VoiceMessage,
|
||||
} from "./contextFormatters";
|
||||
import { getVoiceSession, isVoiceSessionStarted } from "./voiceSession";
|
||||
|
||||
/**
|
||||
* Report a contextual update to the voice session
|
||||
* Internal helper that checks preconditions and handles errors
|
||||
*
|
||||
* @param update - The text update to send, or null/undefined to skip
|
||||
*/
|
||||
function reportContextualUpdate(update: string | null | undefined): void {
|
||||
// Skip empty/null/undefined updates
|
||||
if (!update || update.trim().length === 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Skip if no voice session or not started
|
||||
const voiceSession = getVoiceSession();
|
||||
if (!voiceSession || !isVoiceSessionStarted()) {
|
||||
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
|
||||
console.log("[Voice] Skipping contextual update - no active session");
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
|
||||
console.log("[Voice] Sending contextual update:", update.substring(0, 100));
|
||||
}
|
||||
voiceSession.sendContextualUpdate(update);
|
||||
} catch (error) {
|
||||
// Log error but don't throw - voice updates shouldn't break the app
|
||||
console.error("[Voice] Failed to send contextual update:", error);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Voice hooks - exported functions to route session events to voice
|
||||
*
|
||||
* These hooks should be called when corresponding session events occur.
|
||||
* They respect VOICE_CONFIG feature flags to enable/disable specific
|
||||
* event types.
|
||||
*/
|
||||
export const voiceHooks = {
|
||||
/**
|
||||
* Called when new messages arrive in the session
|
||||
* Formats and sends messages to voice agent (if not disabled)
|
||||
*
|
||||
* @param sessionId - The session ID
|
||||
* @param messages - Array of messages to format and send
|
||||
*/
|
||||
onMessages(sessionId: string, messages: VoiceMessage[]): void {
|
||||
if (VOICE_CONFIG.DISABLE_MESSAGES) {
|
||||
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
|
||||
console.log("[Voice] Message forwarding disabled");
|
||||
}
|
||||
return;
|
||||
}
|
||||
reportContextualUpdate(formatNewMessages(sessionId, messages));
|
||||
},
|
||||
|
||||
/**
|
||||
* Called when a permission request is made
|
||||
* Announces the permission request to the voice agent (if not disabled)
|
||||
*
|
||||
* @param sessionId - The session ID
|
||||
* @param requestId - The permission request ID
|
||||
* @param toolName - Name of the tool requesting permission
|
||||
* @param toolArgs - Arguments for the tool (not sent to voice per LIMITED_TOOL_CALLS)
|
||||
*/
|
||||
onPermissionRequested(
|
||||
sessionId: string,
|
||||
requestId: string,
|
||||
toolName: string,
|
||||
toolArgs: unknown
|
||||
): void {
|
||||
if (VOICE_CONFIG.DISABLE_PERMISSION_REQUESTS) {
|
||||
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
|
||||
console.log("[Voice] Permission request forwarding disabled");
|
||||
}
|
||||
return;
|
||||
}
|
||||
reportContextualUpdate(
|
||||
formatPermissionRequest(sessionId, requestId, toolName, toolArgs)
|
||||
);
|
||||
},
|
||||
|
||||
/**
|
||||
* Called when the AI is ready for the next instruction
|
||||
* Announces ready state to the voice agent (if not disabled)
|
||||
*
|
||||
* @param sessionId - The session ID
|
||||
*/
|
||||
onReady(sessionId: string): void {
|
||||
if (VOICE_CONFIG.DISABLE_READY_EVENTS) {
|
||||
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
|
||||
console.log("[Voice] Ready event forwarding disabled");
|
||||
}
|
||||
return;
|
||||
}
|
||||
reportContextualUpdate(formatReadyEvent(sessionId));
|
||||
},
|
||||
};
|
||||
@@ -1,28 +0,0 @@
|
||||
/**
|
||||
* Voice session interface
|
||||
* Used for type safety without importing ReturnType from SDK
|
||||
*/
|
||||
interface VoiceSession {
|
||||
sendContextualUpdate: (text: string) => void;
|
||||
}
|
||||
|
||||
/**
|
||||
* Global storage for the active voice session.
|
||||
* Used by voiceHooks to send contextual updates to the voice agent.
|
||||
*/
|
||||
const activeVoiceSession: VoiceSession | null = null;
|
||||
|
||||
/**
|
||||
* Get the currently registered voice session.
|
||||
* Used by voiceHooks to send contextual updates.
|
||||
*/
|
||||
export function getVoiceSession(): VoiceSession | null {
|
||||
return activeVoiceSession;
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if a voice session is currently active.
|
||||
*/
|
||||
export function isVoiceSessionStarted(): boolean {
|
||||
return activeVoiceSession !== null;
|
||||
}
|
||||
@@ -1,555 +0,0 @@
|
||||
/**
|
||||
* WASM Speech-to-Text Service
|
||||
*
|
||||
* Local Whisper transcription via Transformers.js (ONNX Runtime Web).
|
||||
* Captures microphone audio, detects utterance boundaries via silence-based
|
||||
* VAD, then transcribes each utterance locally — no cloud API required.
|
||||
*
|
||||
* Works in Electron and all modern browsers that support Web Audio API.
|
||||
* First use downloads a Whisper model (~40–166 MB, cached).
|
||||
*/
|
||||
|
||||
export type WasmModelStatus =
|
||||
| { state: 'unloaded' }
|
||||
| { state: 'downloading'; progress: number }
|
||||
| { state: 'loading' }
|
||||
| { state: 'ready' }
|
||||
| { state: 'error'; error: string };
|
||||
|
||||
export interface WasmModelInfo {
|
||||
id: string;
|
||||
name: string;
|
||||
size: string;
|
||||
languages: string;
|
||||
description: string;
|
||||
}
|
||||
|
||||
export const WASM_MODELS: WasmModelInfo[] = [
|
||||
{
|
||||
id: 'Xenova/whisper-tiny.en',
|
||||
name: 'Whisper Tiny (EN)',
|
||||
size: '~39 MB',
|
||||
languages: 'English',
|
||||
description: 'Fastest, lowest accuracy. Good for quick dictation.',
|
||||
},
|
||||
{
|
||||
id: 'Xenova/whisper-base.en',
|
||||
name: 'Whisper Base (EN)',
|
||||
size: '~73 MB',
|
||||
languages: 'English',
|
||||
description: 'Balanced speed and accuracy. Default for English.',
|
||||
},
|
||||
{
|
||||
id: 'Xenova/whisper-small.en',
|
||||
name: 'Whisper Small (EN)',
|
||||
size: '~166 MB',
|
||||
languages: 'English',
|
||||
description: 'Higher accuracy, slower. Best for noisy environments.',
|
||||
},
|
||||
];
|
||||
|
||||
type SpeechResultCallback = (text: string, isFinal: boolean) => void;
|
||||
type ErrorCallback = (error: string) => void;
|
||||
|
||||
const VAD_POLL_MS = 80;
|
||||
const MIN_UTTERANCE_MS = 300;
|
||||
const WHISPER_SAMPLE_RATE = 16000;
|
||||
|
||||
interface WasmSttConfig {
|
||||
silenceThresholdDb?: number;
|
||||
silenceHoldMs?: number;
|
||||
}
|
||||
|
||||
class WasmSttService {
|
||||
private transcriber: unknown = null;
|
||||
private worker: Worker | null = null;
|
||||
private modelStatus: WasmModelStatus = { state: 'unloaded' };
|
||||
private currentModelId: string | null = null;
|
||||
|
||||
private stream: MediaStream | null = null;
|
||||
private mediaRecorder: MediaRecorder | null = null;
|
||||
private audioContext: AudioContext | null = null;
|
||||
private analyser: AnalyserNode | null = null;
|
||||
private vadTimer: ReturnType<typeof setInterval> | null = null;
|
||||
private chunks: Blob[] = [];
|
||||
private recordingStartMs = 0;
|
||||
private isActive = false;
|
||||
private isSpeaking = false;
|
||||
private silenceSince: number | null = null;
|
||||
private onResult: SpeechResultCallback | null = null;
|
||||
private onError: ErrorCallback | null = null;
|
||||
private finishResolver: (() => void) | null = null;
|
||||
private lang = 'en';
|
||||
|
||||
private cfg: Required<WasmSttConfig> = {
|
||||
silenceThresholdDb: -45,
|
||||
silenceHoldMs: 1500,
|
||||
};
|
||||
|
||||
public onModelStatusChange: ((status: WasmModelStatus) => void) | null = null;
|
||||
|
||||
configure(config: WasmSttConfig): void {
|
||||
this.cfg = { ...this.cfg, ...config };
|
||||
}
|
||||
|
||||
isSupported(): boolean {
|
||||
return (
|
||||
typeof window !== 'undefined' &&
|
||||
typeof navigator !== 'undefined' &&
|
||||
typeof navigator.mediaDevices?.getUserMedia === 'function' &&
|
||||
typeof window.MediaRecorder !== 'undefined' &&
|
||||
typeof window.AudioContext !== 'undefined'
|
||||
);
|
||||
}
|
||||
|
||||
getModelStatus(): WasmModelStatus {
|
||||
return this.modelStatus;
|
||||
}
|
||||
|
||||
getCurrentModelId(): string | null {
|
||||
return this.currentModelId;
|
||||
}
|
||||
|
||||
private setModelStatus(status: WasmModelStatus): void {
|
||||
this.modelStatus = status;
|
||||
this.onModelStatusChange?.(status);
|
||||
}
|
||||
|
||||
async loadModel(modelId: string): Promise<void> {
|
||||
if (this.currentModelId === modelId && this.modelStatus.state === 'ready') {
|
||||
return;
|
||||
}
|
||||
|
||||
if (this.modelStatus.state === 'downloading' || this.modelStatus.state === 'loading') {
|
||||
return;
|
||||
}
|
||||
|
||||
this._terminateWorker();
|
||||
this.transcriber = null;
|
||||
|
||||
this.setModelStatus({ state: 'downloading', progress: 0 });
|
||||
this.currentModelId = modelId;
|
||||
|
||||
// Try Web Worker first — inference off main thread = no UI freeze.
|
||||
try {
|
||||
const WasmWorkerMod = await import('./wasmSttWorker?worker');
|
||||
const WasmWorker = WasmWorkerMod.default as new () => Worker;
|
||||
this.worker = new WasmWorker();
|
||||
|
||||
await new Promise<void>((resolve, reject) => {
|
||||
const timer = setTimeout(() => reject(new Error('Worker init timed out')), 10000);
|
||||
|
||||
this.worker!.onmessage = (e: MessageEvent) => {
|
||||
const data = e.data as { type: string; progress?: number; error?: string; text?: string };
|
||||
if (data.type === 'progress') {
|
||||
this.setModelStatus({ state: 'downloading', progress: data.progress ?? 0 });
|
||||
} else if (data.type === 'loaded') {
|
||||
clearTimeout(timer);
|
||||
resolve();
|
||||
} else if (data.type === 'error') {
|
||||
clearTimeout(timer);
|
||||
reject(new Error(data.error ?? 'Worker load failed'));
|
||||
}
|
||||
};
|
||||
|
||||
this.worker!.onerror = (err) => {
|
||||
clearTimeout(timer);
|
||||
reject(new Error(err.message || 'Worker error'));
|
||||
};
|
||||
|
||||
this.worker!.postMessage({ type: 'load', modelId });
|
||||
});
|
||||
|
||||
this.setModelStatus({ state: 'ready' });
|
||||
return;
|
||||
} catch (err) {
|
||||
console.warn('[WasmStt] Worker failed, using main-thread:', err instanceof Error ? err.message : err);
|
||||
this._terminateWorker();
|
||||
}
|
||||
|
||||
// Fallback: main-thread pipeline (causes brief UI freeze during inference).
|
||||
try {
|
||||
const { pipeline, env } = await import('@xenova/transformers');
|
||||
env.backends.onnx.wasm.numThreads = 1;
|
||||
env.allowLocalModels = false;
|
||||
|
||||
const fileDoneBytes = new Map<string, number>();
|
||||
let totalDone = 0;
|
||||
let totalEstimate = 0;
|
||||
|
||||
this.transcriber = await pipeline('automatic-speech-recognition', modelId, {
|
||||
progress_callback: (info: { status?: string; file?: string; loaded?: number; total?: number }) => {
|
||||
if (info.status === 'progress' && info.file) {
|
||||
const prevDone = fileDoneBytes.get(info.file) ?? 0;
|
||||
const currentDone = info.loaded ?? 0;
|
||||
const delta = Math.max(0, currentDone - prevDone);
|
||||
fileDoneBytes.set(info.file, currentDone);
|
||||
totalDone += delta;
|
||||
if (info.total && info.total > totalEstimate) totalEstimate = info.total;
|
||||
const effectiveTotal = Math.max(totalEstimate, totalDone);
|
||||
const pct = effectiveTotal > 0 ? Math.min(100, Math.round((totalDone / effectiveTotal) * 100)) : 0;
|
||||
this.setModelStatus({ state: 'downloading', progress: pct });
|
||||
}
|
||||
},
|
||||
});
|
||||
|
||||
this.setModelStatus({ state: 'ready' });
|
||||
} catch (err) {
|
||||
const msg = err instanceof Error ? err.message : 'Unknown error loading model';
|
||||
this.setModelStatus({ state: 'error', error: msg });
|
||||
this.transcriber = null;
|
||||
this.currentModelId = null;
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
private _terminateWorker(): void {
|
||||
if (this.worker) {
|
||||
this.worker.terminate();
|
||||
this.worker = null;
|
||||
}
|
||||
}
|
||||
|
||||
async unloadModel(): Promise<void> {
|
||||
this._terminateWorker();
|
||||
this.transcriber = null;
|
||||
this.currentModelId = null;
|
||||
this.setModelStatus({ state: 'unloaded' });
|
||||
}
|
||||
|
||||
async startListening(
|
||||
lang: string,
|
||||
onResult: SpeechResultCallback,
|
||||
onError?: ErrorCallback,
|
||||
): Promise<void> {
|
||||
if (this.isActive) {
|
||||
this.stopListening();
|
||||
}
|
||||
|
||||
if (!this.transcriber && !this.worker) {
|
||||
onError?.('Whisper model not loaded. Select a model in Voice Settings first.');
|
||||
return;
|
||||
}
|
||||
|
||||
this.lang = lang;
|
||||
this.onResult = onResult;
|
||||
this.onError = onError ?? null;
|
||||
this.isActive = true;
|
||||
|
||||
try {
|
||||
this.stream = await navigator.mediaDevices.getUserMedia({ audio: true, video: false });
|
||||
} catch (err) {
|
||||
this.isActive = false;
|
||||
const msg = err instanceof Error ? err.message : 'Microphone access denied';
|
||||
onError?.(msg);
|
||||
return;
|
||||
}
|
||||
|
||||
this._setupAudioContext();
|
||||
this._startRecorder();
|
||||
this._startVAD();
|
||||
}
|
||||
|
||||
stopListening(): void {
|
||||
this._stopVAD();
|
||||
if (this.mediaRecorder && this.mediaRecorder.state === 'recording') {
|
||||
try { this.mediaRecorder.stop(); } catch { /* ignore */ }
|
||||
}
|
||||
this._cleanupAfterStop(true);
|
||||
}
|
||||
|
||||
async finishListening(): Promise<void> {
|
||||
if (!this.isActive) return;
|
||||
|
||||
this._stopVAD();
|
||||
this.isSpeaking = false;
|
||||
this.silenceSince = null;
|
||||
|
||||
if (!this.mediaRecorder || this.mediaRecorder.state === 'inactive') {
|
||||
this._cleanupAfterStop(true);
|
||||
return;
|
||||
}
|
||||
|
||||
await new Promise<void>((resolve) => {
|
||||
this.finishResolver = resolve;
|
||||
this._finaliseUtterance(false);
|
||||
});
|
||||
|
||||
this._cleanupAfterStop(true);
|
||||
}
|
||||
|
||||
getIsListening(): boolean {
|
||||
return this.isActive;
|
||||
}
|
||||
|
||||
// ── Audio capture ────────────────────────────────────────────────────
|
||||
|
||||
private _setupAudioContext(): void {
|
||||
if (!this.stream) return;
|
||||
const AudioContextClass = window.AudioContext ?? (window as unknown as { webkitAudioContext: typeof AudioContext }).webkitAudioContext;
|
||||
this.audioContext = new AudioContextClass();
|
||||
const source = this.audioContext.createMediaStreamSource(this.stream);
|
||||
this.analyser = this.audioContext.createAnalyser();
|
||||
this.analyser.fftSize = 512;
|
||||
source.connect(this.analyser);
|
||||
}
|
||||
|
||||
private _teardownAudioContext(): void {
|
||||
try { this.audioContext?.close(); } catch { /* ignore */ }
|
||||
this.audioContext = null;
|
||||
this.analyser = null;
|
||||
}
|
||||
|
||||
private _startRecorder(): void {
|
||||
if (!this.stream) return;
|
||||
const mimeType = this._pickMimeType();
|
||||
const options: MediaRecorderOptions = {};
|
||||
if (mimeType && MediaRecorder.isTypeSupported(mimeType)) {
|
||||
options.mimeType = mimeType;
|
||||
}
|
||||
this.mediaRecorder = new MediaRecorder(this.stream, options);
|
||||
this.chunks = [];
|
||||
this.recordingStartMs = Date.now();
|
||||
|
||||
this.mediaRecorder.ondataavailable = (e) => {
|
||||
if (e.data && e.data.size > 0) {
|
||||
this.chunks.push(e.data);
|
||||
}
|
||||
};
|
||||
|
||||
this.mediaRecorder.onstop = () => {
|
||||
const blobs = this.chunks.splice(0);
|
||||
const durationMs = Date.now() - this.recordingStartMs;
|
||||
if (blobs.length === 0 || durationMs < MIN_UTTERANCE_MS) {
|
||||
this.finishResolver?.();
|
||||
this.finishResolver = null;
|
||||
return;
|
||||
}
|
||||
const mType = blobs[0].type || mimeType || 'audio/webm';
|
||||
const blob = new Blob(blobs, { type: mType });
|
||||
void this._transcribe(blob).finally(() => {
|
||||
this.finishResolver?.();
|
||||
this.finishResolver = null;
|
||||
});
|
||||
};
|
||||
|
||||
this.mediaRecorder.start(250);
|
||||
}
|
||||
|
||||
private _releaseStream(): void {
|
||||
if (this.stream) {
|
||||
this.stream.getTracks().forEach((t) => t.stop());
|
||||
this.stream = null;
|
||||
}
|
||||
}
|
||||
|
||||
// ── VAD ──────────────────────────────────────────────────────────────
|
||||
|
||||
private _startVAD(): void {
|
||||
this._stopVAD();
|
||||
this.silenceSince = null;
|
||||
this.isSpeaking = false;
|
||||
|
||||
this.vadTimer = setInterval(() => {
|
||||
if (!this.isActive || !this.analyser) return;
|
||||
const db = this._getRmsDb();
|
||||
const isSilent = db < this.cfg.silenceThresholdDb;
|
||||
|
||||
if (!isSilent) {
|
||||
this.silenceSince = null;
|
||||
if (!this.isSpeaking) {
|
||||
this.isSpeaking = true;
|
||||
if (this.mediaRecorder?.state === 'recording') {
|
||||
this.recordingStartMs = Date.now();
|
||||
}
|
||||
}
|
||||
} else {
|
||||
if (this.isSpeaking) {
|
||||
if (this.silenceSince === null) {
|
||||
this.silenceSince = Date.now();
|
||||
} else if (Date.now() - this.silenceSince >= this.cfg.silenceHoldMs) {
|
||||
this.isSpeaking = false;
|
||||
this.silenceSince = null;
|
||||
this._finaliseUtterance(true);
|
||||
}
|
||||
}
|
||||
}
|
||||
}, VAD_POLL_MS);
|
||||
}
|
||||
|
||||
private _stopVAD(): void {
|
||||
if (this.vadTimer !== null) {
|
||||
clearInterval(this.vadTimer);
|
||||
this.vadTimer = null;
|
||||
}
|
||||
}
|
||||
|
||||
private _cleanupAfterStop(clearChunks: boolean): void {
|
||||
const pendingResolver = this.finishResolver;
|
||||
this.isActive = false;
|
||||
this.finishResolver = null;
|
||||
this.mediaRecorder = null;
|
||||
this._teardownAudioContext();
|
||||
this._releaseStream();
|
||||
if (clearChunks) this.chunks = [];
|
||||
this.isSpeaking = false;
|
||||
this.silenceSince = null;
|
||||
this.onResult = null;
|
||||
this.onError = null;
|
||||
pendingResolver?.();
|
||||
}
|
||||
|
||||
private _finaliseUtterance(restart: boolean): void {
|
||||
if (!this.isActive) return;
|
||||
if (this.mediaRecorder && this.mediaRecorder.state === 'recording') {
|
||||
this.mediaRecorder.stop();
|
||||
}
|
||||
if (!restart) return;
|
||||
setTimeout(() => {
|
||||
if (this.isActive && this.stream) {
|
||||
this._startRecorder();
|
||||
}
|
||||
}, 100);
|
||||
}
|
||||
|
||||
private _getRmsDb(): number {
|
||||
if (!this.analyser) return -Infinity;
|
||||
const buf = new Float32Array(this.analyser.fftSize);
|
||||
this.analyser.getFloatTimeDomainData(buf);
|
||||
let sumSq = 0;
|
||||
for (const s of buf) sumSq += s * s;
|
||||
const rms = Math.sqrt(sumSq / buf.length);
|
||||
return rms === 0 ? -Infinity : 20 * Math.log10(rms);
|
||||
}
|
||||
|
||||
// ── Transcription ────────────────────────────────────────────────────
|
||||
|
||||
private async _transcribe(blob: Blob): Promise<void> {
|
||||
if (!this.onResult) return;
|
||||
if (!this.transcriber && !this.worker) {
|
||||
this.onError?.('Model not loaded');
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
const audioData = await this._decodeToFloat32(blob);
|
||||
if (!audioData || audioData.length === 0) {
|
||||
this.onError?.(`Failed to decode audio (${blob.size} bytes)`);
|
||||
return;
|
||||
}
|
||||
|
||||
const langHint = this._resolveLanguageHint();
|
||||
|
||||
// Prefer worker (non-blocking); fall back to main-thread pipeline.
|
||||
const transcript = this.worker
|
||||
? await this._transcribeViaWorker(audioData, langHint)
|
||||
: await this._transcribeMainThread(audioData, langHint);
|
||||
|
||||
if (transcript) {
|
||||
this.onResult(transcript, true);
|
||||
}
|
||||
} catch (err) {
|
||||
if (!this.isActive) return;
|
||||
const msg = err instanceof Error ? err.message : 'Local transcription failed';
|
||||
this.onError?.(msg);
|
||||
}
|
||||
}
|
||||
|
||||
private _transcribeViaWorker(audioData: Float32Array, langHint: string | undefined): Promise<string> {
|
||||
return new Promise((resolve, reject) => {
|
||||
if (!this.worker) return reject(new Error('Worker gone'));
|
||||
|
||||
const onMessage = (e: MessageEvent) => {
|
||||
const data = e.data as { type: string; error?: string; transcript?: string; text?: string };
|
||||
if (data.type === 'result') {
|
||||
this.worker!.removeEventListener('message', onMessage);
|
||||
resolve(data.transcript ?? '');
|
||||
} else if (data.type === 'log') {
|
||||
console.log('[WasmStt Worker]', data.text);
|
||||
} else if (data.type === 'error') {
|
||||
this.worker!.removeEventListener('message', onMessage);
|
||||
reject(new Error(data.error ?? 'Transcription failed'));
|
||||
}
|
||||
};
|
||||
|
||||
this.worker.addEventListener('message', onMessage);
|
||||
this.worker.postMessage(
|
||||
{ type: 'transcribe', audio: audioData.buffer, language: langHint },
|
||||
[audioData.buffer],
|
||||
);
|
||||
|
||||
setTimeout(() => {
|
||||
this.worker?.removeEventListener('message', onMessage);
|
||||
reject(new Error('Transcription timed out'));
|
||||
}, 30000);
|
||||
});
|
||||
}
|
||||
|
||||
private async _transcribeMainThread(audioData: Float32Array, langHint: string | undefined): Promise<string> {
|
||||
const pipelineFn = this.transcriber as (
|
||||
input: Float32Array,
|
||||
options?: Record<string, unknown>,
|
||||
) => Promise<{ text: string }>;
|
||||
|
||||
const result = await pipelineFn(audioData, {
|
||||
task: 'transcribe',
|
||||
...(langHint ? { language: langHint } : {}),
|
||||
});
|
||||
|
||||
return (result?.text ?? '').trim();
|
||||
}
|
||||
|
||||
private async _decodeToFloat32(blob: Blob): Promise<Float32Array | null> {
|
||||
if (!this.audioContext) return null;
|
||||
const arrayBuffer = await blob.arrayBuffer();
|
||||
let audioBuffer: AudioBuffer;
|
||||
try {
|
||||
audioBuffer = await this.audioContext.decodeAudioData(arrayBuffer);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
|
||||
const origRate = audioBuffer.sampleRate;
|
||||
const origData = audioBuffer.getChannelData(0);
|
||||
const targetRate = WHISPER_SAMPLE_RATE;
|
||||
|
||||
if (origRate === targetRate) {
|
||||
return new Float32Array(origData);
|
||||
}
|
||||
|
||||
const ratio = origRate / targetRate;
|
||||
const newLength = Math.ceil(origData.length / ratio);
|
||||
const result = new Float32Array(newLength);
|
||||
for (let i = 0; i < newLength; i++) {
|
||||
const origIdx = i * ratio;
|
||||
const idx0 = Math.floor(origIdx);
|
||||
const idx1 = Math.min(idx0 + 1, origData.length - 1);
|
||||
const frac = origIdx - idx0;
|
||||
result[i] = origData[idx0] * (1 - frac) + origData[idx1] * frac;
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
private _resolveLanguageHint(): string | undefined {
|
||||
if (this.lang && this.lang !== 'auto') {
|
||||
return this.lang.split('-')[0];
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
private _pickMimeType(): string {
|
||||
const candidates = [
|
||||
'audio/webm;codecs=opus',
|
||||
'audio/webm',
|
||||
'audio/ogg;codecs=opus',
|
||||
'audio/ogg',
|
||||
'audio/mp4',
|
||||
];
|
||||
if (typeof MediaRecorder !== 'undefined' && MediaRecorder.isTypeSupported) {
|
||||
return candidates.find((t) => MediaRecorder.isTypeSupported(t)) ?? '';
|
||||
}
|
||||
return '';
|
||||
}
|
||||
}
|
||||
|
||||
export const wasmSttService = new WasmSttService();
|
||||
@@ -1,92 +0,0 @@
|
||||
/**
|
||||
* Web Worker for off-main-thread Whisper transcription.
|
||||
*
|
||||
* Receives `{ type: 'load', modelId }` to load a model, then
|
||||
* `{ type: 'transcribe', audio: Float32Array (transferred buffer), language? }`
|
||||
* to run inference. Posts progress, results, and errors back.
|
||||
*/
|
||||
|
||||
import { pipeline, env } from '@xenova/transformers';
|
||||
|
||||
let transcriber: unknown = null;
|
||||
|
||||
self.onmessage = async (e: MessageEvent) => {
|
||||
const { type } = e.data as { type: string };
|
||||
|
||||
if (type === 'load') {
|
||||
const { modelId } = e.data as { modelId: string };
|
||||
try {
|
||||
env.backends.onnx.wasm.numThreads = 1;
|
||||
|
||||
const fileDoneBytes = new Map<string, number>();
|
||||
let totalDone = 0;
|
||||
let totalEstimate = 0;
|
||||
|
||||
transcriber = await pipeline('automatic-speech-recognition', modelId, {
|
||||
progress_callback: (info: { status?: string; file?: string; loaded?: number; total?: number }) => {
|
||||
if (info.status === 'progress' && info.file) {
|
||||
const prevDone = fileDoneBytes.get(info.file) ?? 0;
|
||||
const currentDone = info.loaded ?? 0;
|
||||
const delta = Math.max(0, currentDone - prevDone);
|
||||
fileDoneBytes.set(info.file, currentDone);
|
||||
totalDone += delta;
|
||||
|
||||
if (info.total && info.total > totalEstimate) {
|
||||
totalEstimate = info.total;
|
||||
}
|
||||
|
||||
const effectiveTotal = Math.max(totalEstimate, totalDone);
|
||||
const pct = effectiveTotal > 0 ? Math.min(100, Math.round((totalDone / effectiveTotal) * 100)) : 0;
|
||||
self.postMessage({ type: 'progress', progress: pct });
|
||||
}
|
||||
},
|
||||
});
|
||||
|
||||
self.postMessage({ type: 'loaded' });
|
||||
} catch (err) {
|
||||
self.postMessage({
|
||||
type: 'error',
|
||||
error: err instanceof Error ? err.message : 'Failed to load model',
|
||||
});
|
||||
}
|
||||
} else if (type === 'transcribe') {
|
||||
if (!transcriber) {
|
||||
self.postMessage({ type: 'error', error: 'Model not loaded', seq: (e.data as { seq?: number }).seq });
|
||||
return;
|
||||
}
|
||||
|
||||
const { audio, language, seq } = e.data as { audio: ArrayBuffer; language?: string; seq?: number };
|
||||
|
||||
try {
|
||||
const samples = new Float32Array(audio);
|
||||
if (samples.length === 0) {
|
||||
self.postMessage({ type: 'error', error: 'Empty audio received', seq });
|
||||
return;
|
||||
}
|
||||
|
||||
self.postMessage({ type: 'log', text: `Transcribing ${samples.length} samples (${(samples.length / 16000).toFixed(1)}s)` });
|
||||
|
||||
const pipelineFn = transcriber as (
|
||||
input: Float32Array,
|
||||
options?: Record<string, unknown>,
|
||||
) => Promise<{ text: string }>;
|
||||
|
||||
const result = await pipelineFn(samples, {
|
||||
task: 'transcribe',
|
||||
...(language ? { language } : {}),
|
||||
});
|
||||
|
||||
self.postMessage({
|
||||
type: 'result',
|
||||
transcript: (result?.text ?? '').trim(),
|
||||
seq,
|
||||
});
|
||||
} catch (err) {
|
||||
self.postMessage({
|
||||
type: 'error',
|
||||
error: err instanceof Error ? err.message : 'Transcription failed',
|
||||
seq,
|
||||
});
|
||||
}
|
||||
}
|
||||
};
|
||||
Reference in New Issue
Block a user