feat(voice): first-class voice input and local TTS across web, desktop, and mobile (#2018)

Complete rebuild of voice input on a server-authoritative streaming
architecture, replacing the legacy Web Speech / whole-blob / WASM engines
and the dead voice-agent layer (~4k lines removed).

Speech-to-text (dictation):
- Client streams 16 kHz mono PCM16 chunks over /api/dictation/ws with
  seq/ack ordering; buffered audio is retained and replayed on reconnect
- Server transcribes and streams live partial transcripts back;
  segments auto-commit every ~15s with silence suppression and adaptive
  finalization timeouts
- Local provider (default, zero config): sherpa-onnx models in a forked
  worker process — auto-download with progress, staged extraction with
  verification, corrupt-model auto-recovery, idle shutdown after 5 min
- Model catalog with settings picker (accuracy/speed ratings, sizes,
  download/delete): Parakeet TDT v2 (English) and v3 (25 European
  languages, auto-detected), Whisper base and tiny (multilingual, light)
- OpenAI-compatible provider for any Whisper endpoint
- Composer overlay with live transcript, volume meter, timer, and
  cancel / insert / insert-and-send actions; failed transcriptions keep
  their audio for retry or accepting the partial text as-is
- Configurable keyboard shortcut (default mod+alt+v) toggles dictation;
  Enter confirms and Escape cancels while recording
- Overlay is pixel-aligned with the composer (measured footer height,
  matching paddings/typography/gaps) — no layout shift when toggling

Text-to-speech:
- Local Kokoro provider (English, 11 voices) synthesized in the same
  worker via /api/dictation/tts/speak, managed by the shared model
  pipeline; sentence-pipelined playback keeps time-to-first-audio at
  ~1 sentence regardless of message length, and stop cancels in-flight
  synthesis
- Sanitizer keeps inline-code content (strips backticks only), reads
  interword slashes aloud, and removes only absolute file paths

Settings:
- Voice page unified: a single read-aloud toggle owns all playback
  options (the confusing "Enable Voice Mode" is gone); a new "Enable
  voice input" toggle (default on, persisted to settings.json) hides
  the composer mic entirely when disabled

Mobile and transport:
- iOS/Android microphone permissions added (dictation was previously
  impossible on mobile)
- Fixed Android WebSocket upgrades: the Capacitor WebView origin
  (https://localhost) was missing from the packaged-client allowlist,
  403-ing every WS connection — root cause of the old mobile SSE lock,
  which is now removed for all transports

Security and conventions:
- All HTTP routes sit behind the global /api auth gate; the WS upgrade
  explicitly validates the UI session and origin, with oc_url_token
  narrowly allowlisted and covered by tests; the dictation socket mints
  a fresh URL token before connecting
- Routes register before the generic OpenCode proxy; the client goes
  through runtimeFetch/getRuntimeUrlResolver, and runtime switches
  reset the dictation socket
- VS Code deliberately reports dictation as unavailable (no server
  process in that runtime)

CI: workflow Node bumped 20 -> 22 to match the repo engines and fix
better-sqlite3 installs broken by node-gyp@latest on Node 20.

New dependency: sherpa-onnx-node (prebuilt N-API; macOS/Linux x64+arm64,
Windows x64 — Windows-on-ARM falls back to the OpenAI-compatible provider)
This commit is contained in:
Bohdan Triapitsyn
2026-07-04 02:48:07 +03:00
committed by GitHub
parent 3f5151d424
commit de1b85ac56
89 changed files with 8740 additions and 6061 deletions
@@ -0,0 +1,447 @@
/**
* WebSocket client for the OpenChamber dictation endpoint (/api/dictation/ws).
*
* One shared client per app. The socket is opened lazily when a dictation
* starts and closed after an idle delay. URLs are resolved at connect time via
* the runtime URL resolver so runtime switches never leak a stale endpoint.
*/
import { getRuntimeUrlResolver } from '@/lib/runtime-url';
import { refreshRuntimeUrlAuthToken } from '@/lib/runtime-auth';
export interface DictationStartOptions {
provider?: 'local' | 'openai-compatible';
language?: string;
localModel?: string;
openaiCompatible?: {
baseUrl?: string;
model?: string;
apiKey?: string;
};
}
interface DictationServerMessage {
type: string;
dictationId?: string;
ackSeq?: number;
text?: string;
timeoutMs?: number;
error?: string;
retryable?: boolean;
reasonCode?: string;
}
interface DictationStreamError extends Error {
retryable: boolean;
reasonCode?: string;
}
const createStreamError = (message: string, retryable: boolean, reasonCode?: string): DictationStreamError => {
const error = new Error(message) as DictationStreamError;
error.name = 'DictationStreamError';
error.retryable = retryable;
if (reasonCode) {
error.reasonCode = reasonCode;
}
return error;
};
const CONNECT_TIMEOUT_MS = 10000;
const START_TIMEOUT_MS = 15000;
const IDLE_CLOSE_DELAY_MS = 30000;
const DEFAULT_FINISH_TIMEOUT_MS = 30000;
type ConnectionStatusListener = (connected: boolean) => void;
type PartialListener = (dictationId: string, text: string) => void;
interface PendingStart {
resolve: () => void;
reject: (error: Error) => void;
timeout: ReturnType<typeof setTimeout>;
}
interface PendingFinish {
resolve: (result: { text: string }) => void;
reject: (error: Error) => void;
timeout: ReturnType<typeof setTimeout> | null;
}
export class DictationClient {
private socket: WebSocket | null = null;
private connectPromise: Promise<void> | null = null;
private idleCloseTimer: ReturnType<typeof setTimeout> | null = null;
private readonly pendingStarts = new Map<string, PendingStart>();
private readonly pendingFinishes = new Map<string, PendingFinish>();
private readonly connectionListeners = new Set<ConnectionStatusListener>();
private readonly partialListeners = new Set<PartialListener>();
private activeDictations = 0;
get isConnected(): boolean {
return this.socket?.readyState === WebSocket.OPEN;
}
subscribeConnectionStatus(listener: ConnectionStatusListener): () => void {
this.connectionListeners.add(listener);
return () => {
this.connectionListeners.delete(listener);
};
}
onPartial(listener: PartialListener): () => void {
this.partialListeners.add(listener);
return () => {
this.partialListeners.delete(listener);
};
}
async ensureConnected(): Promise<void> {
if (this.isConnected) {
return;
}
if (this.connectPromise) {
await this.connectPromise;
return;
}
// A WebSocket upgrade can't carry an Authorization header, so it
// authenticates via the oc_url_token query param. Mint/await a valid
// token BEFORE connecting — the sync getter returns "" while the token
// is unminted or inside its expiry skew, and the server would reject
// the upgrade with 401.
try {
await refreshRuntimeUrlAuthToken();
} catch {
// No auth configured (local runtime) — proceed without a token.
}
this.connectPromise = new Promise<void>((resolve, reject) => {
let settled = false;
let socket: WebSocket;
try {
const url = getRuntimeUrlResolver().websocket('/api/dictation/ws');
socket = new WebSocket(url);
} catch (error) {
this.connectPromise = null;
reject(error instanceof Error ? error : new Error(String(error)));
return;
}
const timeout = setTimeout(() => {
if (!settled) {
settled = true;
this.connectPromise = null;
try {
socket.close();
} catch {
// ignore
}
reject(new Error('Dictation connection timed out'));
}
}, CONNECT_TIMEOUT_MS);
socket.onopen = () => {
// Wait for the server 'ready' frame before resolving.
};
socket.onmessage = (event) => {
let message: DictationServerMessage;
try {
message = JSON.parse(String(event.data));
} catch {
return;
}
if (!settled && message.type === 'ready') {
settled = true;
clearTimeout(timeout);
this.socket = socket;
this.connectPromise = null;
this.notifyConnection(true);
resolve();
return;
}
this.handleMessage(message);
};
socket.onerror = () => {
if (!settled) {
settled = true;
clearTimeout(timeout);
this.connectPromise = null;
reject(new Error('Dictation connection failed'));
}
};
socket.onclose = () => {
if (!settled) {
settled = true;
clearTimeout(timeout);
this.connectPromise = null;
reject(new Error('Dictation connection closed'));
return;
}
if (this.socket === socket) {
this.socket = null;
this.rejectAllPending(new Error('Dictation connection lost'));
this.notifyConnection(false);
}
};
});
await this.connectPromise;
}
/**
* Start a dictation stream. Resolves once the server acks the stream.
*/
async startDictationStream(
dictationId: string,
format: string,
options: DictationStartOptions,
): Promise<void> {
await this.ensureConnected();
this.activeDictations += 1;
this.clearIdleCloseTimer();
return new Promise<void>((resolve, reject) => {
const timeout = setTimeout(() => {
this.pendingStarts.delete(dictationId);
this.releaseDictation();
reject(new Error('Dictation start timed out'));
}, START_TIMEOUT_MS);
this.pendingStarts.set(dictationId, {
resolve: () => {
clearTimeout(timeout);
resolve();
},
reject: (error) => {
clearTimeout(timeout);
this.releaseDictation();
reject(error);
},
timeout });
if (!this.send({ type: 'start', dictationId, format, options })) {
clearTimeout(timeout);
this.pendingStarts.delete(dictationId);
this.releaseDictation();
reject(new Error('Dictation connection lost'));
}
});
}
sendDictationStreamChunk(dictationId: string, seq: number, audioBase64: string): boolean {
return this.send({ type: 'chunk', dictationId, seq, audio: audioBase64 });
}
/**
* Finish a dictation stream. Resolves with the final transcript.
*/
finishDictationStream(dictationId: string, finalSeq: number): Promise<{ text: string }> {
return new Promise<{ text: string }>((resolve, reject) => {
const pending: PendingFinish = {
resolve: (result) => {
if (pending.timeout) {
clearTimeout(pending.timeout);
}
this.releaseDictation();
resolve(result);
},
reject: (error) => {
if (pending.timeout) {
clearTimeout(pending.timeout);
}
this.releaseDictation();
reject(error);
},
timeout: null,
};
pending.timeout = setTimeout(() => {
this.pendingFinishes.delete(dictationId);
this.releaseDictation();
reject(new Error('Timed out waiting for transcription'));
}, DEFAULT_FINISH_TIMEOUT_MS);
this.pendingFinishes.set(dictationId, pending);
if (!this.send({ type: 'finish', dictationId, finalSeq })) {
this.pendingFinishes.delete(dictationId);
pending.reject(new Error('Dictation connection lost'));
}
});
}
cancelDictationStream(dictationId: string): void {
this.send({ type: 'cancel', dictationId });
const start = this.pendingStarts.get(dictationId);
if (start) {
this.pendingStarts.delete(dictationId);
start.reject(new Error('Dictation cancelled'));
}
const finish = this.pendingFinishes.get(dictationId);
if (finish) {
this.pendingFinishes.delete(dictationId);
finish.reject(new Error('Dictation cancelled'));
}
this.releaseDictation();
this.scheduleIdleCloseIfReady();
}
private handleMessage(message: DictationServerMessage): void {
const dictationId = message.dictationId;
if (!dictationId) {
return;
}
switch (message.type) {
case 'ack': {
const pendingStart = this.pendingStarts.get(dictationId);
if (pendingStart) {
this.pendingStarts.delete(dictationId);
pendingStart.resolve();
}
return;
}
case 'partial': {
for (const listener of this.partialListeners) {
listener(dictationId, message.text ?? '');
}
return;
}
case 'finish_accepted': {
const pendingFinish = this.pendingFinishes.get(dictationId);
if (pendingFinish && typeof message.timeoutMs === 'number') {
if (pendingFinish.timeout) {
clearTimeout(pendingFinish.timeout);
}
pendingFinish.timeout = setTimeout(() => {
this.pendingFinishes.delete(dictationId);
pendingFinish.reject(new Error('Timed out waiting for transcription'));
}, message.timeoutMs + 5000);
}
return;
}
case 'final': {
const pendingFinish = this.pendingFinishes.get(dictationId);
if (pendingFinish) {
this.pendingFinishes.delete(dictationId);
pendingFinish.resolve({ text: message.text ?? '' });
}
this.scheduleIdleCloseIfReady();
return;
}
case 'error': {
const error = createStreamError(
message.error || 'Dictation failed',
message.retryable !== false,
message.reasonCode,
);
const pendingStart = this.pendingStarts.get(dictationId);
if (pendingStart) {
this.pendingStarts.delete(dictationId);
pendingStart.reject(error);
}
const pendingFinish = this.pendingFinishes.get(dictationId);
if (pendingFinish) {
this.pendingFinishes.delete(dictationId);
pendingFinish.reject(error);
}
this.scheduleIdleCloseIfReady();
return;
}
default:
}
}
private send(message: object): boolean {
if (!this.isConnected || !this.socket) {
return false;
}
try {
this.socket.send(JSON.stringify(message));
return true;
} catch {
return false;
}
}
private notifyConnection(connected: boolean): void {
for (const listener of this.connectionListeners) {
listener(connected);
}
}
private rejectAllPending(error: Error): void {
for (const [dictationId, pending] of this.pendingStarts) {
this.pendingStarts.delete(dictationId);
pending.reject(error);
}
for (const [dictationId, pending] of this.pendingFinishes) {
this.pendingFinishes.delete(dictationId);
pending.reject(error);
}
this.activeDictations = 0;
}
private releaseDictation(): void {
this.activeDictations = Math.max(0, this.activeDictations - 1);
this.scheduleIdleCloseIfReady();
}
private scheduleIdleCloseIfReady(): void {
if (this.activeDictations > 0 || this.pendingFinishes.size > 0 || this.pendingStarts.size > 0) {
return;
}
this.clearIdleCloseTimer();
this.idleCloseTimer = setTimeout(() => {
if (this.activeDictations === 0 && this.pendingFinishes.size === 0 && this.pendingStarts.size === 0) {
const socket = this.socket;
this.socket = null;
if (socket) {
try {
socket.close(1000, 'idle');
} catch {
// ignore
}
this.notifyConnection(false);
}
}
}, IDLE_CLOSE_DELAY_MS);
}
/**
* Runtime switch: close the socket and fail all in-flight dictations so
* nothing keeps streaming to the previous runtime.
*/
cancelAllForRuntimeSwitch(): void {
this.clearIdleCloseTimer();
const socket = this.socket;
this.socket = null;
this.connectPromise = null;
this.rejectAllPending(new Error('Runtime changed'));
if (socket) {
try {
socket.close(1000, 'runtime switch');
} catch {
// ignore
}
this.notifyConnection(false);
}
}
private clearIdleCloseTimer(): void {
if (this.idleCloseTimer) {
clearTimeout(this.idleCloseTimer);
this.idleCloseTimer = null;
}
}
}
export const dictationClient = new DictationClient();
if (typeof window !== 'undefined') {
window.addEventListener('openchamber:runtime-endpoint-changed', () => {
// Drop the socket so the next dictation reconnects to the new runtime.
dictationClient.cancelAllForRuntimeSwitch();
});
}
@@ -0,0 +1,240 @@
/**
* Small, non-React state machine for dictation streaming.
*
* Responsibilities:
* - Maintain an ordered buffer of base64 PCM segments
* - Start/restart a dictation stream (dictationId)
* - Send missing segments (seq) when connected
* - Finish/cancel the stream
*
* Segments are retained until the dictation completes, which enables replay
* after a connection drop (`resetStreamForReplay()` + `finish()`).
*/
import type { DictationClient, DictationStartOptions } from './dictation-client';
const MAX_CHUNKS_PER_FLUSH_TURN = 128;
const PCM_DICTATION_FORMAT = 'audio/pcm;rate=16000;bits=16';
const waitForNextFlushTurn = (): Promise<void> => new Promise((resolve) => setTimeout(resolve, 0));
const createDictationIdDefault = (): string => {
const rand = Math.random().toString(36).slice(2, 10);
return `dic_${Date.now().toString(16)}${rand}`;
};
export interface DictationFinishResult {
dictationId: string;
text: string;
}
export class DictationStreamSender {
private readonly client: DictationClient;
private readonly format: string;
private readonly createDictationId: () => string;
private getStartOptions: () => DictationStartOptions;
private dictationId: string | null = null;
private sendSeq = 0;
private segments: string[] = [];
private streamReady = false;
private flushTimer: ReturnType<typeof setTimeout> | null = null;
private drainWaiters: Array<() => void> = [];
private startGeneration = 0;
private startPromise: Promise<void> | null = null;
constructor(params: {
client: DictationClient;
getStartOptions: () => DictationStartOptions;
format?: string;
createDictationId?: () => string;
}) {
this.client = params.client;
this.format = params.format ?? PCM_DICTATION_FORMAT;
this.getStartOptions = params.getStartOptions;
this.createDictationId = params.createDictationId ?? createDictationIdDefault;
}
getDictationId(): string | null {
return this.dictationId;
}
getFinalSeq(): number {
return this.segments.length - 1;
}
hasSegments(): boolean {
return this.segments.length > 0;
}
clearAll(): void {
this.clearScheduledFlush();
this.dictationId = null;
this.sendSeq = 0;
this.segments = [];
this.streamReady = false;
this.startPromise = null;
this.startGeneration += 1;
}
resetStreamForReplay(): void {
this.clearScheduledFlush();
this.dictationId = null;
this.sendSeq = 0;
this.streamReady = false;
this.startPromise = null;
this.startGeneration += 1;
}
enqueueSegment(base64Pcm: string): void {
this.segments.push(base64Pcm);
if (!this.client.isConnected) {
return;
}
if (!this.dictationId) {
if (!this.startPromise) {
void this.restartStream().catch(() => {
// Start failures surface through finish(); segments are retained.
});
}
return;
}
this.flush();
}
flush(): number {
const dictationId = this.dictationId;
if (!this.client.isConnected || !dictationId || !this.streamReady) {
return 0;
}
let sent = 0;
while (this.sendSeq < this.segments.length && sent < MAX_CHUNKS_PER_FLUSH_TURN) {
const seq = this.sendSeq;
const audio = this.segments[seq];
if (!this.client.sendDictationStreamChunk(dictationId, seq, audio)) {
break;
}
this.sendSeq = seq + 1;
sent += 1;
}
if (this.hasPendingSegments()) {
this.scheduleFlush();
} else {
this.resolveDrainWaiters();
}
return sent;
}
async restartStream(): Promise<void> {
this.startGeneration += 1;
const generation = this.startGeneration;
const dictationId = this.createDictationId();
this.dictationId = dictationId;
this.sendSeq = 0;
this.streamReady = false;
const start = (async () => {
await this.client.startDictationStream(dictationId, this.format, this.getStartOptions());
if (this.startGeneration !== generation) {
return;
}
if (this.dictationId !== dictationId) {
return;
}
this.streamReady = true;
this.flush();
})()
.catch((error) => {
// Keep segments for retry, but clear the stream so finish can error cleanly.
if (this.startGeneration === generation && this.dictationId === dictationId) {
this.dictationId = null;
this.streamReady = false;
}
throw error;
})
.finally(() => {
if (this.startPromise === start) {
this.startPromise = null;
}
});
this.startPromise = start;
await start;
}
async finish(finalSeq: number): Promise<DictationFinishResult> {
if (!this.dictationId) {
await this.restartStream();
}
if (this.startPromise) {
await this.startPromise;
}
const dictationId = this.dictationId;
if (!dictationId || !this.streamReady) {
throw new Error('Failed to start dictation stream');
}
this.flush();
await this.waitForFlushDrain();
const result = await this.client.finishDictationStream(dictationId, finalSeq);
return { dictationId, text: result.text };
}
cancel(): void {
const dictationId = this.dictationId;
if (this.client.isConnected && dictationId) {
this.client.cancelDictationStream(dictationId);
}
this.resetStreamForReplay();
}
private hasPendingSegments(): boolean {
return this.sendSeq < this.segments.length;
}
private scheduleFlush(): void {
if (this.flushTimer) {
return;
}
this.flushTimer = setTimeout(() => {
this.flushTimer = null;
this.flush();
}, 0);
}
private clearScheduledFlush(): void {
if (!this.flushTimer) {
return;
}
clearTimeout(this.flushTimer);
this.flushTimer = null;
}
private async waitForFlushDrain(): Promise<void> {
while (this.hasPendingSegments()) {
if (!this.client.isConnected || !this.dictationId || !this.streamReady) {
throw new Error('Failed to flush dictation stream');
}
await new Promise<void>((resolve) => {
this.drainWaiters.push(resolve);
});
await waitForNextFlushTurn();
}
}
private resolveDrainWaiters(): void {
const waiters = this.drainWaiters;
this.drainWaiters = [];
for (const resolve of waiters) {
resolve();
}
}
}
@@ -0,0 +1,301 @@
/**
* Microphone capture for dictation.
*
* Captures mono audio via getUserMedia, taps it with a ScriptProcessorNode
* (universally supported, including iOS WKWebView), resamples Float32 to
* 16 kHz PCM16LE, and emits ~1-second base64 chunks plus a normalized RMS
* volume for the level meter.
*/
import { useCallback, useEffect, useMemo, useRef, useState } from 'react';
export interface DictationAudioSourceConfig {
onPcmSegment: (base64Pcm: string) => void;
onError?: (error: Error) => void;
}
export interface DictationAudioSource {
start: () => Promise<void>;
stop: () => Promise<void>;
volume: number;
}
const OUTPUT_RATE = 16000;
const CHUNK_SAMPLES = OUTPUT_RATE; // ~1s per chunk
const getAudioContextCtor = (): typeof AudioContext | null => {
if (typeof window === 'undefined') {
return null;
}
const win = window as typeof window & { webkitAudioContext?: typeof AudioContext };
return win.AudioContext || win.webkitAudioContext || null;
};
const floatToInt16 = (sample: number): number => {
const clamped = Math.max(-1, Math.min(1, sample));
return clamped < 0 ? Math.round(clamped * 0x8000) : Math.round(clamped * 0x7fff);
};
const resampleToPcm16 = (input: Float32Array, inputRate: number, outputRate: number): Int16Array => {
if (input.length === 0) {
return new Int16Array(0);
}
if (inputRate === outputRate) {
const out = new Int16Array(input.length);
for (let i = 0; i < input.length; i++) {
out[i] = floatToInt16(input[i]);
}
return out;
}
const ratio = inputRate / outputRate;
const outputLength = Math.max(1, Math.round(input.length / ratio));
const out = new Int16Array(outputLength);
for (let i = 0; i < outputLength; i++) {
const sourceIndex = i * ratio;
const i0 = Math.floor(sourceIndex);
const i1 = Math.min(input.length - 1, i0 + 1);
const frac = sourceIndex - i0;
out[i] = floatToInt16(input[i0] * (1 - frac) + input[i1] * frac);
}
return out;
};
const concatInt16 = (a: Int16Array, b: Int16Array): Int16Array => {
if (a.length === 0) {
return b;
}
if (b.length === 0) {
return a;
}
const out = new Int16Array(a.length + b.length);
out.set(a, 0);
out.set(b, a.length);
return out;
};
const int16ToBase64 = (pcm: Int16Array): string => {
const bytes = new Uint8Array(pcm.buffer, pcm.byteOffset, pcm.byteLength);
let binary = '';
for (let i = 0; i < bytes.length; i++) {
binary += String.fromCharCode(bytes[i]);
}
return btoa(binary);
};
interface CaptureGraph {
stream: MediaStream | null;
context: AudioContext | null;
source: MediaStreamAudioSourceNode | null;
processor: ScriptProcessorNode | null;
gain: GainNode | null;
pending: Int16Array;
started: boolean;
}
const emptyGraph = (): CaptureGraph => ({
stream: null,
context: null,
source: null,
processor: null,
gain: null,
pending: new Int16Array(0),
started: false,
});
const safeDisconnect = (node: AudioNode | null): void => {
if (!node) {
return;
}
try {
node.disconnect();
} catch {
// no-op
}
};
export const isDictationCaptureSupported = (): boolean => {
if (typeof navigator === 'undefined' || typeof window === 'undefined') {
return false;
}
if (!navigator.mediaDevices || typeof navigator.mediaDevices.getUserMedia !== 'function') {
return false;
}
return getAudioContextCtor() !== null;
};
export function useDictationAudioSource(config: DictationAudioSourceConfig): DictationAudioSource {
const [volume, setVolume] = useState(0);
const onPcmSegmentRef = useRef(config.onPcmSegment);
const onErrorRef = useRef(config.onError);
useEffect(() => {
onPcmSegmentRef.current = config.onPcmSegment;
onErrorRef.current = config.onError;
}, [config.onPcmSegment, config.onError]);
const graphRef = useRef<CaptureGraph>(emptyGraph());
const start = useCallback(async () => {
if (graphRef.current.started) {
return;
}
if (
typeof navigator === 'undefined' ||
!navigator.mediaDevices ||
typeof navigator.mediaDevices.getUserMedia !== 'function'
) {
throw new Error('Microphone capture is not supported in this environment');
}
const AudioContextCtor = getAudioContextCtor();
if (!AudioContextCtor) {
throw new Error('AudioContext unavailable');
}
const stream = await navigator.mediaDevices.getUserMedia({
audio: {
channelCount: 1,
noiseSuppression: true,
echoCancellation: true,
autoGainControl: true,
},
});
const context = new AudioContextCtor();
try {
if (context.state === 'suspended') {
await context.resume().catch(() => undefined);
}
const source = context.createMediaStreamSource(stream);
const processor = context.createScriptProcessor(4096, 1, 1);
const gain = context.createGain();
gain.gain.value = 0;
graphRef.current = {
stream,
context,
source,
processor,
gain,
pending: new Int16Array(0),
started: true,
};
processor.onaudioprocess = (event) => {
const graph = graphRef.current;
if (!graph.started) {
return;
}
const input = event.inputBuffer.getChannelData(0);
let sumSquares = 0;
for (let i = 0; i < input.length; i++) {
sumSquares += input[i] * input[i];
}
const rms = Math.sqrt(sumSquares / Math.max(1, input.length));
setVolume(Math.min(1, Math.max(0, rms * 2)));
const next = resampleToPcm16(input, context.sampleRate, OUTPUT_RATE);
graph.pending = concatInt16(graph.pending, next);
while (graph.pending.length >= CHUNK_SAMPLES) {
const chunk = graph.pending.slice(0, CHUNK_SAMPLES);
graph.pending = graph.pending.slice(CHUNK_SAMPLES);
onPcmSegmentRef.current(int16ToBase64(chunk));
}
};
source.connect(processor);
processor.connect(gain);
gain.connect(context.destination);
} catch (error) {
stream.getTracks().forEach((track) => {
try {
track.stop();
} catch {
// no-op
}
});
try {
await context.close();
} catch {
// no-op
}
graphRef.current = emptyGraph();
throw error instanceof Error ? error : new Error(String(error));
}
}, []);
const stop = useCallback(async () => {
const graph = graphRef.current;
graph.started = false;
setVolume(0);
if (graph.processor) {
try {
graph.processor.onaudioprocess = null;
} catch {
// no-op
}
}
safeDisconnect(graph.processor);
safeDisconnect(graph.source);
safeDisconnect(graph.gain);
if (graph.stream) {
graph.stream.getTracks().forEach((track) => {
try {
track.stop();
} catch {
// no-op
}
});
}
const pending = graph.pending;
graph.pending = new Int16Array(0);
if (pending.length > 0) {
onPcmSegmentRef.current(int16ToBase64(pending));
}
if (graph.context) {
try {
await graph.context.close();
} catch {
// no-op
}
}
// A new capture may have started while the old context was closing;
// only clear the ref if it still points at the graph we tore down.
if (graphRef.current === graph) {
graphRef.current = emptyGraph();
}
}, []);
useEffect(() => {
return () => {
void stop().catch((err) => {
onErrorRef.current?.(err instanceof Error ? err : new Error(String(err)));
});
};
}, [stop]);
return useMemo(
() => ({
start: async () => {
try {
await start();
} catch (err) {
const normalized = err instanceof Error ? err : new Error(String(err));
onErrorRef.current?.(normalized);
throw normalized;
}
},
stop,
volume,
}),
[start, stop, volume],
);
}