feat(voice): first-class voice input and local TTS across web, desktop, and mobile (#2018)
Complete rebuild of voice input on a server-authoritative streaming architecture, replacing the legacy Web Speech / whole-blob / WASM engines and the dead voice-agent layer (~4k lines removed). Speech-to-text (dictation): - Client streams 16 kHz mono PCM16 chunks over /api/dictation/ws with seq/ack ordering; buffered audio is retained and replayed on reconnect - Server transcribes and streams live partial transcripts back; segments auto-commit every ~15s with silence suppression and adaptive finalization timeouts - Local provider (default, zero config): sherpa-onnx models in a forked worker process — auto-download with progress, staged extraction with verification, corrupt-model auto-recovery, idle shutdown after 5 min - Model catalog with settings picker (accuracy/speed ratings, sizes, download/delete): Parakeet TDT v2 (English) and v3 (25 European languages, auto-detected), Whisper base and tiny (multilingual, light) - OpenAI-compatible provider for any Whisper endpoint - Composer overlay with live transcript, volume meter, timer, and cancel / insert / insert-and-send actions; failed transcriptions keep their audio for retry or accepting the partial text as-is - Configurable keyboard shortcut (default mod+alt+v) toggles dictation; Enter confirms and Escape cancels while recording - Overlay is pixel-aligned with the composer (measured footer height, matching paddings/typography/gaps) — no layout shift when toggling Text-to-speech: - Local Kokoro provider (English, 11 voices) synthesized in the same worker via /api/dictation/tts/speak, managed by the shared model pipeline; sentence-pipelined playback keeps time-to-first-audio at ~1 sentence regardless of message length, and stop cancels in-flight synthesis - Sanitizer keeps inline-code content (strips backticks only), reads interword slashes aloud, and removes only absolute file paths Settings: - Voice page unified: a single read-aloud toggle owns all playback options (the confusing "Enable Voice Mode" is gone); a new "Enable voice input" toggle (default on, persisted to settings.json) hides the composer mic entirely when disabled Mobile and transport: - iOS/Android microphone permissions added (dictation was previously impossible on mobile) - Fixed Android WebSocket upgrades: the Capacitor WebView origin (https://localhost) was missing from the packaged-client allowlist, 403-ing every WS connection — root cause of the old mobile SSE lock, which is now removed for all transports Security and conventions: - All HTTP routes sit behind the global /api auth gate; the WS upgrade explicitly validates the UI session and origin, with oc_url_token narrowly allowlisted and covered by tests; the dictation socket mints a fresh URL token before connecting - Routes register before the generic OpenCode proxy; the client goes through runtimeFetch/getRuntimeUrlResolver, and runtime switches reset the dictation socket - VS Code deliberately reports dictation as unavailable (no server process in that runtime) CI: workflow Node bumped 20 -> 22 to match the repo engines and fix better-sqlite3 installs broken by node-gyp@latest on Node 20. New dependency: sherpa-onnx-node (prebuilt N-API; macOS/Linux x64+arm64, Windows x64 — Windows-on-ARM falls back to the OpenAI-compatible provider)
This commit is contained in:
committed by
GitHub
parent
3f5151d424
commit
de1b85ac56
@@ -0,0 +1,447 @@
|
||||
/**
|
||||
* WebSocket client for the OpenChamber dictation endpoint (/api/dictation/ws).
|
||||
*
|
||||
* One shared client per app. The socket is opened lazily when a dictation
|
||||
* starts and closed after an idle delay. URLs are resolved at connect time via
|
||||
* the runtime URL resolver so runtime switches never leak a stale endpoint.
|
||||
*/
|
||||
|
||||
import { getRuntimeUrlResolver } from '@/lib/runtime-url';
|
||||
import { refreshRuntimeUrlAuthToken } from '@/lib/runtime-auth';
|
||||
|
||||
export interface DictationStartOptions {
|
||||
provider?: 'local' | 'openai-compatible';
|
||||
language?: string;
|
||||
localModel?: string;
|
||||
openaiCompatible?: {
|
||||
baseUrl?: string;
|
||||
model?: string;
|
||||
apiKey?: string;
|
||||
};
|
||||
}
|
||||
|
||||
interface DictationServerMessage {
|
||||
type: string;
|
||||
dictationId?: string;
|
||||
ackSeq?: number;
|
||||
text?: string;
|
||||
timeoutMs?: number;
|
||||
error?: string;
|
||||
retryable?: boolean;
|
||||
reasonCode?: string;
|
||||
}
|
||||
|
||||
interface DictationStreamError extends Error {
|
||||
retryable: boolean;
|
||||
reasonCode?: string;
|
||||
}
|
||||
|
||||
const createStreamError = (message: string, retryable: boolean, reasonCode?: string): DictationStreamError => {
|
||||
const error = new Error(message) as DictationStreamError;
|
||||
error.name = 'DictationStreamError';
|
||||
error.retryable = retryable;
|
||||
if (reasonCode) {
|
||||
error.reasonCode = reasonCode;
|
||||
}
|
||||
return error;
|
||||
};
|
||||
|
||||
const CONNECT_TIMEOUT_MS = 10000;
|
||||
const START_TIMEOUT_MS = 15000;
|
||||
const IDLE_CLOSE_DELAY_MS = 30000;
|
||||
const DEFAULT_FINISH_TIMEOUT_MS = 30000;
|
||||
|
||||
type ConnectionStatusListener = (connected: boolean) => void;
|
||||
type PartialListener = (dictationId: string, text: string) => void;
|
||||
|
||||
interface PendingStart {
|
||||
resolve: () => void;
|
||||
reject: (error: Error) => void;
|
||||
timeout: ReturnType<typeof setTimeout>;
|
||||
}
|
||||
|
||||
interface PendingFinish {
|
||||
resolve: (result: { text: string }) => void;
|
||||
reject: (error: Error) => void;
|
||||
timeout: ReturnType<typeof setTimeout> | null;
|
||||
}
|
||||
|
||||
export class DictationClient {
|
||||
private socket: WebSocket | null = null;
|
||||
private connectPromise: Promise<void> | null = null;
|
||||
private idleCloseTimer: ReturnType<typeof setTimeout> | null = null;
|
||||
private readonly pendingStarts = new Map<string, PendingStart>();
|
||||
private readonly pendingFinishes = new Map<string, PendingFinish>();
|
||||
private readonly connectionListeners = new Set<ConnectionStatusListener>();
|
||||
private readonly partialListeners = new Set<PartialListener>();
|
||||
private activeDictations = 0;
|
||||
|
||||
get isConnected(): boolean {
|
||||
return this.socket?.readyState === WebSocket.OPEN;
|
||||
}
|
||||
|
||||
subscribeConnectionStatus(listener: ConnectionStatusListener): () => void {
|
||||
this.connectionListeners.add(listener);
|
||||
return () => {
|
||||
this.connectionListeners.delete(listener);
|
||||
};
|
||||
}
|
||||
|
||||
onPartial(listener: PartialListener): () => void {
|
||||
this.partialListeners.add(listener);
|
||||
return () => {
|
||||
this.partialListeners.delete(listener);
|
||||
};
|
||||
}
|
||||
|
||||
async ensureConnected(): Promise<void> {
|
||||
if (this.isConnected) {
|
||||
return;
|
||||
}
|
||||
if (this.connectPromise) {
|
||||
await this.connectPromise;
|
||||
return;
|
||||
}
|
||||
|
||||
// A WebSocket upgrade can't carry an Authorization header, so it
|
||||
// authenticates via the oc_url_token query param. Mint/await a valid
|
||||
// token BEFORE connecting — the sync getter returns "" while the token
|
||||
// is unminted or inside its expiry skew, and the server would reject
|
||||
// the upgrade with 401.
|
||||
try {
|
||||
await refreshRuntimeUrlAuthToken();
|
||||
} catch {
|
||||
// No auth configured (local runtime) — proceed without a token.
|
||||
}
|
||||
|
||||
this.connectPromise = new Promise<void>((resolve, reject) => {
|
||||
let settled = false;
|
||||
let socket: WebSocket;
|
||||
try {
|
||||
const url = getRuntimeUrlResolver().websocket('/api/dictation/ws');
|
||||
socket = new WebSocket(url);
|
||||
} catch (error) {
|
||||
this.connectPromise = null;
|
||||
reject(error instanceof Error ? error : new Error(String(error)));
|
||||
return;
|
||||
}
|
||||
|
||||
const timeout = setTimeout(() => {
|
||||
if (!settled) {
|
||||
settled = true;
|
||||
this.connectPromise = null;
|
||||
try {
|
||||
socket.close();
|
||||
} catch {
|
||||
// ignore
|
||||
}
|
||||
reject(new Error('Dictation connection timed out'));
|
||||
}
|
||||
}, CONNECT_TIMEOUT_MS);
|
||||
|
||||
socket.onopen = () => {
|
||||
// Wait for the server 'ready' frame before resolving.
|
||||
};
|
||||
|
||||
socket.onmessage = (event) => {
|
||||
let message: DictationServerMessage;
|
||||
try {
|
||||
message = JSON.parse(String(event.data));
|
||||
} catch {
|
||||
return;
|
||||
}
|
||||
if (!settled && message.type === 'ready') {
|
||||
settled = true;
|
||||
clearTimeout(timeout);
|
||||
this.socket = socket;
|
||||
this.connectPromise = null;
|
||||
this.notifyConnection(true);
|
||||
resolve();
|
||||
return;
|
||||
}
|
||||
this.handleMessage(message);
|
||||
};
|
||||
|
||||
socket.onerror = () => {
|
||||
if (!settled) {
|
||||
settled = true;
|
||||
clearTimeout(timeout);
|
||||
this.connectPromise = null;
|
||||
reject(new Error('Dictation connection failed'));
|
||||
}
|
||||
};
|
||||
|
||||
socket.onclose = () => {
|
||||
if (!settled) {
|
||||
settled = true;
|
||||
clearTimeout(timeout);
|
||||
this.connectPromise = null;
|
||||
reject(new Error('Dictation connection closed'));
|
||||
return;
|
||||
}
|
||||
if (this.socket === socket) {
|
||||
this.socket = null;
|
||||
this.rejectAllPending(new Error('Dictation connection lost'));
|
||||
this.notifyConnection(false);
|
||||
}
|
||||
};
|
||||
});
|
||||
|
||||
await this.connectPromise;
|
||||
}
|
||||
|
||||
/**
|
||||
* Start a dictation stream. Resolves once the server acks the stream.
|
||||
*/
|
||||
async startDictationStream(
|
||||
dictationId: string,
|
||||
format: string,
|
||||
options: DictationStartOptions,
|
||||
): Promise<void> {
|
||||
await this.ensureConnected();
|
||||
this.activeDictations += 1;
|
||||
this.clearIdleCloseTimer();
|
||||
|
||||
return new Promise<void>((resolve, reject) => {
|
||||
const timeout = setTimeout(() => {
|
||||
this.pendingStarts.delete(dictationId);
|
||||
this.releaseDictation();
|
||||
reject(new Error('Dictation start timed out'));
|
||||
}, START_TIMEOUT_MS);
|
||||
|
||||
this.pendingStarts.set(dictationId, {
|
||||
resolve: () => {
|
||||
clearTimeout(timeout);
|
||||
resolve();
|
||||
},
|
||||
reject: (error) => {
|
||||
clearTimeout(timeout);
|
||||
this.releaseDictation();
|
||||
reject(error);
|
||||
},
|
||||
timeout });
|
||||
|
||||
if (!this.send({ type: 'start', dictationId, format, options })) {
|
||||
clearTimeout(timeout);
|
||||
this.pendingStarts.delete(dictationId);
|
||||
this.releaseDictation();
|
||||
reject(new Error('Dictation connection lost'));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
sendDictationStreamChunk(dictationId: string, seq: number, audioBase64: string): boolean {
|
||||
return this.send({ type: 'chunk', dictationId, seq, audio: audioBase64 });
|
||||
}
|
||||
|
||||
/**
|
||||
* Finish a dictation stream. Resolves with the final transcript.
|
||||
*/
|
||||
finishDictationStream(dictationId: string, finalSeq: number): Promise<{ text: string }> {
|
||||
return new Promise<{ text: string }>((resolve, reject) => {
|
||||
const pending: PendingFinish = {
|
||||
resolve: (result) => {
|
||||
if (pending.timeout) {
|
||||
clearTimeout(pending.timeout);
|
||||
}
|
||||
this.releaseDictation();
|
||||
resolve(result);
|
||||
},
|
||||
reject: (error) => {
|
||||
if (pending.timeout) {
|
||||
clearTimeout(pending.timeout);
|
||||
}
|
||||
this.releaseDictation();
|
||||
reject(error);
|
||||
},
|
||||
timeout: null,
|
||||
};
|
||||
pending.timeout = setTimeout(() => {
|
||||
this.pendingFinishes.delete(dictationId);
|
||||
this.releaseDictation();
|
||||
reject(new Error('Timed out waiting for transcription'));
|
||||
}, DEFAULT_FINISH_TIMEOUT_MS);
|
||||
|
||||
this.pendingFinishes.set(dictationId, pending);
|
||||
|
||||
if (!this.send({ type: 'finish', dictationId, finalSeq })) {
|
||||
this.pendingFinishes.delete(dictationId);
|
||||
pending.reject(new Error('Dictation connection lost'));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
cancelDictationStream(dictationId: string): void {
|
||||
this.send({ type: 'cancel', dictationId });
|
||||
const start = this.pendingStarts.get(dictationId);
|
||||
if (start) {
|
||||
this.pendingStarts.delete(dictationId);
|
||||
start.reject(new Error('Dictation cancelled'));
|
||||
}
|
||||
const finish = this.pendingFinishes.get(dictationId);
|
||||
if (finish) {
|
||||
this.pendingFinishes.delete(dictationId);
|
||||
finish.reject(new Error('Dictation cancelled'));
|
||||
}
|
||||
this.releaseDictation();
|
||||
this.scheduleIdleCloseIfReady();
|
||||
}
|
||||
|
||||
private handleMessage(message: DictationServerMessage): void {
|
||||
const dictationId = message.dictationId;
|
||||
if (!dictationId) {
|
||||
return;
|
||||
}
|
||||
|
||||
switch (message.type) {
|
||||
case 'ack': {
|
||||
const pendingStart = this.pendingStarts.get(dictationId);
|
||||
if (pendingStart) {
|
||||
this.pendingStarts.delete(dictationId);
|
||||
pendingStart.resolve();
|
||||
}
|
||||
return;
|
||||
}
|
||||
case 'partial': {
|
||||
for (const listener of this.partialListeners) {
|
||||
listener(dictationId, message.text ?? '');
|
||||
}
|
||||
return;
|
||||
}
|
||||
case 'finish_accepted': {
|
||||
const pendingFinish = this.pendingFinishes.get(dictationId);
|
||||
if (pendingFinish && typeof message.timeoutMs === 'number') {
|
||||
if (pendingFinish.timeout) {
|
||||
clearTimeout(pendingFinish.timeout);
|
||||
}
|
||||
pendingFinish.timeout = setTimeout(() => {
|
||||
this.pendingFinishes.delete(dictationId);
|
||||
pendingFinish.reject(new Error('Timed out waiting for transcription'));
|
||||
}, message.timeoutMs + 5000);
|
||||
}
|
||||
return;
|
||||
}
|
||||
case 'final': {
|
||||
const pendingFinish = this.pendingFinishes.get(dictationId);
|
||||
if (pendingFinish) {
|
||||
this.pendingFinishes.delete(dictationId);
|
||||
pendingFinish.resolve({ text: message.text ?? '' });
|
||||
}
|
||||
this.scheduleIdleCloseIfReady();
|
||||
return;
|
||||
}
|
||||
case 'error': {
|
||||
const error = createStreamError(
|
||||
message.error || 'Dictation failed',
|
||||
message.retryable !== false,
|
||||
message.reasonCode,
|
||||
);
|
||||
const pendingStart = this.pendingStarts.get(dictationId);
|
||||
if (pendingStart) {
|
||||
this.pendingStarts.delete(dictationId);
|
||||
pendingStart.reject(error);
|
||||
}
|
||||
const pendingFinish = this.pendingFinishes.get(dictationId);
|
||||
if (pendingFinish) {
|
||||
this.pendingFinishes.delete(dictationId);
|
||||
pendingFinish.reject(error);
|
||||
}
|
||||
this.scheduleIdleCloseIfReady();
|
||||
return;
|
||||
}
|
||||
default:
|
||||
}
|
||||
}
|
||||
|
||||
private send(message: object): boolean {
|
||||
if (!this.isConnected || !this.socket) {
|
||||
return false;
|
||||
}
|
||||
try {
|
||||
this.socket.send(JSON.stringify(message));
|
||||
return true;
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
private notifyConnection(connected: boolean): void {
|
||||
for (const listener of this.connectionListeners) {
|
||||
listener(connected);
|
||||
}
|
||||
}
|
||||
|
||||
private rejectAllPending(error: Error): void {
|
||||
for (const [dictationId, pending] of this.pendingStarts) {
|
||||
this.pendingStarts.delete(dictationId);
|
||||
pending.reject(error);
|
||||
}
|
||||
for (const [dictationId, pending] of this.pendingFinishes) {
|
||||
this.pendingFinishes.delete(dictationId);
|
||||
pending.reject(error);
|
||||
}
|
||||
this.activeDictations = 0;
|
||||
}
|
||||
|
||||
private releaseDictation(): void {
|
||||
this.activeDictations = Math.max(0, this.activeDictations - 1);
|
||||
this.scheduleIdleCloseIfReady();
|
||||
}
|
||||
|
||||
private scheduleIdleCloseIfReady(): void {
|
||||
if (this.activeDictations > 0 || this.pendingFinishes.size > 0 || this.pendingStarts.size > 0) {
|
||||
return;
|
||||
}
|
||||
this.clearIdleCloseTimer();
|
||||
this.idleCloseTimer = setTimeout(() => {
|
||||
if (this.activeDictations === 0 && this.pendingFinishes.size === 0 && this.pendingStarts.size === 0) {
|
||||
const socket = this.socket;
|
||||
this.socket = null;
|
||||
if (socket) {
|
||||
try {
|
||||
socket.close(1000, 'idle');
|
||||
} catch {
|
||||
// ignore
|
||||
}
|
||||
this.notifyConnection(false);
|
||||
}
|
||||
}
|
||||
}, IDLE_CLOSE_DELAY_MS);
|
||||
}
|
||||
|
||||
/**
|
||||
* Runtime switch: close the socket and fail all in-flight dictations so
|
||||
* nothing keeps streaming to the previous runtime.
|
||||
*/
|
||||
cancelAllForRuntimeSwitch(): void {
|
||||
this.clearIdleCloseTimer();
|
||||
const socket = this.socket;
|
||||
this.socket = null;
|
||||
this.connectPromise = null;
|
||||
this.rejectAllPending(new Error('Runtime changed'));
|
||||
if (socket) {
|
||||
try {
|
||||
socket.close(1000, 'runtime switch');
|
||||
} catch {
|
||||
// ignore
|
||||
}
|
||||
this.notifyConnection(false);
|
||||
}
|
||||
}
|
||||
|
||||
private clearIdleCloseTimer(): void {
|
||||
if (this.idleCloseTimer) {
|
||||
clearTimeout(this.idleCloseTimer);
|
||||
this.idleCloseTimer = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
export const dictationClient = new DictationClient();
|
||||
|
||||
if (typeof window !== 'undefined') {
|
||||
window.addEventListener('openchamber:runtime-endpoint-changed', () => {
|
||||
// Drop the socket so the next dictation reconnects to the new runtime.
|
||||
dictationClient.cancelAllForRuntimeSwitch();
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,240 @@
|
||||
/**
|
||||
* Small, non-React state machine for dictation streaming.
|
||||
*
|
||||
* Responsibilities:
|
||||
* - Maintain an ordered buffer of base64 PCM segments
|
||||
* - Start/restart a dictation stream (dictationId)
|
||||
* - Send missing segments (seq) when connected
|
||||
* - Finish/cancel the stream
|
||||
*
|
||||
* Segments are retained until the dictation completes, which enables replay
|
||||
* after a connection drop (`resetStreamForReplay()` + `finish()`).
|
||||
*/
|
||||
|
||||
import type { DictationClient, DictationStartOptions } from './dictation-client';
|
||||
|
||||
const MAX_CHUNKS_PER_FLUSH_TURN = 128;
|
||||
|
||||
const PCM_DICTATION_FORMAT = 'audio/pcm;rate=16000;bits=16';
|
||||
|
||||
const waitForNextFlushTurn = (): Promise<void> => new Promise((resolve) => setTimeout(resolve, 0));
|
||||
|
||||
const createDictationIdDefault = (): string => {
|
||||
const rand = Math.random().toString(36).slice(2, 10);
|
||||
return `dic_${Date.now().toString(16)}${rand}`;
|
||||
};
|
||||
|
||||
export interface DictationFinishResult {
|
||||
dictationId: string;
|
||||
text: string;
|
||||
}
|
||||
|
||||
export class DictationStreamSender {
|
||||
private readonly client: DictationClient;
|
||||
private readonly format: string;
|
||||
private readonly createDictationId: () => string;
|
||||
private getStartOptions: () => DictationStartOptions;
|
||||
|
||||
private dictationId: string | null = null;
|
||||
private sendSeq = 0;
|
||||
private segments: string[] = [];
|
||||
private streamReady = false;
|
||||
private flushTimer: ReturnType<typeof setTimeout> | null = null;
|
||||
private drainWaiters: Array<() => void> = [];
|
||||
|
||||
private startGeneration = 0;
|
||||
private startPromise: Promise<void> | null = null;
|
||||
|
||||
constructor(params: {
|
||||
client: DictationClient;
|
||||
getStartOptions: () => DictationStartOptions;
|
||||
format?: string;
|
||||
createDictationId?: () => string;
|
||||
}) {
|
||||
this.client = params.client;
|
||||
this.format = params.format ?? PCM_DICTATION_FORMAT;
|
||||
this.getStartOptions = params.getStartOptions;
|
||||
this.createDictationId = params.createDictationId ?? createDictationIdDefault;
|
||||
}
|
||||
|
||||
getDictationId(): string | null {
|
||||
return this.dictationId;
|
||||
}
|
||||
|
||||
getFinalSeq(): number {
|
||||
return this.segments.length - 1;
|
||||
}
|
||||
|
||||
hasSegments(): boolean {
|
||||
return this.segments.length > 0;
|
||||
}
|
||||
|
||||
clearAll(): void {
|
||||
this.clearScheduledFlush();
|
||||
this.dictationId = null;
|
||||
this.sendSeq = 0;
|
||||
this.segments = [];
|
||||
this.streamReady = false;
|
||||
this.startPromise = null;
|
||||
this.startGeneration += 1;
|
||||
}
|
||||
|
||||
resetStreamForReplay(): void {
|
||||
this.clearScheduledFlush();
|
||||
this.dictationId = null;
|
||||
this.sendSeq = 0;
|
||||
this.streamReady = false;
|
||||
this.startPromise = null;
|
||||
this.startGeneration += 1;
|
||||
}
|
||||
|
||||
enqueueSegment(base64Pcm: string): void {
|
||||
this.segments.push(base64Pcm);
|
||||
|
||||
if (!this.client.isConnected) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (!this.dictationId) {
|
||||
if (!this.startPromise) {
|
||||
void this.restartStream().catch(() => {
|
||||
// Start failures surface through finish(); segments are retained.
|
||||
});
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
this.flush();
|
||||
}
|
||||
|
||||
flush(): number {
|
||||
const dictationId = this.dictationId;
|
||||
if (!this.client.isConnected || !dictationId || !this.streamReady) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
let sent = 0;
|
||||
while (this.sendSeq < this.segments.length && sent < MAX_CHUNKS_PER_FLUSH_TURN) {
|
||||
const seq = this.sendSeq;
|
||||
const audio = this.segments[seq];
|
||||
if (!this.client.sendDictationStreamChunk(dictationId, seq, audio)) {
|
||||
break;
|
||||
}
|
||||
this.sendSeq = seq + 1;
|
||||
sent += 1;
|
||||
}
|
||||
if (this.hasPendingSegments()) {
|
||||
this.scheduleFlush();
|
||||
} else {
|
||||
this.resolveDrainWaiters();
|
||||
}
|
||||
return sent;
|
||||
}
|
||||
|
||||
async restartStream(): Promise<void> {
|
||||
this.startGeneration += 1;
|
||||
const generation = this.startGeneration;
|
||||
|
||||
const dictationId = this.createDictationId();
|
||||
this.dictationId = dictationId;
|
||||
this.sendSeq = 0;
|
||||
this.streamReady = false;
|
||||
|
||||
const start = (async () => {
|
||||
await this.client.startDictationStream(dictationId, this.format, this.getStartOptions());
|
||||
if (this.startGeneration !== generation) {
|
||||
return;
|
||||
}
|
||||
if (this.dictationId !== dictationId) {
|
||||
return;
|
||||
}
|
||||
this.streamReady = true;
|
||||
this.flush();
|
||||
})()
|
||||
.catch((error) => {
|
||||
// Keep segments for retry, but clear the stream so finish can error cleanly.
|
||||
if (this.startGeneration === generation && this.dictationId === dictationId) {
|
||||
this.dictationId = null;
|
||||
this.streamReady = false;
|
||||
}
|
||||
throw error;
|
||||
})
|
||||
.finally(() => {
|
||||
if (this.startPromise === start) {
|
||||
this.startPromise = null;
|
||||
}
|
||||
});
|
||||
|
||||
this.startPromise = start;
|
||||
await start;
|
||||
}
|
||||
|
||||
async finish(finalSeq: number): Promise<DictationFinishResult> {
|
||||
if (!this.dictationId) {
|
||||
await this.restartStream();
|
||||
}
|
||||
if (this.startPromise) {
|
||||
await this.startPromise;
|
||||
}
|
||||
|
||||
const dictationId = this.dictationId;
|
||||
if (!dictationId || !this.streamReady) {
|
||||
throw new Error('Failed to start dictation stream');
|
||||
}
|
||||
|
||||
this.flush();
|
||||
await this.waitForFlushDrain();
|
||||
const result = await this.client.finishDictationStream(dictationId, finalSeq);
|
||||
return { dictationId, text: result.text };
|
||||
}
|
||||
|
||||
cancel(): void {
|
||||
const dictationId = this.dictationId;
|
||||
if (this.client.isConnected && dictationId) {
|
||||
this.client.cancelDictationStream(dictationId);
|
||||
}
|
||||
this.resetStreamForReplay();
|
||||
}
|
||||
|
||||
private hasPendingSegments(): boolean {
|
||||
return this.sendSeq < this.segments.length;
|
||||
}
|
||||
|
||||
private scheduleFlush(): void {
|
||||
if (this.flushTimer) {
|
||||
return;
|
||||
}
|
||||
this.flushTimer = setTimeout(() => {
|
||||
this.flushTimer = null;
|
||||
this.flush();
|
||||
}, 0);
|
||||
}
|
||||
|
||||
private clearScheduledFlush(): void {
|
||||
if (!this.flushTimer) {
|
||||
return;
|
||||
}
|
||||
clearTimeout(this.flushTimer);
|
||||
this.flushTimer = null;
|
||||
}
|
||||
|
||||
private async waitForFlushDrain(): Promise<void> {
|
||||
while (this.hasPendingSegments()) {
|
||||
if (!this.client.isConnected || !this.dictationId || !this.streamReady) {
|
||||
throw new Error('Failed to flush dictation stream');
|
||||
}
|
||||
await new Promise<void>((resolve) => {
|
||||
this.drainWaiters.push(resolve);
|
||||
});
|
||||
await waitForNextFlushTurn();
|
||||
}
|
||||
}
|
||||
|
||||
private resolveDrainWaiters(): void {
|
||||
const waiters = this.drainWaiters;
|
||||
this.drainWaiters = [];
|
||||
for (const resolve of waiters) {
|
||||
resolve();
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,301 @@
|
||||
/**
|
||||
* Microphone capture for dictation.
|
||||
*
|
||||
* Captures mono audio via getUserMedia, taps it with a ScriptProcessorNode
|
||||
* (universally supported, including iOS WKWebView), resamples Float32 to
|
||||
* 16 kHz PCM16LE, and emits ~1-second base64 chunks plus a normalized RMS
|
||||
* volume for the level meter.
|
||||
*/
|
||||
|
||||
import { useCallback, useEffect, useMemo, useRef, useState } from 'react';
|
||||
|
||||
export interface DictationAudioSourceConfig {
|
||||
onPcmSegment: (base64Pcm: string) => void;
|
||||
onError?: (error: Error) => void;
|
||||
}
|
||||
|
||||
export interface DictationAudioSource {
|
||||
start: () => Promise<void>;
|
||||
stop: () => Promise<void>;
|
||||
volume: number;
|
||||
}
|
||||
|
||||
const OUTPUT_RATE = 16000;
|
||||
const CHUNK_SAMPLES = OUTPUT_RATE; // ~1s per chunk
|
||||
|
||||
const getAudioContextCtor = (): typeof AudioContext | null => {
|
||||
if (typeof window === 'undefined') {
|
||||
return null;
|
||||
}
|
||||
const win = window as typeof window & { webkitAudioContext?: typeof AudioContext };
|
||||
return win.AudioContext || win.webkitAudioContext || null;
|
||||
};
|
||||
|
||||
const floatToInt16 = (sample: number): number => {
|
||||
const clamped = Math.max(-1, Math.min(1, sample));
|
||||
return clamped < 0 ? Math.round(clamped * 0x8000) : Math.round(clamped * 0x7fff);
|
||||
};
|
||||
|
||||
const resampleToPcm16 = (input: Float32Array, inputRate: number, outputRate: number): Int16Array => {
|
||||
if (input.length === 0) {
|
||||
return new Int16Array(0);
|
||||
}
|
||||
if (inputRate === outputRate) {
|
||||
const out = new Int16Array(input.length);
|
||||
for (let i = 0; i < input.length; i++) {
|
||||
out[i] = floatToInt16(input[i]);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
const ratio = inputRate / outputRate;
|
||||
const outputLength = Math.max(1, Math.round(input.length / ratio));
|
||||
const out = new Int16Array(outputLength);
|
||||
for (let i = 0; i < outputLength; i++) {
|
||||
const sourceIndex = i * ratio;
|
||||
const i0 = Math.floor(sourceIndex);
|
||||
const i1 = Math.min(input.length - 1, i0 + 1);
|
||||
const frac = sourceIndex - i0;
|
||||
out[i] = floatToInt16(input[i0] * (1 - frac) + input[i1] * frac);
|
||||
}
|
||||
return out;
|
||||
};
|
||||
|
||||
const concatInt16 = (a: Int16Array, b: Int16Array): Int16Array => {
|
||||
if (a.length === 0) {
|
||||
return b;
|
||||
}
|
||||
if (b.length === 0) {
|
||||
return a;
|
||||
}
|
||||
const out = new Int16Array(a.length + b.length);
|
||||
out.set(a, 0);
|
||||
out.set(b, a.length);
|
||||
return out;
|
||||
};
|
||||
|
||||
const int16ToBase64 = (pcm: Int16Array): string => {
|
||||
const bytes = new Uint8Array(pcm.buffer, pcm.byteOffset, pcm.byteLength);
|
||||
let binary = '';
|
||||
for (let i = 0; i < bytes.length; i++) {
|
||||
binary += String.fromCharCode(bytes[i]);
|
||||
}
|
||||
return btoa(binary);
|
||||
};
|
||||
|
||||
interface CaptureGraph {
|
||||
stream: MediaStream | null;
|
||||
context: AudioContext | null;
|
||||
source: MediaStreamAudioSourceNode | null;
|
||||
processor: ScriptProcessorNode | null;
|
||||
gain: GainNode | null;
|
||||
pending: Int16Array;
|
||||
started: boolean;
|
||||
}
|
||||
|
||||
const emptyGraph = (): CaptureGraph => ({
|
||||
stream: null,
|
||||
context: null,
|
||||
source: null,
|
||||
processor: null,
|
||||
gain: null,
|
||||
pending: new Int16Array(0),
|
||||
started: false,
|
||||
});
|
||||
|
||||
const safeDisconnect = (node: AudioNode | null): void => {
|
||||
if (!node) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
node.disconnect();
|
||||
} catch {
|
||||
// no-op
|
||||
}
|
||||
};
|
||||
|
||||
export const isDictationCaptureSupported = (): boolean => {
|
||||
if (typeof navigator === 'undefined' || typeof window === 'undefined') {
|
||||
return false;
|
||||
}
|
||||
if (!navigator.mediaDevices || typeof navigator.mediaDevices.getUserMedia !== 'function') {
|
||||
return false;
|
||||
}
|
||||
return getAudioContextCtor() !== null;
|
||||
};
|
||||
|
||||
export function useDictationAudioSource(config: DictationAudioSourceConfig): DictationAudioSource {
|
||||
const [volume, setVolume] = useState(0);
|
||||
|
||||
const onPcmSegmentRef = useRef(config.onPcmSegment);
|
||||
const onErrorRef = useRef(config.onError);
|
||||
useEffect(() => {
|
||||
onPcmSegmentRef.current = config.onPcmSegment;
|
||||
onErrorRef.current = config.onError;
|
||||
}, [config.onPcmSegment, config.onError]);
|
||||
|
||||
const graphRef = useRef<CaptureGraph>(emptyGraph());
|
||||
|
||||
const start = useCallback(async () => {
|
||||
if (graphRef.current.started) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (
|
||||
typeof navigator === 'undefined' ||
|
||||
!navigator.mediaDevices ||
|
||||
typeof navigator.mediaDevices.getUserMedia !== 'function'
|
||||
) {
|
||||
throw new Error('Microphone capture is not supported in this environment');
|
||||
}
|
||||
|
||||
const AudioContextCtor = getAudioContextCtor();
|
||||
if (!AudioContextCtor) {
|
||||
throw new Error('AudioContext unavailable');
|
||||
}
|
||||
|
||||
const stream = await navigator.mediaDevices.getUserMedia({
|
||||
audio: {
|
||||
channelCount: 1,
|
||||
noiseSuppression: true,
|
||||
echoCancellation: true,
|
||||
autoGainControl: true,
|
||||
},
|
||||
});
|
||||
|
||||
const context = new AudioContextCtor();
|
||||
try {
|
||||
if (context.state === 'suspended') {
|
||||
await context.resume().catch(() => undefined);
|
||||
}
|
||||
|
||||
const source = context.createMediaStreamSource(stream);
|
||||
const processor = context.createScriptProcessor(4096, 1, 1);
|
||||
const gain = context.createGain();
|
||||
gain.gain.value = 0;
|
||||
|
||||
graphRef.current = {
|
||||
stream,
|
||||
context,
|
||||
source,
|
||||
processor,
|
||||
gain,
|
||||
pending: new Int16Array(0),
|
||||
started: true,
|
||||
};
|
||||
|
||||
processor.onaudioprocess = (event) => {
|
||||
const graph = graphRef.current;
|
||||
if (!graph.started) {
|
||||
return;
|
||||
}
|
||||
const input = event.inputBuffer.getChannelData(0);
|
||||
|
||||
let sumSquares = 0;
|
||||
for (let i = 0; i < input.length; i++) {
|
||||
sumSquares += input[i] * input[i];
|
||||
}
|
||||
const rms = Math.sqrt(sumSquares / Math.max(1, input.length));
|
||||
setVolume(Math.min(1, Math.max(0, rms * 2)));
|
||||
|
||||
const next = resampleToPcm16(input, context.sampleRate, OUTPUT_RATE);
|
||||
graph.pending = concatInt16(graph.pending, next);
|
||||
|
||||
while (graph.pending.length >= CHUNK_SAMPLES) {
|
||||
const chunk = graph.pending.slice(0, CHUNK_SAMPLES);
|
||||
graph.pending = graph.pending.slice(CHUNK_SAMPLES);
|
||||
onPcmSegmentRef.current(int16ToBase64(chunk));
|
||||
}
|
||||
};
|
||||
|
||||
source.connect(processor);
|
||||
processor.connect(gain);
|
||||
gain.connect(context.destination);
|
||||
} catch (error) {
|
||||
stream.getTracks().forEach((track) => {
|
||||
try {
|
||||
track.stop();
|
||||
} catch {
|
||||
// no-op
|
||||
}
|
||||
});
|
||||
try {
|
||||
await context.close();
|
||||
} catch {
|
||||
// no-op
|
||||
}
|
||||
graphRef.current = emptyGraph();
|
||||
throw error instanceof Error ? error : new Error(String(error));
|
||||
}
|
||||
}, []);
|
||||
|
||||
const stop = useCallback(async () => {
|
||||
const graph = graphRef.current;
|
||||
graph.started = false;
|
||||
setVolume(0);
|
||||
|
||||
if (graph.processor) {
|
||||
try {
|
||||
graph.processor.onaudioprocess = null;
|
||||
} catch {
|
||||
// no-op
|
||||
}
|
||||
}
|
||||
safeDisconnect(graph.processor);
|
||||
safeDisconnect(graph.source);
|
||||
safeDisconnect(graph.gain);
|
||||
if (graph.stream) {
|
||||
graph.stream.getTracks().forEach((track) => {
|
||||
try {
|
||||
track.stop();
|
||||
} catch {
|
||||
// no-op
|
||||
}
|
||||
});
|
||||
}
|
||||
const pending = graph.pending;
|
||||
graph.pending = new Int16Array(0);
|
||||
if (pending.length > 0) {
|
||||
onPcmSegmentRef.current(int16ToBase64(pending));
|
||||
}
|
||||
|
||||
if (graph.context) {
|
||||
try {
|
||||
await graph.context.close();
|
||||
} catch {
|
||||
// no-op
|
||||
}
|
||||
}
|
||||
|
||||
// A new capture may have started while the old context was closing;
|
||||
// only clear the ref if it still points at the graph we tore down.
|
||||
if (graphRef.current === graph) {
|
||||
graphRef.current = emptyGraph();
|
||||
}
|
||||
}, []);
|
||||
|
||||
useEffect(() => {
|
||||
return () => {
|
||||
void stop().catch((err) => {
|
||||
onErrorRef.current?.(err instanceof Error ? err : new Error(String(err)));
|
||||
});
|
||||
};
|
||||
}, [stop]);
|
||||
|
||||
return useMemo(
|
||||
() => ({
|
||||
start: async () => {
|
||||
try {
|
||||
await start();
|
||||
} catch (err) {
|
||||
const normalized = err instanceof Error ? err : new Error(String(err));
|
||||
onErrorRef.current?.(normalized);
|
||||
throw normalized;
|
||||
}
|
||||
},
|
||||
stop,
|
||||
volume,
|
||||
}),
|
||||
[start, stop, volume],
|
||||
);
|
||||
}
|
||||
Reference in New Issue
Block a user