feat(voice): add voice input/output support with multiple providers (#281)
* feat(voice): add voice input/output support with multiple providers - Add BrowserVoiceButton component for Web Speech API voice input - Add VoiceProvider context for managing voice state across the app - Add TTS (Text-to-Speech) support with browser, macOS Say, and OpenAI providers - Add message TTS buttons to read assistant messages aloud - Add VoiceSettings page in OpenChamber settings - Add server endpoints for TTS and summarization services - Include slider component for voice rate/pitch/volume controls - Add hidden session support for background voice operations - Add Caddyfile for HTTPS support (required for microphone access) * fix: Build errors fixed and removed outdated ElevenLabs test code. * refactor(voice): use zen API with gpt-5-nano for TTS summarization Replace the hidden session + OpenCode SDK approach with direct calls to the opencode.ai zen API (same pattern used for commit message and PR description generation). - Rewrite summarization-service.js to call zen/v1/responses with gpt-5-nano - Remove hidden session logic (hiddenSession.ts, sessionStore filtering) - Remove summarizeModel setting and model selector from VoiceSettings - Simplify client-side summarize.ts to no longer pass model params - Clean up callers in useMessageTTS and useBrowserVoice * fix(voice): remove false 'voice not supported' warning in settings Mobile Safari does support voice but the isSupported check was incorrectly flagging it. Remove the warning banner entirely. * feat(voice): add configurable summary length limit for TTS output Add a slider (50-2000 chars) in voice settings to control max summary length. The limit is passed through the summarize endpoint and speak endpoint to the zen API prompt, with token budget scaled accordingly. * fix(voice): add diagnostic logging and sanitize TTS fallback Add console logging throughout the summarization flow (client + server) to trace why text may not be summarized. Fix silent error swallowing in /api/tts/speak. Always apply sanitizeForTTS even when summarization is disabled so raw markdown/code is never spoken verbatim. * fix(voice): fix token budget starving model of output tokens max_output_tokens includes both reasoning and output tokens. With effort:'low', reasoning alone consumes ~128 tokens, so a budget of 100 left zero tokens for the actual summary text. Use a fixed 1000 token budget (matching commit message generation) and control output length via the prompt's character limit instruction instead. * chore(voice): remove diagnostic logging from summarization flow * fix(voice): don't request mic permission on mobile page load Remove the useEffect that pre-requested microphone permission when the BrowserVoiceButton component mounted on mobile. This caused an unwanted permission prompt immediately on page load before the user tapped the mic icon. Permission is now only requested on explicit user interaction. * fix(voice): remove unused BrowserVoiceButton binding * fix(voice): desktop mic flow + non-continuous draft mode * fix(voice): stabilize continuous loop and polish controls * feat(settings): mark voice section experimental --------- Co-authored-by: Bohdan Triapitsyn <artmore@protonmail.com>
This commit is contained in:
committed by
GitHub
co-authored by
Bohdan Triapitsyn
parent
6776ac31c2
commit
1ed5316ac7
@@ -0,0 +1,654 @@
|
||||
/**
|
||||
* Browser Voice Service - Web Speech API wrapper
|
||||
*
|
||||
* Provides speech recognition (STT) and speech synthesis (TTS) using
|
||||
* browser-native Web Speech API. No external dependencies required.
|
||||
*
|
||||
* @example
|
||||
* ```typescript
|
||||
* import { browserVoiceService } from './browserVoiceService';
|
||||
*
|
||||
* // Check support
|
||||
* if (browserVoiceService.isSupported()) {
|
||||
* // Start listening
|
||||
* browserVoiceService.startListening('en-US', (text, isFinal) => {
|
||||
* if (isFinal) {
|
||||
* console.log('Final transcript:', text);
|
||||
* }
|
||||
* });
|
||||
*
|
||||
* // Speak text
|
||||
* browserVoiceService.speakText('Hello world', 'en-US', () => {
|
||||
* console.log('Speech finished');
|
||||
* });
|
||||
* }
|
||||
* ```
|
||||
*/
|
||||
|
||||
// Extend Window interface for SpeechRecognition
|
||||
declare global {
|
||||
interface Window {
|
||||
SpeechRecognition: { new(): SpeechRecognition };
|
||||
webkitSpeechRecognition: { new(): SpeechRecognition };
|
||||
}
|
||||
}
|
||||
|
||||
// Callback types
|
||||
export type SpeechResultCallback = (text: string, isFinal: boolean) => void;
|
||||
export type SpeechEndCallback = () => void;
|
||||
export type ErrorCallback = (error: string) => void;
|
||||
|
||||
/**
|
||||
* Browser Voice Service class
|
||||
* Wraps Web Speech API with a clean interface
|
||||
*/
|
||||
class BrowserVoiceService {
|
||||
private recognition: SpeechRecognition | null = null;
|
||||
private isListening = false;
|
||||
private currentLang = 'en-US';
|
||||
private onResultCallback: SpeechResultCallback | null = null;
|
||||
private onErrorCallback: ErrorCallback | null = null;
|
||||
private restartOnEnd = false;
|
||||
private conversationMode = false;
|
||||
private isSpeaking = false;
|
||||
private audioContext: AudioContext | null = null;
|
||||
private audioUnlockRequired = false;
|
||||
|
||||
/**
|
||||
* Check if browser supports Web Speech API
|
||||
*/
|
||||
isSupported(): boolean {
|
||||
const hasRecognition = 'SpeechRecognition' in window || 'webkitSpeechRecognition' in window;
|
||||
const hasSynthesis = 'speechSynthesis' in window;
|
||||
return hasRecognition && hasSynthesis;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get detailed support information
|
||||
*/
|
||||
getSupportDetails(): {
|
||||
recognition: boolean;
|
||||
synthesis: boolean;
|
||||
prefixed: boolean;
|
||||
secureContext: boolean;
|
||||
} {
|
||||
return {
|
||||
recognition: 'SpeechRecognition' in window || 'webkitSpeechRecognition' in window,
|
||||
synthesis: 'speechSynthesis' in window,
|
||||
prefixed: !('SpeechRecognition' in window) && 'webkitSpeechRecognition' in window,
|
||||
secureContext: window.isSecureContext,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Set conversation mode (continuous back-and-forth)
|
||||
* @param enabled - Whether to enable continuous conversation mode
|
||||
*
|
||||
* Note: This only sets the flag - it does NOT auto-start listening.
|
||||
* The user must explicitly start voice mode first. Conversation mode
|
||||
* only affects whether listening auto-resumes after AI finishes speaking.
|
||||
*/
|
||||
setConversationMode(enabled: boolean): void {
|
||||
this.conversationMode = enabled;
|
||||
// If disabling, stop auto-restart
|
||||
if (!enabled) {
|
||||
this.restartOnEnd = false;
|
||||
}
|
||||
// Note: We intentionally do NOT auto-start listening here.
|
||||
// The user must explicitly click the microphone button to start.
|
||||
// Conversation mode only controls auto-resume after speech ends.
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if conversation mode is active
|
||||
*/
|
||||
isConversationMode(): boolean {
|
||||
return this.conversationMode;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pause listening temporarily (e.g., while AI is speaking)
|
||||
*/
|
||||
pauseListening(): void {
|
||||
this.restartOnEnd = false;
|
||||
if (this.recognition) {
|
||||
try {
|
||||
this.recognition.stop();
|
||||
} catch {
|
||||
// Ignore stop errors
|
||||
}
|
||||
}
|
||||
this.isListening = false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resume listening after being paused
|
||||
*/
|
||||
resumeListening(): void {
|
||||
console.log('[BrowserVoiceService] resumeListening called:', {
|
||||
conversationMode: this.conversationMode,
|
||||
hasCallback: !!this.onResultCallback,
|
||||
isSpeaking: this.isSpeaking,
|
||||
currentLang: this.currentLang
|
||||
});
|
||||
|
||||
if (this.conversationMode && this.onResultCallback && !this.isSpeaking) {
|
||||
// Use sync version for resume (should already have permission)
|
||||
try {
|
||||
console.log('[BrowserVoiceService] Resuming listening...');
|
||||
this.startListeningSync(this.currentLang, this.onResultCallback, this.onErrorCallback || undefined);
|
||||
} catch (err) {
|
||||
console.error('[BrowserVoiceService] Failed to resume listening:', err);
|
||||
}
|
||||
} else {
|
||||
console.log('[BrowserVoiceService] Not resuming - conditions not met');
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if microphone permission is already granted
|
||||
* @returns Promise<boolean> - true if permission granted
|
||||
*/
|
||||
async checkMicrophonePermission(): Promise<boolean> {
|
||||
try {
|
||||
if ('permissions' in navigator) {
|
||||
const result = await navigator.permissions.query({ name: 'microphone' as PermissionName });
|
||||
return result.state === 'granted';
|
||||
}
|
||||
return false;
|
||||
} catch {
|
||||
// permissions API not supported or failed
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Prepare listening by requesting microphone permission
|
||||
* This should be called BEFORE user gesture on mobile to pre-request permission
|
||||
* @returns Promise<boolean> - true if permission granted
|
||||
*/
|
||||
async prepareListening(): Promise<boolean> {
|
||||
if (!this.isSupported()) {
|
||||
throw new Error('Web Speech API not supported in this browser');
|
||||
}
|
||||
|
||||
if (typeof navigator === 'undefined' || typeof navigator.mediaDevices?.getUserMedia !== 'function') {
|
||||
// Some embedded runtimes (e.g. desktop webviews) may not expose mediaDevices,
|
||||
// while SpeechRecognition can still request mic permission on start().
|
||||
return true;
|
||||
}
|
||||
|
||||
try {
|
||||
const stream = await navigator.mediaDevices.getUserMedia({ audio: true });
|
||||
stream.getTracks().forEach((track) => track.stop());
|
||||
return true;
|
||||
} catch (err) {
|
||||
const name = typeof err === 'object' && err && 'name' in err ? String((err as { name?: unknown }).name) : '';
|
||||
if (name === 'NotAllowedError') {
|
||||
throw new Error('Microphone permission denied');
|
||||
}
|
||||
if (name === 'NotFoundError') {
|
||||
throw new Error('No microphone found');
|
||||
}
|
||||
const errorMsg = err instanceof Error ? err.message : 'Unable to access microphone';
|
||||
throw new Error(errorMsg);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Start speech recognition SYNCHRONOUSLY
|
||||
* Must be called within a user gesture handler on mobile (iOS Safari)
|
||||
* @param lang - BCP 47 language tag (e.g., 'en-US', 'es-ES')
|
||||
* @param onResult - Callback for speech results
|
||||
* @param onError - Optional callback for errors
|
||||
*/
|
||||
startListeningSync(
|
||||
lang: string,
|
||||
onResult: SpeechResultCallback,
|
||||
onError?: ErrorCallback
|
||||
): void {
|
||||
if (!this.isSupported()) {
|
||||
const errorMsg = 'Web Speech API not supported in this browser';
|
||||
onError?.(errorMsg);
|
||||
throw new Error(errorMsg);
|
||||
}
|
||||
|
||||
// Stop any existing recognition
|
||||
this.stopListening();
|
||||
|
||||
// Create new recognition instance
|
||||
const SpeechRecognitionConstructor = window.SpeechRecognition || window.webkitSpeechRecognition;
|
||||
this.recognition = new SpeechRecognitionConstructor();
|
||||
this.currentLang = lang;
|
||||
this.onResultCallback = onResult;
|
||||
this.onErrorCallback = onError || null;
|
||||
|
||||
// Configure recognition
|
||||
this.recognition.continuous = true;
|
||||
this.recognition.interimResults = true;
|
||||
this.recognition.lang = lang;
|
||||
|
||||
// Set up event handlers
|
||||
this.recognition.onstart = () => {
|
||||
console.log('[BrowserVoiceService] Recognition started');
|
||||
this.isListening = true;
|
||||
this.restartOnEnd = true;
|
||||
};
|
||||
|
||||
this.recognition.onaudiostart = () => {
|
||||
console.log('[BrowserVoiceService] Audio recording started');
|
||||
};
|
||||
|
||||
this.recognition.onsoundstart = () => {
|
||||
console.log('[BrowserVoiceService] Sound detected');
|
||||
};
|
||||
|
||||
this.recognition.onresult = (event: SpeechRecognitionEvent) => {
|
||||
console.log('[BrowserVoiceService] Got result:', event.results.length, 'results');
|
||||
let finalTranscript = '';
|
||||
let interimTranscript = '';
|
||||
|
||||
for (let i = event.resultIndex; i < event.results.length; i++) {
|
||||
const result = event.results[i];
|
||||
if (result.isFinal) {
|
||||
finalTranscript += result[0].transcript;
|
||||
} else {
|
||||
interimTranscript += result[0].transcript;
|
||||
}
|
||||
}
|
||||
|
||||
console.log('[BrowserVoiceService] Transcripts - interim:', interimTranscript, 'final:', finalTranscript);
|
||||
|
||||
// Send interim results
|
||||
if (interimTranscript) {
|
||||
console.log('[BrowserVoiceService] Calling onResultCallback with interim');
|
||||
this.onResultCallback?.(interimTranscript, false);
|
||||
}
|
||||
|
||||
// Send final results
|
||||
if (finalTranscript) {
|
||||
console.log('[BrowserVoiceService] Calling onResultCallback with final');
|
||||
this.onResultCallback?.(finalTranscript, true);
|
||||
}
|
||||
};
|
||||
|
||||
this.recognition.onerror = (event: SpeechRecognitionErrorEvent) => {
|
||||
// "aborted" is commonly emitted when we intentionally stop/pause recognition.
|
||||
// Treat it as non-fatal to avoid noisy error loops in continuous mode.
|
||||
if (event.error === 'aborted') {
|
||||
return;
|
||||
}
|
||||
|
||||
const errorMessage = this.getErrorMessage(event.error);
|
||||
this.onErrorCallback?.(errorMessage);
|
||||
|
||||
// Don't restart on certain errors
|
||||
if (event.error === 'not-allowed' || event.error === 'service-not-allowed') {
|
||||
this.restartOnEnd = false;
|
||||
this.isListening = false;
|
||||
}
|
||||
};
|
||||
|
||||
this.recognition.onend = () => {
|
||||
this.isListening = false;
|
||||
|
||||
// Auto-restart if still supposed to be listening and not speaking
|
||||
if (this.restartOnEnd && this.recognition && !this.isSpeaking) {
|
||||
try {
|
||||
this.recognition.start();
|
||||
} catch {
|
||||
// Ignore restart errors
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// Start recognition - MUST be synchronous for iOS Safari
|
||||
try {
|
||||
this.recognition.start();
|
||||
} catch (err) {
|
||||
const errorMsg = err instanceof Error ? err.message : 'Failed to start speech recognition';
|
||||
onError?.(errorMsg);
|
||||
throw new Error(errorMsg);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Start speech recognition (async version for desktop/backward compatibility)
|
||||
* @param lang - BCP 47 language tag (e.g., 'en-US', 'es-ES')
|
||||
* @param onResult - Callback for speech results
|
||||
* @param onError - Optional callback for errors
|
||||
* @returns Promise that resolves when recognition starts
|
||||
*/
|
||||
async startListening(
|
||||
lang: string,
|
||||
onResult: SpeechResultCallback,
|
||||
onError?: ErrorCallback
|
||||
): Promise<void> {
|
||||
// Start recognition directly from the user gesture path.
|
||||
// Some webview runtimes reject preflight getUserMedia and then never show
|
||||
// permission prompt, while SpeechRecognition.start() can still trigger it.
|
||||
this.startListeningSync(lang, onResult, onError);
|
||||
}
|
||||
|
||||
/**
|
||||
* Stop speech recognition
|
||||
*/
|
||||
stopListening(): void {
|
||||
this.restartOnEnd = false;
|
||||
|
||||
if (this.recognition) {
|
||||
try {
|
||||
this.recognition.stop();
|
||||
} catch {
|
||||
// Ignore stop errors
|
||||
}
|
||||
this.recognition = null;
|
||||
}
|
||||
|
||||
this.isListening = false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if currently listening
|
||||
*/
|
||||
getIsListening(): boolean {
|
||||
return this.isListening;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get current language
|
||||
*/
|
||||
getCurrentLang(): string {
|
||||
return this.currentLang;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resume audio context to unlock audio for playback
|
||||
* Must be called within a user gesture handler
|
||||
*/
|
||||
async resumeAudioContext(): Promise<void> {
|
||||
if (!this.audioContext) {
|
||||
this.audioContext = new (window.AudioContext || (window as unknown as { webkitAudioContext: typeof AudioContext }).webkitAudioContext)();
|
||||
}
|
||||
if (this.audioContext.state === 'suspended') {
|
||||
await this.audioContext.resume();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if audio unlock is required (autoplay policy blocked audio)
|
||||
*/
|
||||
isAudioUnlockRequired(): boolean {
|
||||
return this.audioUnlockRequired;
|
||||
}
|
||||
|
||||
/**
|
||||
* Manually unlock audio by playing a silent sound
|
||||
* Call this from a button click handler for stubborn browsers
|
||||
*/
|
||||
async unlockAudio(): Promise<boolean> {
|
||||
try {
|
||||
// Create and play silent audio to unlock Web Audio API
|
||||
const audio = new Audio();
|
||||
// 1ms silence WAV file (base64 encoded)
|
||||
audio.src = 'data:audio/wav;base64,UklGRigAAABXQVZFZm10IBIAAAABAAEARKwAAIhYAQACABAAAABkYXRhAgAAAAEA';
|
||||
audio.volume = 0.01;
|
||||
await audio.play();
|
||||
|
||||
// Resume audio context
|
||||
await this.resumeAudioContext();
|
||||
|
||||
// Also unlock speech synthesis on mobile Safari by speaking a silent utterance
|
||||
// This must be done within a user gesture to allow future speech
|
||||
if (this.isMobileDevice() && 'speechSynthesis' in window) {
|
||||
const unlockUtterance = new SpeechSynthesisUtterance('');
|
||||
unlockUtterance.volume = 0;
|
||||
window.speechSynthesis.speak(unlockUtterance);
|
||||
window.speechSynthesis.cancel(); // Cancel immediately
|
||||
console.log('[BrowserVoiceService] Speech synthesis unlocked for mobile');
|
||||
}
|
||||
|
||||
this.audioUnlockRequired = false;
|
||||
console.log('[BrowserVoiceService] Audio unlocked successfully');
|
||||
return true;
|
||||
} catch (err) {
|
||||
console.error('[BrowserVoiceService] Failed to unlock audio:', err);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Speak text using speech synthesis
|
||||
* @param text - Text to speak
|
||||
* @param lang - BCP 47 language tag for voice selection
|
||||
* @param onEnd - Optional callback when speech ends
|
||||
* @param options - Optional TTS configuration (rate, pitch, volume, voiceName)
|
||||
* @returns Promise that resolves when speech starts
|
||||
*/
|
||||
async speakText(
|
||||
text: string,
|
||||
lang: string,
|
||||
onEnd?: SpeechEndCallback,
|
||||
options?: { rate?: number; pitch?: number; volume?: number; voiceName?: string }
|
||||
): Promise<void> {
|
||||
if (!('speechSynthesis' in window)) {
|
||||
throw new Error('Speech synthesis not supported');
|
||||
}
|
||||
|
||||
// Resume audio context first (user gesture must have happened)
|
||||
await this.resumeAudioContext();
|
||||
|
||||
// Wait for voices to be available (Chrome requires this)
|
||||
const voices = await this.waitForVoices();
|
||||
|
||||
// Small delay to ensure audio context is ready
|
||||
await new Promise(resolve => setTimeout(resolve, 50));
|
||||
|
||||
// Set speaking state and pause listening to avoid hearing ourselves
|
||||
this.isSpeaking = true;
|
||||
this.pauseListening();
|
||||
|
||||
// Cancel any ongoing speech
|
||||
window.speechSynthesis.cancel();
|
||||
|
||||
const utterance = new SpeechSynthesisUtterance(text);
|
||||
utterance.lang = lang;
|
||||
utterance.rate = options?.rate ?? 1;
|
||||
utterance.pitch = options?.pitch ?? 1;
|
||||
utterance.volume = options?.volume ?? 1;
|
||||
|
||||
// Try to find voice by name first (user-selected), then fallback to language match
|
||||
let selectedVoice: SpeechSynthesisVoice | null = null;
|
||||
|
||||
if (options?.voiceName) {
|
||||
selectedVoice = voices.find(v => v.name === options.voiceName) || null;
|
||||
if (selectedVoice) {
|
||||
console.log(`[BrowserVoiceService] Using selected voice: ${selectedVoice.name} (${selectedVoice.lang})`);
|
||||
} else {
|
||||
console.warn(`[BrowserVoiceService] Selected voice "${options.voiceName}" not found, falling back to language match`);
|
||||
}
|
||||
}
|
||||
|
||||
if (!selectedVoice) {
|
||||
selectedVoice = this.findBestVoice(voices, lang);
|
||||
if (selectedVoice) {
|
||||
console.log(`[BrowserVoiceService] Using language-matched voice: ${selectedVoice.name} (${selectedVoice.lang})`);
|
||||
} else {
|
||||
console.warn(`[BrowserVoiceService] No voice found for language: ${lang}, using default`);
|
||||
}
|
||||
}
|
||||
|
||||
if (selectedVoice) {
|
||||
utterance.voice = selectedVoice;
|
||||
}
|
||||
|
||||
console.log(`[BrowserVoiceService] Speaking text (${text.length} chars) in ${lang}`);
|
||||
|
||||
return new Promise((resolve, reject) => {
|
||||
let hasStarted = false;
|
||||
|
||||
utterance.onstart = () => {
|
||||
hasStarted = true;
|
||||
console.log('[BrowserVoiceService] Speech started');
|
||||
resolve();
|
||||
};
|
||||
|
||||
utterance.onend = () => {
|
||||
this.isSpeaking = false;
|
||||
console.log('[BrowserVoiceService] Speech ended');
|
||||
onEnd?.();
|
||||
};
|
||||
|
||||
utterance.onerror = (event) => {
|
||||
this.isSpeaking = false;
|
||||
console.error('[BrowserVoiceService] Speech synthesis error:', event.error);
|
||||
|
||||
// Track autoplay policy violations
|
||||
if (event.error === 'not-allowed' || event.error === 'interrupted') {
|
||||
this.audioUnlockRequired = true;
|
||||
}
|
||||
|
||||
// Provide more specific error messages
|
||||
let errorMessage = `Speech synthesis error: ${event.error || 'unknown'}`;
|
||||
if (event.error === 'not-allowed') {
|
||||
errorMessage = 'Audio blocked by browser autoplay policy. Please interact with the page first.';
|
||||
} else if (event.error === 'interrupted') {
|
||||
errorMessage = 'Speech was interrupted. Please try again.';
|
||||
}
|
||||
|
||||
reject(new Error(errorMessage));
|
||||
};
|
||||
|
||||
// Safety timeout - if onstart doesn't fire within 2 seconds, something is wrong
|
||||
setTimeout(() => {
|
||||
if (!hasStarted) {
|
||||
console.warn('[BrowserVoiceService] Speech start timeout - audio may be blocked');
|
||||
this.audioUnlockRequired = true;
|
||||
}
|
||||
}, 2000);
|
||||
|
||||
window.speechSynthesis.speak(utterance);
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Cancel ongoing speech
|
||||
*/
|
||||
cancelSpeech(): void {
|
||||
if ('speechSynthesis' in window) {
|
||||
window.speechSynthesis.cancel();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Get available voices
|
||||
*/
|
||||
getVoices(): SpeechSynthesisVoice[] {
|
||||
if (!('speechSynthesis' in window)) {
|
||||
return [];
|
||||
}
|
||||
return window.speechSynthesis.getVoices();
|
||||
}
|
||||
|
||||
/**
|
||||
* Wait for voices to load (needed for Chrome)
|
||||
*/
|
||||
async waitForVoices(): Promise<SpeechSynthesisVoice[]> {
|
||||
if (!('speechSynthesis' in window)) {
|
||||
return [];
|
||||
}
|
||||
|
||||
const voices = window.speechSynthesis.getVoices();
|
||||
if (voices.length > 0) {
|
||||
return voices;
|
||||
}
|
||||
|
||||
return new Promise((resolve) => {
|
||||
const handleVoicesChanged = () => {
|
||||
resolve(window.speechSynthesis.getVoices());
|
||||
window.speechSynthesis.onvoiceschanged = null;
|
||||
};
|
||||
|
||||
window.speechSynthesis.onvoiceschanged = handleVoicesChanged;
|
||||
|
||||
// Timeout fallback
|
||||
setTimeout(() => {
|
||||
resolve(window.speechSynthesis.getVoices());
|
||||
window.speechSynthesis.onvoiceschanged = null;
|
||||
}, 1000);
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Find the best voice for a given language
|
||||
*/
|
||||
private findBestVoice(voices: SpeechSynthesisVoice[], lang: string): SpeechSynthesisVoice | null {
|
||||
// First try exact match
|
||||
let voice = voices.find(v => v.lang === lang);
|
||||
|
||||
if (!voice) {
|
||||
// Try language code match (e.g., 'en' for 'en-US')
|
||||
const langCode = lang.split('-')[0];
|
||||
voice = voices.find(v => v.lang.startsWith(langCode));
|
||||
}
|
||||
|
||||
if (!voice) {
|
||||
// Prefer local voices
|
||||
voice = voices.find(v => v.lang.startsWith(lang.split('-')[0]) && v.localService);
|
||||
}
|
||||
|
||||
return voice || null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if running on mobile device
|
||||
*/
|
||||
private isMobileDevice(): boolean {
|
||||
if (typeof navigator === 'undefined') return false;
|
||||
const userAgent = navigator.userAgent.toLowerCase();
|
||||
return /iphone|ipad|ipod|android|mobile|webos|blackberry|iemobile|opera mini/i.test(userAgent);
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if running on iOS Safari
|
||||
*/
|
||||
private isIOSSafari(): boolean {
|
||||
if (typeof navigator === 'undefined') return false;
|
||||
const userAgent = navigator.userAgent.toLowerCase();
|
||||
const isIOS = /iphone|ipad|ipod/i.test(userAgent);
|
||||
const isSafari = /safari/i.test(userAgent) && !/chrome|crios|crmo/i.test(userAgent);
|
||||
return isIOS && isSafari;
|
||||
}
|
||||
|
||||
/**
|
||||
* Get human-readable error message
|
||||
* @param error - Error code from SpeechRecognition
|
||||
*/
|
||||
private getErrorMessage(error: string): string {
|
||||
const isMobileDevice = this.isMobileDevice();
|
||||
const isIOS = this.isIOSSafari();
|
||||
|
||||
const errorMessages: Record<string, string> = {
|
||||
'no-speech': 'No speech detected',
|
||||
'aborted': 'Speech recognition aborted',
|
||||
'audio-capture': 'No microphone found',
|
||||
'network': 'Network error - check connection',
|
||||
'not-allowed': isMobileDevice
|
||||
? 'Microphone permission denied. Check Settings > Safari > Microphone'
|
||||
: 'Microphone permission denied',
|
||||
'service-not-allowed': isIOS
|
||||
? 'Speech recognition requires a user gesture. Please tap the microphone button again.'
|
||||
: 'Speech recognition service not allowed',
|
||||
'bad-grammar': 'Grammar error',
|
||||
'language-not-supported': 'Language not supported',
|
||||
};
|
||||
|
||||
return errorMessages[error] || `Speech recognition error: ${error}`;
|
||||
}
|
||||
}
|
||||
|
||||
// Export singleton instance
|
||||
export const browserVoiceService = new BrowserVoiceService();
|
||||
|
||||
// Also export the class for testing/customization
|
||||
export { BrowserVoiceService };
|
||||
@@ -0,0 +1,131 @@
|
||||
/**
|
||||
* Context formatters for voice-native output
|
||||
* Formats session events (messages, permissions, ready events) into natural language
|
||||
* for the ElevenLabs voice agent to speak aloud.
|
||||
*
|
||||
* @example
|
||||
* ```typescript
|
||||
* import { formatMessage, formatPermissionRequest } from '@/lib/voice';
|
||||
*
|
||||
* const voiceText = formatMessage({ role: 'assistant', content: 'Hello!' });
|
||||
* // Returns: "Claude Code: Hello!"
|
||||
* ```
|
||||
*/
|
||||
|
||||
import { VOICE_CONFIG } from "./voiceConfig";
|
||||
|
||||
/** Message type for voice formatting */
|
||||
export interface VoiceMessage {
|
||||
role: string;
|
||||
content: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Format a single message for voice output
|
||||
* - Assistant messages: Code blocks replaced with "[code block]", prefixed with "Claude Code: "
|
||||
* - User messages: Prefixed with "User: "
|
||||
* - Other roles: Returns null (not spoken)
|
||||
*
|
||||
* @param message - The message to format
|
||||
* @returns Formatted text for voice, or null if should not be spoken
|
||||
*/
|
||||
export function formatMessage(message: VoiceMessage): string | null {
|
||||
// Handle edge cases
|
||||
if (!message || typeof message.content !== "string") {
|
||||
return null;
|
||||
}
|
||||
|
||||
const content = message.content.trim();
|
||||
if (!content) {
|
||||
return null;
|
||||
}
|
||||
|
||||
if (message.role === "assistant") {
|
||||
// Replace code blocks with description (don't read code aloud)
|
||||
const textOnly = content.replace(/```[\s\S]*?```/g, "[code block]");
|
||||
return `Claude Code: ${textOnly}`;
|
||||
}
|
||||
|
||||
if (message.role === "user") {
|
||||
return `User: ${content}`;
|
||||
}
|
||||
|
||||
// Skip system, tool, and other roles for voice
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Format multiple new messages for voice output
|
||||
* - Maps messages through formatMessage
|
||||
* - Filters out nulls (unspoken roles)
|
||||
* - Joins with newlines
|
||||
*
|
||||
* @param sessionId - The session ID (for future use/debugging)
|
||||
* @param messages - Array of messages to format
|
||||
* @returns Formatted text for voice, or null if no speakable messages
|
||||
*/
|
||||
export function formatNewMessages(
|
||||
sessionId: string,
|
||||
messages: VoiceMessage[]
|
||||
): string | null {
|
||||
// Handle edge cases
|
||||
if (!Array.isArray(messages) || messages.length === 0) {
|
||||
return null;
|
||||
}
|
||||
|
||||
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
|
||||
console.log(`[Voice] Formatting ${messages.length} messages for session ${sessionId}`);
|
||||
}
|
||||
|
||||
// Format each message and filter out nulls
|
||||
const formattedMessages = messages
|
||||
.map(formatMessage)
|
||||
.filter((msg): msg is string => msg !== null);
|
||||
|
||||
if (formattedMessages.length === 0) {
|
||||
return null;
|
||||
}
|
||||
|
||||
return formattedMessages.join("\n");
|
||||
}
|
||||
|
||||
/**
|
||||
* Format a permission request for voice announcement
|
||||
* - Per CONTEXT.md: Only tool name, not arguments (LIMITED_TOOL_CALLS)
|
||||
* - Prompts user to say "allow" or "deny"
|
||||
*
|
||||
* @param sessionId - The session ID
|
||||
* @param requestId - The permission request ID
|
||||
* @param toolName - Name of the tool requesting permission
|
||||
* @param toolArgs - Tool arguments (not included in voice output per config)
|
||||
* @returns Formatted permission request for voice
|
||||
*/
|
||||
export function formatPermissionRequest(
|
||||
sessionId: string,
|
||||
requestId: string,
|
||||
toolName: string,
|
||||
// eslint-disable-next-line @typescript-eslint/no-unused-vars
|
||||
toolArgs: unknown
|
||||
): string {
|
||||
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
|
||||
console.log(`[Voice] Formatting permission request ${requestId} for session ${sessionId}`);
|
||||
}
|
||||
|
||||
// Per VOICE_CONFIG.LIMITED_TOOL_CALLS, we don't include toolArgs in voice output
|
||||
return `Claude Code is requesting permission to use ${toolName}. Say "allow" or "deny".`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Format a ready event for voice announcement
|
||||
* - Indicates the AI has finished working and is ready for next instruction
|
||||
*
|
||||
* @param sessionId - The session ID
|
||||
* @returns Formatted ready event for voice
|
||||
*/
|
||||
export function formatReadyEvent(sessionId: string): string {
|
||||
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
|
||||
console.log(`[Voice] Formatting ready event for session ${sessionId}`);
|
||||
}
|
||||
|
||||
return "Claude Code finished working. Ready for next instruction.";
|
||||
}
|
||||
@@ -0,0 +1,36 @@
|
||||
/**
|
||||
* Voice module barrel export
|
||||
* Provides clean import path for voice configuration and client tools
|
||||
*
|
||||
* @example
|
||||
* ```typescript
|
||||
* import { VOICE_CONFIG, realtimeClientTools, voiceHooks } from '@/lib/voice';
|
||||
* ```
|
||||
*/
|
||||
|
||||
// Configuration
|
||||
export { VOICE_CONFIG } from "./voiceConfig";
|
||||
|
||||
// Client tools for ElevenLabs voice agent
|
||||
export { realtimeClientTools } from "./realtimeClientTools";
|
||||
export type { RealtimeClientTools } from "./realtimeClientTools";
|
||||
|
||||
// Voice session registry (from voiceSession.ts)
|
||||
export {
|
||||
registerVoiceSession,
|
||||
unregisterVoiceSession,
|
||||
getVoiceSession,
|
||||
isVoiceSessionStarted,
|
||||
} from "./voiceSession";
|
||||
|
||||
// Voice hooks for session-to-voice event routing (from voiceHooks.ts)
|
||||
export { voiceHooks } from "./voiceHooks";
|
||||
|
||||
// Context formatters for voice-native output
|
||||
export {
|
||||
formatMessage,
|
||||
formatNewMessages,
|
||||
formatPermissionRequest,
|
||||
formatReadyEvent,
|
||||
type VoiceMessage,
|
||||
} from "./contextFormatters";
|
||||
@@ -0,0 +1,106 @@
|
||||
import { z } from "zod";
|
||||
import { useSessionStore } from "@/stores/useSessionStore";
|
||||
import { useConfigStore } from "@/stores/useConfigStore";
|
||||
import { usePermissionStore } from "@/stores/permissionStore";
|
||||
|
||||
/**
|
||||
* Static client tools for the realtime voice interface.
|
||||
* These tools allow the voice agent to interact with Claude Code.
|
||||
*/
|
||||
export const realtimeClientTools = {
|
||||
/**
|
||||
* Send a message to Claude Code via the current session.
|
||||
* Validates parameters with Zod and returns status strings.
|
||||
*/
|
||||
messageClaudeCode: async (parameters: unknown): Promise<string> => {
|
||||
// Validate parameters with Zod
|
||||
const schema = z.object({
|
||||
message: z.string().min(1, "Message cannot be empty"),
|
||||
});
|
||||
|
||||
const parsed = schema.safeParse(parameters);
|
||||
if (!parsed.success) {
|
||||
console.error("[Voice] Invalid message parameter:", parsed.error);
|
||||
return "error (invalid message parameter)";
|
||||
}
|
||||
|
||||
// Get current session ID from store
|
||||
const sessionId = useSessionStore.getState().currentSessionId;
|
||||
if (!sessionId) {
|
||||
console.error("[Voice] No active session");
|
||||
return "error (no active session)";
|
||||
}
|
||||
|
||||
// Get current provider and model from config store
|
||||
const { currentProviderId, currentModelId, currentAgentName } = useConfigStore.getState();
|
||||
if (!currentProviderId || !currentModelId) {
|
||||
console.error("[Voice] No provider/model selected");
|
||||
return "error (no provider or model selected)";
|
||||
}
|
||||
|
||||
try {
|
||||
console.log("[Voice] Sending message to session:", sessionId);
|
||||
await useSessionStore
|
||||
.getState()
|
||||
.sendMessage(parsed.data.message, currentProviderId, currentModelId, currentAgentName ?? undefined);
|
||||
return "sent";
|
||||
} catch (error) {
|
||||
console.error("[Voice] Failed to send message:", error);
|
||||
return "error (failed to send message)";
|
||||
}
|
||||
},
|
||||
|
||||
/**
|
||||
* Process a permission request from voice.
|
||||
* Validates decision with Zod enum and interacts with permission store.
|
||||
*/
|
||||
processPermissionRequest: async (parameters: unknown): Promise<string> => {
|
||||
// Validate parameters with Zod
|
||||
const schema = z.object({
|
||||
decision: z.enum(["allow", "deny"]),
|
||||
});
|
||||
|
||||
const parsed = schema.safeParse(parameters);
|
||||
if (!parsed.success) {
|
||||
console.error("[Voice] Invalid decision parameter:", parsed.error);
|
||||
return "error (invalid decision parameter, expected 'allow' or 'deny')";
|
||||
}
|
||||
|
||||
// Get current session ID from store
|
||||
const sessionId = useSessionStore.getState().currentSessionId;
|
||||
if (!sessionId) {
|
||||
console.error("[Voice] No active session");
|
||||
return "error (no active session)";
|
||||
}
|
||||
|
||||
// Get pending permissions for this session
|
||||
const permissions = usePermissionStore.getState().permissions.get(sessionId);
|
||||
if (!permissions || permissions.length === 0) {
|
||||
console.error("[Voice] No pending permission requests");
|
||||
return "error (no pending permission request)";
|
||||
}
|
||||
|
||||
// Get the first pending permission request
|
||||
const request = permissions[0];
|
||||
if (!request) {
|
||||
return "error (no pending permission request)";
|
||||
}
|
||||
|
||||
try {
|
||||
const decision = parsed.data.decision;
|
||||
console.log(`[Voice] Processing permission request ${request.id}: ${decision}`);
|
||||
|
||||
// Respond to the permission based on decision
|
||||
const response: "once" | "always" | "reject" = decision === "allow" ? "once" : "reject";
|
||||
await usePermissionStore.getState().respondToPermission(sessionId, request.id, response);
|
||||
|
||||
return "done";
|
||||
} catch (error) {
|
||||
console.error("[Voice] Failed to process permission:", error);
|
||||
return `error (failed to ${parsed.data.decision} permission)`;
|
||||
}
|
||||
},
|
||||
};
|
||||
|
||||
/** Type for the realtime client tools */
|
||||
export type RealtimeClientTools = typeof realtimeClientTools;
|
||||
@@ -0,0 +1,116 @@
|
||||
/**
|
||||
* Text summarization utility for TTS
|
||||
*
|
||||
* Calls the server-side summarization endpoint which uses
|
||||
* the opencode.ai zen API with gpt-5-nano.
|
||||
*/
|
||||
|
||||
import { useConfigStore } from '@/stores/useConfigStore';
|
||||
|
||||
/**
|
||||
* Summarize text using the server-side zen API endpoint
|
||||
*
|
||||
* @param text - The text to summarize
|
||||
* @param options - Optional configuration
|
||||
* @returns The summarized text, or original text if summarization fails
|
||||
*/
|
||||
export async function summarizeText(
|
||||
text: string,
|
||||
options?: {
|
||||
/** Character threshold - don't summarize if under this length */
|
||||
threshold?: number;
|
||||
/** Max characters for the summary output */
|
||||
maxLength?: number;
|
||||
}
|
||||
): Promise<string> {
|
||||
const store = useConfigStore.getState();
|
||||
const threshold = options?.threshold ?? store.summarizeCharacterThreshold;
|
||||
const maxLength = options?.maxLength ?? store.summarizeMaxLength;
|
||||
|
||||
// Don't summarize if text is under threshold
|
||||
if (text.length <= threshold) {
|
||||
return text;
|
||||
}
|
||||
|
||||
try {
|
||||
const response = await fetch('/api/tts/summarize', {
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
},
|
||||
body: JSON.stringify({ text, threshold, maxLength }),
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const errorText = await response.text();
|
||||
console.error(`[summarize] HTTP error ${response.status}:`, errorText);
|
||||
throw new Error(`Summarization failed: ${response.status}`);
|
||||
}
|
||||
|
||||
const data = await response.json() as {
|
||||
summarized: boolean;
|
||||
summary?: string;
|
||||
reason?: string;
|
||||
originalLength?: number;
|
||||
summaryLength?: number;
|
||||
};
|
||||
|
||||
if (data.summarized && data.summary) {
|
||||
return data.summary;
|
||||
}
|
||||
|
||||
// Return original text if not summarized
|
||||
return text;
|
||||
} catch (err) {
|
||||
console.error('[summarize] Failed to summarize:', err);
|
||||
// Return original text on error
|
||||
return text;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if text should be summarized based on settings
|
||||
*/
|
||||
export function shouldSummarize(
|
||||
text: string,
|
||||
context: 'message' | 'voice'
|
||||
): boolean {
|
||||
const store = useConfigStore.getState();
|
||||
|
||||
const isEnabled = context === 'message'
|
||||
? store.summarizeMessageTTS
|
||||
: store.summarizeVoiceConversation;
|
||||
|
||||
if (!isEnabled) {
|
||||
return false;
|
||||
}
|
||||
|
||||
return text.length > store.summarizeCharacterThreshold;
|
||||
}
|
||||
|
||||
/**
|
||||
* Client-side text sanitization for TTS output.
|
||||
* Removes markdown, URLs, file paths, and other non-speakable content.
|
||||
* Applied as a fallback when server-side summarization is skipped.
|
||||
*/
|
||||
export function sanitizeForTTS(text: string): string {
|
||||
if (!text) return '';
|
||||
return text
|
||||
// Remove code blocks
|
||||
.replace(/```[\s\S]*?```/g, '')
|
||||
.replace(/`[^`]*`/g, '')
|
||||
// Remove markdown formatting
|
||||
.replace(/[*_~#]/g, '')
|
||||
// Remove URLs
|
||||
.replace(/https?:\/\/[^\s]+/g, '')
|
||||
// Remove file paths
|
||||
.replace(/\/[\w\-./]+/g, '')
|
||||
// Remove shell-like patterns
|
||||
.replace(/^\s*[$#>]\s*/gm, '')
|
||||
// Remove brackets and special chars
|
||||
.replace(/[[\]{}()<>|&;]/g, ' ')
|
||||
.replace(/\\/g, '')
|
||||
// Collapse whitespace
|
||||
.replace(/\s+/g, ' ')
|
||||
.trim();
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
/**
|
||||
* Static voice context configuration
|
||||
* Controls voice behavior and feature flags for the ElevenLabs voice agent
|
||||
*/
|
||||
export const VOICE_CONFIG = {
|
||||
/** Disable all tool call information from being sent to voice context */
|
||||
DISABLE_TOOL_CALLS: false,
|
||||
|
||||
/** Send only tool names and descriptions, exclude arguments */
|
||||
LIMITED_TOOL_CALLS: true,
|
||||
|
||||
/** Disable permission request forwarding */
|
||||
DISABLE_PERMISSION_REQUESTS: false,
|
||||
|
||||
/** Disable session online/offline notifications */
|
||||
DISABLE_SESSION_STATUS: true,
|
||||
|
||||
/** Disable message forwarding */
|
||||
DISABLE_MESSAGES: false,
|
||||
|
||||
/** Disable session focus notifications */
|
||||
DISABLE_SESSION_FOCUS: false,
|
||||
|
||||
/** Disable ready event notifications */
|
||||
DISABLE_READY_EVENTS: false,
|
||||
|
||||
/** Maximum number of messages to include in session history */
|
||||
MAX_HISTORY_MESSAGES: 50,
|
||||
|
||||
/** Enable debug logging for voice context updates */
|
||||
ENABLE_DEBUG_LOGGING: true,
|
||||
} as const;
|
||||
|
||||
/** Type for VOICE_CONFIG keys */
|
||||
export type VoiceConfigKey = keyof typeof VOICE_CONFIG;
|
||||
@@ -0,0 +1,133 @@
|
||||
/**
|
||||
* Voice hooks for session-to-voice event routing
|
||||
* Routes session events (messages, permissions, ready events) to the ElevenLabs
|
||||
* voice agent via contextual updates.
|
||||
*
|
||||
* This module provides hooks that can be called when session events occur,
|
||||
* using the voice session registry from voiceSession.ts.
|
||||
*
|
||||
* @example
|
||||
* ```typescript
|
||||
* import { voiceHooks } from '@/lib/voice';
|
||||
*
|
||||
* // Route session messages to voice
|
||||
* voiceHooks.onMessages(sessionId, messages);
|
||||
* ```
|
||||
*/
|
||||
|
||||
import { VOICE_CONFIG } from "./voiceConfig";
|
||||
import {
|
||||
formatNewMessages,
|
||||
formatPermissionRequest,
|
||||
formatReadyEvent,
|
||||
type VoiceMessage,
|
||||
} from "./contextFormatters";
|
||||
import { getVoiceSession, isVoiceSessionStarted } from "./voiceSession";
|
||||
|
||||
// Re-export registry functions from voiceSession.ts for convenience
|
||||
export {
|
||||
registerVoiceSession,
|
||||
unregisterVoiceSession,
|
||||
getVoiceSession,
|
||||
isVoiceSessionStarted,
|
||||
} from "./voiceSession";
|
||||
|
||||
/**
|
||||
* Report a contextual update to the voice session
|
||||
* Internal helper that checks preconditions and handles errors
|
||||
*
|
||||
* @param update - The text update to send, or null/undefined to skip
|
||||
*/
|
||||
function reportContextualUpdate(update: string | null | undefined): void {
|
||||
// Skip empty/null/undefined updates
|
||||
if (!update || update.trim().length === 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Skip if no voice session or not started
|
||||
const voiceSession = getVoiceSession();
|
||||
if (!voiceSession || !isVoiceSessionStarted()) {
|
||||
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
|
||||
console.log("[Voice] Skipping contextual update - no active session");
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
|
||||
console.log("[Voice] Sending contextual update:", update.substring(0, 100));
|
||||
}
|
||||
voiceSession.sendContextualUpdate(update);
|
||||
} catch (error) {
|
||||
// Log error but don't throw - voice updates shouldn't break the app
|
||||
console.error("[Voice] Failed to send contextual update:", error);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Voice hooks - exported functions to route session events to voice
|
||||
*
|
||||
* These hooks should be called when corresponding session events occur.
|
||||
* They respect VOICE_CONFIG feature flags to enable/disable specific
|
||||
* event types.
|
||||
*/
|
||||
export const voiceHooks = {
|
||||
/**
|
||||
* Called when new messages arrive in the session
|
||||
* Formats and sends messages to voice agent (if not disabled)
|
||||
*
|
||||
* @param sessionId - The session ID
|
||||
* @param messages - Array of messages to format and send
|
||||
*/
|
||||
onMessages(sessionId: string, messages: VoiceMessage[]): void {
|
||||
if (VOICE_CONFIG.DISABLE_MESSAGES) {
|
||||
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
|
||||
console.log("[Voice] Message forwarding disabled");
|
||||
}
|
||||
return;
|
||||
}
|
||||
reportContextualUpdate(formatNewMessages(sessionId, messages));
|
||||
},
|
||||
|
||||
/**
|
||||
* Called when a permission request is made
|
||||
* Announces the permission request to the voice agent (if not disabled)
|
||||
*
|
||||
* @param sessionId - The session ID
|
||||
* @param requestId - The permission request ID
|
||||
* @param toolName - Name of the tool requesting permission
|
||||
* @param toolArgs - Arguments for the tool (not sent to voice per LIMITED_TOOL_CALLS)
|
||||
*/
|
||||
onPermissionRequested(
|
||||
sessionId: string,
|
||||
requestId: string,
|
||||
toolName: string,
|
||||
toolArgs: unknown
|
||||
): void {
|
||||
if (VOICE_CONFIG.DISABLE_PERMISSION_REQUESTS) {
|
||||
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
|
||||
console.log("[Voice] Permission request forwarding disabled");
|
||||
}
|
||||
return;
|
||||
}
|
||||
reportContextualUpdate(
|
||||
formatPermissionRequest(sessionId, requestId, toolName, toolArgs)
|
||||
);
|
||||
},
|
||||
|
||||
/**
|
||||
* Called when the AI is ready for the next instruction
|
||||
* Announces ready state to the voice agent (if not disabled)
|
||||
*
|
||||
* @param sessionId - The session ID
|
||||
*/
|
||||
onReady(sessionId: string): void {
|
||||
if (VOICE_CONFIG.DISABLE_READY_EVENTS) {
|
||||
if (VOICE_CONFIG.ENABLE_DEBUG_LOGGING) {
|
||||
console.log("[Voice] Ready event forwarding disabled");
|
||||
}
|
||||
return;
|
||||
}
|
||||
reportContextualUpdate(formatReadyEvent(sessionId));
|
||||
},
|
||||
};
|
||||
@@ -0,0 +1,46 @@
|
||||
/**
|
||||
* Voice session interface
|
||||
* Used for type safety without importing ReturnType from SDK
|
||||
*/
|
||||
interface VoiceSession {
|
||||
sendContextualUpdate: (text: string) => void;
|
||||
}
|
||||
|
||||
/**
|
||||
* Global storage for the active voice session.
|
||||
* Used by voiceHooks to send contextual updates to the voice agent.
|
||||
*/
|
||||
let activeVoiceSession: VoiceSession | null = null;
|
||||
|
||||
/**
|
||||
* Register a voice session for use by voiceHooks.
|
||||
* Called by useVoice when a conversation is established.
|
||||
*/
|
||||
export function registerVoiceSession(session: VoiceSession): void {
|
||||
activeVoiceSession = session;
|
||||
console.log("[Voice] Session registered");
|
||||
}
|
||||
|
||||
/**
|
||||
* Unregister the active voice session.
|
||||
* Called by useVoice when the session ends.
|
||||
*/
|
||||
export function unregisterVoiceSession(): void {
|
||||
activeVoiceSession = null;
|
||||
console.log("[Voice] Session unregistered");
|
||||
}
|
||||
|
||||
/**
|
||||
* Get the currently registered voice session.
|
||||
* Used by voiceHooks to send contextual updates.
|
||||
*/
|
||||
export function getVoiceSession(): VoiceSession | null {
|
||||
return activeVoiceSession;
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if a voice session is currently active.
|
||||
*/
|
||||
export function isVoiceSessionStarted(): boolean {
|
||||
return activeVoiceSession !== null;
|
||||
}
|
||||
Reference in New Issue
Block a user