feat(voice): add voice input/output support with multiple providers (#281)
* feat(voice): add voice input/output support with multiple providers - Add BrowserVoiceButton component for Web Speech API voice input - Add VoiceProvider context for managing voice state across the app - Add TTS (Text-to-Speech) support with browser, macOS Say, and OpenAI providers - Add message TTS buttons to read assistant messages aloud - Add VoiceSettings page in OpenChamber settings - Add server endpoints for TTS and summarization services - Include slider component for voice rate/pitch/volume controls - Add hidden session support for background voice operations - Add Caddyfile for HTTPS support (required for microphone access) * fix: Build errors fixed and removed outdated ElevenLabs test code. * refactor(voice): use zen API with gpt-5-nano for TTS summarization Replace the hidden session + OpenCode SDK approach with direct calls to the opencode.ai zen API (same pattern used for commit message and PR description generation). - Rewrite summarization-service.js to call zen/v1/responses with gpt-5-nano - Remove hidden session logic (hiddenSession.ts, sessionStore filtering) - Remove summarizeModel setting and model selector from VoiceSettings - Simplify client-side summarize.ts to no longer pass model params - Clean up callers in useMessageTTS and useBrowserVoice * fix(voice): remove false 'voice not supported' warning in settings Mobile Safari does support voice but the isSupported check was incorrectly flagging it. Remove the warning banner entirely. * feat(voice): add configurable summary length limit for TTS output Add a slider (50-2000 chars) in voice settings to control max summary length. The limit is passed through the summarize endpoint and speak endpoint to the zen API prompt, with token budget scaled accordingly. * fix(voice): add diagnostic logging and sanitize TTS fallback Add console logging throughout the summarization flow (client + server) to trace why text may not be summarized. Fix silent error swallowing in /api/tts/speak. Always apply sanitizeForTTS even when summarization is disabled so raw markdown/code is never spoken verbatim. * fix(voice): fix token budget starving model of output tokens max_output_tokens includes both reasoning and output tokens. With effort:'low', reasoning alone consumes ~128 tokens, so a budget of 100 left zero tokens for the actual summary text. Use a fixed 1000 token budget (matching commit message generation) and control output length via the prompt's character limit instruction instead. * chore(voice): remove diagnostic logging from summarization flow * fix(voice): don't request mic permission on mobile page load Remove the useEffect that pre-requested microphone permission when the BrowserVoiceButton component mounted on mobile. This caused an unwanted permission prompt immediately on page load before the user tapped the mic icon. Permission is now only requested on explicit user interaction. * fix(voice): remove unused BrowserVoiceButton binding * fix(voice): desktop mic flow + non-continuous draft mode * fix(voice): stabilize continuous loop and polish controls * feat(settings): mark voice section experimental --------- Co-authored-by: Bohdan Triapitsyn <artmore@protonmail.com>
This commit is contained in:
committed by
GitHub
co-authored by
Bohdan Triapitsyn
parent
6776ac31c2
commit
1ed5316ac7
@@ -0,0 +1,247 @@
|
||||
/**
|
||||
* useSayTTS Hook
|
||||
*
|
||||
* React hook for macOS 'say' command text-to-speech playback.
|
||||
* Uses the native macOS speech synthesis via server API.
|
||||
* Uses Web Audio API for playback (better iOS Safari support).
|
||||
*
|
||||
* @example
|
||||
* ```typescript
|
||||
* const { speak, isPlaying, stop, isAvailable } = useSayTTS();
|
||||
*
|
||||
* // Speak text
|
||||
* await speak('Hello, this is a test');
|
||||
*
|
||||
* // Stop playback
|
||||
* stop();
|
||||
* ```
|
||||
*/
|
||||
|
||||
import { useCallback, useEffect, useRef, useState } from 'react';
|
||||
|
||||
export interface UseSayTTSReturn {
|
||||
/** Whether TTS is currently playing */
|
||||
isPlaying: boolean;
|
||||
/** Whether the macOS say command is available */
|
||||
isAvailable: boolean;
|
||||
/** Available voices */
|
||||
voices: Array<{ name: string; locale: string }>;
|
||||
/** Current error if any */
|
||||
error: string | null;
|
||||
/** Speak the given text */
|
||||
speak: (text: string, options?: SpeakOptions) => Promise<void>;
|
||||
/** Stop current playback */
|
||||
stop: () => void;
|
||||
/** Check if service is available */
|
||||
checkAvailability: () => Promise<boolean>;
|
||||
/** Unlock audio for mobile Safari - call this on user gesture */
|
||||
unlockAudio: () => Promise<void>;
|
||||
}
|
||||
|
||||
export interface SpeakOptions {
|
||||
/** Voice to use (defaults to Samantha) */
|
||||
voice?: string;
|
||||
/** Speech rate in words per minute (defaults to 200) */
|
||||
rate?: number;
|
||||
/** Callback when playback starts */
|
||||
onStart?: () => void;
|
||||
/** Callback when playback ends */
|
||||
onEnd?: () => void;
|
||||
/** Callback on error */
|
||||
onError?: (error: string) => void;
|
||||
}
|
||||
|
||||
// Shared AudioContext for Web Audio API playback (better iOS support)
|
||||
let sharedAudioContext: AudioContext | null = null;
|
||||
|
||||
function getAudioContext(): AudioContext {
|
||||
if (!sharedAudioContext) {
|
||||
sharedAudioContext = new (window.AudioContext || (window as unknown as { webkitAudioContext: typeof AudioContext }).webkitAudioContext)();
|
||||
}
|
||||
return sharedAudioContext;
|
||||
}
|
||||
|
||||
export function useSayTTS(): UseSayTTSReturn {
|
||||
const [isPlaying, setIsPlaying] = useState(false);
|
||||
const [isAvailable, setIsAvailable] = useState(false);
|
||||
const [voices, setVoices] = useState<Array<{ name: string; locale: string }>>([]);
|
||||
const [error, setError] = useState<string | null>(null);
|
||||
|
||||
const audioSourceRef = useRef<AudioBufferSourceNode | null>(null);
|
||||
const abortControllerRef = useRef<AbortController | null>(null);
|
||||
|
||||
// Unlock audio for mobile Safari - must be called within user gesture
|
||||
const unlockAudio = useCallback(async (): Promise<void> => {
|
||||
try {
|
||||
// Get or create AudioContext
|
||||
const ctx = getAudioContext();
|
||||
|
||||
// Resume if suspended (required for iOS Safari)
|
||||
if (ctx.state === 'suspended') {
|
||||
await ctx.resume();
|
||||
console.log('[useSayTTS] AudioContext resumed');
|
||||
}
|
||||
|
||||
// Play a tiny silent buffer to fully unlock
|
||||
const buffer = ctx.createBuffer(1, 1, 22050);
|
||||
const source = ctx.createBufferSource();
|
||||
source.buffer = buffer;
|
||||
source.connect(ctx.destination);
|
||||
source.start(0);
|
||||
|
||||
console.log('[useSayTTS] Audio unlocked for mobile playback');
|
||||
} catch (err) {
|
||||
console.error('[useSayTTS] Failed to unlock audio:', err);
|
||||
}
|
||||
}, []);
|
||||
|
||||
// Check if macOS say is available
|
||||
const checkAvailability = useCallback(async (): Promise<boolean> => {
|
||||
try {
|
||||
const response = await fetch('/api/tts/say/status');
|
||||
if (response.ok) {
|
||||
const data = await response.json();
|
||||
setIsAvailable(data.available);
|
||||
if (data.voices) {
|
||||
setVoices(data.voices);
|
||||
}
|
||||
return data.available;
|
||||
}
|
||||
setIsAvailable(false);
|
||||
return false;
|
||||
} catch (err) {
|
||||
console.error('[useSayTTS] Failed to check availability:', err);
|
||||
setIsAvailable(false);
|
||||
return false;
|
||||
}
|
||||
}, []);
|
||||
|
||||
// Check availability on mount
|
||||
useEffect(() => {
|
||||
checkAvailability();
|
||||
}, [checkAvailability]);
|
||||
|
||||
// Stop current playback
|
||||
const stop = useCallback(() => {
|
||||
if (audioSourceRef.current) {
|
||||
try {
|
||||
audioSourceRef.current.stop();
|
||||
} catch {
|
||||
// Already stopped
|
||||
}
|
||||
audioSourceRef.current = null;
|
||||
}
|
||||
|
||||
if (abortControllerRef.current) {
|
||||
abortControllerRef.current.abort();
|
||||
abortControllerRef.current = null;
|
||||
}
|
||||
|
||||
setIsPlaying(false);
|
||||
}, []);
|
||||
|
||||
// Speak text using macOS say
|
||||
const speak = useCallback(async (text: string, options?: SpeakOptions): Promise<void> => {
|
||||
// Stop any existing playback
|
||||
stop();
|
||||
|
||||
if (!text.trim()) {
|
||||
setError('No text to speak');
|
||||
options?.onError?.('No text to speak');
|
||||
return;
|
||||
}
|
||||
|
||||
setError(null);
|
||||
|
||||
try {
|
||||
// Create abort controller for this request
|
||||
abortControllerRef.current = new AbortController();
|
||||
|
||||
// Fetch audio from server
|
||||
const response = await fetch('/api/tts/say/speak', {
|
||||
method: 'POST',
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
},
|
||||
body: JSON.stringify({
|
||||
text: text.trim(),
|
||||
voice: options?.voice || 'Samantha',
|
||||
rate: options?.rate || 200,
|
||||
}),
|
||||
signal: abortControllerRef.current.signal,
|
||||
});
|
||||
|
||||
if (!response.ok) {
|
||||
const errorData = await response.json().catch(() => ({ error: 'Unknown error' }));
|
||||
throw new Error(errorData.error || `HTTP ${response.status}`);
|
||||
}
|
||||
|
||||
// Get audio data from response
|
||||
const audioBlob = await response.blob();
|
||||
const arrayBuffer = await audioBlob.arrayBuffer();
|
||||
console.log('[useSayTTS] Received audio:', audioBlob.size, 'bytes');
|
||||
|
||||
// Use Web Audio API for playback (same as useServerTTS)
|
||||
const ctx = getAudioContext();
|
||||
|
||||
// Resume context if suspended
|
||||
if (ctx.state === 'suspended') {
|
||||
await ctx.resume();
|
||||
console.log('[useSayTTS] AudioContext resumed before playback');
|
||||
}
|
||||
|
||||
// Decode audio data
|
||||
console.log('[useSayTTS] Decoding audio data...');
|
||||
const audioBuffer = await ctx.decodeAudioData(arrayBuffer);
|
||||
|
||||
// Create source node
|
||||
const source = ctx.createBufferSource();
|
||||
source.buffer = audioBuffer;
|
||||
source.connect(ctx.destination);
|
||||
audioSourceRef.current = source;
|
||||
|
||||
// Set up event handlers
|
||||
source.onended = () => {
|
||||
console.log('[useSayTTS] Audio playback ended');
|
||||
setIsPlaying(false);
|
||||
audioSourceRef.current = null;
|
||||
options?.onEnd?.();
|
||||
};
|
||||
|
||||
// Start playback
|
||||
console.log('[useSayTTS] Starting audio playback via Web Audio API...');
|
||||
setIsPlaying(true);
|
||||
options?.onStart?.();
|
||||
source.start(0);
|
||||
|
||||
} catch (err) {
|
||||
if ((err as Error).name === 'AbortError') {
|
||||
return;
|
||||
}
|
||||
|
||||
const errorMsg = err instanceof Error ? err.message : 'Failed to speak';
|
||||
console.error('[useSayTTS] Error:', errorMsg);
|
||||
setError(errorMsg);
|
||||
options?.onError?.(errorMsg);
|
||||
setIsPlaying(false);
|
||||
}
|
||||
}, [stop]);
|
||||
|
||||
// Cleanup on unmount
|
||||
useEffect(() => {
|
||||
return () => {
|
||||
stop();
|
||||
};
|
||||
}, [stop]);
|
||||
|
||||
return {
|
||||
isPlaying,
|
||||
isAvailable,
|
||||
voices,
|
||||
error,
|
||||
speak,
|
||||
stop,
|
||||
checkAvailability,
|
||||
unlockAudio,
|
||||
};
|
||||
}
|
||||
Reference in New Issue
Block a user