feat(voice): add voice input/output support with multiple providers (#281)

* feat(voice): add voice input/output support with multiple providers

- Add BrowserVoiceButton component for Web Speech API voice input
- Add VoiceProvider context for managing voice state across the app
- Add TTS (Text-to-Speech) support with browser, macOS Say, and OpenAI providers
- Add message TTS buttons to read assistant messages aloud
- Add VoiceSettings page in OpenChamber settings
- Add server endpoints for TTS and summarization services
- Include slider component for voice rate/pitch/volume controls
- Add hidden session support for background voice operations
- Add Caddyfile for HTTPS support (required for microphone access)

* fix: Build errors fixed and removed outdated ElevenLabs test code.

* refactor(voice): use zen API with gpt-5-nano for TTS summarization

Replace the hidden session + OpenCode SDK approach with direct calls
to the opencode.ai zen API (same pattern used for commit message and
PR description generation).

- Rewrite summarization-service.js to call zen/v1/responses with gpt-5-nano
- Remove hidden session logic (hiddenSession.ts, sessionStore filtering)
- Remove summarizeModel setting and model selector from VoiceSettings
- Simplify client-side summarize.ts to no longer pass model params
- Clean up callers in useMessageTTS and useBrowserVoice

* fix(voice): remove false 'voice not supported' warning in settings

Mobile Safari does support voice but the isSupported check was
incorrectly flagging it. Remove the warning banner entirely.

* feat(voice): add configurable summary length limit for TTS output

Add a slider (50-2000 chars) in voice settings to control max summary
length. The limit is passed through the summarize endpoint and speak
endpoint to the zen API prompt, with token budget scaled accordingly.

* fix(voice): add diagnostic logging and sanitize TTS fallback

Add console logging throughout the summarization flow (client + server)
to trace why text may not be summarized. Fix silent error swallowing in
/api/tts/speak. Always apply sanitizeForTTS even when summarization is
disabled so raw markdown/code is never spoken verbatim.

* fix(voice): fix token budget starving model of output tokens

max_output_tokens includes both reasoning and output tokens. With
effort:'low', reasoning alone consumes ~128 tokens, so a budget of
100 left zero tokens for the actual summary text. Use a fixed 1000
token budget (matching commit message generation) and control output
length via the prompt's character limit instruction instead.

* chore(voice): remove diagnostic logging from summarization flow

* fix(voice): don't request mic permission on mobile page load

Remove the useEffect that pre-requested microphone permission when the
BrowserVoiceButton component mounted on mobile. This caused an unwanted
permission prompt immediately on page load before the user tapped the
mic icon. Permission is now only requested on explicit user interaction.

* fix(voice): remove unused BrowserVoiceButton binding

* fix(voice): desktop mic flow + non-continuous draft mode

* fix(voice): stabilize continuous loop and polish controls

* feat(settings): mark voice section experimental

---------

Co-authored-by: Bohdan Triapitsyn <artmore@protonmail.com>
This commit is contained in:
gsxdsm
2026-02-09 23:55:10 +02:00
committed by GitHub
co-authored by Bohdan Triapitsyn
parent 6776ac31c2
commit 1ed5316ac7
40 changed files with 5518 additions and 736 deletions
+279
View File
@@ -0,0 +1,279 @@
/**
* useServerTTS Hook
*
* React hook for server-side text-to-speech playback.
* Fetches audio from the server and plays it, bypassing mobile Safari restrictions.
*
* @example
* ```typescript
* const { speak, isPlaying, stop, isAvailable } = useServerTTS();
*
* // Speak text
* await speak('Hello, this is a test');
*
* // Stop playback
* stop();
* ```
*/
import { useCallback, useEffect, useRef, useState } from 'react';
import { useConfigStore } from '@/stores/useConfigStore';
export interface UseServerTTSReturn {
/** Whether TTS is currently playing */
isPlaying: boolean;
/** Whether the server TTS service is available */
isAvailable: boolean;
/** Current error if any */
error: string | null;
/** Speak the given text */
speak: (text: string, options?: SpeakOptions) => Promise<void>;
/** Stop current playback */
stop: () => void;
/** Check if service is available */
checkAvailability: () => Promise<boolean>;
/** Unlock audio for mobile Safari - call this on user gesture before speaking */
unlockAudio: () => Promise<void>;
}
export interface SpeakOptions {
/** Voice to use (defaults to coral) */
voice?: string;
/** Speech speed (0.25 to 4.0, defaults to 1.0) */
speed?: number;
/** Optional instructions for the voice */
instructions?: string;
/** Summarize long text before speaking (defaults to true) */
summarize?: boolean;
/** Provider ID for summarization model */
providerId?: string;
/** Model ID for summarization */
modelId?: string;
/** Character threshold for summarization (defaults to 200) */
threshold?: number;
/** Callback when playback starts */
onStart?: () => void;
/** Callback when playback ends */
onEnd?: () => void;
/** Callback on error */
onError?: (error: string) => void;
}
// Shared AudioContext for Web Audio API playback (better iOS support)
let sharedAudioContext: AudioContext | null = null;
function getAudioContext(): AudioContext {
if (!sharedAudioContext) {
sharedAudioContext = new (window.AudioContext || (window as unknown as { webkitAudioContext: typeof AudioContext }).webkitAudioContext)();
}
return sharedAudioContext;
}
export function useServerTTS(): UseServerTTSReturn {
const [isPlaying, setIsPlaying] = useState(false);
const [isAvailable, setIsAvailable] = useState(false);
const [error, setError] = useState<string | null>(null);
const audioSourceRef = useRef<AudioBufferSourceNode | null>(null);
const abortControllerRef = useRef<AbortController | null>(null);
// Get current model, threshold, and max length from config store for summarization
const { currentProviderId, currentModelId, summarizeCharacterThreshold, summarizeMaxLength, openaiApiKey } = useConfigStore();
// Check if server TTS is available
const checkAvailability = useCallback(async (): Promise<boolean> => {
try {
const response = await fetch('/api/tts/status');
if (response.ok) {
const data = await response.json();
// Available if server has key OR user has provided their own key
const hasServerKey = data.available;
const hasClientKey = openaiApiKey && openaiApiKey.trim().length > 0;
const available = hasServerKey || hasClientKey;
setIsAvailable(available);
return available;
}
setIsAvailable(false);
return false;
} catch (err) {
console.error('[useServerTTS] Failed to check availability:', err);
setIsAvailable(false);
return false;
}
}, [openaiApiKey]);
// Check availability on mount and when API key changes
useEffect(() => {
checkAvailability();
}, [checkAvailability]);
// Stop current playback
const stop = useCallback(() => {
// Stop Web Audio API source
if (audioSourceRef.current) {
try {
audioSourceRef.current.stop();
} catch {
// Already stopped
}
audioSourceRef.current = null;
}
if (abortControllerRef.current) {
abortControllerRef.current.abort();
abortControllerRef.current = null;
}
setIsPlaying(false);
}, []);
// Pre-unlock audio for mobile Safari
// This must be called within a user gesture context
const unlockAudio = useCallback(async (): Promise<void> => {
try {
// Get or create AudioContext
const ctx = getAudioContext();
// Resume if suspended (required for iOS Safari)
if (ctx.state === 'suspended') {
await ctx.resume();
console.log('[useServerTTS] AudioContext resumed');
}
// Play a tiny silent buffer to fully unlock
const buffer = ctx.createBuffer(1, 1, 22050);
const source = ctx.createBufferSource();
source.buffer = buffer;
source.connect(ctx.destination);
source.start(0);
console.log('[useServerTTS] Audio unlocked for mobile playback');
} catch (err) {
console.error('[useServerTTS] Failed to unlock audio:', err);
}
}, []);
// Speak text using server TTS
const speak = useCallback(async (text: string, options?: SpeakOptions): Promise<void> => {
// Stop any existing playback
stop();
if (!text.trim()) {
setError('No text to speak');
options?.onError?.('No text to speak');
return;
}
setError(null);
try {
// Unlock audio context first (required for mobile Safari)
// Must be done before any async operations to stay within user gesture context
const ctx = getAudioContext();
if (ctx.state === 'suspended') {
await ctx.resume();
console.log('[useServerTTS] AudioContext resumed');
}
// Play a silent buffer to fully unlock audio on iOS
const silentBuffer = ctx.createBuffer(1, 1, 22050);
const silentSource = ctx.createBufferSource();
silentSource.buffer = silentBuffer;
silentSource.connect(ctx.destination);
silentSource.start(0);
// Create abort controller for this request
abortControllerRef.current = new AbortController();
const voice = options?.voice || 'nova';
console.log('[useServerTTS] Speaking with voice:', voice, 'options:', options);
// Fetch audio from server
const response = await fetch('/api/tts/speak', {
method: 'POST',
headers: {
'Content-Type': 'application/json',
},
body: JSON.stringify({
text: text.trim(),
voice,
speed: options?.speed || 0.9,
instructions: options?.instructions,
summarize: options?.summarize ?? true, // Summarize by default for voice output
// Use provided provider/model, or fall back to current chat model
providerId: options?.providerId || currentProviderId || undefined,
modelId: options?.modelId || currentModelId || undefined,
// Use provided threshold, or fall back to user setting, or default to 200
threshold: options?.threshold ?? summarizeCharacterThreshold ?? 200,
// Max character length for summaries
maxLength: summarizeMaxLength ?? 500,
// Send API key from settings if available
apiKey: openaiApiKey || undefined,
}),
signal: abortControllerRef.current.signal,
});
if (!response.ok) {
const errorData = await response.json().catch(() => ({ error: 'Unknown error' }));
throw new Error(errorData.error || `HTTP ${response.status}`);
}
// Get audio data from response
const audioBlob = await response.blob();
const arrayBuffer = await audioBlob.arrayBuffer();
// Decode audio data using the same context we unlocked earlier
console.log('[useServerTTS] Decoding audio data...');
const audioBuffer = await ctx.decodeAudioData(arrayBuffer);
// Create source node
const source = ctx.createBufferSource();
source.buffer = audioBuffer;
source.connect(ctx.destination);
audioSourceRef.current = source;
// Set up event handlers
source.onended = () => {
console.log('[useServerTTS] Audio playback ended');
setIsPlaying(false);
audioSourceRef.current = null;
options?.onEnd?.();
};
// Start playback
console.log('[useServerTTS] Starting audio playback via Web Audio API...');
setIsPlaying(true);
options?.onStart?.();
source.start(0);
} catch (err) {
if ((err as Error).name === 'AbortError') {
// Request was aborted, don't show error
return;
}
const errorMsg = err instanceof Error ? err.message : 'Failed to speak';
console.error('[useServerTTS] Error:', errorMsg);
setError(errorMsg);
options?.onError?.(errorMsg);
setIsPlaying(false);
}
}, [stop, currentProviderId, currentModelId, summarizeCharacterThreshold, summarizeMaxLength, openaiApiKey]);
// Cleanup on unmount
useEffect(() => {
return () => {
stop();
};
}, [stop]);
return {
isPlaying,
isAvailable,
error,
speak,
stop,
checkAvailability,
unlockAudio,
};
}