feat(tts): add OpenAI-compatible custom server TTS provider with configurable model, pitch, and volume (#859)

Co-authored-by: Alexander Busse <alex@ableph.net>
This commit is contained in:
Alexander Busse
2026-04-12 10:33:15 +03:00
committed by GitHub
co-authored by Alexander Busse
parent 8a836d4aed
commit 055b6f6af0
9 changed files with 285 additions and 77 deletions
@@ -715,7 +715,7 @@ const AssistantMessageBody: React.FC<Omit<MessageBodyProps, 'isUser'>> = ({
if (isTTSPlaying) { if (isTTSPlaying) {
return 'Stop speaking'; return 'Stop speaking';
} }
const providerLabel = voiceProvider === 'browser' ? 'Browser' : voiceProvider === 'openai' ? 'OpenAI' : 'Say'; const providerLabel = voiceProvider === 'browser' ? 'Browser' : voiceProvider === 'openai' ? 'OpenAI' : voiceProvider === 'openai-compatible' ? 'Custom' : 'Say';
return `Read aloud (${providerLabel} voice)`; return `Read aloud (${providerLabel} voice)`;
}, [isTTSPlaying, voiceProvider]); }, [isTTSPlaying, voiceProvider]);
@@ -18,7 +18,6 @@ import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip
import { browserVoiceService } from '@/lib/voice/browserVoiceService'; import { browserVoiceService } from '@/lib/voice/browserVoiceService';
import { audioStreamService } from '@/lib/voice/audioStreamService'; import { audioStreamService } from '@/lib/voice/audioStreamService';
import { cn } from '@/lib/utils'; import { cn } from '@/lib/utils';
const LANGUAGE_OPTIONS = [ const LANGUAGE_OPTIONS = [
{ value: 'en-US', label: 'English' }, { value: 'en-US', label: 'English' },
{ value: 'es-ES', label: 'Español' }, { value: 'es-ES', label: 'Español' },
@@ -71,6 +70,12 @@ export const VoiceSettings: React.FC = () => {
const setOpenaiVoice = useConfigStore((state) => state.setOpenaiVoice); const setOpenaiVoice = useConfigStore((state) => state.setOpenaiVoice);
const openaiApiKey = useConfigStore((state) => state.openaiApiKey); const openaiApiKey = useConfigStore((state) => state.openaiApiKey);
const setOpenaiApiKey = useConfigStore((state) => state.setOpenaiApiKey); const setOpenaiApiKey = useConfigStore((state) => state.setOpenaiApiKey);
const openaiCompatibleUrl = useConfigStore((state) => state.openaiCompatibleUrl);
const setOpenaiCompatibleUrl = useConfigStore((state) => state.setOpenaiCompatibleUrl);
const openaiCompatibleVoice = useConfigStore((state) => state.openaiCompatibleVoice);
const setOpenaiCompatibleVoice = useConfigStore((state) => state.setOpenaiCompatibleVoice);
const openaiCompatibleTtsModel = useConfigStore((state) => state.openaiCompatibleTtsModel);
const setOpenaiCompatibleTtsModel = useConfigStore((state) => state.setOpenaiCompatibleTtsModel);
const showMessageTTSButtons = useConfigStore((state) => state.showMessageTTSButtons); const showMessageTTSButtons = useConfigStore((state) => state.showMessageTTSButtons);
// STT settings // STT settings
const sttProvider = useConfigStore((state) => state.sttProvider); const sttProvider = useConfigStore((state) => state.sttProvider);
@@ -106,6 +111,9 @@ export const VoiceSettings: React.FC = () => {
const [isOpenAIPreviewPlaying, setIsOpenAIPreviewPlaying] = useState(false); const [isOpenAIPreviewPlaying, setIsOpenAIPreviewPlaying] = useState(false);
const [openaiPreviewAudio, setOpenaiPreviewAudio] = useState<HTMLAudioElement | null>(null); const [openaiPreviewAudio, setOpenaiPreviewAudio] = useState<HTMLAudioElement | null>(null);
const [isCompatiblePreviewPlaying, setIsCompatiblePreviewPlaying] = useState(false);
const [compatiblePreviewAudio, setCompatiblePreviewAudio] = useState<HTMLAudioElement | null>(null);
const [browserVoices, setBrowserVoices] = useState<SpeechSynthesisVoice[]>([]); const [browserVoices, setBrowserVoices] = useState<SpeechSynthesisVoice[]>([]);
const [isBrowserPreviewPlaying, setIsBrowserPreviewPlaying] = useState(false); const [isBrowserPreviewPlaying, setIsBrowserPreviewPlaying] = useState(false);
@@ -182,7 +190,7 @@ export const VoiceSettings: React.FC = () => {
}, [isBrowserPreviewPlaying]); }, [isBrowserPreviewPlaying]);
useEffect(() => { useEffect(() => {
if (!voiceModeEnabled || voiceProvider !== 'openai') { if (!voiceModeEnabled || (voiceProvider !== 'openai' && voiceProvider !== 'openai-compatible')) {
setIsOpenAIAvailable(openaiApiKey.trim().length > 0); setIsOpenAIAvailable(openaiApiKey.trim().length > 0);
return; return;
} }
@@ -339,6 +347,67 @@ export const VoiceSettings: React.FC = () => {
}; };
}, [openaiPreviewAudio]); }, [openaiPreviewAudio]);
const previewCompatibleVoice = useCallback(async () => {
if (compatiblePreviewAudio) {
compatiblePreviewAudio.pause();
compatiblePreviewAudio.currentTime = 0;
setCompatiblePreviewAudio(null);
setIsCompatiblePreviewPlaying(false);
return;
}
if (!openaiCompatibleUrl.trim()) return;
setIsCompatiblePreviewPlaying(true);
try {
const response = await fetch('/api/tts/speak', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({
text: `Hello! This is a preview of the custom TTS server.`,
voice: openaiCompatibleVoice,
model: openaiCompatibleTtsModel || undefined,
speed: speechRate,
baseURL: openaiCompatibleUrl,
}),
});
if (!response.ok) {
const errorData = await response.json().catch(() => ({ error: 'Unknown error' }));
throw new Error(errorData.error || `HTTP ${response.status}`);
}
const blob = await response.blob();
const url = URL.createObjectURL(blob);
const audio = new Audio(url);
audio.onended = () => {
URL.revokeObjectURL(url);
setCompatiblePreviewAudio(null);
setIsCompatiblePreviewPlaying(false);
};
audio.onerror = () => {
URL.revokeObjectURL(url);
setCompatiblePreviewAudio(null);
setIsCompatiblePreviewPlaying(false);
};
setCompatiblePreviewAudio(audio);
await audio.play();
} catch {
setIsCompatiblePreviewPlaying(false);
}
}, [openaiCompatibleUrl, openaiCompatibleVoice, openaiCompatibleTtsModel, speechRate, compatiblePreviewAudio]);
useEffect(() => {
return () => {
if (compatiblePreviewAudio) {
compatiblePreviewAudio.pause();
}
};
}, [compatiblePreviewAudio]);
const sliderClass = "flex-1 min-w-0 h-1.5 bg-[var(--interactive-border)] rounded-full appearance-none cursor-pointer [&::-webkit-slider-thumb]:appearance-none [&::-webkit-slider-thumb]:w-4 [&::-webkit-slider-thumb]:h-4 [&::-webkit-slider-thumb]:rounded-full [&::-webkit-slider-thumb]:bg-[var(--primary-base)] [&::-moz-range-thumb]:w-4 [&::-moz-range-thumb]:h-4 [&::-moz-range-thumb]:rounded-full [&::-moz-range-thumb]:bg-[var(--primary-base)] [&::-moz-range-thumb]:border-0 disabled:opacity-50"; const sliderClass = "flex-1 min-w-0 h-1.5 bg-[var(--interactive-border)] rounded-full appearance-none cursor-pointer [&::-webkit-slider-thumb]:appearance-none [&::-webkit-slider-thumb]:w-4 [&::-webkit-slider-thumb]:h-4 [&::-webkit-slider-thumb]:rounded-full [&::-webkit-slider-thumb]:bg-[var(--primary-base)] [&::-moz-range-thumb]:w-4 [&::-moz-range-thumb]:h-4 [&::-moz-range-thumb]:rounded-full [&::-moz-range-thumb]:bg-[var(--primary-base)] [&::-moz-range-thumb]:border-0 disabled:opacity-50";
return ( return (
@@ -380,6 +449,7 @@ export const VoiceSettings: React.FC = () => {
<ul className="space-y-1"> <ul className="space-y-1">
<li><strong>Browser:</strong> Free, offline, limited mobile support.</li> <li><strong>Browser:</strong> Free, offline, limited mobile support.</li>
<li><strong>OpenAI:</strong> High quality, mobile ready, needs API key.</li> <li><strong>OpenAI:</strong> High quality, mobile ready, needs API key.</li>
<li><strong>Custom:</strong> OpenAI-compatible server (e.g. Kokoro).</li>
<li><strong>Say:</strong> macOS native. Fast, free, offline.</li> <li><strong>Say:</strong> macOS native. Fast, free, offline.</li>
</ul> </ul>
</TooltipContent> </TooltipContent>
@@ -412,6 +482,19 @@ export const VoiceSettings: React.FC = () => {
> >
OpenAI OpenAI
</Button> </Button>
<Button
variant="outline"
size="xs"
onClick={() => setVoiceProvider('openai-compatible')}
className={cn(
'!font-normal',
voiceProvider === 'openai-compatible'
? 'border-[var(--primary-base)] text-[var(--primary-base)] bg-[var(--primary-base)]/10 hover:text-[var(--primary-base)]'
: 'text-foreground'
)}
>
Custom
</Button>
{isSayAvailable && ( {isSayAvailable && (
<Button <Button
variant="outline" variant="outline"
@@ -462,6 +545,70 @@ export const VoiceSettings: React.FC = () => {
</div> </div>
)} )}
{/* OpenAI-compatible custom server */}
{voiceProvider === 'openai-compatible' && (
<div className="py-1.5 space-y-2">
<div>
<span className={cn("typography-ui-label text-foreground", !openaiCompatibleUrl.trim() && "text-[var(--status-error)]")}>
Server URL
</span>
<span className="typography-meta ml-2 text-muted-foreground">
Base URL of the OpenAI-compatible TTS server
</span>
<div className="relative mt-1.5 max-w-xs">
<input
type="text"
value={openaiCompatibleUrl}
onChange={(e) => setOpenaiCompatibleUrl(e.target.value)}
placeholder="http://localhost:8880/v1"
className="w-full h-7 rounded-lg border border-input bg-transparent px-2 typography-ui-label text-foreground placeholder:text-muted-foreground focus:outline-none focus:ring-1 focus:ring-primary/50 focus:border-primary/70"
/>
{openaiCompatibleUrl && (
<button
type="button"
onClick={() => setOpenaiCompatibleUrl('')}
className="absolute right-2 top-1/2 -translate-y-1/2 text-muted-foreground hover:text-foreground"
>
<RiCloseLine className="w-3.5 h-3.5" />
</button>
)}
</div>
</div>
<div>
<span className="typography-ui-label text-foreground">Model</span>
<div className="relative mt-1.5 max-w-xs">
<input
type="text"
value={openaiCompatibleTtsModel}
onChange={(e) => setOpenaiCompatibleTtsModel(e.target.value)}
placeholder="speaches-ai/Kokoro-82M-v1.0-ONNX"
className="w-full h-7 rounded-lg border border-input bg-transparent px-2 typography-ui-label text-foreground placeholder:text-muted-foreground focus:outline-none focus:ring-1 focus:ring-primary/50 focus:border-primary/70"
/>
</div>
</div>
<div>
<span className="typography-ui-label text-foreground">Voice</span>
<span className="typography-meta ml-2 text-muted-foreground">
Voice identifier supported by the server
</span>
<div className="flex items-center gap-2 mt-1.5">
<div className="relative max-w-xs flex-1">
<input
type="text"
value={openaiCompatibleVoice}
onChange={(e) => setOpenaiCompatibleVoice(e.target.value)}
placeholder="af_sky"
className="w-full h-7 rounded-lg border border-input bg-transparent px-2 typography-ui-label text-foreground placeholder:text-muted-foreground focus:outline-none focus:ring-1 focus:ring-primary/50 focus:border-primary/70"
/>
</div>
<Button size="xs" variant="ghost" onClick={previewCompatibleVoice} title="Preview" disabled={!openaiCompatibleUrl.trim()}>
{isCompatiblePreviewPlaying ? <RiStopLine className="w-3.5 h-3.5" /> : <RiPlayLine className="w-3.5 h-3.5" />}
</Button>
</div>
</div>
</div>
)}
{/* Voice Selection */} {/* Voice Selection */}
<div className="flex items-center gap-8 py-1.5"> <div className="flex items-center gap-8 py-1.5">
<span className="typography-ui-label text-foreground sm:w-56 shrink-0">Voice</span> <span className="typography-ui-label text-foreground sm:w-56 shrink-0">Voice</span>
@@ -484,6 +631,10 @@ export const VoiceSettings: React.FC = () => {
</> </>
)} )}
{voiceProvider === 'openai-compatible' && (
<span className="typography-meta text-muted-foreground">Configured above</span>
)}
{voiceProvider === 'say' && isSayAvailable && sayVoices.length > 0 && ( {voiceProvider === 'say' && isSayAvailable && sayVoices.length > 0 && (
<> <>
<Select value={sayVoice} onValueChange={setSayVoice}> <Select value={sayVoice} onValueChange={setSayVoice}>
+16 -9
View File
@@ -62,7 +62,7 @@ export interface UseBrowserVoiceReturn {
/** Whether the device is mobile */ /** Whether the device is mobile */
isMobile: boolean; isMobile: boolean;
/** Current voice provider */ /** Current voice provider */
voiceProvider: 'browser' | 'openai' | 'say'; voiceProvider: 'browser' | 'openai' | 'openai-compatible' | 'say';
} }
// Storage key for persisting language preference // Storage key for persisting language preference
@@ -144,10 +144,11 @@ export function useBrowserVoice(): UseBrowserVoiceReturn {
const openaiVoice = useConfigStore((state) => state.openaiVoice); const openaiVoice = useConfigStore((state) => state.openaiVoice);
const openaiCompatibleVoice = useConfigStore((state) => state.openaiCompatibleVoice); const openaiCompatibleVoice = useConfigStore((state) => state.openaiCompatibleVoice);
const openaiCompatibleUrl = useConfigStore((state) => state.openaiCompatibleUrl); const openaiCompatibleUrl = useConfigStore((state) => state.openaiCompatibleUrl);
const openaiCompatibleTtsModel = useConfigStore((state) => state.openaiCompatibleTtsModel);
const summarizeVoiceConversation = useConfigStore((state) => state.summarizeVoiceConversation); const summarizeVoiceConversation = useConfigStore((state) => state.summarizeVoiceConversation);
const summarizeCharacterThreshold = useConfigStore((state) => state.summarizeCharacterThreshold); const summarizeCharacterThreshold = useConfigStore((state) => state.summarizeCharacterThreshold);
const shouldCheckOpenAIAvailability = voiceModeEnabled && voiceProvider === 'openai'; const shouldCheckOpenAIAvailability = voiceModeEnabled && (voiceProvider === 'openai' || voiceProvider === 'openai-compatible');
const shouldCheckSayAvailability = voiceModeEnabled && voiceProvider === 'say'; const shouldCheckSayAvailability = voiceModeEnabled && voiceProvider === 'say';
// STT provider config // STT provider config
@@ -462,12 +463,19 @@ export function useBrowserVoice(): UseBrowserVoiceReturn {
} }
}; };
// Use server TTS when OpenAI provider is selected and available // Use server TTS when OpenAI (or OpenAI-compatible) provider is selected and available
if (voiceProvider === 'openai' && isServerTTSAvailable) { if ((voiceProvider === 'openai' || voiceProvider === 'openai-compatible') && isServerTTSAvailable) {
console.log('[useBrowserVoice] Using OpenAI server TTS with voice:', openaiVoice); const ttsVoice = voiceProvider === 'openai-compatible' ? openaiCompatibleVoice : openaiVoice;
const ttsBaseURL = voiceProvider === 'openai-compatible' ? openaiCompatibleUrl : undefined;
const ttsModel = voiceProvider === 'openai-compatible' ? openaiCompatibleTtsModel : undefined;
console.log('[useBrowserVoice] Using server TTS with voice:', ttsVoice, 'provider:', voiceProvider);
await speakServerTTS(textToSpeak, { await speakServerTTS(textToSpeak, {
voice: openaiVoice, voice: ttsVoice,
model: ttsModel,
speed: speechRate, speed: speechRate,
pitch: speechPitch,
volume: speechVolume,
baseURL: ttsBaseURL,
onStart: () => console.log('[useBrowserVoice] Server TTS started'), onStart: () => console.log('[useBrowserVoice] Server TTS started'),
onEnd: () => { onEnd: () => {
console.log('[useBrowserVoice] Server TTS ended'); console.log('[useBrowserVoice] Server TTS ended');
@@ -475,8 +483,7 @@ export function useBrowserVoice(): UseBrowserVoiceReturn {
}, },
onError: (errorMsg) => { onError: (errorMsg) => {
console.error('[useBrowserVoice] Server TTS error:', errorMsg); console.error('[useBrowserVoice] Server TTS error:', errorMsg);
// Show error to user when OpenAI voice fails setError(`Voice TTS failed: ${errorMsg}. Please check your settings or switch to Browser voice.`);
setError(`OpenAI voice failed: ${errorMsg}. Please check your OpenAI API key or switch to Browser voice.`);
setStatus('error'); setStatus('error');
restartListening(); restartListening();
} }
@@ -573,7 +580,7 @@ export function useBrowserVoice(): UseBrowserVoiceReturn {
setStatus('error'); setStatus('error');
processingMessageRef.current = false; processingMessageRef.current = false;
} }
}, [currentSessionId, currentProviderId, currentModelId, currentAgentName, language, sendMessage, setPendingInputText, createSession, speechRate, speechPitch, speechVolume, isMobile, isServerTTSAvailable, speakServerTTS, isSayTTSAvailable, speakSayTTS, voiceProvider, sayVoice, browserVoice, openaiVoice, openaiCompatibleVoice, openaiCompatibleUrl, summarizeVoiceConversation, summarizeCharacterThreshold, conversationMode, sttProvider]); }, [currentSessionId, currentProviderId, currentModelId, currentAgentName, language, sendMessage, setPendingInputText, createSession, speechRate, speechPitch, speechVolume, isMobile, isServerTTSAvailable, speakServerTTS, isSayTTSAvailable, speakSayTTS, voiceProvider, sayVoice, browserVoice, openaiVoice, openaiCompatibleVoice, openaiCompatibleUrl, openaiCompatibleTtsModel, summarizeVoiceConversation, summarizeCharacterThreshold, conversationMode, sttProvider]);
// Handle speech recognition result // Handle speech recognition result
const handleSpeechResult = useCallback(async (text: string, isFinal: boolean) => { const handleSpeechResult = useCallback(async (text: string, isFinal: boolean) => {
+18 -3
View File
@@ -31,11 +31,15 @@ export function useMessageTTS(): UseMessageTTSReturn {
const sayVoice = useConfigStore((state) => state.sayVoice); const sayVoice = useConfigStore((state) => state.sayVoice);
const browserVoice = useConfigStore((state) => state.browserVoice); const browserVoice = useConfigStore((state) => state.browserVoice);
const openaiVoice = useConfigStore((state) => state.openaiVoice); const openaiVoice = useConfigStore((state) => state.openaiVoice);
const openaiCompatibleVoice = useConfigStore((state) => state.openaiCompatibleVoice);
const openaiCompatibleUrl = useConfigStore((state) => state.openaiCompatibleUrl);
const openaiCompatibleTtsModel = useConfigStore((state) => state.openaiCompatibleTtsModel);
const summarizeMessageTTS = useConfigStore((state) => state.summarizeMessageTTS); const summarizeMessageTTS = useConfigStore((state) => state.summarizeMessageTTS);
const summarizeCharacterThreshold = useConfigStore((state) => state.summarizeCharacterThreshold); const summarizeCharacterThreshold = useConfigStore((state) => state.summarizeCharacterThreshold);
const showMessageTTSButtons = useConfigStore((state) => state.showMessageTTSButtons); const showMessageTTSButtons = useConfigStore((state) => state.showMessageTTSButtons);
const shouldCheckOpenAIAvailability = showMessageTTSButtons && voiceProvider === 'openai'; const isServerProvider = voiceProvider === 'openai' || voiceProvider === 'openai-compatible';
const shouldCheckOpenAIAvailability = showMessageTTSButtons && isServerProvider;
const shouldCheckSayAvailability = showMessageTTSButtons && voiceProvider === 'say'; const shouldCheckSayAvailability = showMessageTTSButtons && voiceProvider === 'say';
const { speak: speakServerTTS, stop: stopServerTTS, isAvailable: isServerTTSAvailable } = useServerTTS({ const { speak: speakServerTTS, stop: stopServerTTS, isAvailable: isServerTTSAvailable } = useServerTTS({
@@ -72,11 +76,18 @@ export function useMessageTTS(): UseMessageTTSReturn {
textToSpeak = sanitizeForTTS(text); textToSpeak = sanitizeForTTS(text);
} }
if (voiceProvider === 'openai' && isServerTTSAvailable) { if (isServerProvider && isServerTTSAvailable) {
const voice = voiceProvider === 'openai-compatible' ? openaiCompatibleVoice : openaiVoice;
const baseURL = voiceProvider === 'openai-compatible' ? openaiCompatibleUrl : undefined;
const model = voiceProvider === 'openai-compatible' ? openaiCompatibleTtsModel : undefined;
await speakServerTTS(textToSpeak, { await speakServerTTS(textToSpeak, {
voice: openaiVoice, voice,
model,
speed: speechRate, speed: speechRate,
pitch: speechPitch,
volume: speechVolume,
summarize: false, // We already summarized client-side summarize: false, // We already summarized client-side
baseURL,
onEnd: () => setIsPlaying(false), onEnd: () => setIsPlaying(false),
onError: () => setIsPlaying(false), onError: () => setIsPlaying(false),
}); });
@@ -110,12 +121,16 @@ export function useMessageTTS(): UseMessageTTSReturn {
} }
}, [ }, [
voiceProvider, voiceProvider,
isServerProvider,
speechRate, speechRate,
speechPitch, speechPitch,
speechVolume, speechVolume,
sayVoice, sayVoice,
browserVoice, browserVoice,
openaiVoice, openaiVoice,
openaiCompatibleVoice,
openaiCompatibleUrl,
openaiCompatibleTtsModel,
summarizeMessageTTS, summarizeMessageTTS,
summarizeCharacterThreshold, summarizeCharacterThreshold,
isServerTTSAvailable, isServerTTSAvailable,
+29 -3
View File
@@ -85,8 +85,14 @@ export interface UseServerTTSReturn {
export interface SpeakOptions { export interface SpeakOptions {
/** Voice to use (defaults to coral) */ /** Voice to use (defaults to coral) */
voice?: string; voice?: string;
/** Model to use (defaults to gpt-4o-mini-tts) */
model?: string;
/** Speech speed (0.25 to 4.0, defaults to 1.0) */ /** Speech speed (0.25 to 4.0, defaults to 1.0) */
speed?: number; speed?: number;
/** Speech pitch shift (0.5 to 2.0, mapped to cents; 1.0 = no shift) */
pitch?: number;
/** Playback volume (0 to 1, defaults to 1.0) */
volume?: number;
/** Optional instructions for the voice */ /** Optional instructions for the voice */
instructions?: string; instructions?: string;
/** Summarize long text before speaking (defaults to true) */ /** Summarize long text before speaking (defaults to true) */
@@ -97,6 +103,8 @@ export interface SpeakOptions {
modelId?: string; modelId?: string;
/** Character threshold for summarization (defaults to 200) */ /** Character threshold for summarization (defaults to 200) */
threshold?: number; threshold?: number;
/** Custom base URL for OpenAI-compatible server */
baseURL?: string;
/** Callback when playback starts */ /** Callback when playback starts */
onStart?: () => void; onStart?: () => void;
/** Callback when playback ends */ /** Callback when playback ends */
@@ -130,6 +138,7 @@ export function useServerTTS(options: UseServerTTSOptions = {}): UseServerTTSRet
const summarizeCharacterThreshold = useConfigStore((state) => state.summarizeCharacterThreshold); const summarizeCharacterThreshold = useConfigStore((state) => state.summarizeCharacterThreshold);
const summarizeMaxLength = useConfigStore((state) => state.summarizeMaxLength); const summarizeMaxLength = useConfigStore((state) => state.summarizeMaxLength);
const openaiApiKey = useConfigStore((state) => state.openaiApiKey); const openaiApiKey = useConfigStore((state) => state.openaiApiKey);
const openaiCompatibleUrl = useConfigStore((state) => state.openaiCompatibleUrl);
const settingsZenModel = useConfigStore((state) => state.settingsZenModel); const settingsZenModel = useConfigStore((state) => state.settingsZenModel);
// Check if server TTS is available // Check if server TTS is available
@@ -140,7 +149,8 @@ export function useServerTTS(options: UseServerTTSOptions = {}): UseServerTTSRet
} }
const hasClientKey = Boolean(openaiApiKey && openaiApiKey.trim().length > 0); const hasClientKey = Boolean(openaiApiKey && openaiApiKey.trim().length > 0);
if (hasClientKey) { const hasCustomUrl = Boolean(openaiCompatibleUrl && openaiCompatibleUrl.trim().length > 0);
if (hasClientKey || hasCustomUrl) {
setIsAvailable(true); setIsAvailable(true);
return true; return true;
} }
@@ -153,7 +163,7 @@ export function useServerTTS(options: UseServerTTSOptions = {}): UseServerTTSRet
setIsAvailable(false); setIsAvailable(false);
return false; return false;
} }
}, [enabled, openaiApiKey]); }, [enabled, openaiApiKey, openaiCompatibleUrl]);
// Check availability on mount and when API key changes // Check availability on mount and when API key changes
useEffect(() => { useEffect(() => {
@@ -250,6 +260,7 @@ export function useServerTTS(options: UseServerTTSOptions = {}): UseServerTTSRet
body: JSON.stringify({ body: JSON.stringify({
text: text.trim(), text: text.trim(),
voice, voice,
model: options?.model || undefined,
speed: options?.speed || 0.9, speed: options?.speed || 0.9,
instructions: options?.instructions, instructions: options?.instructions,
summarize: options?.summarize ?? true, // Summarize by default for voice output summarize: options?.summarize ?? true, // Summarize by default for voice output
@@ -262,6 +273,8 @@ export function useServerTTS(options: UseServerTTSOptions = {}): UseServerTTSRet
maxLength: summarizeMaxLength ?? 500, maxLength: summarizeMaxLength ?? 500,
// Send API key from settings if available // Send API key from settings if available
apiKey: openaiApiKey || undefined, apiKey: openaiApiKey || undefined,
// Send custom base URL for OpenAI-compatible servers
baseURL: options?.baseURL || undefined,
...(settingsZenModel ? { zenModel: settingsZenModel } : {}), ...(settingsZenModel ? { zenModel: settingsZenModel } : {}),
}), }),
signal: abortControllerRef.current.signal, signal: abortControllerRef.current.signal,
@@ -283,7 +296,20 @@ export function useServerTTS(options: UseServerTTSOptions = {}): UseServerTTSRet
// Create source node // Create source node
const source = ctx.createBufferSource(); const source = ctx.createBufferSource();
source.buffer = audioBuffer; source.buffer = audioBuffer;
source.connect(ctx.destination);
// Apply pitch shift via detune (cents): 1200 cents = 1 octave
const pitch = options?.pitch ?? 1.0;
if (pitch !== 1.0) {
source.detune.value = (pitch - 1.0) * 1200;
}
// Apply volume via GainNode
const volume = options?.volume ?? 1.0;
const gainNode = ctx.createGain();
gainNode.gain.value = volume;
source.connect(gainNode);
gainNode.connect(ctx.destination);
audioSourceRef.current = source; audioSourceRef.current = source;
// Set up event handlers // Set up event handlers
+22 -5
View File
@@ -466,9 +466,9 @@ interface ConfigStore {
settingsAutoCreateWorktree: boolean; settingsAutoCreateWorktree: boolean;
settingsGitmojiEnabled: boolean; settingsGitmojiEnabled: boolean;
settingsZenModel: string | undefined; settingsZenModel: string | undefined;
// Voice provider preference ('browser', 'openai', or 'say' for macOS) // Voice provider preference ('browser', 'openai', 'openai-compatible', or 'say' for macOS)
voiceProvider: 'browser' | 'openai' | 'say'; voiceProvider: 'browser' | 'openai' | 'openai-compatible' | 'say';
setVoiceProvider: (provider: 'browser' | 'openai' | 'say') => void; setVoiceProvider: (provider: 'browser' | 'openai' | 'openai-compatible' | 'say') => void;
// TTS settings // TTS settings
speechRate: number; speechRate: number;
speechPitch: number; speechPitch: number;
@@ -479,6 +479,7 @@ interface ConfigStore {
openaiApiKey: string; openaiApiKey: string;
openaiCompatibleUrl: string; openaiCompatibleUrl: string;
openaiCompatibleVoice: string; openaiCompatibleVoice: string;
openaiCompatibleTtsModel: string;
// STT (speech-to-text) settings // STT (speech-to-text) settings
sttProvider: 'browser' | 'server'; sttProvider: 'browser' | 'server';
sttServerUrl: string; sttServerUrl: string;
@@ -502,6 +503,7 @@ interface ConfigStore {
setOpenaiApiKey: (apiKey: string) => void; setOpenaiApiKey: (apiKey: string) => void;
setOpenaiCompatibleUrl: (url: string) => void; setOpenaiCompatibleUrl: (url: string) => void;
setOpenaiCompatibleVoice: (voice: string) => void; setOpenaiCompatibleVoice: (voice: string) => void;
setOpenaiCompatibleTtsModel: (model: string) => void;
setSttProvider: (provider: 'browser' | 'server') => void; setSttProvider: (provider: 'browser' | 'server') => void;
setSttServerUrl: (url: string) => void; setSttServerUrl: (url: string) => void;
setSttModel: (model: string) => void; setSttModel: (model: string) => void;
@@ -585,7 +587,7 @@ export const useConfigStore = create<ConfigStore>()(
voiceProvider: (() => { voiceProvider: (() => {
if (typeof window !== 'undefined') { if (typeof window !== 'undefined') {
const saved = localStorage.getItem('voiceProvider'); const saved = localStorage.getItem('voiceProvider');
if (saved === 'openai' || saved === 'browser' || saved === 'say') return saved; if (saved === 'openai' || saved === 'browser' || saved === 'say' || saved === 'openai-compatible') return saved;
} }
return 'browser'; return 'browser';
})(), })(),
@@ -668,6 +670,14 @@ export const useConfigStore = create<ConfigStore>()(
} }
return 'af_sky'; return 'af_sky';
})(), })(),
// OpenAI-compatible custom server TTS model
openaiCompatibleTtsModel: (() => {
if (typeof window !== 'undefined') {
const saved = localStorage.getItem('openaiCompatibleTtsModel');
if (saved && saved !== 'speaches-ai/Kokoro-82M-v1.0-ONNX') return saved;
}
return 'kokoro';
})(),
// STT provider: 'browser' (Web Speech API) or 'server' (OpenAI-compat) // STT provider: 'browser' (Web Speech API) or 'server' (OpenAI-compat)
sttProvider: (() => { sttProvider: (() => {
if (typeof window !== 'undefined') { if (typeof window !== 'undefined') {
@@ -1685,7 +1695,7 @@ export const useConfigStore = create<ConfigStore>()(
}); });
}, },
setVoiceProvider: (provider: 'browser' | 'openai' | 'say') => { setVoiceProvider: (provider: 'browser' | 'openai' | 'openai-compatible' | 'say') => {
set({ voiceProvider: provider }); set({ voiceProvider: provider });
if (typeof window !== 'undefined') { if (typeof window !== 'undefined') {
localStorage.setItem('voiceProvider', provider); localStorage.setItem('voiceProvider', provider);
@@ -1758,6 +1768,13 @@ export const useConfigStore = create<ConfigStore>()(
} }
}, },
setOpenaiCompatibleTtsModel: (model: string) => {
set({ openaiCompatibleTtsModel: model });
if (typeof window !== 'undefined') {
localStorage.setItem('openaiCompatibleTtsModel', model);
}
},
setSttProvider: (provider: 'browser' | 'server') => { setSttProvider: (provider: 'browser' | 'server') => {
set({ sttProvider: provider }); set({ sttProvider: provider });
if (typeof window !== 'undefined') { if (typeof window !== 'undefined') {
+2
View File
@@ -14,3 +14,5 @@ export {
summarizeText, summarizeText,
sanitizeForTTS, sanitizeForTTS,
} from './summarization.js'; } from './summarization.js';
export { transcribeAudio } from './stt.js';
+17 -36
View File
@@ -42,9 +42,9 @@ export function registerTtsRoutes(app, { resolveZenModel, sayTTSCapability }) {
// Server-side TTS endpoint - streams audio from OpenAI TTS API // Server-side TTS endpoint - streams audio from OpenAI TTS API
app.post('/api/tts/speak', async (req, res) => { app.post('/api/tts/speak', async (req, res) => {
try { try {
const { text, voice = 'nova', model = 'gpt-4o-mini-tts', speed = 0.9, instructions, summarize = false, providerId, modelId, threshold = 200, maxLength = 500, apiKey } = req.body || {}; const { text, voice = 'nova', model = 'gpt-4o-mini-tts', speed = 0.9, instructions, summarize = false, providerId, modelId, threshold = 200, maxLength = 500, apiKey, baseURL } = req.body || {};
console.log('[TTS] Request received:', { voice, model, speed, textLength: text?.length, hasApiKey: !!apiKey }); console.log('[TTS] Request received:', { voice, model, speed, textLength: text?.length, hasApiKey: !!apiKey, hasBaseURL: !!baseURL });
if (!text || typeof text !== 'string' || !text.trim()) { if (!text || typeof text !== 'string' || !text.trim()) {
return res.status(400).json({ error: 'Text is required' }); return res.status(400).json({ error: 'Text is required' });
@@ -53,13 +53,14 @@ export function registerTtsRoutes(app, { resolveZenModel, sayTTSCapability }) {
// Dynamically import the TTS service (ESM) // Dynamically import the TTS service (ESM)
const { ttsService } = await getTtsModule(); const { ttsService } = await getTtsModule();
// Check availability - either server-configured or client-provided API key // Check availability - server-configured key, client-provided key, or custom server URL
const hasServerKey = ttsService.isAvailable(); const hasServerKey = ttsService.isAvailable();
const hasClientKey = apiKey && typeof apiKey === 'string' && apiKey.trim().length > 0; const hasClientKey = apiKey && typeof apiKey === 'string' && apiKey.trim().length > 0;
const hasCustomBaseURL = baseURL && typeof baseURL === 'string' && baseURL.trim().length > 0;
if (!hasServerKey && !hasClientKey) { if (!hasServerKey && !hasClientKey && !hasCustomBaseURL) {
return res.status(503).json({ return res.status(503).json({
error: 'TTS service not available. Please configure OpenAI in OpenCode or provide an API key in settings.' error: 'TTS service not available. Please configure OpenAI in OpenCode, provide an API key, or set a custom server URL in settings.'
}); });
} }
@@ -87,44 +88,24 @@ export function registerTtsRoutes(app, { resolveZenModel, sayTTSCapability }) {
model, model,
speed, speed,
instructions, instructions,
apiKey: hasClientKey ? apiKey.trim() : undefined apiKey: hasClientKey ? apiKey.trim() : undefined,
baseURL: hasCustomBaseURL ? baseURL.trim() : undefined,
}); });
// Set headers for audio streaming
// Note: Don't set Transfer-Encoding manually - Express handles it automatically
res.setHeader('Content-Type', result.contentType); res.setHeader('Content-Type', result.contentType);
res.setHeader('Cache-Control', 'no-cache'); res.setHeader('Cache-Control', 'no-cache');
res.setHeader('Content-Length', result.buffer.length);
// Collect the full audio buffer and send it res.send(result.buffer);
// This avoids chunked encoding issues with proxies } catch (error) {
const reader = result.stream.getReader(); console.error('[TTS] Error:', error);
const chunks = [];
try {
while (true) {
const { done, value } = await reader.read();
if (done) break;
chunks.push(Buffer.from(value));
}
const audioBuffer = Buffer.concat(chunks);
res.setHeader('Content-Length', audioBuffer.length);
res.send(audioBuffer);
} catch (streamError) {
console.error('[TTS] Stream error:', streamError);
if (!res.headersSent) { if (!res.headersSent) {
res.status(500).json({ error: 'Stream error' }); const { model: m, voice: v, baseURL: b } = req.body || {};
} else { res.status(500).json({
res.end(); error: error instanceof Error ? error.message : 'TTS generation failed',
detail: { model: m, voice: v, hasBaseURL: !!b },
});
} }
} }
} catch (error) {
console.error('[TTS] Error:', error);
if (!res.headersSent) {
res.status(500).json({
error: error instanceof Error ? error.message : 'TTS generation failed'
});
}
}
}); });
app.post('/api/tts/summarize', async (req, res) => { app.post('/api/tts/summarize', async (req, res) => {
+27 -18
View File
@@ -78,19 +78,24 @@ class TTSService {
model = 'gpt-4o-mini-tts', model = 'gpt-4o-mini-tts',
speed = 1.0, speed = 1.0,
instructions, instructions,
apiKey apiKey,
baseURL,
} = options; } = options;
// Use provided API key or fall back to configured key // Use provided API key / baseURL or fall back to configured key
let client; let client;
if (apiKey) { if (baseURL || apiKey) {
client = new OpenAI({ apiKey }); const clientOpts = {};
if (apiKey) clientOpts.apiKey = apiKey;
if (!apiKey) clientOpts.apiKey = 'not-required';
if (baseURL) clientOpts.baseURL = baseURL;
client = new OpenAI(clientOpts);
} else { } else {
client = this._getClient(); client = this._getClient();
} }
if (!client) { if (!client) {
throw new Error('OpenAI API key not configured. Set OPENAI_API_KEY environment variable, configure OpenAI in OpenCode, or provide an API key in settings.'); throw new Error('TTS service not available. Configure OpenAI in OpenCode, provide an API key, or set a custom server URL in settings.');
} }
if (!text.trim()) { if (!text.trim()) {
@@ -98,21 +103,25 @@ class TTSService {
} }
try { try {
console.log('[TTSService] Generating speech with voice:', voice, 'model:', model); // OpenAI-compatible servers (custom baseURL) may not support `instructions`
const response = await client.audio.speech.create({ // or `response_format`, but do support `speed`. Send the safe subset.
model, const speechParams = baseURL
voice, ? { model, voice, input: text, speed }
input: text, : {
speed, model,
...(instructions && { instructions }), voice,
response_format: 'mp3', input: text,
}); speed,
...(instructions && { instructions }),
response_format: 'mp3',
};
// Convert the response to a web stream console.log('[TTSService] Generating speech — model:', model, 'voice:', voice, 'baseURL:', baseURL ?? '(openai)');
const stream = response.body; const response = await client.audio.speech.create(speechParams);
const arrayBuffer = await response.arrayBuffer();
return { return {
stream, buffer: Buffer.from(arrayBuffer),
contentType: 'audio/mpeg', contentType: 'audio/mpeg',
}; };
} catch (error) { } catch (error) {