feat(tts): add OpenAI-compatible custom server TTS provider with configurable model, pitch, and volume (#859)
Co-authored-by: Alexander Busse <alex@ableph.net>
This commit is contained in:
committed by
GitHub
co-authored by
Alexander Busse
parent
8a836d4aed
commit
055b6f6af0
@@ -715,7 +715,7 @@ const AssistantMessageBody: React.FC<Omit<MessageBodyProps, 'isUser'>> = ({
|
|||||||
if (isTTSPlaying) {
|
if (isTTSPlaying) {
|
||||||
return 'Stop speaking';
|
return 'Stop speaking';
|
||||||
}
|
}
|
||||||
const providerLabel = voiceProvider === 'browser' ? 'Browser' : voiceProvider === 'openai' ? 'OpenAI' : 'Say';
|
const providerLabel = voiceProvider === 'browser' ? 'Browser' : voiceProvider === 'openai' ? 'OpenAI' : voiceProvider === 'openai-compatible' ? 'Custom' : 'Say';
|
||||||
return `Read aloud (${providerLabel} voice)`;
|
return `Read aloud (${providerLabel} voice)`;
|
||||||
}, [isTTSPlaying, voiceProvider]);
|
}, [isTTSPlaying, voiceProvider]);
|
||||||
|
|
||||||
|
|||||||
@@ -18,7 +18,6 @@ import { Tooltip, TooltipContent, TooltipTrigger } from '@/components/ui/tooltip
|
|||||||
import { browserVoiceService } from '@/lib/voice/browserVoiceService';
|
import { browserVoiceService } from '@/lib/voice/browserVoiceService';
|
||||||
import { audioStreamService } from '@/lib/voice/audioStreamService';
|
import { audioStreamService } from '@/lib/voice/audioStreamService';
|
||||||
import { cn } from '@/lib/utils';
|
import { cn } from '@/lib/utils';
|
||||||
|
|
||||||
const LANGUAGE_OPTIONS = [
|
const LANGUAGE_OPTIONS = [
|
||||||
{ value: 'en-US', label: 'English' },
|
{ value: 'en-US', label: 'English' },
|
||||||
{ value: 'es-ES', label: 'Español' },
|
{ value: 'es-ES', label: 'Español' },
|
||||||
@@ -71,6 +70,12 @@ export const VoiceSettings: React.FC = () => {
|
|||||||
const setOpenaiVoice = useConfigStore((state) => state.setOpenaiVoice);
|
const setOpenaiVoice = useConfigStore((state) => state.setOpenaiVoice);
|
||||||
const openaiApiKey = useConfigStore((state) => state.openaiApiKey);
|
const openaiApiKey = useConfigStore((state) => state.openaiApiKey);
|
||||||
const setOpenaiApiKey = useConfigStore((state) => state.setOpenaiApiKey);
|
const setOpenaiApiKey = useConfigStore((state) => state.setOpenaiApiKey);
|
||||||
|
const openaiCompatibleUrl = useConfigStore((state) => state.openaiCompatibleUrl);
|
||||||
|
const setOpenaiCompatibleUrl = useConfigStore((state) => state.setOpenaiCompatibleUrl);
|
||||||
|
const openaiCompatibleVoice = useConfigStore((state) => state.openaiCompatibleVoice);
|
||||||
|
const setOpenaiCompatibleVoice = useConfigStore((state) => state.setOpenaiCompatibleVoice);
|
||||||
|
const openaiCompatibleTtsModel = useConfigStore((state) => state.openaiCompatibleTtsModel);
|
||||||
|
const setOpenaiCompatibleTtsModel = useConfigStore((state) => state.setOpenaiCompatibleTtsModel);
|
||||||
const showMessageTTSButtons = useConfigStore((state) => state.showMessageTTSButtons);
|
const showMessageTTSButtons = useConfigStore((state) => state.showMessageTTSButtons);
|
||||||
// STT settings
|
// STT settings
|
||||||
const sttProvider = useConfigStore((state) => state.sttProvider);
|
const sttProvider = useConfigStore((state) => state.sttProvider);
|
||||||
@@ -106,6 +111,9 @@ export const VoiceSettings: React.FC = () => {
|
|||||||
const [isOpenAIPreviewPlaying, setIsOpenAIPreviewPlaying] = useState(false);
|
const [isOpenAIPreviewPlaying, setIsOpenAIPreviewPlaying] = useState(false);
|
||||||
const [openaiPreviewAudio, setOpenaiPreviewAudio] = useState<HTMLAudioElement | null>(null);
|
const [openaiPreviewAudio, setOpenaiPreviewAudio] = useState<HTMLAudioElement | null>(null);
|
||||||
|
|
||||||
|
const [isCompatiblePreviewPlaying, setIsCompatiblePreviewPlaying] = useState(false);
|
||||||
|
const [compatiblePreviewAudio, setCompatiblePreviewAudio] = useState<HTMLAudioElement | null>(null);
|
||||||
|
|
||||||
const [browserVoices, setBrowserVoices] = useState<SpeechSynthesisVoice[]>([]);
|
const [browserVoices, setBrowserVoices] = useState<SpeechSynthesisVoice[]>([]);
|
||||||
const [isBrowserPreviewPlaying, setIsBrowserPreviewPlaying] = useState(false);
|
const [isBrowserPreviewPlaying, setIsBrowserPreviewPlaying] = useState(false);
|
||||||
|
|
||||||
@@ -182,7 +190,7 @@ export const VoiceSettings: React.FC = () => {
|
|||||||
}, [isBrowserPreviewPlaying]);
|
}, [isBrowserPreviewPlaying]);
|
||||||
|
|
||||||
useEffect(() => {
|
useEffect(() => {
|
||||||
if (!voiceModeEnabled || voiceProvider !== 'openai') {
|
if (!voiceModeEnabled || (voiceProvider !== 'openai' && voiceProvider !== 'openai-compatible')) {
|
||||||
setIsOpenAIAvailable(openaiApiKey.trim().length > 0);
|
setIsOpenAIAvailable(openaiApiKey.trim().length > 0);
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
@@ -339,6 +347,67 @@ export const VoiceSettings: React.FC = () => {
|
|||||||
};
|
};
|
||||||
}, [openaiPreviewAudio]);
|
}, [openaiPreviewAudio]);
|
||||||
|
|
||||||
|
const previewCompatibleVoice = useCallback(async () => {
|
||||||
|
if (compatiblePreviewAudio) {
|
||||||
|
compatiblePreviewAudio.pause();
|
||||||
|
compatiblePreviewAudio.currentTime = 0;
|
||||||
|
setCompatiblePreviewAudio(null);
|
||||||
|
setIsCompatiblePreviewPlaying(false);
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!openaiCompatibleUrl.trim()) return;
|
||||||
|
|
||||||
|
setIsCompatiblePreviewPlaying(true);
|
||||||
|
try {
|
||||||
|
const response = await fetch('/api/tts/speak', {
|
||||||
|
method: 'POST',
|
||||||
|
headers: { 'Content-Type': 'application/json' },
|
||||||
|
body: JSON.stringify({
|
||||||
|
text: `Hello! This is a preview of the custom TTS server.`,
|
||||||
|
voice: openaiCompatibleVoice,
|
||||||
|
model: openaiCompatibleTtsModel || undefined,
|
||||||
|
speed: speechRate,
|
||||||
|
baseURL: openaiCompatibleUrl,
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
|
||||||
|
if (!response.ok) {
|
||||||
|
const errorData = await response.json().catch(() => ({ error: 'Unknown error' }));
|
||||||
|
throw new Error(errorData.error || `HTTP ${response.status}`);
|
||||||
|
}
|
||||||
|
|
||||||
|
const blob = await response.blob();
|
||||||
|
const url = URL.createObjectURL(blob);
|
||||||
|
const audio = new Audio(url);
|
||||||
|
|
||||||
|
audio.onended = () => {
|
||||||
|
URL.revokeObjectURL(url);
|
||||||
|
setCompatiblePreviewAudio(null);
|
||||||
|
setIsCompatiblePreviewPlaying(false);
|
||||||
|
};
|
||||||
|
|
||||||
|
audio.onerror = () => {
|
||||||
|
URL.revokeObjectURL(url);
|
||||||
|
setCompatiblePreviewAudio(null);
|
||||||
|
setIsCompatiblePreviewPlaying(false);
|
||||||
|
};
|
||||||
|
|
||||||
|
setCompatiblePreviewAudio(audio);
|
||||||
|
await audio.play();
|
||||||
|
} catch {
|
||||||
|
setIsCompatiblePreviewPlaying(false);
|
||||||
|
}
|
||||||
|
}, [openaiCompatibleUrl, openaiCompatibleVoice, openaiCompatibleTtsModel, speechRate, compatiblePreviewAudio]);
|
||||||
|
|
||||||
|
useEffect(() => {
|
||||||
|
return () => {
|
||||||
|
if (compatiblePreviewAudio) {
|
||||||
|
compatiblePreviewAudio.pause();
|
||||||
|
}
|
||||||
|
};
|
||||||
|
}, [compatiblePreviewAudio]);
|
||||||
|
|
||||||
const sliderClass = "flex-1 min-w-0 h-1.5 bg-[var(--interactive-border)] rounded-full appearance-none cursor-pointer [&::-webkit-slider-thumb]:appearance-none [&::-webkit-slider-thumb]:w-4 [&::-webkit-slider-thumb]:h-4 [&::-webkit-slider-thumb]:rounded-full [&::-webkit-slider-thumb]:bg-[var(--primary-base)] [&::-moz-range-thumb]:w-4 [&::-moz-range-thumb]:h-4 [&::-moz-range-thumb]:rounded-full [&::-moz-range-thumb]:bg-[var(--primary-base)] [&::-moz-range-thumb]:border-0 disabled:opacity-50";
|
const sliderClass = "flex-1 min-w-0 h-1.5 bg-[var(--interactive-border)] rounded-full appearance-none cursor-pointer [&::-webkit-slider-thumb]:appearance-none [&::-webkit-slider-thumb]:w-4 [&::-webkit-slider-thumb]:h-4 [&::-webkit-slider-thumb]:rounded-full [&::-webkit-slider-thumb]:bg-[var(--primary-base)] [&::-moz-range-thumb]:w-4 [&::-moz-range-thumb]:h-4 [&::-moz-range-thumb]:rounded-full [&::-moz-range-thumb]:bg-[var(--primary-base)] [&::-moz-range-thumb]:border-0 disabled:opacity-50";
|
||||||
|
|
||||||
return (
|
return (
|
||||||
@@ -380,6 +449,7 @@ export const VoiceSettings: React.FC = () => {
|
|||||||
<ul className="space-y-1">
|
<ul className="space-y-1">
|
||||||
<li><strong>Browser:</strong> Free, offline, limited mobile support.</li>
|
<li><strong>Browser:</strong> Free, offline, limited mobile support.</li>
|
||||||
<li><strong>OpenAI:</strong> High quality, mobile ready, needs API key.</li>
|
<li><strong>OpenAI:</strong> High quality, mobile ready, needs API key.</li>
|
||||||
|
<li><strong>Custom:</strong> OpenAI-compatible server (e.g. Kokoro).</li>
|
||||||
<li><strong>Say:</strong> macOS native. Fast, free, offline.</li>
|
<li><strong>Say:</strong> macOS native. Fast, free, offline.</li>
|
||||||
</ul>
|
</ul>
|
||||||
</TooltipContent>
|
</TooltipContent>
|
||||||
@@ -412,6 +482,19 @@ export const VoiceSettings: React.FC = () => {
|
|||||||
>
|
>
|
||||||
OpenAI
|
OpenAI
|
||||||
</Button>
|
</Button>
|
||||||
|
<Button
|
||||||
|
variant="outline"
|
||||||
|
size="xs"
|
||||||
|
onClick={() => setVoiceProvider('openai-compatible')}
|
||||||
|
className={cn(
|
||||||
|
'!font-normal',
|
||||||
|
voiceProvider === 'openai-compatible'
|
||||||
|
? 'border-[var(--primary-base)] text-[var(--primary-base)] bg-[var(--primary-base)]/10 hover:text-[var(--primary-base)]'
|
||||||
|
: 'text-foreground'
|
||||||
|
)}
|
||||||
|
>
|
||||||
|
Custom
|
||||||
|
</Button>
|
||||||
{isSayAvailable && (
|
{isSayAvailable && (
|
||||||
<Button
|
<Button
|
||||||
variant="outline"
|
variant="outline"
|
||||||
@@ -462,6 +545,70 @@ export const VoiceSettings: React.FC = () => {
|
|||||||
</div>
|
</div>
|
||||||
)}
|
)}
|
||||||
|
|
||||||
|
{/* OpenAI-compatible custom server */}
|
||||||
|
{voiceProvider === 'openai-compatible' && (
|
||||||
|
<div className="py-1.5 space-y-2">
|
||||||
|
<div>
|
||||||
|
<span className={cn("typography-ui-label text-foreground", !openaiCompatibleUrl.trim() && "text-[var(--status-error)]")}>
|
||||||
|
Server URL
|
||||||
|
</span>
|
||||||
|
<span className="typography-meta ml-2 text-muted-foreground">
|
||||||
|
Base URL of the OpenAI-compatible TTS server
|
||||||
|
</span>
|
||||||
|
<div className="relative mt-1.5 max-w-xs">
|
||||||
|
<input
|
||||||
|
type="text"
|
||||||
|
value={openaiCompatibleUrl}
|
||||||
|
onChange={(e) => setOpenaiCompatibleUrl(e.target.value)}
|
||||||
|
placeholder="http://localhost:8880/v1"
|
||||||
|
className="w-full h-7 rounded-lg border border-input bg-transparent px-2 typography-ui-label text-foreground placeholder:text-muted-foreground focus:outline-none focus:ring-1 focus:ring-primary/50 focus:border-primary/70"
|
||||||
|
/>
|
||||||
|
{openaiCompatibleUrl && (
|
||||||
|
<button
|
||||||
|
type="button"
|
||||||
|
onClick={() => setOpenaiCompatibleUrl('')}
|
||||||
|
className="absolute right-2 top-1/2 -translate-y-1/2 text-muted-foreground hover:text-foreground"
|
||||||
|
>
|
||||||
|
<RiCloseLine className="w-3.5 h-3.5" />
|
||||||
|
</button>
|
||||||
|
)}
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
<div>
|
||||||
|
<span className="typography-ui-label text-foreground">Model</span>
|
||||||
|
<div className="relative mt-1.5 max-w-xs">
|
||||||
|
<input
|
||||||
|
type="text"
|
||||||
|
value={openaiCompatibleTtsModel}
|
||||||
|
onChange={(e) => setOpenaiCompatibleTtsModel(e.target.value)}
|
||||||
|
placeholder="speaches-ai/Kokoro-82M-v1.0-ONNX"
|
||||||
|
className="w-full h-7 rounded-lg border border-input bg-transparent px-2 typography-ui-label text-foreground placeholder:text-muted-foreground focus:outline-none focus:ring-1 focus:ring-primary/50 focus:border-primary/70"
|
||||||
|
/>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
<div>
|
||||||
|
<span className="typography-ui-label text-foreground">Voice</span>
|
||||||
|
<span className="typography-meta ml-2 text-muted-foreground">
|
||||||
|
Voice identifier supported by the server
|
||||||
|
</span>
|
||||||
|
<div className="flex items-center gap-2 mt-1.5">
|
||||||
|
<div className="relative max-w-xs flex-1">
|
||||||
|
<input
|
||||||
|
type="text"
|
||||||
|
value={openaiCompatibleVoice}
|
||||||
|
onChange={(e) => setOpenaiCompatibleVoice(e.target.value)}
|
||||||
|
placeholder="af_sky"
|
||||||
|
className="w-full h-7 rounded-lg border border-input bg-transparent px-2 typography-ui-label text-foreground placeholder:text-muted-foreground focus:outline-none focus:ring-1 focus:ring-primary/50 focus:border-primary/70"
|
||||||
|
/>
|
||||||
|
</div>
|
||||||
|
<Button size="xs" variant="ghost" onClick={previewCompatibleVoice} title="Preview" disabled={!openaiCompatibleUrl.trim()}>
|
||||||
|
{isCompatiblePreviewPlaying ? <RiStopLine className="w-3.5 h-3.5" /> : <RiPlayLine className="w-3.5 h-3.5" />}
|
||||||
|
</Button>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
</div>
|
||||||
|
)}
|
||||||
|
|
||||||
{/* Voice Selection */}
|
{/* Voice Selection */}
|
||||||
<div className="flex items-center gap-8 py-1.5">
|
<div className="flex items-center gap-8 py-1.5">
|
||||||
<span className="typography-ui-label text-foreground sm:w-56 shrink-0">Voice</span>
|
<span className="typography-ui-label text-foreground sm:w-56 shrink-0">Voice</span>
|
||||||
@@ -484,6 +631,10 @@ export const VoiceSettings: React.FC = () => {
|
|||||||
</>
|
</>
|
||||||
)}
|
)}
|
||||||
|
|
||||||
|
{voiceProvider === 'openai-compatible' && (
|
||||||
|
<span className="typography-meta text-muted-foreground">Configured above</span>
|
||||||
|
)}
|
||||||
|
|
||||||
{voiceProvider === 'say' && isSayAvailable && sayVoices.length > 0 && (
|
{voiceProvider === 'say' && isSayAvailable && sayVoices.length > 0 && (
|
||||||
<>
|
<>
|
||||||
<Select value={sayVoice} onValueChange={setSayVoice}>
|
<Select value={sayVoice} onValueChange={setSayVoice}>
|
||||||
|
|||||||
@@ -62,7 +62,7 @@ export interface UseBrowserVoiceReturn {
|
|||||||
/** Whether the device is mobile */
|
/** Whether the device is mobile */
|
||||||
isMobile: boolean;
|
isMobile: boolean;
|
||||||
/** Current voice provider */
|
/** Current voice provider */
|
||||||
voiceProvider: 'browser' | 'openai' | 'say';
|
voiceProvider: 'browser' | 'openai' | 'openai-compatible' | 'say';
|
||||||
}
|
}
|
||||||
|
|
||||||
// Storage key for persisting language preference
|
// Storage key for persisting language preference
|
||||||
@@ -144,10 +144,11 @@ export function useBrowserVoice(): UseBrowserVoiceReturn {
|
|||||||
const openaiVoice = useConfigStore((state) => state.openaiVoice);
|
const openaiVoice = useConfigStore((state) => state.openaiVoice);
|
||||||
const openaiCompatibleVoice = useConfigStore((state) => state.openaiCompatibleVoice);
|
const openaiCompatibleVoice = useConfigStore((state) => state.openaiCompatibleVoice);
|
||||||
const openaiCompatibleUrl = useConfigStore((state) => state.openaiCompatibleUrl);
|
const openaiCompatibleUrl = useConfigStore((state) => state.openaiCompatibleUrl);
|
||||||
|
const openaiCompatibleTtsModel = useConfigStore((state) => state.openaiCompatibleTtsModel);
|
||||||
const summarizeVoiceConversation = useConfigStore((state) => state.summarizeVoiceConversation);
|
const summarizeVoiceConversation = useConfigStore((state) => state.summarizeVoiceConversation);
|
||||||
const summarizeCharacterThreshold = useConfigStore((state) => state.summarizeCharacterThreshold);
|
const summarizeCharacterThreshold = useConfigStore((state) => state.summarizeCharacterThreshold);
|
||||||
|
|
||||||
const shouldCheckOpenAIAvailability = voiceModeEnabled && voiceProvider === 'openai';
|
const shouldCheckOpenAIAvailability = voiceModeEnabled && (voiceProvider === 'openai' || voiceProvider === 'openai-compatible');
|
||||||
const shouldCheckSayAvailability = voiceModeEnabled && voiceProvider === 'say';
|
const shouldCheckSayAvailability = voiceModeEnabled && voiceProvider === 'say';
|
||||||
|
|
||||||
// STT provider config
|
// STT provider config
|
||||||
@@ -462,12 +463,19 @@ export function useBrowserVoice(): UseBrowserVoiceReturn {
|
|||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
// Use server TTS when OpenAI provider is selected and available
|
// Use server TTS when OpenAI (or OpenAI-compatible) provider is selected and available
|
||||||
if (voiceProvider === 'openai' && isServerTTSAvailable) {
|
if ((voiceProvider === 'openai' || voiceProvider === 'openai-compatible') && isServerTTSAvailable) {
|
||||||
console.log('[useBrowserVoice] Using OpenAI server TTS with voice:', openaiVoice);
|
const ttsVoice = voiceProvider === 'openai-compatible' ? openaiCompatibleVoice : openaiVoice;
|
||||||
|
const ttsBaseURL = voiceProvider === 'openai-compatible' ? openaiCompatibleUrl : undefined;
|
||||||
|
const ttsModel = voiceProvider === 'openai-compatible' ? openaiCompatibleTtsModel : undefined;
|
||||||
|
console.log('[useBrowserVoice] Using server TTS with voice:', ttsVoice, 'provider:', voiceProvider);
|
||||||
await speakServerTTS(textToSpeak, {
|
await speakServerTTS(textToSpeak, {
|
||||||
voice: openaiVoice,
|
voice: ttsVoice,
|
||||||
|
model: ttsModel,
|
||||||
speed: speechRate,
|
speed: speechRate,
|
||||||
|
pitch: speechPitch,
|
||||||
|
volume: speechVolume,
|
||||||
|
baseURL: ttsBaseURL,
|
||||||
onStart: () => console.log('[useBrowserVoice] Server TTS started'),
|
onStart: () => console.log('[useBrowserVoice] Server TTS started'),
|
||||||
onEnd: () => {
|
onEnd: () => {
|
||||||
console.log('[useBrowserVoice] Server TTS ended');
|
console.log('[useBrowserVoice] Server TTS ended');
|
||||||
@@ -475,8 +483,7 @@ export function useBrowserVoice(): UseBrowserVoiceReturn {
|
|||||||
},
|
},
|
||||||
onError: (errorMsg) => {
|
onError: (errorMsg) => {
|
||||||
console.error('[useBrowserVoice] Server TTS error:', errorMsg);
|
console.error('[useBrowserVoice] Server TTS error:', errorMsg);
|
||||||
// Show error to user when OpenAI voice fails
|
setError(`Voice TTS failed: ${errorMsg}. Please check your settings or switch to Browser voice.`);
|
||||||
setError(`OpenAI voice failed: ${errorMsg}. Please check your OpenAI API key or switch to Browser voice.`);
|
|
||||||
setStatus('error');
|
setStatus('error');
|
||||||
restartListening();
|
restartListening();
|
||||||
}
|
}
|
||||||
@@ -573,7 +580,7 @@ export function useBrowserVoice(): UseBrowserVoiceReturn {
|
|||||||
setStatus('error');
|
setStatus('error');
|
||||||
processingMessageRef.current = false;
|
processingMessageRef.current = false;
|
||||||
}
|
}
|
||||||
}, [currentSessionId, currentProviderId, currentModelId, currentAgentName, language, sendMessage, setPendingInputText, createSession, speechRate, speechPitch, speechVolume, isMobile, isServerTTSAvailable, speakServerTTS, isSayTTSAvailable, speakSayTTS, voiceProvider, sayVoice, browserVoice, openaiVoice, openaiCompatibleVoice, openaiCompatibleUrl, summarizeVoiceConversation, summarizeCharacterThreshold, conversationMode, sttProvider]);
|
}, [currentSessionId, currentProviderId, currentModelId, currentAgentName, language, sendMessage, setPendingInputText, createSession, speechRate, speechPitch, speechVolume, isMobile, isServerTTSAvailable, speakServerTTS, isSayTTSAvailable, speakSayTTS, voiceProvider, sayVoice, browserVoice, openaiVoice, openaiCompatibleVoice, openaiCompatibleUrl, openaiCompatibleTtsModel, summarizeVoiceConversation, summarizeCharacterThreshold, conversationMode, sttProvider]);
|
||||||
|
|
||||||
// Handle speech recognition result
|
// Handle speech recognition result
|
||||||
const handleSpeechResult = useCallback(async (text: string, isFinal: boolean) => {
|
const handleSpeechResult = useCallback(async (text: string, isFinal: boolean) => {
|
||||||
|
|||||||
@@ -31,11 +31,15 @@ export function useMessageTTS(): UseMessageTTSReturn {
|
|||||||
const sayVoice = useConfigStore((state) => state.sayVoice);
|
const sayVoice = useConfigStore((state) => state.sayVoice);
|
||||||
const browserVoice = useConfigStore((state) => state.browserVoice);
|
const browserVoice = useConfigStore((state) => state.browserVoice);
|
||||||
const openaiVoice = useConfigStore((state) => state.openaiVoice);
|
const openaiVoice = useConfigStore((state) => state.openaiVoice);
|
||||||
|
const openaiCompatibleVoice = useConfigStore((state) => state.openaiCompatibleVoice);
|
||||||
|
const openaiCompatibleUrl = useConfigStore((state) => state.openaiCompatibleUrl);
|
||||||
|
const openaiCompatibleTtsModel = useConfigStore((state) => state.openaiCompatibleTtsModel);
|
||||||
const summarizeMessageTTS = useConfigStore((state) => state.summarizeMessageTTS);
|
const summarizeMessageTTS = useConfigStore((state) => state.summarizeMessageTTS);
|
||||||
const summarizeCharacterThreshold = useConfigStore((state) => state.summarizeCharacterThreshold);
|
const summarizeCharacterThreshold = useConfigStore((state) => state.summarizeCharacterThreshold);
|
||||||
const showMessageTTSButtons = useConfigStore((state) => state.showMessageTTSButtons);
|
const showMessageTTSButtons = useConfigStore((state) => state.showMessageTTSButtons);
|
||||||
|
|
||||||
const shouldCheckOpenAIAvailability = showMessageTTSButtons && voiceProvider === 'openai';
|
const isServerProvider = voiceProvider === 'openai' || voiceProvider === 'openai-compatible';
|
||||||
|
const shouldCheckOpenAIAvailability = showMessageTTSButtons && isServerProvider;
|
||||||
const shouldCheckSayAvailability = showMessageTTSButtons && voiceProvider === 'say';
|
const shouldCheckSayAvailability = showMessageTTSButtons && voiceProvider === 'say';
|
||||||
|
|
||||||
const { speak: speakServerTTS, stop: stopServerTTS, isAvailable: isServerTTSAvailable } = useServerTTS({
|
const { speak: speakServerTTS, stop: stopServerTTS, isAvailable: isServerTTSAvailable } = useServerTTS({
|
||||||
@@ -72,11 +76,18 @@ export function useMessageTTS(): UseMessageTTSReturn {
|
|||||||
textToSpeak = sanitizeForTTS(text);
|
textToSpeak = sanitizeForTTS(text);
|
||||||
}
|
}
|
||||||
|
|
||||||
if (voiceProvider === 'openai' && isServerTTSAvailable) {
|
if (isServerProvider && isServerTTSAvailable) {
|
||||||
|
const voice = voiceProvider === 'openai-compatible' ? openaiCompatibleVoice : openaiVoice;
|
||||||
|
const baseURL = voiceProvider === 'openai-compatible' ? openaiCompatibleUrl : undefined;
|
||||||
|
const model = voiceProvider === 'openai-compatible' ? openaiCompatibleTtsModel : undefined;
|
||||||
await speakServerTTS(textToSpeak, {
|
await speakServerTTS(textToSpeak, {
|
||||||
voice: openaiVoice,
|
voice,
|
||||||
|
model,
|
||||||
speed: speechRate,
|
speed: speechRate,
|
||||||
|
pitch: speechPitch,
|
||||||
|
volume: speechVolume,
|
||||||
summarize: false, // We already summarized client-side
|
summarize: false, // We already summarized client-side
|
||||||
|
baseURL,
|
||||||
onEnd: () => setIsPlaying(false),
|
onEnd: () => setIsPlaying(false),
|
||||||
onError: () => setIsPlaying(false),
|
onError: () => setIsPlaying(false),
|
||||||
});
|
});
|
||||||
@@ -110,12 +121,16 @@ export function useMessageTTS(): UseMessageTTSReturn {
|
|||||||
}
|
}
|
||||||
}, [
|
}, [
|
||||||
voiceProvider,
|
voiceProvider,
|
||||||
|
isServerProvider,
|
||||||
speechRate,
|
speechRate,
|
||||||
speechPitch,
|
speechPitch,
|
||||||
speechVolume,
|
speechVolume,
|
||||||
sayVoice,
|
sayVoice,
|
||||||
browserVoice,
|
browserVoice,
|
||||||
openaiVoice,
|
openaiVoice,
|
||||||
|
openaiCompatibleVoice,
|
||||||
|
openaiCompatibleUrl,
|
||||||
|
openaiCompatibleTtsModel,
|
||||||
summarizeMessageTTS,
|
summarizeMessageTTS,
|
||||||
summarizeCharacterThreshold,
|
summarizeCharacterThreshold,
|
||||||
isServerTTSAvailable,
|
isServerTTSAvailable,
|
||||||
|
|||||||
@@ -85,8 +85,14 @@ export interface UseServerTTSReturn {
|
|||||||
export interface SpeakOptions {
|
export interface SpeakOptions {
|
||||||
/** Voice to use (defaults to coral) */
|
/** Voice to use (defaults to coral) */
|
||||||
voice?: string;
|
voice?: string;
|
||||||
|
/** Model to use (defaults to gpt-4o-mini-tts) */
|
||||||
|
model?: string;
|
||||||
/** Speech speed (0.25 to 4.0, defaults to 1.0) */
|
/** Speech speed (0.25 to 4.0, defaults to 1.0) */
|
||||||
speed?: number;
|
speed?: number;
|
||||||
|
/** Speech pitch shift (0.5 to 2.0, mapped to cents; 1.0 = no shift) */
|
||||||
|
pitch?: number;
|
||||||
|
/** Playback volume (0 to 1, defaults to 1.0) */
|
||||||
|
volume?: number;
|
||||||
/** Optional instructions for the voice */
|
/** Optional instructions for the voice */
|
||||||
instructions?: string;
|
instructions?: string;
|
||||||
/** Summarize long text before speaking (defaults to true) */
|
/** Summarize long text before speaking (defaults to true) */
|
||||||
@@ -97,6 +103,8 @@ export interface SpeakOptions {
|
|||||||
modelId?: string;
|
modelId?: string;
|
||||||
/** Character threshold for summarization (defaults to 200) */
|
/** Character threshold for summarization (defaults to 200) */
|
||||||
threshold?: number;
|
threshold?: number;
|
||||||
|
/** Custom base URL for OpenAI-compatible server */
|
||||||
|
baseURL?: string;
|
||||||
/** Callback when playback starts */
|
/** Callback when playback starts */
|
||||||
onStart?: () => void;
|
onStart?: () => void;
|
||||||
/** Callback when playback ends */
|
/** Callback when playback ends */
|
||||||
@@ -130,6 +138,7 @@ export function useServerTTS(options: UseServerTTSOptions = {}): UseServerTTSRet
|
|||||||
const summarizeCharacterThreshold = useConfigStore((state) => state.summarizeCharacterThreshold);
|
const summarizeCharacterThreshold = useConfigStore((state) => state.summarizeCharacterThreshold);
|
||||||
const summarizeMaxLength = useConfigStore((state) => state.summarizeMaxLength);
|
const summarizeMaxLength = useConfigStore((state) => state.summarizeMaxLength);
|
||||||
const openaiApiKey = useConfigStore((state) => state.openaiApiKey);
|
const openaiApiKey = useConfigStore((state) => state.openaiApiKey);
|
||||||
|
const openaiCompatibleUrl = useConfigStore((state) => state.openaiCompatibleUrl);
|
||||||
const settingsZenModel = useConfigStore((state) => state.settingsZenModel);
|
const settingsZenModel = useConfigStore((state) => state.settingsZenModel);
|
||||||
|
|
||||||
// Check if server TTS is available
|
// Check if server TTS is available
|
||||||
@@ -140,7 +149,8 @@ export function useServerTTS(options: UseServerTTSOptions = {}): UseServerTTSRet
|
|||||||
}
|
}
|
||||||
|
|
||||||
const hasClientKey = Boolean(openaiApiKey && openaiApiKey.trim().length > 0);
|
const hasClientKey = Boolean(openaiApiKey && openaiApiKey.trim().length > 0);
|
||||||
if (hasClientKey) {
|
const hasCustomUrl = Boolean(openaiCompatibleUrl && openaiCompatibleUrl.trim().length > 0);
|
||||||
|
if (hasClientKey || hasCustomUrl) {
|
||||||
setIsAvailable(true);
|
setIsAvailable(true);
|
||||||
return true;
|
return true;
|
||||||
}
|
}
|
||||||
@@ -153,7 +163,7 @@ export function useServerTTS(options: UseServerTTSOptions = {}): UseServerTTSRet
|
|||||||
setIsAvailable(false);
|
setIsAvailable(false);
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
}, [enabled, openaiApiKey]);
|
}, [enabled, openaiApiKey, openaiCompatibleUrl]);
|
||||||
|
|
||||||
// Check availability on mount and when API key changes
|
// Check availability on mount and when API key changes
|
||||||
useEffect(() => {
|
useEffect(() => {
|
||||||
@@ -250,6 +260,7 @@ export function useServerTTS(options: UseServerTTSOptions = {}): UseServerTTSRet
|
|||||||
body: JSON.stringify({
|
body: JSON.stringify({
|
||||||
text: text.trim(),
|
text: text.trim(),
|
||||||
voice,
|
voice,
|
||||||
|
model: options?.model || undefined,
|
||||||
speed: options?.speed || 0.9,
|
speed: options?.speed || 0.9,
|
||||||
instructions: options?.instructions,
|
instructions: options?.instructions,
|
||||||
summarize: options?.summarize ?? true, // Summarize by default for voice output
|
summarize: options?.summarize ?? true, // Summarize by default for voice output
|
||||||
@@ -262,6 +273,8 @@ export function useServerTTS(options: UseServerTTSOptions = {}): UseServerTTSRet
|
|||||||
maxLength: summarizeMaxLength ?? 500,
|
maxLength: summarizeMaxLength ?? 500,
|
||||||
// Send API key from settings if available
|
// Send API key from settings if available
|
||||||
apiKey: openaiApiKey || undefined,
|
apiKey: openaiApiKey || undefined,
|
||||||
|
// Send custom base URL for OpenAI-compatible servers
|
||||||
|
baseURL: options?.baseURL || undefined,
|
||||||
...(settingsZenModel ? { zenModel: settingsZenModel } : {}),
|
...(settingsZenModel ? { zenModel: settingsZenModel } : {}),
|
||||||
}),
|
}),
|
||||||
signal: abortControllerRef.current.signal,
|
signal: abortControllerRef.current.signal,
|
||||||
@@ -283,7 +296,20 @@ export function useServerTTS(options: UseServerTTSOptions = {}): UseServerTTSRet
|
|||||||
// Create source node
|
// Create source node
|
||||||
const source = ctx.createBufferSource();
|
const source = ctx.createBufferSource();
|
||||||
source.buffer = audioBuffer;
|
source.buffer = audioBuffer;
|
||||||
source.connect(ctx.destination);
|
|
||||||
|
// Apply pitch shift via detune (cents): 1200 cents = 1 octave
|
||||||
|
const pitch = options?.pitch ?? 1.0;
|
||||||
|
if (pitch !== 1.0) {
|
||||||
|
source.detune.value = (pitch - 1.0) * 1200;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Apply volume via GainNode
|
||||||
|
const volume = options?.volume ?? 1.0;
|
||||||
|
const gainNode = ctx.createGain();
|
||||||
|
gainNode.gain.value = volume;
|
||||||
|
|
||||||
|
source.connect(gainNode);
|
||||||
|
gainNode.connect(ctx.destination);
|
||||||
audioSourceRef.current = source;
|
audioSourceRef.current = source;
|
||||||
|
|
||||||
// Set up event handlers
|
// Set up event handlers
|
||||||
|
|||||||
@@ -466,9 +466,9 @@ interface ConfigStore {
|
|||||||
settingsAutoCreateWorktree: boolean;
|
settingsAutoCreateWorktree: boolean;
|
||||||
settingsGitmojiEnabled: boolean;
|
settingsGitmojiEnabled: boolean;
|
||||||
settingsZenModel: string | undefined;
|
settingsZenModel: string | undefined;
|
||||||
// Voice provider preference ('browser', 'openai', or 'say' for macOS)
|
// Voice provider preference ('browser', 'openai', 'openai-compatible', or 'say' for macOS)
|
||||||
voiceProvider: 'browser' | 'openai' | 'say';
|
voiceProvider: 'browser' | 'openai' | 'openai-compatible' | 'say';
|
||||||
setVoiceProvider: (provider: 'browser' | 'openai' | 'say') => void;
|
setVoiceProvider: (provider: 'browser' | 'openai' | 'openai-compatible' | 'say') => void;
|
||||||
// TTS settings
|
// TTS settings
|
||||||
speechRate: number;
|
speechRate: number;
|
||||||
speechPitch: number;
|
speechPitch: number;
|
||||||
@@ -479,6 +479,7 @@ interface ConfigStore {
|
|||||||
openaiApiKey: string;
|
openaiApiKey: string;
|
||||||
openaiCompatibleUrl: string;
|
openaiCompatibleUrl: string;
|
||||||
openaiCompatibleVoice: string;
|
openaiCompatibleVoice: string;
|
||||||
|
openaiCompatibleTtsModel: string;
|
||||||
// STT (speech-to-text) settings
|
// STT (speech-to-text) settings
|
||||||
sttProvider: 'browser' | 'server';
|
sttProvider: 'browser' | 'server';
|
||||||
sttServerUrl: string;
|
sttServerUrl: string;
|
||||||
@@ -502,6 +503,7 @@ interface ConfigStore {
|
|||||||
setOpenaiApiKey: (apiKey: string) => void;
|
setOpenaiApiKey: (apiKey: string) => void;
|
||||||
setOpenaiCompatibleUrl: (url: string) => void;
|
setOpenaiCompatibleUrl: (url: string) => void;
|
||||||
setOpenaiCompatibleVoice: (voice: string) => void;
|
setOpenaiCompatibleVoice: (voice: string) => void;
|
||||||
|
setOpenaiCompatibleTtsModel: (model: string) => void;
|
||||||
setSttProvider: (provider: 'browser' | 'server') => void;
|
setSttProvider: (provider: 'browser' | 'server') => void;
|
||||||
setSttServerUrl: (url: string) => void;
|
setSttServerUrl: (url: string) => void;
|
||||||
setSttModel: (model: string) => void;
|
setSttModel: (model: string) => void;
|
||||||
@@ -585,7 +587,7 @@ export const useConfigStore = create<ConfigStore>()(
|
|||||||
voiceProvider: (() => {
|
voiceProvider: (() => {
|
||||||
if (typeof window !== 'undefined') {
|
if (typeof window !== 'undefined') {
|
||||||
const saved = localStorage.getItem('voiceProvider');
|
const saved = localStorage.getItem('voiceProvider');
|
||||||
if (saved === 'openai' || saved === 'browser' || saved === 'say') return saved;
|
if (saved === 'openai' || saved === 'browser' || saved === 'say' || saved === 'openai-compatible') return saved;
|
||||||
}
|
}
|
||||||
return 'browser';
|
return 'browser';
|
||||||
})(),
|
})(),
|
||||||
@@ -668,6 +670,14 @@ export const useConfigStore = create<ConfigStore>()(
|
|||||||
}
|
}
|
||||||
return 'af_sky';
|
return 'af_sky';
|
||||||
})(),
|
})(),
|
||||||
|
// OpenAI-compatible custom server TTS model
|
||||||
|
openaiCompatibleTtsModel: (() => {
|
||||||
|
if (typeof window !== 'undefined') {
|
||||||
|
const saved = localStorage.getItem('openaiCompatibleTtsModel');
|
||||||
|
if (saved && saved !== 'speaches-ai/Kokoro-82M-v1.0-ONNX') return saved;
|
||||||
|
}
|
||||||
|
return 'kokoro';
|
||||||
|
})(),
|
||||||
// STT provider: 'browser' (Web Speech API) or 'server' (OpenAI-compat)
|
// STT provider: 'browser' (Web Speech API) or 'server' (OpenAI-compat)
|
||||||
sttProvider: (() => {
|
sttProvider: (() => {
|
||||||
if (typeof window !== 'undefined') {
|
if (typeof window !== 'undefined') {
|
||||||
@@ -1685,7 +1695,7 @@ export const useConfigStore = create<ConfigStore>()(
|
|||||||
});
|
});
|
||||||
},
|
},
|
||||||
|
|
||||||
setVoiceProvider: (provider: 'browser' | 'openai' | 'say') => {
|
setVoiceProvider: (provider: 'browser' | 'openai' | 'openai-compatible' | 'say') => {
|
||||||
set({ voiceProvider: provider });
|
set({ voiceProvider: provider });
|
||||||
if (typeof window !== 'undefined') {
|
if (typeof window !== 'undefined') {
|
||||||
localStorage.setItem('voiceProvider', provider);
|
localStorage.setItem('voiceProvider', provider);
|
||||||
@@ -1758,6 +1768,13 @@ export const useConfigStore = create<ConfigStore>()(
|
|||||||
}
|
}
|
||||||
},
|
},
|
||||||
|
|
||||||
|
setOpenaiCompatibleTtsModel: (model: string) => {
|
||||||
|
set({ openaiCompatibleTtsModel: model });
|
||||||
|
if (typeof window !== 'undefined') {
|
||||||
|
localStorage.setItem('openaiCompatibleTtsModel', model);
|
||||||
|
}
|
||||||
|
},
|
||||||
|
|
||||||
setSttProvider: (provider: 'browser' | 'server') => {
|
setSttProvider: (provider: 'browser' | 'server') => {
|
||||||
set({ sttProvider: provider });
|
set({ sttProvider: provider });
|
||||||
if (typeof window !== 'undefined') {
|
if (typeof window !== 'undefined') {
|
||||||
|
|||||||
@@ -14,3 +14,5 @@ export {
|
|||||||
summarizeText,
|
summarizeText,
|
||||||
sanitizeForTTS,
|
sanitizeForTTS,
|
||||||
} from './summarization.js';
|
} from './summarization.js';
|
||||||
|
|
||||||
|
export { transcribeAudio } from './stt.js';
|
||||||
|
|||||||
@@ -42,9 +42,9 @@ export function registerTtsRoutes(app, { resolveZenModel, sayTTSCapability }) {
|
|||||||
// Server-side TTS endpoint - streams audio from OpenAI TTS API
|
// Server-side TTS endpoint - streams audio from OpenAI TTS API
|
||||||
app.post('/api/tts/speak', async (req, res) => {
|
app.post('/api/tts/speak', async (req, res) => {
|
||||||
try {
|
try {
|
||||||
const { text, voice = 'nova', model = 'gpt-4o-mini-tts', speed = 0.9, instructions, summarize = false, providerId, modelId, threshold = 200, maxLength = 500, apiKey } = req.body || {};
|
const { text, voice = 'nova', model = 'gpt-4o-mini-tts', speed = 0.9, instructions, summarize = false, providerId, modelId, threshold = 200, maxLength = 500, apiKey, baseURL } = req.body || {};
|
||||||
|
|
||||||
console.log('[TTS] Request received:', { voice, model, speed, textLength: text?.length, hasApiKey: !!apiKey });
|
console.log('[TTS] Request received:', { voice, model, speed, textLength: text?.length, hasApiKey: !!apiKey, hasBaseURL: !!baseURL });
|
||||||
|
|
||||||
if (!text || typeof text !== 'string' || !text.trim()) {
|
if (!text || typeof text !== 'string' || !text.trim()) {
|
||||||
return res.status(400).json({ error: 'Text is required' });
|
return res.status(400).json({ error: 'Text is required' });
|
||||||
@@ -53,13 +53,14 @@ export function registerTtsRoutes(app, { resolveZenModel, sayTTSCapability }) {
|
|||||||
// Dynamically import the TTS service (ESM)
|
// Dynamically import the TTS service (ESM)
|
||||||
const { ttsService } = await getTtsModule();
|
const { ttsService } = await getTtsModule();
|
||||||
|
|
||||||
// Check availability - either server-configured or client-provided API key
|
// Check availability - server-configured key, client-provided key, or custom server URL
|
||||||
const hasServerKey = ttsService.isAvailable();
|
const hasServerKey = ttsService.isAvailable();
|
||||||
const hasClientKey = apiKey && typeof apiKey === 'string' && apiKey.trim().length > 0;
|
const hasClientKey = apiKey && typeof apiKey === 'string' && apiKey.trim().length > 0;
|
||||||
|
const hasCustomBaseURL = baseURL && typeof baseURL === 'string' && baseURL.trim().length > 0;
|
||||||
|
|
||||||
if (!hasServerKey && !hasClientKey) {
|
if (!hasServerKey && !hasClientKey && !hasCustomBaseURL) {
|
||||||
return res.status(503).json({
|
return res.status(503).json({
|
||||||
error: 'TTS service not available. Please configure OpenAI in OpenCode or provide an API key in settings.'
|
error: 'TTS service not available. Please configure OpenAI in OpenCode, provide an API key, or set a custom server URL in settings.'
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -87,44 +88,24 @@ export function registerTtsRoutes(app, { resolveZenModel, sayTTSCapability }) {
|
|||||||
model,
|
model,
|
||||||
speed,
|
speed,
|
||||||
instructions,
|
instructions,
|
||||||
apiKey: hasClientKey ? apiKey.trim() : undefined
|
apiKey: hasClientKey ? apiKey.trim() : undefined,
|
||||||
|
baseURL: hasCustomBaseURL ? baseURL.trim() : undefined,
|
||||||
});
|
});
|
||||||
|
|
||||||
// Set headers for audio streaming
|
|
||||||
// Note: Don't set Transfer-Encoding manually - Express handles it automatically
|
|
||||||
res.setHeader('Content-Type', result.contentType);
|
res.setHeader('Content-Type', result.contentType);
|
||||||
res.setHeader('Cache-Control', 'no-cache');
|
res.setHeader('Cache-Control', 'no-cache');
|
||||||
|
res.setHeader('Content-Length', result.buffer.length);
|
||||||
// Collect the full audio buffer and send it
|
res.send(result.buffer);
|
||||||
// This avoids chunked encoding issues with proxies
|
} catch (error) {
|
||||||
const reader = result.stream.getReader();
|
console.error('[TTS] Error:', error);
|
||||||
const chunks = [];
|
|
||||||
|
|
||||||
try {
|
|
||||||
while (true) {
|
|
||||||
const { done, value } = await reader.read();
|
|
||||||
if (done) break;
|
|
||||||
chunks.push(Buffer.from(value));
|
|
||||||
}
|
|
||||||
const audioBuffer = Buffer.concat(chunks);
|
|
||||||
res.setHeader('Content-Length', audioBuffer.length);
|
|
||||||
res.send(audioBuffer);
|
|
||||||
} catch (streamError) {
|
|
||||||
console.error('[TTS] Stream error:', streamError);
|
|
||||||
if (!res.headersSent) {
|
if (!res.headersSent) {
|
||||||
res.status(500).json({ error: 'Stream error' });
|
const { model: m, voice: v, baseURL: b } = req.body || {};
|
||||||
} else {
|
res.status(500).json({
|
||||||
res.end();
|
error: error instanceof Error ? error.message : 'TTS generation failed',
|
||||||
|
detail: { model: m, voice: v, hasBaseURL: !!b },
|
||||||
|
});
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
} catch (error) {
|
|
||||||
console.error('[TTS] Error:', error);
|
|
||||||
if (!res.headersSent) {
|
|
||||||
res.status(500).json({
|
|
||||||
error: error instanceof Error ? error.message : 'TTS generation failed'
|
|
||||||
});
|
|
||||||
}
|
|
||||||
}
|
|
||||||
});
|
});
|
||||||
|
|
||||||
app.post('/api/tts/summarize', async (req, res) => {
|
app.post('/api/tts/summarize', async (req, res) => {
|
||||||
|
|||||||
@@ -78,19 +78,24 @@ class TTSService {
|
|||||||
model = 'gpt-4o-mini-tts',
|
model = 'gpt-4o-mini-tts',
|
||||||
speed = 1.0,
|
speed = 1.0,
|
||||||
instructions,
|
instructions,
|
||||||
apiKey
|
apiKey,
|
||||||
|
baseURL,
|
||||||
} = options;
|
} = options;
|
||||||
|
|
||||||
// Use provided API key or fall back to configured key
|
// Use provided API key / baseURL or fall back to configured key
|
||||||
let client;
|
let client;
|
||||||
if (apiKey) {
|
if (baseURL || apiKey) {
|
||||||
client = new OpenAI({ apiKey });
|
const clientOpts = {};
|
||||||
|
if (apiKey) clientOpts.apiKey = apiKey;
|
||||||
|
if (!apiKey) clientOpts.apiKey = 'not-required';
|
||||||
|
if (baseURL) clientOpts.baseURL = baseURL;
|
||||||
|
client = new OpenAI(clientOpts);
|
||||||
} else {
|
} else {
|
||||||
client = this._getClient();
|
client = this._getClient();
|
||||||
}
|
}
|
||||||
|
|
||||||
if (!client) {
|
if (!client) {
|
||||||
throw new Error('OpenAI API key not configured. Set OPENAI_API_KEY environment variable, configure OpenAI in OpenCode, or provide an API key in settings.');
|
throw new Error('TTS service not available. Configure OpenAI in OpenCode, provide an API key, or set a custom server URL in settings.');
|
||||||
}
|
}
|
||||||
|
|
||||||
if (!text.trim()) {
|
if (!text.trim()) {
|
||||||
@@ -98,21 +103,25 @@ class TTSService {
|
|||||||
}
|
}
|
||||||
|
|
||||||
try {
|
try {
|
||||||
console.log('[TTSService] Generating speech with voice:', voice, 'model:', model);
|
// OpenAI-compatible servers (custom baseURL) may not support `instructions`
|
||||||
const response = await client.audio.speech.create({
|
// or `response_format`, but do support `speed`. Send the safe subset.
|
||||||
model,
|
const speechParams = baseURL
|
||||||
voice,
|
? { model, voice, input: text, speed }
|
||||||
input: text,
|
: {
|
||||||
speed,
|
model,
|
||||||
...(instructions && { instructions }),
|
voice,
|
||||||
response_format: 'mp3',
|
input: text,
|
||||||
});
|
speed,
|
||||||
|
...(instructions && { instructions }),
|
||||||
|
response_format: 'mp3',
|
||||||
|
};
|
||||||
|
|
||||||
// Convert the response to a web stream
|
console.log('[TTSService] Generating speech — model:', model, 'voice:', voice, 'baseURL:', baseURL ?? '(openai)');
|
||||||
const stream = response.body;
|
const response = await client.audio.speech.create(speechParams);
|
||||||
|
|
||||||
|
const arrayBuffer = await response.arrayBuffer();
|
||||||
return {
|
return {
|
||||||
stream,
|
buffer: Buffer.from(arrayBuffer),
|
||||||
contentType: 'audio/mpeg',
|
contentType: 'audio/mpeg',
|
||||||
};
|
};
|
||||||
} catch (error) {
|
} catch (error) {
|
||||||
|
|||||||
Reference in New Issue
Block a user