Files
openchamber/packages/web/server/lib/tts/routes.js
T
yangyaofeiandBohdan Triapitsyn 06526767a2 feat(tts/stt): add API key support for OpenAI-compatible custom providers (#1361)
* feat(tts/stt): add API key support for OpenAI-compatible custom providers

## Problem
Custom (OpenAI-compatible) TTS/STT provider in Voice Settings has no way to
pass an API key or bearer token. Many self-hosted or third-party compatible
servers require authentication, making them unreachable from OpenChamber.

The server-side TTS route already accepts an `apiKey` parameter, but the
frontend never sends it. The STT route hardcodes `'not-required'`.

## Implementation
- Add `openaiCompatibleApiKey` to Zustand config store, persisted to localStorage
- Add API Key input field in VoiceSettings.tsx under the custom provider section
- Wire `openaiCompatibleApiKey` through useServerTTS to the TTS backend
- Add `apiKey` field to AudioStreamConfig for STT, forwarded as X-API-Key header
- Update server STT route to accept and forward X-API-Key to transcribeAudio
- Update stt.js to use client-provided apiKey before falling back to env var

## Files changed
- packages/ui/src/stores/useConfigStore.ts
- packages/ui/src/components/sections/openchamber/VoiceSettings.tsx
- packages/ui/src/hooks/useServerTTS.ts
- packages/ui/src/hooks/useBrowserVoice.ts
- packages/ui/src/lib/voice/audioStreamService.ts
- packages/web/server/lib/tts/routes.js
- packages/web/server/lib/tts/stt.js

* feat(tts/stt): add separate API key support for custom TTS and STT providers

## Problem
Custom (OpenAI-compatible) TTS and STT providers in Voice Settings have no way
to pass API keys. Many self-hosted or third-party compatible servers require
authentication, making them unreachable from OpenChamber Desktop (Electron).

## Implementation
- Add `openaiCompatibleApiKey` for TTS (persisted to localStorage, passed in JSON body)
- Add `sttApiKey` for STT (persisted to localStorage, passed via Authorization: Bearer header)
- Two independent keys: TTS and STT are configured separately
- STT authentication follows OpenAI standard (Authorization: Bearer <token>)
- TTS authentication follows existing pattern (apiKey in JSON body)
- Backend STT route extracts bearer token from Authorization header
- Backend STT service prefers client-provided key over OPENAI_API_KEY env var

## Fixes
- Fixed P1: ConfigStore interface now declares setOpenaiCompatibleApiKey setter
- STT API key is only forwarded when sttProvider === 'server' (not leaked to other providers)

## Files changed (7)
- packages/ui/src/stores/useConfigStore.ts
- packages/ui/src/components/sections/openchamber/VoiceSettings.tsx
- packages/ui/src/hooks/useServerTTS.ts
- packages/ui/src/hooks/useBrowserVoice.ts
- packages/ui/src/lib/voice/audioStreamService.ts
- packages/web/server/lib/tts/routes.js
- packages/web/server/lib/tts/stt.js

* fix: refresh server STT callback when API key changes

---------

Co-authored-by: Bohdan Triapitsyn <artmore@protonmail.com>
2026-05-24 00:58:22 +03:00

266 lines
9.5 KiB
JavaScript

import express from 'express';
import { normalizeCustomOpenAIBaseURL } from './base-url.js';
import { summarizeText, sanitizeForTTS, sanitizeForNote } from '../text/summarization.js';
export function registerTtsRoutes(app, { sayTTSCapability }) {
let ttsModulePromise = null;
const getTtsModule = async () => {
if (!ttsModulePromise) {
ttsModulePromise = import('./index.js');
}
return ttsModulePromise;
};
app.post('/api/voice/token', async (req, res) => {
console.log('[Voice] Token request received:', {
contentType: req.headers['content-type'] || null,
});
try {
const openaiApiKey = process.env.OPENAI_API_KEY;
console.log('[Voice] OpenAI API Key present:', !!openaiApiKey);
if (!openaiApiKey) {
return res.status(503).json({
allowed: false,
error: 'OpenAI voice service not configured. Set OPENAI_API_KEY environment variable.'
});
}
// Return success - OpenAI TTS is available
res.json({
allowed: true,
provider: 'openai',
message: 'OpenAI TTS is available'
});
} catch (error) {
console.error('[Voice] Token generation error:', error);
res.status(500).json({
allowed: false,
error: 'Voice service error'
});
}
});
// Server-side TTS endpoint - streams audio from OpenAI TTS API
app.post('/api/tts/speak', async (req, res) => {
try {
const { text, voice = 'nova', model = 'gpt-4o-mini-tts', speed = 0.9, instructions, providerId, modelId, apiKey, baseURL } = req.body || {};
const normalizedBaseURLResult = normalizeCustomOpenAIBaseURL(baseURL);
if (normalizedBaseURLResult.error) {
return res.status(400).json({ error: normalizedBaseURLResult.error });
}
const normalizedBaseURL = normalizedBaseURLResult.value;
console.log('[TTS] Request received:', { voice, model, speed, textLength: text?.length, hasApiKey: !!apiKey, hasBaseURL: !!baseURL });
if (!text || typeof text !== 'string' || !text.trim()) {
return res.status(400).json({ error: 'Text is required' });
}
// Dynamically import the TTS service (ESM)
const { ttsService } = await getTtsModule();
// Check availability - server-configured key, client-provided key, or custom server URL
const hasServerKey = ttsService.isAvailable();
const hasClientKey = apiKey && typeof apiKey === 'string' && apiKey.trim().length > 0;
const hasCustomBaseURL = typeof normalizedBaseURL === 'string' && normalizedBaseURL.length > 0;
if (!hasServerKey && !hasClientKey && !hasCustomBaseURL) {
return res.status(503).json({
error: 'TTS service not available. Please configure OpenAI in OpenCode, provide an API key, or set a custom server URL in settings.'
});
}
let textToSpeak = text.trim();
// Historical summarize request fields are intentionally ignored. The
// model-backed summarization provider is retired.
const result = await ttsService.generateSpeechStream({
text: textToSpeak,
voice,
model,
speed,
instructions,
apiKey: hasClientKey ? apiKey.trim() : undefined,
baseURL: hasCustomBaseURL ? normalizedBaseURL : undefined,
});
res.setHeader('Content-Type', result.contentType);
res.setHeader('Cache-Control', 'no-cache');
res.setHeader('Content-Length', result.buffer.length);
res.send(result.buffer);
} catch (error) {
console.error('[TTS] Error:', error);
if (!res.headersSent) {
const { model: m, voice: v, baseURL: b } = req.body || {};
res.status(500).json({
error: error instanceof Error ? error.message : 'TTS generation failed',
detail: { model: m, voice: v, hasBaseURL: !!b },
});
}
}
});
app.post('/api/text/summarize', async (req, res) => {
try {
const { text, threshold = 200, maxLength = 500, mode } = req.body || {};
if (!text || typeof text !== 'string' || !text.trim()) {
return res.status(400).json({ error: 'Text is required' });
}
const result = await summarizeText({
text,
threshold,
maxLength,
mode: typeof mode === 'string' ? mode : 'tts',
});
return res.json(result);
} catch (error) {
console.error('[Summarize] Error:', error);
const sanitized = typeof req.body?.mode === 'string' && req.body.mode === 'note'
? sanitizeForNote(req.body?.text || '')
: sanitizeForTTS(req.body?.text || '');
return res.json({ summary: sanitized, summarized: false, reason: error.message });
}
});
// TTS status endpoint
app.get('/api/tts/status', async (_req, res) => {
try {
const { ttsService } = await getTtsModule();
res.json({
available: ttsService.isAvailable(),
voices: [
'alloy', 'ash', 'ballad', 'coral', 'echo', 'fable',
'nova', 'onyx', 'sage', 'shimmer', 'verse', 'marin', 'cedar'
]
});
} catch (error) {
res.status(500).json({ error: 'Failed to check TTS status' });
}
});
// macOS 'say' command TTS status endpoint - returns cached capability from startup
app.get('/api/tts/say/status', (_req, res) => {
res.json(sayTTSCapability);
});
// macOS 'say' command TTS speak endpoint
app.post('/api/tts/say/speak', async (req, res) => {
try {
const { text, voice = 'Samantha', rate = 200 } = req.body || {};
if (!text || typeof text !== 'string' || !text.trim()) {
return res.status(400).json({ error: 'Text is required' });
}
// Check if we're on macOS
if (process.platform !== 'darwin') {
return res.status(503).json({ error: 'macOS say command not available on this platform' });
}
const { exec } = await import('child_process');
const { promisify } = await import('util');
const fs = await import('fs');
const os = await import('os');
const path = await import('path');
const execAsync = promisify(exec);
// Create temp file for audio output (use m4a for browser compatibility)
const tempDir = os.tmpdir();
const tempFile = path.join(tempDir, `say-${Date.now()}.m4a`);
// Escape text for shell - escape both single quotes and double quotes
const escapedText = text.trim().replace(/'/g, "'\\''").replace(/"/g, '\\"');
// Generate audio file using 'say' command
// -o outputs to file, -r sets rate (words per minute)
// --data-format=aac outputs as m4a which browsers can decode
const cmd = `say -v "${voice}" -r ${rate} -o "${tempFile}" --data-format=aac '${escapedText}'`;
console.log('[TTS-Say] Generating speech:', { textLength: text.length, voice, rate });
await execAsync(cmd);
// Read the generated audio file
const audioBuffer = await fs.promises.readFile(tempFile);
// Clean up temp file
fs.promises.unlink(tempFile).catch(() => {});
// Send audio response
res.setHeader('Content-Type', 'audio/mp4');
res.setHeader('Content-Length', audioBuffer.length);
res.send(audioBuffer);
} catch (error) {
console.error('[TTS-Say] Error:', error);
res.status(500).json({
error: error instanceof Error ? error.message : 'Say command failed'
});
}
});
// Server-side STT: receive raw audio, proxy to OpenAI-compatible transcription endpoint
app.post(
'/api/stt/transcribe',
express.raw({ type: (req) => (req.headers['content-type'] || '').startsWith('audio/'), limit: '20mb' }),
async (req, res) => {
try {
const { transcribeAudio } = await import('./stt.js');
const mimeType = (req.headers['content-type'] || 'audio/webm').split(',')[0].trim();
const baseURL = typeof req.headers['x-base-url'] === 'string' ? req.headers['x-base-url'].trim() : '';
const model = typeof req.headers['x-model'] === 'string' && req.headers['x-model'].trim().length > 0
? req.headers['x-model'].trim()
: 'deepdml/faster-whisper-large-v3-turbo-ct2';
const language = typeof req.headers['x-language'] === 'string' && req.headers['x-language'].trim().length > 0
? req.headers['x-language'].trim()
: undefined;
const authHeader = typeof req.headers['authorization'] === 'string' ? req.headers['authorization'].trim() : '';
const apiKey = authHeader.startsWith('Bearer ') ? authHeader.slice(7).trim() : undefined;
if (!req.body || !Buffer.isBuffer(req.body) || req.body.length === 0) {
return res.status(400).json({ error: 'Audio data is required' });
}
if (!baseURL) {
return res.status(400).json({ error: 'X-Base-URL header is required' });
}
console.log('[STT] Transcribing audio:', {
bytes: req.body.length,
mimeType,
model,
baseURL,
language,
hasApiKey: !!apiKey,
});
const transcript = await transcribeAudio({
audioBuffer: req.body,
mimeType,
model,
baseURL,
apiKey,
language,
});
console.log('[STT] Transcript:', transcript?.slice(0, 120));
res.json({ transcript: transcript ?? '' });
} catch (error) {
console.error('[STT] Error:', error);
if (!res.headersSent) {
res.status(500).json({
error: error instanceof Error ? error.message : 'Transcription failed',
});
}
}
}
);
}