* perf: optimize session loading and startup * fix(chat): stabilize history prepend virtualization * perf: unblock first session open from startup network contention Opening the first session after app start waited seconds for its message fetch. Three independent contributors, each measured via CDP network capture and Chromium net-log against the packaged desktop app: - The active-session watchdog fired an uncapped per-directory status poll and child-session discovery burst at startup, and other subsystems (git checks, global session pages, command/skill discovery) fanned out alongside it, saturating the browser's ~6 HTTP/1.1 sockets per origin. Add a shared background-network gate (concurrency 3) and route the watchdog, poll-shaped git reads (also priority: low), global session pages, command/skill loads, and the background update check through it. - The packaged renderer is cross-origin to the loopback backend, so every API call needs a CORS preflight; a few slow OpenCode-proxied requests held the whole pool while preflights and interactive traffic queued behind them. Lift Chromium's per-host connection cap for loopback via ignore-connections-limit in the Electron shell. - OpenCode initializes each directory lazily on its first request, so the first click paid that cost interactively. Warm the last-used directory and the three most recently opened projects right after OpenCode readiness, sequentially and best-effort, overlapping UI startup. Validation: new background-network tests, lifecycle warmup test, focused store/sync tests, UI type-check and lint, dead-code report, node --check plus electron type-check/lint, and CDP first-open measurements on the packaged app (message fetch socket queue 5.4s -> 0.03s). * fix(ui): keep interactive git reads out of background queue --------- Co-authored-by: Bohdan Triapitsyn <artmore@protonmail.com>
267 lines
9.6 KiB
JavaScript
267 lines
9.6 KiB
JavaScript
import express from 'express';
|
|
import { normalizeCustomOpenAIBaseURL } from './base-url.js';
|
|
import { summarizeText, sanitizeForTTS, sanitizeForNote } from '../text/summarization.js';
|
|
|
|
export function registerTtsRoutes(app, { sayTTSCapability }) {
|
|
let ttsModulePromise = null;
|
|
const getTtsModule = async () => {
|
|
if (!ttsModulePromise) {
|
|
ttsModulePromise = import('./index.js');
|
|
}
|
|
return ttsModulePromise;
|
|
};
|
|
|
|
app.post('/api/voice/token', async (req, res) => {
|
|
console.log('[Voice] Token request received:', {
|
|
contentType: req.headers['content-type'] || null,
|
|
});
|
|
try {
|
|
const openaiApiKey = process.env.OPENAI_API_KEY;
|
|
console.log('[Voice] OpenAI API Key present:', !!openaiApiKey);
|
|
|
|
if (!openaiApiKey) {
|
|
return res.status(503).json({
|
|
allowed: false,
|
|
error: 'OpenAI voice service not configured. Set OPENAI_API_KEY environment variable.'
|
|
});
|
|
}
|
|
|
|
// Return success - OpenAI TTS is available
|
|
res.json({
|
|
allowed: true,
|
|
provider: 'openai',
|
|
message: 'OpenAI TTS is available'
|
|
});
|
|
} catch (error) {
|
|
console.error('[Voice] Token generation error:', error);
|
|
res.status(500).json({
|
|
allowed: false,
|
|
error: 'Voice service error'
|
|
});
|
|
}
|
|
});
|
|
|
|
// Server-side TTS endpoint - streams audio from OpenAI TTS API
|
|
app.post('/api/tts/speak', async (req, res) => {
|
|
try {
|
|
const { text, voice = 'nova', model = 'gpt-4o-mini-tts', speed = 0.9, instructions, providerId, modelId, apiKey, baseURL } = req.body || {};
|
|
|
|
const normalizedBaseURLResult = normalizeCustomOpenAIBaseURL(baseURL);
|
|
if (normalizedBaseURLResult.error) {
|
|
return res.status(400).json({ error: normalizedBaseURLResult.error });
|
|
}
|
|
const normalizedBaseURL = normalizedBaseURLResult.value;
|
|
|
|
console.log('[TTS] Request received:', { voice, model, speed, textLength: text?.length, hasApiKey: !!apiKey, hasBaseURL: !!baseURL });
|
|
|
|
if (!text || typeof text !== 'string' || !text.trim()) {
|
|
return res.status(400).json({ error: 'Text is required' });
|
|
}
|
|
|
|
// Dynamically import the TTS service (ESM)
|
|
const { ttsService } = await getTtsModule();
|
|
|
|
// Check availability - server-configured key, client-provided key, or custom server URL
|
|
const hasServerKey = ttsService.isAvailable();
|
|
const hasClientKey = apiKey && typeof apiKey === 'string' && apiKey.trim().length > 0;
|
|
const hasCustomBaseURL = typeof normalizedBaseURL === 'string' && normalizedBaseURL.length > 0;
|
|
|
|
if (!hasServerKey && !hasClientKey && !hasCustomBaseURL) {
|
|
return res.status(503).json({
|
|
error: 'TTS service not available. Please configure OpenAI in OpenCode, provide an API key, or set a custom server URL in settings.'
|
|
});
|
|
}
|
|
|
|
let textToSpeak = text.trim();
|
|
|
|
// Historical summarize request fields are intentionally ignored. The
|
|
// model-backed summarization provider is retired.
|
|
|
|
const result = await ttsService.generateSpeechStream({
|
|
text: textToSpeak,
|
|
voice,
|
|
model,
|
|
speed,
|
|
instructions,
|
|
apiKey: hasClientKey ? apiKey.trim() : undefined,
|
|
baseURL: hasCustomBaseURL ? normalizedBaseURL : undefined,
|
|
});
|
|
|
|
res.setHeader('Content-Type', result.contentType);
|
|
res.setHeader('Cache-Control', 'no-cache');
|
|
res.setHeader('Content-Length', result.buffer.length);
|
|
res.send(result.buffer);
|
|
} catch (error) {
|
|
console.error('[TTS] Error:', error);
|
|
if (!res.headersSent) {
|
|
const { model: m, voice: v, baseURL: b } = req.body || {};
|
|
res.status(500).json({
|
|
error: error instanceof Error ? error.message : 'TTS generation failed',
|
|
detail: { model: m, voice: v, hasBaseURL: !!b },
|
|
});
|
|
}
|
|
}
|
|
});
|
|
|
|
app.post('/api/text/summarize', async (req, res) => {
|
|
try {
|
|
const { text, threshold = 200, maxLength = 500, mode } = req.body || {};
|
|
|
|
if (!text || typeof text !== 'string' || !text.trim()) {
|
|
return res.status(400).json({ error: 'Text is required' });
|
|
}
|
|
|
|
const result = await summarizeText({
|
|
text,
|
|
threshold,
|
|
maxLength,
|
|
mode: typeof mode === 'string' ? mode : 'tts',
|
|
});
|
|
|
|
return res.json(result);
|
|
} catch (error) {
|
|
console.error('[Summarize] Error:', error);
|
|
const sanitized = typeof req.body?.mode === 'string' && req.body.mode === 'note'
|
|
? sanitizeForNote(req.body?.text || '')
|
|
: sanitizeForTTS(req.body?.text || '');
|
|
return res.json({ summary: sanitized, summarized: false, reason: error.message });
|
|
}
|
|
});
|
|
|
|
|
|
// TTS status endpoint
|
|
app.get('/api/tts/status', async (_req, res) => {
|
|
try {
|
|
const { ttsService } = await getTtsModule();
|
|
res.json({
|
|
available: ttsService.isAvailable(),
|
|
voices: [
|
|
'alloy', 'ash', 'ballad', 'coral', 'echo', 'fable',
|
|
'nova', 'onyx', 'sage', 'shimmer', 'verse', 'marin', 'cedar'
|
|
]
|
|
});
|
|
} catch (error) {
|
|
res.status(500).json({ error: 'Failed to check TTS status' });
|
|
}
|
|
});
|
|
|
|
// The startup probe runs concurrently with server bootstrap. An unusually
|
|
// early status request waits for that same authoritative result.
|
|
app.get('/api/tts/say/status', async (_req, res) => {
|
|
res.json(await sayTTSCapability);
|
|
});
|
|
|
|
// macOS 'say' command TTS speak endpoint
|
|
app.post('/api/tts/say/speak', async (req, res) => {
|
|
try {
|
|
const { text, voice = 'Samantha', rate = 200 } = req.body || {};
|
|
|
|
if (!text || typeof text !== 'string' || !text.trim()) {
|
|
return res.status(400).json({ error: 'Text is required' });
|
|
}
|
|
|
|
// Check if we're on macOS
|
|
if (process.platform !== 'darwin') {
|
|
return res.status(503).json({ error: 'macOS say command not available on this platform' });
|
|
}
|
|
|
|
const { exec } = await import('child_process');
|
|
const { promisify } = await import('util');
|
|
const fs = await import('fs');
|
|
const os = await import('os');
|
|
const path = await import('path');
|
|
const execAsync = promisify(exec);
|
|
|
|
// Create temp file for audio output (use m4a for browser compatibility)
|
|
const tempDir = os.tmpdir();
|
|
const tempFile = path.join(tempDir, `say-${Date.now()}.m4a`);
|
|
|
|
// Escape text for shell - escape both single quotes and double quotes
|
|
const escapedText = text.trim().replace(/'/g, "'\\''").replace(/"/g, '\\"');
|
|
|
|
// Generate audio file using 'say' command
|
|
// -o outputs to file, -r sets rate (words per minute)
|
|
// --data-format=aac outputs as m4a which browsers can decode
|
|
const cmd = `say -v "${voice}" -r ${rate} -o "${tempFile}" --data-format=aac '${escapedText}'`;
|
|
console.log('[TTS-Say] Generating speech:', { textLength: text.length, voice, rate });
|
|
|
|
await execAsync(cmd);
|
|
|
|
// Read the generated audio file
|
|
const audioBuffer = await fs.promises.readFile(tempFile);
|
|
|
|
// Clean up temp file
|
|
fs.promises.unlink(tempFile).catch(() => {});
|
|
|
|
// Send audio response
|
|
res.setHeader('Content-Type', 'audio/mp4');
|
|
res.setHeader('Content-Length', audioBuffer.length);
|
|
res.send(audioBuffer);
|
|
|
|
} catch (error) {
|
|
console.error('[TTS-Say] Error:', error);
|
|
res.status(500).json({
|
|
error: error instanceof Error ? error.message : 'Say command failed'
|
|
});
|
|
}
|
|
});
|
|
|
|
// Server-side STT: receive raw audio, proxy to OpenAI-compatible transcription endpoint
|
|
app.post(
|
|
'/api/stt/transcribe',
|
|
express.raw({ type: (req) => (req.headers['content-type'] || '').startsWith('audio/'), limit: '20mb' }),
|
|
async (req, res) => {
|
|
try {
|
|
const { transcribeAudio } = await import('./stt.js');
|
|
|
|
const mimeType = (req.headers['content-type'] || 'audio/webm').split(',')[0].trim();
|
|
const baseURL = typeof req.headers['x-base-url'] === 'string' ? req.headers['x-base-url'].trim() : '';
|
|
const model = typeof req.headers['x-model'] === 'string' && req.headers['x-model'].trim().length > 0
|
|
? req.headers['x-model'].trim()
|
|
: 'deepdml/faster-whisper-large-v3-turbo-ct2';
|
|
const language = typeof req.headers['x-language'] === 'string' && req.headers['x-language'].trim().length > 0
|
|
? req.headers['x-language'].trim()
|
|
: undefined;
|
|
const authHeader = typeof req.headers['authorization'] === 'string' ? req.headers['authorization'].trim() : '';
|
|
const apiKey = authHeader.startsWith('Bearer ') ? authHeader.slice(7).trim() : undefined;
|
|
|
|
if (!req.body || !Buffer.isBuffer(req.body) || req.body.length === 0) {
|
|
return res.status(400).json({ error: 'Audio data is required' });
|
|
}
|
|
|
|
if (!baseURL) {
|
|
return res.status(400).json({ error: 'X-Base-URL header is required' });
|
|
}
|
|
|
|
console.log('[STT] Transcribing audio:', {
|
|
bytes: req.body.length,
|
|
mimeType,
|
|
model,
|
|
baseURL,
|
|
language,
|
|
hasApiKey: !!apiKey,
|
|
});
|
|
|
|
const transcript = await transcribeAudio({
|
|
audioBuffer: req.body,
|
|
mimeType,
|
|
model,
|
|
baseURL,
|
|
apiKey,
|
|
language,
|
|
});
|
|
|
|
console.log('[STT] Transcript:', transcript?.slice(0, 120));
|
|
res.json({ transcript: transcript ?? '' });
|
|
} catch (error) {
|
|
console.error('[STT] Error:', error);
|
|
if (!res.headersSent) {
|
|
res.status(500).json({
|
|
error: error instanceof Error ? error.message : 'Transcription failed',
|
|
});
|
|
}
|
|
}
|
|
}
|
|
);
|
|
}
|