Files
openchamber/packages/web/server/lib/tts/routes.js
T
𝖎𝖚𝖑𝖎𝖎𝖆andBohdan Triapitsyn aae889b904 perf: optimize session loading and desktop startup (#2545)
* perf: optimize session loading and startup

* fix(chat): stabilize history prepend virtualization

* perf: unblock first session open from startup network contention

Opening the first session after app start waited seconds for its message
fetch. Three independent contributors, each measured via CDP network
capture and Chromium net-log against the packaged desktop app:

- The active-session watchdog fired an uncapped per-directory status poll
  and child-session discovery burst at startup, and other subsystems
  (git checks, global session pages, command/skill discovery) fanned out
  alongside it, saturating the browser's ~6 HTTP/1.1 sockets per origin.
  Add a shared background-network gate (concurrency 3) and route the
  watchdog, poll-shaped git reads (also priority: low), global session
  pages, command/skill loads, and the background update check through it.

- The packaged renderer is cross-origin to the loopback backend, so every
  API call needs a CORS preflight; a few slow OpenCode-proxied requests
  held the whole pool while preflights and interactive traffic queued
  behind them. Lift Chromium's per-host connection cap for loopback via
  ignore-connections-limit in the Electron shell.

- OpenCode initializes each directory lazily on its first request, so the
  first click paid that cost interactively. Warm the last-used directory
  and the three most recently opened projects right after OpenCode
  readiness, sequentially and best-effort, overlapping UI startup.

Validation: new background-network tests, lifecycle warmup test, focused
store/sync tests, UI type-check and lint, dead-code report, node --check
plus electron type-check/lint, and CDP first-open measurements on the
packaged app (message fetch socket queue 5.4s -> 0.03s).

* fix(ui): keep interactive git reads out of background queue

---------

Co-authored-by: Bohdan Triapitsyn <artmore@protonmail.com>
2026-07-31 12:51:15 +03:00

267 lines
9.6 KiB
JavaScript

import express from 'express';
import { normalizeCustomOpenAIBaseURL } from './base-url.js';
import { summarizeText, sanitizeForTTS, sanitizeForNote } from '../text/summarization.js';
export function registerTtsRoutes(app, { sayTTSCapability }) {
let ttsModulePromise = null;
const getTtsModule = async () => {
if (!ttsModulePromise) {
ttsModulePromise = import('./index.js');
}
return ttsModulePromise;
};
app.post('/api/voice/token', async (req, res) => {
console.log('[Voice] Token request received:', {
contentType: req.headers['content-type'] || null,
});
try {
const openaiApiKey = process.env.OPENAI_API_KEY;
console.log('[Voice] OpenAI API Key present:', !!openaiApiKey);
if (!openaiApiKey) {
return res.status(503).json({
allowed: false,
error: 'OpenAI voice service not configured. Set OPENAI_API_KEY environment variable.'
});
}
// Return success - OpenAI TTS is available
res.json({
allowed: true,
provider: 'openai',
message: 'OpenAI TTS is available'
});
} catch (error) {
console.error('[Voice] Token generation error:', error);
res.status(500).json({
allowed: false,
error: 'Voice service error'
});
}
});
// Server-side TTS endpoint - streams audio from OpenAI TTS API
app.post('/api/tts/speak', async (req, res) => {
try {
const { text, voice = 'nova', model = 'gpt-4o-mini-tts', speed = 0.9, instructions, providerId, modelId, apiKey, baseURL } = req.body || {};
const normalizedBaseURLResult = normalizeCustomOpenAIBaseURL(baseURL);
if (normalizedBaseURLResult.error) {
return res.status(400).json({ error: normalizedBaseURLResult.error });
}
const normalizedBaseURL = normalizedBaseURLResult.value;
console.log('[TTS] Request received:', { voice, model, speed, textLength: text?.length, hasApiKey: !!apiKey, hasBaseURL: !!baseURL });
if (!text || typeof text !== 'string' || !text.trim()) {
return res.status(400).json({ error: 'Text is required' });
}
// Dynamically import the TTS service (ESM)
const { ttsService } = await getTtsModule();
// Check availability - server-configured key, client-provided key, or custom server URL
const hasServerKey = ttsService.isAvailable();
const hasClientKey = apiKey && typeof apiKey === 'string' && apiKey.trim().length > 0;
const hasCustomBaseURL = typeof normalizedBaseURL === 'string' && normalizedBaseURL.length > 0;
if (!hasServerKey && !hasClientKey && !hasCustomBaseURL) {
return res.status(503).json({
error: 'TTS service not available. Please configure OpenAI in OpenCode, provide an API key, or set a custom server URL in settings.'
});
}
let textToSpeak = text.trim();
// Historical summarize request fields are intentionally ignored. The
// model-backed summarization provider is retired.
const result = await ttsService.generateSpeechStream({
text: textToSpeak,
voice,
model,
speed,
instructions,
apiKey: hasClientKey ? apiKey.trim() : undefined,
baseURL: hasCustomBaseURL ? normalizedBaseURL : undefined,
});
res.setHeader('Content-Type', result.contentType);
res.setHeader('Cache-Control', 'no-cache');
res.setHeader('Content-Length', result.buffer.length);
res.send(result.buffer);
} catch (error) {
console.error('[TTS] Error:', error);
if (!res.headersSent) {
const { model: m, voice: v, baseURL: b } = req.body || {};
res.status(500).json({
error: error instanceof Error ? error.message : 'TTS generation failed',
detail: { model: m, voice: v, hasBaseURL: !!b },
});
}
}
});
app.post('/api/text/summarize', async (req, res) => {
try {
const { text, threshold = 200, maxLength = 500, mode } = req.body || {};
if (!text || typeof text !== 'string' || !text.trim()) {
return res.status(400).json({ error: 'Text is required' });
}
const result = await summarizeText({
text,
threshold,
maxLength,
mode: typeof mode === 'string' ? mode : 'tts',
});
return res.json(result);
} catch (error) {
console.error('[Summarize] Error:', error);
const sanitized = typeof req.body?.mode === 'string' && req.body.mode === 'note'
? sanitizeForNote(req.body?.text || '')
: sanitizeForTTS(req.body?.text || '');
return res.json({ summary: sanitized, summarized: false, reason: error.message });
}
});
// TTS status endpoint
app.get('/api/tts/status', async (_req, res) => {
try {
const { ttsService } = await getTtsModule();
res.json({
available: ttsService.isAvailable(),
voices: [
'alloy', 'ash', 'ballad', 'coral', 'echo', 'fable',
'nova', 'onyx', 'sage', 'shimmer', 'verse', 'marin', 'cedar'
]
});
} catch (error) {
res.status(500).json({ error: 'Failed to check TTS status' });
}
});
// The startup probe runs concurrently with server bootstrap. An unusually
// early status request waits for that same authoritative result.
app.get('/api/tts/say/status', async (_req, res) => {
res.json(await sayTTSCapability);
});
// macOS 'say' command TTS speak endpoint
app.post('/api/tts/say/speak', async (req, res) => {
try {
const { text, voice = 'Samantha', rate = 200 } = req.body || {};
if (!text || typeof text !== 'string' || !text.trim()) {
return res.status(400).json({ error: 'Text is required' });
}
// Check if we're on macOS
if (process.platform !== 'darwin') {
return res.status(503).json({ error: 'macOS say command not available on this platform' });
}
const { exec } = await import('child_process');
const { promisify } = await import('util');
const fs = await import('fs');
const os = await import('os');
const path = await import('path');
const execAsync = promisify(exec);
// Create temp file for audio output (use m4a for browser compatibility)
const tempDir = os.tmpdir();
const tempFile = path.join(tempDir, `say-${Date.now()}.m4a`);
// Escape text for shell - escape both single quotes and double quotes
const escapedText = text.trim().replace(/'/g, "'\\''").replace(/"/g, '\\"');
// Generate audio file using 'say' command
// -o outputs to file, -r sets rate (words per minute)
// --data-format=aac outputs as m4a which browsers can decode
const cmd = `say -v "${voice}" -r ${rate} -o "${tempFile}" --data-format=aac '${escapedText}'`;
console.log('[TTS-Say] Generating speech:', { textLength: text.length, voice, rate });
await execAsync(cmd);
// Read the generated audio file
const audioBuffer = await fs.promises.readFile(tempFile);
// Clean up temp file
fs.promises.unlink(tempFile).catch(() => {});
// Send audio response
res.setHeader('Content-Type', 'audio/mp4');
res.setHeader('Content-Length', audioBuffer.length);
res.send(audioBuffer);
} catch (error) {
console.error('[TTS-Say] Error:', error);
res.status(500).json({
error: error instanceof Error ? error.message : 'Say command failed'
});
}
});
// Server-side STT: receive raw audio, proxy to OpenAI-compatible transcription endpoint
app.post(
'/api/stt/transcribe',
express.raw({ type: (req) => (req.headers['content-type'] || '').startsWith('audio/'), limit: '20mb' }),
async (req, res) => {
try {
const { transcribeAudio } = await import('./stt.js');
const mimeType = (req.headers['content-type'] || 'audio/webm').split(',')[0].trim();
const baseURL = typeof req.headers['x-base-url'] === 'string' ? req.headers['x-base-url'].trim() : '';
const model = typeof req.headers['x-model'] === 'string' && req.headers['x-model'].trim().length > 0
? req.headers['x-model'].trim()
: 'deepdml/faster-whisper-large-v3-turbo-ct2';
const language = typeof req.headers['x-language'] === 'string' && req.headers['x-language'].trim().length > 0
? req.headers['x-language'].trim()
: undefined;
const authHeader = typeof req.headers['authorization'] === 'string' ? req.headers['authorization'].trim() : '';
const apiKey = authHeader.startsWith('Bearer ') ? authHeader.slice(7).trim() : undefined;
if (!req.body || !Buffer.isBuffer(req.body) || req.body.length === 0) {
return res.status(400).json({ error: 'Audio data is required' });
}
if (!baseURL) {
return res.status(400).json({ error: 'X-Base-URL header is required' });
}
console.log('[STT] Transcribing audio:', {
bytes: req.body.length,
mimeType,
model,
baseURL,
language,
hasApiKey: !!apiKey,
});
const transcript = await transcribeAudio({
audioBuffer: req.body,
mimeType,
model,
baseURL,
apiKey,
language,
});
console.log('[STT] Transcript:', transcript?.slice(0, 120));
res.json({ transcript: transcript ?? '' });
} catch (error) {
console.error('[STT] Error:', error);
if (!res.headersSent) {
res.status(500).json({
error: error instanceof Error ? error.message : 'Transcription failed',
});
}
}
}
);
}