Complete rebuild of voice input on a server-authoritative streaming architecture, replacing the legacy Web Speech / whole-blob / WASM engines and the dead voice-agent layer (~4k lines removed). Speech-to-text (dictation): - Client streams 16 kHz mono PCM16 chunks over /api/dictation/ws with seq/ack ordering; buffered audio is retained and replayed on reconnect - Server transcribes and streams live partial transcripts back; segments auto-commit every ~15s with silence suppression and adaptive finalization timeouts - Local provider (default, zero config): sherpa-onnx models in a forked worker process — auto-download with progress, staged extraction with verification, corrupt-model auto-recovery, idle shutdown after 5 min - Model catalog with settings picker (accuracy/speed ratings, sizes, download/delete): Parakeet TDT v2 (English) and v3 (25 European languages, auto-detected), Whisper base and tiny (multilingual, light) - OpenAI-compatible provider for any Whisper endpoint - Composer overlay with live transcript, volume meter, timer, and cancel / insert / insert-and-send actions; failed transcriptions keep their audio for retry or accepting the partial text as-is - Configurable keyboard shortcut (default mod+alt+v) toggles dictation; Enter confirms and Escape cancels while recording - Overlay is pixel-aligned with the composer (measured footer height, matching paddings/typography/gaps) — no layout shift when toggling Text-to-speech: - Local Kokoro provider (English, 11 voices) synthesized in the same worker via /api/dictation/tts/speak, managed by the shared model pipeline; sentence-pipelined playback keeps time-to-first-audio at ~1 sentence regardless of message length, and stop cancels in-flight synthesis - Sanitizer keeps inline-code content (strips backticks only), reads interword slashes aloud, and removes only absolute file paths Settings: - Voice page unified: a single read-aloud toggle owns all playback options (the confusing "Enable Voice Mode" is gone); a new "Enable voice input" toggle (default on, persisted to settings.json) hides the composer mic entirely when disabled Mobile and transport: - iOS/Android microphone permissions added (dictation was previously impossible on mobile) - Fixed Android WebSocket upgrades: the Capacitor WebView origin (https://localhost) was missing from the packaged-client allowlist, 403-ing every WS connection — root cause of the old mobile SSE lock, which is now removed for all transports Security and conventions: - All HTTP routes sit behind the global /api auth gate; the WS upgrade explicitly validates the UI session and origin, with oc_url_token narrowly allowlisted and covered by tests; the dictation socket mints a fresh URL token before connecting - Routes register before the generic OpenCode proxy; the client goes through runtimeFetch/getRuntimeUrlResolver, and runtime switches reset the dictation socket - VS Code deliberately reports dictation as unavailable (no server process in that runtime) CI: workflow Node bumped 20 -> 22 to match the repo engines and fix better-sqlite3 installs broken by node-gyp@latest on Node 20. New dependency: sherpa-onnx-node (prebuilt N-API; macOS/Linux x64+arm64, Windows x64 — Windows-on-ARM falls back to the OpenAI-compatible provider)
149 lines
3.6 KiB
JavaScript
149 lines
3.6 KiB
JavaScript
export const createStartupPipelineRuntime = (dependencies) => {
|
|
const {
|
|
createTerminalRuntime,
|
|
createDictationRuntime,
|
|
createMessageStreamWsRuntime,
|
|
createServerStartupRuntime,
|
|
} = dependencies;
|
|
|
|
const run = async (options) => {
|
|
const {
|
|
app,
|
|
server,
|
|
express,
|
|
fs,
|
|
path,
|
|
uiAuthController,
|
|
buildAugmentedPath,
|
|
searchPathFor,
|
|
isExecutable,
|
|
isRequestOriginAllowed,
|
|
rejectWebSocketUpgrade,
|
|
buildOpenCodeUrl,
|
|
getOpenCodeAuthHeaders,
|
|
globalEventHub,
|
|
processForwardedEventPayload,
|
|
messageStreamWsClients,
|
|
triggerHealthCheck,
|
|
upstreamStallTimeoutMs,
|
|
terminalHeartbeatIntervalMs,
|
|
terminalRebindWindowMs,
|
|
terminalMaxRebindsPerWindow,
|
|
setupProxy,
|
|
scheduleOpenCodeApiDetection,
|
|
bootstrapOpenCodeAtStartup,
|
|
staticRoutesRuntime,
|
|
process,
|
|
crypto,
|
|
normalizeTunnelBootstrapTtlMs,
|
|
readSettingsFromDiskMigrated,
|
|
tunnelAuthController,
|
|
startTunnelWithNormalizedRequest,
|
|
gracefulShutdown,
|
|
getSignalsAttached,
|
|
setSignalsAttached,
|
|
syncToHmrState,
|
|
TUNNEL_MODE_QUICK,
|
|
TUNNEL_MODE_MANAGED_LOCAL,
|
|
TUNNEL_MODE_MANAGED_REMOTE,
|
|
host,
|
|
port,
|
|
startupTunnelRequest,
|
|
onTunnelReady,
|
|
tunnelRuntimeContext,
|
|
attachSignals,
|
|
apiOnly,
|
|
dictationModelsDir,
|
|
} = options;
|
|
|
|
const terminalRuntime = createTerminalRuntime({
|
|
app,
|
|
server,
|
|
express,
|
|
fs,
|
|
path,
|
|
uiAuthController,
|
|
buildAugmentedPath,
|
|
searchPathFor,
|
|
isExecutable,
|
|
isRequestOriginAllowed,
|
|
rejectWebSocketUpgrade,
|
|
TERMINAL_INPUT_WS_HEARTBEAT_INTERVAL_MS: terminalHeartbeatIntervalMs,
|
|
TERMINAL_INPUT_WS_REBIND_WINDOW_MS: terminalRebindWindowMs,
|
|
TERMINAL_INPUT_WS_MAX_REBINDS_PER_WINDOW: terminalMaxRebindsPerWindow,
|
|
});
|
|
|
|
const dictationRuntime = createDictationRuntime({
|
|
app,
|
|
server,
|
|
express,
|
|
uiAuthController,
|
|
isRequestOriginAllowed,
|
|
rejectWebSocketUpgrade,
|
|
modelsDir: dictationModelsDir,
|
|
});
|
|
|
|
const messageStreamRuntime = createMessageStreamWsRuntime({
|
|
server,
|
|
uiAuthController,
|
|
isRequestOriginAllowed,
|
|
rejectWebSocketUpgrade,
|
|
buildOpenCodeUrl,
|
|
getOpenCodeAuthHeaders,
|
|
globalEventHub,
|
|
processForwardedEventPayload,
|
|
wsClients: messageStreamWsClients,
|
|
triggerHealthCheck,
|
|
upstreamStallTimeoutMs,
|
|
});
|
|
|
|
setupProxy(app);
|
|
scheduleOpenCodeApiDetection();
|
|
void bootstrapOpenCodeAtStartup();
|
|
|
|
if (apiOnly) {
|
|
staticRoutesRuntime.registerApiOnlyFallbackRoutes(app);
|
|
} else {
|
|
staticRoutesRuntime.registerStaticRoutes(app);
|
|
}
|
|
|
|
const serverStartupRuntime = createServerStartupRuntime({
|
|
process,
|
|
crypto,
|
|
server,
|
|
normalizeTunnelBootstrapTtlMs,
|
|
readSettingsFromDiskMigrated,
|
|
tunnelAuthController,
|
|
startTunnelWithNormalizedRequest,
|
|
gracefulShutdown,
|
|
getSignalsAttached,
|
|
setSignalsAttached,
|
|
syncToHmrState,
|
|
TUNNEL_MODE_QUICK,
|
|
TUNNEL_MODE_MANAGED_LOCAL,
|
|
TUNNEL_MODE_MANAGED_REMOTE,
|
|
});
|
|
|
|
const bindHost = serverStartupRuntime.resolveBindHost(host);
|
|
const startupResult = await serverStartupRuntime.startListeningAndMaybeTunnel({
|
|
port,
|
|
bindHost,
|
|
startupTunnelRequest,
|
|
onTunnelReady,
|
|
});
|
|
tunnelRuntimeContext.setActivePort(startupResult.activePort);
|
|
|
|
serverStartupRuntime.attachProcessHandlers({ attachSignals });
|
|
|
|
return {
|
|
terminalRuntime,
|
|
dictationRuntime,
|
|
messageStreamRuntime,
|
|
};
|
|
};
|
|
|
|
return {
|
|
run,
|
|
};
|
|
};
|