Files
Bohdan Triapitsyn 391f938334 feat(voice): match local and macOS voices to the language of the text
Text-to-speech picked one voice regardless of what language a reply was in.
A dependency-free language detector (script, marker letters, function words)
now decides the language of the whole message once; with the new
"Match the voice to the language of the text" setting the local provider
switches to a catalog model for that language (Kokoro zh/en and Piper models
for 12 languages, downloaded on first use like the existing model) and macOS
say switches to an installed voice whose locale matches. The local voice
picker lists voices of every installed model, and the settings show which
language models are on disk.

The Ukrainian Piper medium build is a character-level model that sherpa-onnx
turns into noise, so the espeak-based Lada build is used instead.

Claude-Session: https://claude.ai/code/session_017TK5JAYDfT3Fotc23UEg98
2026-08-30 02:24:22 +03:00

132 lines
4.1 KiB
JavaScript

/**
* Sherpa-onnx offline TTS (Kokoro and Piper/VITS). Runs inside the dictation worker process
* only — never load the native addon in the main server process.
*/
import { existsSync } from 'fs';
import path from 'path';
import { loadSherpaOnnxNode } from './sherpa-loader.js';
function assertFileExists(filePath, label) {
if (!existsSync(filePath)) {
throw new Error(`Missing ${label}: ${filePath}`);
}
}
function float32ToPcm16le(samples) {
const out = new Int16Array(samples.length);
for (let i = 0; i < samples.length; i += 1) {
const clamped = Math.max(-1, Math.min(1, samples[i]));
out[i] = Math.round(clamped * 32767);
}
return Buffer.from(out.buffer, out.byteOffset, out.byteLength);
}
/**
* sherpa-onnx model config for one catalog entry. Kokoro carries a voices
* bank (speaker ids) and optional lexicons; a Piper/VITS model is a single
* voice with espeak-ng phonemization.
* @param {{ modelDir: string, type?: string, files: Record<string, string>, lexicon?: string[] }} config
*/
function buildModelConfig(config) {
const file = (key, label) => {
const filePath = path.join(config.modelDir, config.files[key]);
assertFileExists(filePath, label);
return filePath;
};
const modelPath = file('model', 'TTS model');
const tokensPath = file('tokens', 'TTS tokens');
if (config.type === 'vits') {
// Piper models phonemize through espeak-ng (`espeakData`); character
// models (Coqui) read the text directly and carry no espeak data.
const dataDir = config.files.espeakData ? file('espeakData', 'TTS espeak-ng dataDir') : '';
return { vits: { model: modelPath, tokens: tokensPath, ...(dataDir ? { dataDir } : {}), lengthScale: 1.0 } };
}
const dataDir = file('espeakData', 'TTS espeak-ng dataDir');
const voicesPath = file('voices', 'TTS voices');
const lexicon = (config.lexicon ?? []).map((key) => file(key, 'TTS lexicon')).join(',');
return {
kokoro: {
model: modelPath,
voices: voicesPath,
tokens: tokensPath,
dataDir,
lengthScale: 1.0,
...(lexicon ? { lexicon } : {}),
},
};
}
export class SherpaTtsEngine {
/**
* @param {{ modelDir: string, type?: string, files: Record<string, string>, lexicon?: string[], numThreads?: number }} config
*/
constructor(config) {
const model = buildModelConfig(config);
const sherpa = loadSherpaOnnxNode();
if (typeof sherpa.OfflineTts !== 'function') {
throw new Error('sherpa-onnx-node OfflineTts is unavailable');
}
this.tts = new sherpa.OfflineTts({
model,
numThreads: config.numThreads ?? 2,
provider: 'cpu',
maxNumSentences: 1,
});
}
/**
* Synthesize text to PCM16LE.
* @param {string} text
* @param {{ speakerId?: number, speed?: number }} [options]
* @returns {{ pcm16: Buffer, sampleRate: number }}
*/
synthesize(text, options = {}) {
const trimmed = String(text || '').trim();
if (!trimmed) {
throw new Error('Cannot synthesize empty text');
}
const audio = this.tts.generate({
text: trimmed,
sid: Number.isInteger(options.speakerId) ? options.speakerId : 0,
speed: typeof options.speed === 'number' && options.speed > 0 ? options.speed : 1.0,
// Request a copied buffer from sherpa itself: native external-backed
// typed arrays are rejected by Electron.
enableExternalBuffer: false,
});
let samples = null;
if (audio && audio.samples instanceof Float32Array) {
samples = Float32Array.from(audio.samples);
} else if (audio && Array.isArray(audio.samples)) {
samples = Float32Array.from(audio.samples);
}
if (!samples) {
throw new Error('Unexpected sherpa TTS output: missing Float32 samples');
}
const sampleRate =
audio && typeof audio.sampleRate === 'number' && audio.sampleRate > 0
? audio.sampleRate
: typeof this.tts.sampleRate === 'number' && this.tts.sampleRate > 0
? this.tts.sampleRate
: 24000;
return { pcm16: float32ToPcm16le(samples), sampleRate };
}
free() {
try {
this.tts?.free?.();
} catch {
// ignore
}
}
}