138 lines
5.2 KiB
JavaScript
138 lines
5.2 KiB
JavaScript
// ============================================
|
|
// Speech models — all ONNX, all CPU, all in this Node process.
|
|
//
|
|
// There is no GPU and no Python. That is the whole point: the AWS host has
|
|
// neither, and a second service was one more thing to deploy and keep alive.
|
|
//
|
|
// VAD Silero 2 MB endpointing
|
|
// STT Whisper base ~80 MB Tamil + English + language detection
|
|
// TTS MMS-TTS VITS ~114 MB per language, feed-forward
|
|
//
|
|
// Measured on an i7-10850H, CPU only:
|
|
// TTS RTF 0.28x (3.5x faster than realtime)
|
|
// STT ~1.2 s for 4 s of audio
|
|
//
|
|
// Two findings worth keeping:
|
|
//
|
|
// * VITS is feed-forward. The earlier Parler-TTS attempt was autoregressive
|
|
// and ran at RTF ~5x — i.e. 5x SLOWER than realtime — which is why voice was
|
|
// unusable even on a GPU. Architecture mattered far more than hardware here.
|
|
//
|
|
// * int8 is a trap for a model this small: dynamic quantisation made TTS 5.7x
|
|
// SLOWER than fp32 (RTF 1.67x vs 0.28x) because the quantise/dequantise
|
|
// overhead dominates. We ship fp32 deliberately.
|
|
// ============================================
|
|
import path from 'node:path';
|
|
import { fileURLToPath } from 'node:url';
|
|
import { pipeline, env } from '@huggingface/transformers';
|
|
import logger from '../utils/logger.js';
|
|
|
|
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..');
|
|
|
|
// Hub downloads are cached here so a container restart does not re-fetch.
|
|
env.cacheDir = process.env.SPEECH_CACHE_DIR || path.join(ROOT, '.transformers-cache');
|
|
// Tamil is loaded from a folder we exported ourselves — no public ONNX build
|
|
// of mms-tts-tam exists. See scripts/export-tamil-tts.py.
|
|
env.localModelPath = path.join(ROOT, 'assets/tts');
|
|
|
|
// whisper-tiny.en by default: measured, Whisper is the memory hog, not TTS.
|
|
// base cost 554 MB of an 879 MB total, which overran the 768 MB container cap.
|
|
// The .en build is half the size and, being English-only, cannot detect a
|
|
// language — which is fine when VOICE_LANGUAGES is just `en`.
|
|
const STT_MODEL = process.env.STT_MODEL || 'onnx-community/whisper-tiny.en';
|
|
|
|
/** True when the STT checkpoint is English-only and cannot identify languages. */
|
|
export const sttIsEnglishOnly = () => /\.en$/.test(STT_MODEL);
|
|
|
|
/** TTS voice per language. Tamil is exported locally; English is on the Hub. */
|
|
const ALL_VOICES = {
|
|
en: { id: 'Xenova/mms-tts-eng', local: false, label: 'English', native: 'English' },
|
|
ta: { id: 'mms-tts-tam', local: true, label: 'Tamil', native: 'தமிழ்' },
|
|
};
|
|
|
|
// Each extra language is a further ~200 MB resident. Enable only what the
|
|
// deployment actually speaks — English alone on the current AWS box.
|
|
const ENABLED = (process.env.VOICE_LANGUAGES || 'en')
|
|
.split(',').map((s) => s.trim()).filter((c) => ALL_VOICES[c]);
|
|
|
|
const VOICES = Object.fromEntries(ENABLED.map((c) => [c, ALL_VOICES[c]]));
|
|
|
|
export const LANGUAGES = [
|
|
// Auto-detect is only offered when there is a choice to make AND the STT
|
|
// model can actually detect — offering it otherwise is a lie.
|
|
...(ENABLED.length > 1 && !sttIsEnglishOnly()
|
|
? [{ code: 'auto', label: 'Auto-detect', native: 'Auto' }] : []),
|
|
...ENABLED.map((c) => ({ code: c, label: ALL_VOICES[c].label, native: ALL_VOICES[c].native })),
|
|
];
|
|
|
|
export const defaultLanguage = () => (LANGUAGES[0]?.code || 'en');
|
|
|
|
const cache = new Map();
|
|
let sttPromise = null;
|
|
|
|
/**
|
|
* Models load on first use, not at boot. A CRM restart should not wait ~10 s
|
|
* for speech models that most sessions never touch.
|
|
*/
|
|
async function loadOnce(key, build) {
|
|
if (!cache.has(key)) {
|
|
const t0 = Date.now();
|
|
cache.set(key, build().then((m) => {
|
|
logger.info(`🔊 loaded ${key} in ${((Date.now() - t0) / 1000).toFixed(1)}s`);
|
|
return m;
|
|
}).catch((e) => {
|
|
cache.delete(key); // let the next attempt retry
|
|
throw e;
|
|
}));
|
|
}
|
|
return cache.get(key);
|
|
}
|
|
|
|
export async function getSTT() {
|
|
if (!sttPromise) {
|
|
sttPromise = loadOnce(STT_MODEL, () =>
|
|
// q8 is the right call for Whisper — unlike VITS it is big enough that
|
|
// quantisation is a clear win.
|
|
pipeline('automatic-speech-recognition', STT_MODEL, { dtype: 'q8' }),
|
|
).catch((e) => { sttPromise = null; throw e; });
|
|
}
|
|
return sttPromise;
|
|
}
|
|
|
|
export async function getTTS(lang) {
|
|
const voice = VOICES[lang] || VOICES[ENABLED[0]];
|
|
const prev = env.allowRemoteModels;
|
|
try {
|
|
// Local folders must not be looked up on the Hub, and vice versa.
|
|
env.allowRemoteModels = !voice.local;
|
|
return await loadOnce(`tts:${voice.id}`, () =>
|
|
pipeline('text-to-speech', voice.id, { dtype: 'fp32' }),
|
|
);
|
|
} finally {
|
|
env.allowRemoteModels = prev;
|
|
}
|
|
}
|
|
|
|
export const supportsTTS = (lang) => Boolean(VOICES[lang]);
|
|
|
|
/** Warm the models the deployment actually expects to use. */
|
|
export async function warmup(langs = ENABLED) {
|
|
try {
|
|
await getSTT();
|
|
for (const l of langs) await getTTS(l);
|
|
logger.info('🔊 speech models warm');
|
|
} catch (e) {
|
|
logger.warn(`speech warmup failed (will retry on first use): ${e.message}`);
|
|
}
|
|
}
|
|
|
|
export function speechStatus() {
|
|
return {
|
|
stt_model: STT_MODEL,
|
|
english_only_stt: sttIsEnglishOnly(),
|
|
tts_voices: Object.fromEntries(Object.entries(VOICES).map(([k, v]) => [k, v.id])),
|
|
loaded: [...cache.keys()],
|
|
languages: LANGUAGES,
|
|
};
|
|
}
|