// ============================================ // Speech models — all ONNX, all CPU, all in this Node process. // // There is no GPU and no Python. That is the whole point: the AWS host has // neither, and a second service was one more thing to deploy and keep alive. // // VAD Silero 2 MB endpointing // STT Whisper base ~80 MB Tamil + English + language detection // TTS MMS-TTS VITS ~114 MB per language, feed-forward // // Measured on an i7-10850H, CPU only: // TTS RTF 0.28x (3.5x faster than realtime) // STT ~1.2 s for 4 s of audio // // Two findings worth keeping: // // * VITS is feed-forward. The earlier Parler-TTS attempt was autoregressive // and ran at RTF ~5x — i.e. 5x SLOWER than realtime — which is why voice was // unusable even on a GPU. Architecture mattered far more than hardware here. // // * int8 is a trap for a model this small: dynamic quantisation made TTS 5.7x // SLOWER than fp32 (RTF 1.67x vs 0.28x) because the quantise/dequantise // overhead dominates. We ship fp32 deliberately. // ============================================ import path from 'node:path'; import { fileURLToPath } from 'node:url'; import { pipeline, env } from '@huggingface/transformers'; import logger from '../utils/logger.js'; const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..'); // Hub downloads are cached here so a container restart does not re-fetch. env.cacheDir = process.env.SPEECH_CACHE_DIR || path.join(ROOT, '.transformers-cache'); // Tamil is loaded from a folder we exported ourselves — no public ONNX build // of mms-tts-tam exists. See scripts/export-tamil-tts.py. env.localModelPath = path.join(ROOT, 'assets/tts'); // whisper-tiny.en by default: measured, Whisper is the memory hog, not TTS. // base cost 554 MB of an 879 MB total, which overran the 768 MB container cap. // The .en build is half the size and, being English-only, cannot detect a // language — which is fine when VOICE_LANGUAGES is just `en`. const STT_MODEL = process.env.STT_MODEL || 'onnx-community/whisper-tiny.en'; /** True when the STT checkpoint is English-only and cannot identify languages. */ export const sttIsEnglishOnly = () => /\.en$/.test(STT_MODEL); /** TTS voice per language. Tamil is exported locally; English is on the Hub. */ const ALL_VOICES = { en: { id: 'Xenova/mms-tts-eng', local: false, label: 'English', native: 'English' }, ta: { id: 'mms-tts-tam', local: true, label: 'Tamil', native: 'தமிழ்' }, }; // Each extra language is a further ~200 MB resident. Enable only what the // deployment actually speaks — English alone on the current AWS box. const ENABLED = (process.env.VOICE_LANGUAGES || 'en') .split(',').map((s) => s.trim()).filter((c) => ALL_VOICES[c]); const VOICES = Object.fromEntries(ENABLED.map((c) => [c, ALL_VOICES[c]])); export const LANGUAGES = [ // Auto-detect is only offered when there is a choice to make AND the STT // model can actually detect — offering it otherwise is a lie. ...(ENABLED.length > 1 && !sttIsEnglishOnly() ? [{ code: 'auto', label: 'Auto-detect', native: 'Auto' }] : []), ...ENABLED.map((c) => ({ code: c, label: ALL_VOICES[c].label, native: ALL_VOICES[c].native })), ]; export const defaultLanguage = () => (LANGUAGES[0]?.code || 'en'); const cache = new Map(); let sttPromise = null; /** * Models load on first use, not at boot. A CRM restart should not wait ~10 s * for speech models that most sessions never touch. */ async function loadOnce(key, build) { if (!cache.has(key)) { const t0 = Date.now(); cache.set(key, build().then((m) => { logger.info(`🔊 loaded ${key} in ${((Date.now() - t0) / 1000).toFixed(1)}s`); return m; }).catch((e) => { cache.delete(key); // let the next attempt retry throw e; })); } return cache.get(key); } export async function getSTT() { if (!sttPromise) { sttPromise = loadOnce(STT_MODEL, () => // q8 is the right call for Whisper — unlike VITS it is big enough that // quantisation is a clear win. pipeline('automatic-speech-recognition', STT_MODEL, { dtype: 'q8' }), ).catch((e) => { sttPromise = null; throw e; }); } return sttPromise; } export async function getTTS(lang) { const voice = VOICES[lang] || VOICES[ENABLED[0]]; const prev = env.allowRemoteModels; try { // Local folders must not be looked up on the Hub, and vice versa. env.allowRemoteModels = !voice.local; return await loadOnce(`tts:${voice.id}`, () => pipeline('text-to-speech', voice.id, { dtype: 'fp32' }), ); } finally { env.allowRemoteModels = prev; } } export const supportsTTS = (lang) => Boolean(VOICES[lang]); /** Warm the models the deployment actually expects to use. */ export async function warmup(langs = ENABLED) { try { await getSTT(); for (const l of langs) await getTTS(l); logger.info('🔊 speech models warm'); } catch (e) { logger.warn(`speech warmup failed (will retry on first use): ${e.message}`); } } export function speechStatus() { return { stt_model: STT_MODEL, english_only_stt: sttIsEnglishOnly(), tts_voices: Object.fromEntries(Object.entries(VOICES).map(([k, v]) => [k, v.id])), loaded: [...cache.keys()], languages: LANGUAGES, }; }