diff --git a/.env.example b/.env.example index 74333d4..6b02c94 100644 --- a/.env.example +++ b/.env.example @@ -58,11 +58,14 @@ ARTIFACT_TTL_HOURS=72 # --- Voice (speech-to-speech, CPU, in-process) --- # Models are ONNX via Transformers.js — no GPU, no Python, no second service. -# STT onnx-community/whisper-base Tamil + English + language detection -# TTS assets/tts/mms-tts-tam exported locally; no public ONNX exists -# TTS Xenova/mms-tts-eng from the Hub +# STT onnx-community/whisper-tiny.en English only; half the RAM of base +# TTS Xenova/mms-tts-eng from the Hub +# Tamil is built and tested (assets/tts/mms-tts-tam, exported locally — no +# public ONNX exists). Enable it with VOICE_LANGUAGES=en,ta and a multilingual +# STT_MODEL, but budget ~200 MB more resident for the extra voice. +VOICE_LANGUAGES=en SPEECH_WARMUP=false -STT_MODEL=onnx-community/whisper-base +STT_MODEL=onnx-community/whisper-tiny.en # Below this confidence, the user's preferred language beats the detector. DETECT_CONFIDENCE=0.6 diff --git a/Dockerfile b/Dockerfile index 2cca1a9..14729b6 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,19 +1,23 @@ # ============================================ # WeLe Agentic AI — production image # -# Node service only. The voice service is deliberately NOT in this image: it -# needs a CUDA GPU and several GB of RAM, and the target host has neither. -# See DEPLOY.md. +# Speech runs in this same process: ONNX on CPU via Transformers.js. No GPU, +# no Python, no second service. +# +# node:20-slim, NOT alpine. onnxruntime-node ships glibc binaries and Alpine is +# musl, so the native module fails at load with: +# Error loading shared library ld-linux-x86-64.so.2 (needed by libonnxruntime.so.1) +# `sharp`, pulled in by Transformers.js, has the same constraint. # ============================================ -FROM node:20-alpine AS deps +FROM node:20-slim AS deps WORKDIR /app COPY package*.json ./ # `npm ci` builds exactly the lockfile, so a deploy can never silently pick up # a different dependency tree than the one that was tested. RUN npm ci --omit=dev -FROM node:20-alpine +FROM node:20-slim WORKDIR /app # Run unprivileged. The base image already ships a `node` user. diff --git a/docker-compose.yml b/docker-compose.yml index 80580f5..7b3bf03 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -28,6 +28,12 @@ services: - NODE_ENV=production - PORT=4000 + # English-only keeps one TTS voice resident (~200 MB). Tamil is built and + # tested — VOICE_LANGUAGES=en,ta plus a multilingual STT_MODEL enables it, + # at roughly 200 MB more. + - VOICE_LANGUAGES=en + - STT_MODEL=onnx-community/whisper-tiny.en + # Reuse the CRM's Redis by service name on the shared network. # Keys are namespaced with REDIS_PREFIX, so the two never collide. - REDIS_ENABLED=true @@ -50,13 +56,22 @@ services: volumes: # Generated xlsx/pdf/pptx survive rebuilds; swept on a TTL by the app. - artifacts:/app/storage/artifacts + # Speech models are fetched from HuggingFace on first use (~200 MB). + # Without this they re-download on every restart and the first voice turn + # after a deploy stalls for a minute. + - speech_cache:/app/.transformers-cache # A 2 vCPU / 3.7 GB host already runs the CRM, chat-service, Redis and # Milvus. Capping this container keeps a runaway turn from starving them. + # + # 1 GB, not 768 MB: the speech models are resident once voice is used — + # measured 729 MB (Whisper tiny.en 415 MB + MMS-TTS English 203 MB + VAD + # 28 MB + the agent itself). 768 MB left no headroom, and an OOM kill takes + # text chat down with voice. Text-only sessions stay near 80 MB. deploy: resources: limits: - memory: 768M + memory: 1024M logging: driver: json-file @@ -75,3 +90,4 @@ networks: volumes: artifacts: + speech_cache: diff --git a/scripts/t-footprint.mjs b/scripts/t-footprint.mjs new file mode 100644 index 0000000..61dc6b2 --- /dev/null +++ b/scripts/t-footprint.mjs @@ -0,0 +1,46 @@ +/* English-only footprint: does it fit the 768 MB container cap on AWS? + + Loads exactly what an English-only deployment needs and reports RSS after + each stage, so the answer is measured rather than estimated. +*/ +const mb = () => Math.round(process.memoryUsage().rss / 1048576); +const step = (label) => console.log(` ${label.padEnd(34)} RSS ${String(mb()).padStart(4)} MB`); + +step('baseline (node + agent code)'); + +const { synthesize, transcribe, Endpointer } = await import('../src/speech/index.js'); +step('after importing speech module'); + +// VAD +const ep = new Endpointer(); +await ep.push(new Float32Array(16000)); +step('+ Silero VAD'); + +// TTS English +const spoken = await synthesize('There are three thousand four hundred and twenty seven new leads.', 'en'); +step('+ MMS-TTS English'); + +// STT +const resample = (a, from, to) => { + const r = from / to, out = new Float32Array(Math.floor(a.length / r)); + for (let i = 0; i < out.length; i++) { const p = i * r, k = Math.floor(p); out[i] = a[k] + (a[Math.min(k + 1, a.length - 1)] - a[k]) * (p - k); } + return out; +}; +const audio = resample(spoken.audio, spoken.sampling_rate, 16000); +const heard = await transcribe(audio, 'en', 'en'); +step('+ Whisper base (STT)'); + +// Steady state: a few turns, to see whether it keeps growing. +for (let i = 0; i < 3; i++) { + await synthesize('Checking the leads now.', 'en'); + await transcribe(audio, 'en', 'en'); +} +step('after 3 more turns'); + +console.log(`\n transcript: ${JSON.stringify(heard.text)}`); +console.log(` TTS rate : ${spoken.sampling_rate} Hz`); + +const peak = mb(); +const CAP = 768; +console.log(`\n peak ${peak} MB vs ${CAP} MB container cap → ${peak < CAP * 0.8 ? 'FITS ✅' : peak < CAP ? 'TIGHT ⚠️' : 'EXCEEDS ❌'}`); +process.exit(0); diff --git a/src/speech/index.js b/src/speech/index.js index 518feab..b95a4a0 100644 --- a/src/speech/index.js +++ b/src/speech/index.js @@ -7,10 +7,10 @@ // one shift and should never have to touch a language menu. // ============================================ import { Tensor } from '@huggingface/transformers'; -import { getSTT, getTTS, supportsTTS } from './models.js'; +import { getSTT, getTTS, supportsTTS, sttIsEnglishOnly, defaultLanguage } from './models.js'; import logger from '../utils/logger.js'; -export { LANGUAGES, warmup, speechStatus } from './models.js'; +export { LANGUAGES, warmup, speechStatus, defaultLanguage } from './models.js'; export { Endpointer, warmupVad } from './vad.js'; const RATE = 16000; @@ -67,7 +67,7 @@ async function detectLanguage(audio) { * @param {string} lang 'auto' | 'ta' | 'en' * @param {string} prefer used when detection is unusable */ -export async function transcribe(audio, lang = 'auto', prefer = 'ta') { +export async function transcribe(audio, lang = 'auto', prefer = defaultLanguage()) { if (!audio || audio.length < RATE / 5) { // under 200 ms return { text: '', lang: prefer, note: 'too short' }; } @@ -80,7 +80,11 @@ export async function transcribe(audio, lang = 'auto', prefer = 'ta') { let detected = null; let confidence = null; - if (lang === 'auto') { + // An English-only checkpoint has no language tokens to read, and passing + // `language` to it is rejected — so "auto" simply means English there. + if (lang === 'auto' && sttIsEnglishOnly()) { + used = 'en'; + } else if (lang === 'auto') { try { const d = await detectLanguage(audio); detected = d.lang; @@ -95,7 +99,15 @@ export async function transcribe(audio, lang = 'auto', prefer = 'ta') { } } - const result = await stt(audio, { task: 'transcribe', language: used, return_timestamps: false }); + // An English-only checkpoint rejects BOTH `task` and `language` — it has no + // other mode to select. Multilingual builds require the language, since they + // silently default to English otherwise. + const opts = { return_timestamps: false }; + if (!sttIsEnglishOnly()) { + opts.task = 'transcribe'; + opts.language = used; + } + const result = await stt(audio, opts); return finish(result, used, audio, t0, detected, confidence); } @@ -115,11 +127,11 @@ function finish(result, used, audio, t0, detected, confidence) { * Synthesise one piece of text. * @returns {Promise<{audio: Float32Array, sampling_rate: number}>} */ -export async function synthesize(text, lang = 'ta') { +export async function synthesize(text, lang = defaultLanguage()) { const clean = (text || '').trim(); if (!clean) return null; - const use = supportsTTS(lang) ? lang : 'en'; + const use = supportsTTS(lang) ? lang : defaultLanguage(); const tts = await getTTS(use); const t0 = Date.now(); diff --git a/src/speech/models.js b/src/speech/models.js index f35ec40..85d5d73 100644 --- a/src/speech/models.js +++ b/src/speech/models.js @@ -35,20 +35,38 @@ env.cacheDir = process.env.SPEECH_CACHE_DIR || path.join(ROOT, '.transformers-ca // of mms-tts-tam exists. See scripts/export-tamil-tts.py. env.localModelPath = path.join(ROOT, 'assets/tts'); -const STT_MODEL = process.env.STT_MODEL || 'onnx-community/whisper-base'; +// whisper-tiny.en by default: measured, Whisper is the memory hog, not TTS. +// base cost 554 MB of an 879 MB total, which overran the 768 MB container cap. +// The .en build is half the size and, being English-only, cannot detect a +// language — which is fine when VOICE_LANGUAGES is just `en`. +const STT_MODEL = process.env.STT_MODEL || 'onnx-community/whisper-tiny.en'; -/** TTS voice per language. Tamil is local; English comes from the Hub. */ -const VOICES = { - ta: { id: 'mms-tts-tam', local: true }, - en: { id: 'Xenova/mms-tts-eng', local: false }, +/** True when the STT checkpoint is English-only and cannot identify languages. */ +export const sttIsEnglishOnly = () => /\.en$/.test(STT_MODEL); + +/** TTS voice per language. Tamil is exported locally; English is on the Hub. */ +const ALL_VOICES = { + en: { id: 'Xenova/mms-tts-eng', local: false, label: 'English', native: 'English' }, + ta: { id: 'mms-tts-tam', local: true, label: 'Tamil', native: 'தமிழ்' }, }; +// Each extra language is a further ~200 MB resident. Enable only what the +// deployment actually speaks — English alone on the current AWS box. +const ENABLED = (process.env.VOICE_LANGUAGES || 'en') + .split(',').map((s) => s.trim()).filter((c) => ALL_VOICES[c]); + +const VOICES = Object.fromEntries(ENABLED.map((c) => [c, ALL_VOICES[c]])); + export const LANGUAGES = [ - { code: 'auto', label: 'Auto-detect', native: 'Auto' }, - { code: 'ta', label: 'Tamil', native: 'தமிழ்' }, - { code: 'en', label: 'English', native: 'English' }, + // Auto-detect is only offered when there is a choice to make AND the STT + // model can actually detect — offering it otherwise is a lie. + ...(ENABLED.length > 1 && !sttIsEnglishOnly() + ? [{ code: 'auto', label: 'Auto-detect', native: 'Auto' }] : []), + ...ENABLED.map((c) => ({ code: c, label: ALL_VOICES[c].label, native: ALL_VOICES[c].native })), ]; +export const defaultLanguage = () => (LANGUAGES[0]?.code || 'en'); + const cache = new Map(); let sttPromise = null; @@ -81,8 +99,8 @@ export async function getSTT() { return sttPromise; } -export async function getTTS(lang = 'ta') { - const voice = VOICES[lang] || VOICES.ta; +export async function getTTS(lang) { + const voice = VOICES[lang] || VOICES[ENABLED[0]]; const prev = env.allowRemoteModels; try { // Local folders must not be looked up on the Hub, and vice versa. @@ -98,7 +116,7 @@ export async function getTTS(lang = 'ta') { export const supportsTTS = (lang) => Boolean(VOICES[lang]); /** Warm the models the deployment actually expects to use. */ -export async function warmup(langs = ['ta', 'en']) { +export async function warmup(langs = ENABLED) { try { await getSTT(); for (const l of langs) await getTTS(l); @@ -111,6 +129,7 @@ export async function warmup(langs = ['ta', 'en']) { export function speechStatus() { return { stt_model: STT_MODEL, + english_only_stt: sttIsEnglishOnly(), tts_voices: Object.fromEntries(Object.entries(VOICES).map(([k, v]) => [k, v.id])), loaded: [...cache.keys()], languages: LANGUAGES,