WeLe Agentic AI with Docker deployment
This commit is contained in:
+7
-4
@@ -58,11 +58,14 @@ ARTIFACT_TTL_HOURS=72
|
|||||||
|
|
||||||
# --- Voice (speech-to-speech, CPU, in-process) ---
|
# --- Voice (speech-to-speech, CPU, in-process) ---
|
||||||
# Models are ONNX via Transformers.js — no GPU, no Python, no second service.
|
# Models are ONNX via Transformers.js — no GPU, no Python, no second service.
|
||||||
# STT onnx-community/whisper-base Tamil + English + language detection
|
# STT onnx-community/whisper-tiny.en English only; half the RAM of base
|
||||||
# TTS assets/tts/mms-tts-tam exported locally; no public ONNX exists
|
# TTS Xenova/mms-tts-eng from the Hub
|
||||||
# TTS Xenova/mms-tts-eng from the Hub
|
# Tamil is built and tested (assets/tts/mms-tts-tam, exported locally — no
|
||||||
|
# public ONNX exists). Enable it with VOICE_LANGUAGES=en,ta and a multilingual
|
||||||
|
# STT_MODEL, but budget ~200 MB more resident for the extra voice.
|
||||||
|
VOICE_LANGUAGES=en
|
||||||
SPEECH_WARMUP=false
|
SPEECH_WARMUP=false
|
||||||
STT_MODEL=onnx-community/whisper-base
|
STT_MODEL=onnx-community/whisper-tiny.en
|
||||||
# Below this confidence, the user's preferred language beats the detector.
|
# Below this confidence, the user's preferred language beats the detector.
|
||||||
DETECT_CONFIDENCE=0.6
|
DETECT_CONFIDENCE=0.6
|
||||||
|
|
||||||
|
|||||||
+9
-5
@@ -1,19 +1,23 @@
|
|||||||
# ============================================
|
# ============================================
|
||||||
# WeLe Agentic AI — production image
|
# WeLe Agentic AI — production image
|
||||||
#
|
#
|
||||||
# Node service only. The voice service is deliberately NOT in this image: it
|
# Speech runs in this same process: ONNX on CPU via Transformers.js. No GPU,
|
||||||
# needs a CUDA GPU and several GB of RAM, and the target host has neither.
|
# no Python, no second service.
|
||||||
# See DEPLOY.md.
|
#
|
||||||
|
# node:20-slim, NOT alpine. onnxruntime-node ships glibc binaries and Alpine is
|
||||||
|
# musl, so the native module fails at load with:
|
||||||
|
# Error loading shared library ld-linux-x86-64.so.2 (needed by libonnxruntime.so.1)
|
||||||
|
# `sharp`, pulled in by Transformers.js, has the same constraint.
|
||||||
# ============================================
|
# ============================================
|
||||||
|
|
||||||
FROM node:20-alpine AS deps
|
FROM node:20-slim AS deps
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
COPY package*.json ./
|
COPY package*.json ./
|
||||||
# `npm ci` builds exactly the lockfile, so a deploy can never silently pick up
|
# `npm ci` builds exactly the lockfile, so a deploy can never silently pick up
|
||||||
# a different dependency tree than the one that was tested.
|
# a different dependency tree than the one that was tested.
|
||||||
RUN npm ci --omit=dev
|
RUN npm ci --omit=dev
|
||||||
|
|
||||||
FROM node:20-alpine
|
FROM node:20-slim
|
||||||
WORKDIR /app
|
WORKDIR /app
|
||||||
|
|
||||||
# Run unprivileged. The base image already ships a `node` user.
|
# Run unprivileged. The base image already ships a `node` user.
|
||||||
|
|||||||
+17
-1
@@ -28,6 +28,12 @@ services:
|
|||||||
- NODE_ENV=production
|
- NODE_ENV=production
|
||||||
- PORT=4000
|
- PORT=4000
|
||||||
|
|
||||||
|
# English-only keeps one TTS voice resident (~200 MB). Tamil is built and
|
||||||
|
# tested — VOICE_LANGUAGES=en,ta plus a multilingual STT_MODEL enables it,
|
||||||
|
# at roughly 200 MB more.
|
||||||
|
- VOICE_LANGUAGES=en
|
||||||
|
- STT_MODEL=onnx-community/whisper-tiny.en
|
||||||
|
|
||||||
# Reuse the CRM's Redis by service name on the shared network.
|
# Reuse the CRM's Redis by service name on the shared network.
|
||||||
# Keys are namespaced with REDIS_PREFIX, so the two never collide.
|
# Keys are namespaced with REDIS_PREFIX, so the two never collide.
|
||||||
- REDIS_ENABLED=true
|
- REDIS_ENABLED=true
|
||||||
@@ -50,13 +56,22 @@ services:
|
|||||||
volumes:
|
volumes:
|
||||||
# Generated xlsx/pdf/pptx survive rebuilds; swept on a TTL by the app.
|
# Generated xlsx/pdf/pptx survive rebuilds; swept on a TTL by the app.
|
||||||
- artifacts:/app/storage/artifacts
|
- artifacts:/app/storage/artifacts
|
||||||
|
# Speech models are fetched from HuggingFace on first use (~200 MB).
|
||||||
|
# Without this they re-download on every restart and the first voice turn
|
||||||
|
# after a deploy stalls for a minute.
|
||||||
|
- speech_cache:/app/.transformers-cache
|
||||||
|
|
||||||
# A 2 vCPU / 3.7 GB host already runs the CRM, chat-service, Redis and
|
# A 2 vCPU / 3.7 GB host already runs the CRM, chat-service, Redis and
|
||||||
# Milvus. Capping this container keeps a runaway turn from starving them.
|
# Milvus. Capping this container keeps a runaway turn from starving them.
|
||||||
|
#
|
||||||
|
# 1 GB, not 768 MB: the speech models are resident once voice is used —
|
||||||
|
# measured 729 MB (Whisper tiny.en 415 MB + MMS-TTS English 203 MB + VAD
|
||||||
|
# 28 MB + the agent itself). 768 MB left no headroom, and an OOM kill takes
|
||||||
|
# text chat down with voice. Text-only sessions stay near 80 MB.
|
||||||
deploy:
|
deploy:
|
||||||
resources:
|
resources:
|
||||||
limits:
|
limits:
|
||||||
memory: 768M
|
memory: 1024M
|
||||||
|
|
||||||
logging:
|
logging:
|
||||||
driver: json-file
|
driver: json-file
|
||||||
@@ -75,3 +90,4 @@ networks:
|
|||||||
|
|
||||||
volumes:
|
volumes:
|
||||||
artifacts:
|
artifacts:
|
||||||
|
speech_cache:
|
||||||
|
|||||||
@@ -0,0 +1,46 @@
|
|||||||
|
/* English-only footprint: does it fit the 768 MB container cap on AWS?
|
||||||
|
|
||||||
|
Loads exactly what an English-only deployment needs and reports RSS after
|
||||||
|
each stage, so the answer is measured rather than estimated.
|
||||||
|
*/
|
||||||
|
const mb = () => Math.round(process.memoryUsage().rss / 1048576);
|
||||||
|
const step = (label) => console.log(` ${label.padEnd(34)} RSS ${String(mb()).padStart(4)} MB`);
|
||||||
|
|
||||||
|
step('baseline (node + agent code)');
|
||||||
|
|
||||||
|
const { synthesize, transcribe, Endpointer } = await import('../src/speech/index.js');
|
||||||
|
step('after importing speech module');
|
||||||
|
|
||||||
|
// VAD
|
||||||
|
const ep = new Endpointer();
|
||||||
|
await ep.push(new Float32Array(16000));
|
||||||
|
step('+ Silero VAD');
|
||||||
|
|
||||||
|
// TTS English
|
||||||
|
const spoken = await synthesize('There are three thousand four hundred and twenty seven new leads.', 'en');
|
||||||
|
step('+ MMS-TTS English');
|
||||||
|
|
||||||
|
// STT
|
||||||
|
const resample = (a, from, to) => {
|
||||||
|
const r = from / to, out = new Float32Array(Math.floor(a.length / r));
|
||||||
|
for (let i = 0; i < out.length; i++) { const p = i * r, k = Math.floor(p); out[i] = a[k] + (a[Math.min(k + 1, a.length - 1)] - a[k]) * (p - k); }
|
||||||
|
return out;
|
||||||
|
};
|
||||||
|
const audio = resample(spoken.audio, spoken.sampling_rate, 16000);
|
||||||
|
const heard = await transcribe(audio, 'en', 'en');
|
||||||
|
step('+ Whisper base (STT)');
|
||||||
|
|
||||||
|
// Steady state: a few turns, to see whether it keeps growing.
|
||||||
|
for (let i = 0; i < 3; i++) {
|
||||||
|
await synthesize('Checking the leads now.', 'en');
|
||||||
|
await transcribe(audio, 'en', 'en');
|
||||||
|
}
|
||||||
|
step('after 3 more turns');
|
||||||
|
|
||||||
|
console.log(`\n transcript: ${JSON.stringify(heard.text)}`);
|
||||||
|
console.log(` TTS rate : ${spoken.sampling_rate} Hz`);
|
||||||
|
|
||||||
|
const peak = mb();
|
||||||
|
const CAP = 768;
|
||||||
|
console.log(`\n peak ${peak} MB vs ${CAP} MB container cap → ${peak < CAP * 0.8 ? 'FITS ✅' : peak < CAP ? 'TIGHT ⚠️' : 'EXCEEDS ❌'}`);
|
||||||
|
process.exit(0);
|
||||||
+19
-7
@@ -7,10 +7,10 @@
|
|||||||
// one shift and should never have to touch a language menu.
|
// one shift and should never have to touch a language menu.
|
||||||
// ============================================
|
// ============================================
|
||||||
import { Tensor } from '@huggingface/transformers';
|
import { Tensor } from '@huggingface/transformers';
|
||||||
import { getSTT, getTTS, supportsTTS } from './models.js';
|
import { getSTT, getTTS, supportsTTS, sttIsEnglishOnly, defaultLanguage } from './models.js';
|
||||||
import logger from '../utils/logger.js';
|
import logger from '../utils/logger.js';
|
||||||
|
|
||||||
export { LANGUAGES, warmup, speechStatus } from './models.js';
|
export { LANGUAGES, warmup, speechStatus, defaultLanguage } from './models.js';
|
||||||
export { Endpointer, warmupVad } from './vad.js';
|
export { Endpointer, warmupVad } from './vad.js';
|
||||||
|
|
||||||
const RATE = 16000;
|
const RATE = 16000;
|
||||||
@@ -67,7 +67,7 @@ async function detectLanguage(audio) {
|
|||||||
* @param {string} lang 'auto' | 'ta' | 'en'
|
* @param {string} lang 'auto' | 'ta' | 'en'
|
||||||
* @param {string} prefer used when detection is unusable
|
* @param {string} prefer used when detection is unusable
|
||||||
*/
|
*/
|
||||||
export async function transcribe(audio, lang = 'auto', prefer = 'ta') {
|
export async function transcribe(audio, lang = 'auto', prefer = defaultLanguage()) {
|
||||||
if (!audio || audio.length < RATE / 5) { // under 200 ms
|
if (!audio || audio.length < RATE / 5) { // under 200 ms
|
||||||
return { text: '', lang: prefer, note: 'too short' };
|
return { text: '', lang: prefer, note: 'too short' };
|
||||||
}
|
}
|
||||||
@@ -80,7 +80,11 @@ export async function transcribe(audio, lang = 'auto', prefer = 'ta') {
|
|||||||
let detected = null;
|
let detected = null;
|
||||||
let confidence = null;
|
let confidence = null;
|
||||||
|
|
||||||
if (lang === 'auto') {
|
// An English-only checkpoint has no language tokens to read, and passing
|
||||||
|
// `language` to it is rejected — so "auto" simply means English there.
|
||||||
|
if (lang === 'auto' && sttIsEnglishOnly()) {
|
||||||
|
used = 'en';
|
||||||
|
} else if (lang === 'auto') {
|
||||||
try {
|
try {
|
||||||
const d = await detectLanguage(audio);
|
const d = await detectLanguage(audio);
|
||||||
detected = d.lang;
|
detected = d.lang;
|
||||||
@@ -95,7 +99,15 @@ export async function transcribe(audio, lang = 'auto', prefer = 'ta') {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
const result = await stt(audio, { task: 'transcribe', language: used, return_timestamps: false });
|
// An English-only checkpoint rejects BOTH `task` and `language` — it has no
|
||||||
|
// other mode to select. Multilingual builds require the language, since they
|
||||||
|
// silently default to English otherwise.
|
||||||
|
const opts = { return_timestamps: false };
|
||||||
|
if (!sttIsEnglishOnly()) {
|
||||||
|
opts.task = 'transcribe';
|
||||||
|
opts.language = used;
|
||||||
|
}
|
||||||
|
const result = await stt(audio, opts);
|
||||||
return finish(result, used, audio, t0, detected, confidence);
|
return finish(result, used, audio, t0, detected, confidence);
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -115,11 +127,11 @@ function finish(result, used, audio, t0, detected, confidence) {
|
|||||||
* Synthesise one piece of text.
|
* Synthesise one piece of text.
|
||||||
* @returns {Promise<{audio: Float32Array, sampling_rate: number}>}
|
* @returns {Promise<{audio: Float32Array, sampling_rate: number}>}
|
||||||
*/
|
*/
|
||||||
export async function synthesize(text, lang = 'ta') {
|
export async function synthesize(text, lang = defaultLanguage()) {
|
||||||
const clean = (text || '').trim();
|
const clean = (text || '').trim();
|
||||||
if (!clean) return null;
|
if (!clean) return null;
|
||||||
|
|
||||||
const use = supportsTTS(lang) ? lang : 'en';
|
const use = supportsTTS(lang) ? lang : defaultLanguage();
|
||||||
const tts = await getTTS(use);
|
const tts = await getTTS(use);
|
||||||
|
|
||||||
const t0 = Date.now();
|
const t0 = Date.now();
|
||||||
|
|||||||
+30
-11
@@ -35,20 +35,38 @@ env.cacheDir = process.env.SPEECH_CACHE_DIR || path.join(ROOT, '.transformers-ca
|
|||||||
// of mms-tts-tam exists. See scripts/export-tamil-tts.py.
|
// of mms-tts-tam exists. See scripts/export-tamil-tts.py.
|
||||||
env.localModelPath = path.join(ROOT, 'assets/tts');
|
env.localModelPath = path.join(ROOT, 'assets/tts');
|
||||||
|
|
||||||
const STT_MODEL = process.env.STT_MODEL || 'onnx-community/whisper-base';
|
// whisper-tiny.en by default: measured, Whisper is the memory hog, not TTS.
|
||||||
|
// base cost 554 MB of an 879 MB total, which overran the 768 MB container cap.
|
||||||
|
// The .en build is half the size and, being English-only, cannot detect a
|
||||||
|
// language — which is fine when VOICE_LANGUAGES is just `en`.
|
||||||
|
const STT_MODEL = process.env.STT_MODEL || 'onnx-community/whisper-tiny.en';
|
||||||
|
|
||||||
/** TTS voice per language. Tamil is local; English comes from the Hub. */
|
/** True when the STT checkpoint is English-only and cannot identify languages. */
|
||||||
const VOICES = {
|
export const sttIsEnglishOnly = () => /\.en$/.test(STT_MODEL);
|
||||||
ta: { id: 'mms-tts-tam', local: true },
|
|
||||||
en: { id: 'Xenova/mms-tts-eng', local: false },
|
/** TTS voice per language. Tamil is exported locally; English is on the Hub. */
|
||||||
|
const ALL_VOICES = {
|
||||||
|
en: { id: 'Xenova/mms-tts-eng', local: false, label: 'English', native: 'English' },
|
||||||
|
ta: { id: 'mms-tts-tam', local: true, label: 'Tamil', native: 'தமிழ்' },
|
||||||
};
|
};
|
||||||
|
|
||||||
|
// Each extra language is a further ~200 MB resident. Enable only what the
|
||||||
|
// deployment actually speaks — English alone on the current AWS box.
|
||||||
|
const ENABLED = (process.env.VOICE_LANGUAGES || 'en')
|
||||||
|
.split(',').map((s) => s.trim()).filter((c) => ALL_VOICES[c]);
|
||||||
|
|
||||||
|
const VOICES = Object.fromEntries(ENABLED.map((c) => [c, ALL_VOICES[c]]));
|
||||||
|
|
||||||
export const LANGUAGES = [
|
export const LANGUAGES = [
|
||||||
{ code: 'auto', label: 'Auto-detect', native: 'Auto' },
|
// Auto-detect is only offered when there is a choice to make AND the STT
|
||||||
{ code: 'ta', label: 'Tamil', native: 'தமிழ்' },
|
// model can actually detect — offering it otherwise is a lie.
|
||||||
{ code: 'en', label: 'English', native: 'English' },
|
...(ENABLED.length > 1 && !sttIsEnglishOnly()
|
||||||
|
? [{ code: 'auto', label: 'Auto-detect', native: 'Auto' }] : []),
|
||||||
|
...ENABLED.map((c) => ({ code: c, label: ALL_VOICES[c].label, native: ALL_VOICES[c].native })),
|
||||||
];
|
];
|
||||||
|
|
||||||
|
export const defaultLanguage = () => (LANGUAGES[0]?.code || 'en');
|
||||||
|
|
||||||
const cache = new Map();
|
const cache = new Map();
|
||||||
let sttPromise = null;
|
let sttPromise = null;
|
||||||
|
|
||||||
@@ -81,8 +99,8 @@ export async function getSTT() {
|
|||||||
return sttPromise;
|
return sttPromise;
|
||||||
}
|
}
|
||||||
|
|
||||||
export async function getTTS(lang = 'ta') {
|
export async function getTTS(lang) {
|
||||||
const voice = VOICES[lang] || VOICES.ta;
|
const voice = VOICES[lang] || VOICES[ENABLED[0]];
|
||||||
const prev = env.allowRemoteModels;
|
const prev = env.allowRemoteModels;
|
||||||
try {
|
try {
|
||||||
// Local folders must not be looked up on the Hub, and vice versa.
|
// Local folders must not be looked up on the Hub, and vice versa.
|
||||||
@@ -98,7 +116,7 @@ export async function getTTS(lang = 'ta') {
|
|||||||
export const supportsTTS = (lang) => Boolean(VOICES[lang]);
|
export const supportsTTS = (lang) => Boolean(VOICES[lang]);
|
||||||
|
|
||||||
/** Warm the models the deployment actually expects to use. */
|
/** Warm the models the deployment actually expects to use. */
|
||||||
export async function warmup(langs = ['ta', 'en']) {
|
export async function warmup(langs = ENABLED) {
|
||||||
try {
|
try {
|
||||||
await getSTT();
|
await getSTT();
|
||||||
for (const l of langs) await getTTS(l);
|
for (const l of langs) await getTTS(l);
|
||||||
@@ -111,6 +129,7 @@ export async function warmup(langs = ['ta', 'en']) {
|
|||||||
export function speechStatus() {
|
export function speechStatus() {
|
||||||
return {
|
return {
|
||||||
stt_model: STT_MODEL,
|
stt_model: STT_MODEL,
|
||||||
|
english_only_stt: sttIsEnglishOnly(),
|
||||||
tts_voices: Object.fromEntries(Object.entries(VOICES).map(([k, v]) => [k, v.id])),
|
tts_voices: Object.fromEntries(Object.entries(VOICES).map(([k, v]) => [k, v.id])),
|
||||||
loaded: [...cache.keys()],
|
loaded: [...cache.keys()],
|
||||||
languages: LANGUAGES,
|
languages: LANGUAGES,
|
||||||
|
|||||||
Reference in New Issue
Block a user