WeLe Agentic AI with Docker deployment

This commit is contained in:
2026-08-28 09:19:05 +05:30
parent 6bb0ef25ca
commit 2fb6a0640c
6 changed files with 128 additions and 28 deletions
+19 -7
View File
@@ -7,10 +7,10 @@
// one shift and should never have to touch a language menu.
// ============================================
import { Tensor } from '@huggingface/transformers';
import { getSTT, getTTS, supportsTTS } from './models.js';
import { getSTT, getTTS, supportsTTS, sttIsEnglishOnly, defaultLanguage } from './models.js';
import logger from '../utils/logger.js';
export { LANGUAGES, warmup, speechStatus } from './models.js';
export { LANGUAGES, warmup, speechStatus, defaultLanguage } from './models.js';
export { Endpointer, warmupVad } from './vad.js';
const RATE = 16000;
@@ -67,7 +67,7 @@ async function detectLanguage(audio) {
* @param {string} lang 'auto' | 'ta' | 'en'
* @param {string} prefer used when detection is unusable
*/
export async function transcribe(audio, lang = 'auto', prefer = 'ta') {
export async function transcribe(audio, lang = 'auto', prefer = defaultLanguage()) {
if (!audio || audio.length < RATE / 5) { // under 200 ms
return { text: '', lang: prefer, note: 'too short' };
}
@@ -80,7 +80,11 @@ export async function transcribe(audio, lang = 'auto', prefer = 'ta') {
let detected = null;
let confidence = null;
if (lang === 'auto') {
// An English-only checkpoint has no language tokens to read, and passing
// `language` to it is rejected — so "auto" simply means English there.
if (lang === 'auto' && sttIsEnglishOnly()) {
used = 'en';
} else if (lang === 'auto') {
try {
const d = await detectLanguage(audio);
detected = d.lang;
@@ -95,7 +99,15 @@ export async function transcribe(audio, lang = 'auto', prefer = 'ta') {
}
}
const result = await stt(audio, { task: 'transcribe', language: used, return_timestamps: false });
// An English-only checkpoint rejects BOTH `task` and `language` — it has no
// other mode to select. Multilingual builds require the language, since they
// silently default to English otherwise.
const opts = { return_timestamps: false };
if (!sttIsEnglishOnly()) {
opts.task = 'transcribe';
opts.language = used;
}
const result = await stt(audio, opts);
return finish(result, used, audio, t0, detected, confidence);
}
@@ -115,11 +127,11 @@ function finish(result, used, audio, t0, detected, confidence) {
* Synthesise one piece of text.
* @returns {Promise<{audio: Float32Array, sampling_rate: number}>}
*/
export async function synthesize(text, lang = 'ta') {
export async function synthesize(text, lang = defaultLanguage()) {
const clean = (text || '').trim();
if (!clean) return null;
const use = supportsTTS(lang) ? lang : 'en';
const use = supportsTTS(lang) ? lang : defaultLanguage();
const tts = await getTTS(use);
const t0 = Date.now();
+30 -11
View File
@@ -35,20 +35,38 @@ env.cacheDir = process.env.SPEECH_CACHE_DIR || path.join(ROOT, '.transformers-ca
// of mms-tts-tam exists. See scripts/export-tamil-tts.py.
env.localModelPath = path.join(ROOT, 'assets/tts');
const STT_MODEL = process.env.STT_MODEL || 'onnx-community/whisper-base';
// whisper-tiny.en by default: measured, Whisper is the memory hog, not TTS.
// base cost 554 MB of an 879 MB total, which overran the 768 MB container cap.
// The .en build is half the size and, being English-only, cannot detect a
// language — which is fine when VOICE_LANGUAGES is just `en`.
const STT_MODEL = process.env.STT_MODEL || 'onnx-community/whisper-tiny.en';
/** TTS voice per language. Tamil is local; English comes from the Hub. */
const VOICES = {
ta: { id: 'mms-tts-tam', local: true },
en: { id: 'Xenova/mms-tts-eng', local: false },
/** True when the STT checkpoint is English-only and cannot identify languages. */
export const sttIsEnglishOnly = () => /\.en$/.test(STT_MODEL);
/** TTS voice per language. Tamil is exported locally; English is on the Hub. */
const ALL_VOICES = {
en: { id: 'Xenova/mms-tts-eng', local: false, label: 'English', native: 'English' },
ta: { id: 'mms-tts-tam', local: true, label: 'Tamil', native: 'தமிழ்' },
};
// Each extra language is a further ~200 MB resident. Enable only what the
// deployment actually speaks — English alone on the current AWS box.
const ENABLED = (process.env.VOICE_LANGUAGES || 'en')
.split(',').map((s) => s.trim()).filter((c) => ALL_VOICES[c]);
const VOICES = Object.fromEntries(ENABLED.map((c) => [c, ALL_VOICES[c]]));
export const LANGUAGES = [
{ code: 'auto', label: 'Auto-detect', native: 'Auto' },
{ code: 'ta', label: 'Tamil', native: 'தமிழ்' },
{ code: 'en', label: 'English', native: 'English' },
// Auto-detect is only offered when there is a choice to make AND the STT
// model can actually detect — offering it otherwise is a lie.
...(ENABLED.length > 1 && !sttIsEnglishOnly()
? [{ code: 'auto', label: 'Auto-detect', native: 'Auto' }] : []),
...ENABLED.map((c) => ({ code: c, label: ALL_VOICES[c].label, native: ALL_VOICES[c].native })),
];
export const defaultLanguage = () => (LANGUAGES[0]?.code || 'en');
const cache = new Map();
let sttPromise = null;
@@ -81,8 +99,8 @@ export async function getSTT() {
return sttPromise;
}
export async function getTTS(lang = 'ta') {
const voice = VOICES[lang] || VOICES.ta;
export async function getTTS(lang) {
const voice = VOICES[lang] || VOICES[ENABLED[0]];
const prev = env.allowRemoteModels;
try {
// Local folders must not be looked up on the Hub, and vice versa.
@@ -98,7 +116,7 @@ export async function getTTS(lang = 'ta') {
export const supportsTTS = (lang) => Boolean(VOICES[lang]);
/** Warm the models the deployment actually expects to use. */
export async function warmup(langs = ['ta', 'en']) {
export async function warmup(langs = ENABLED) {
try {
await getSTT();
for (const l of langs) await getTTS(l);
@@ -111,6 +129,7 @@ export async function warmup(langs = ['ta', 'en']) {
export function speechStatus() {
return {
stt_model: STT_MODEL,
english_only_stt: sttIsEnglishOnly(),
tts_voices: Object.fromEntries(Object.entries(VOICES).map(([k, v]) => [k, v.id])),
loaded: [...cache.keys()],
languages: LANGUAGES,