WeLe Agentic AI with Docker deployment
This commit is contained in:
+30
-11
@@ -35,20 +35,38 @@ env.cacheDir = process.env.SPEECH_CACHE_DIR || path.join(ROOT, '.transformers-ca
|
||||
// of mms-tts-tam exists. See scripts/export-tamil-tts.py.
|
||||
env.localModelPath = path.join(ROOT, 'assets/tts');
|
||||
|
||||
const STT_MODEL = process.env.STT_MODEL || 'onnx-community/whisper-base';
|
||||
// whisper-tiny.en by default: measured, Whisper is the memory hog, not TTS.
|
||||
// base cost 554 MB of an 879 MB total, which overran the 768 MB container cap.
|
||||
// The .en build is half the size and, being English-only, cannot detect a
|
||||
// language — which is fine when VOICE_LANGUAGES is just `en`.
|
||||
const STT_MODEL = process.env.STT_MODEL || 'onnx-community/whisper-tiny.en';
|
||||
|
||||
/** TTS voice per language. Tamil is local; English comes from the Hub. */
|
||||
const VOICES = {
|
||||
ta: { id: 'mms-tts-tam', local: true },
|
||||
en: { id: 'Xenova/mms-tts-eng', local: false },
|
||||
/** True when the STT checkpoint is English-only and cannot identify languages. */
|
||||
export const sttIsEnglishOnly = () => /\.en$/.test(STT_MODEL);
|
||||
|
||||
/** TTS voice per language. Tamil is exported locally; English is on the Hub. */
|
||||
const ALL_VOICES = {
|
||||
en: { id: 'Xenova/mms-tts-eng', local: false, label: 'English', native: 'English' },
|
||||
ta: { id: 'mms-tts-tam', local: true, label: 'Tamil', native: 'தமிழ்' },
|
||||
};
|
||||
|
||||
// Each extra language is a further ~200 MB resident. Enable only what the
|
||||
// deployment actually speaks — English alone on the current AWS box.
|
||||
const ENABLED = (process.env.VOICE_LANGUAGES || 'en')
|
||||
.split(',').map((s) => s.trim()).filter((c) => ALL_VOICES[c]);
|
||||
|
||||
const VOICES = Object.fromEntries(ENABLED.map((c) => [c, ALL_VOICES[c]]));
|
||||
|
||||
export const LANGUAGES = [
|
||||
{ code: 'auto', label: 'Auto-detect', native: 'Auto' },
|
||||
{ code: 'ta', label: 'Tamil', native: 'தமிழ்' },
|
||||
{ code: 'en', label: 'English', native: 'English' },
|
||||
// Auto-detect is only offered when there is a choice to make AND the STT
|
||||
// model can actually detect — offering it otherwise is a lie.
|
||||
...(ENABLED.length > 1 && !sttIsEnglishOnly()
|
||||
? [{ code: 'auto', label: 'Auto-detect', native: 'Auto' }] : []),
|
||||
...ENABLED.map((c) => ({ code: c, label: ALL_VOICES[c].label, native: ALL_VOICES[c].native })),
|
||||
];
|
||||
|
||||
export const defaultLanguage = () => (LANGUAGES[0]?.code || 'en');
|
||||
|
||||
const cache = new Map();
|
||||
let sttPromise = null;
|
||||
|
||||
@@ -81,8 +99,8 @@ export async function getSTT() {
|
||||
return sttPromise;
|
||||
}
|
||||
|
||||
export async function getTTS(lang = 'ta') {
|
||||
const voice = VOICES[lang] || VOICES.ta;
|
||||
export async function getTTS(lang) {
|
||||
const voice = VOICES[lang] || VOICES[ENABLED[0]];
|
||||
const prev = env.allowRemoteModels;
|
||||
try {
|
||||
// Local folders must not be looked up on the Hub, and vice versa.
|
||||
@@ -98,7 +116,7 @@ export async function getTTS(lang = 'ta') {
|
||||
export const supportsTTS = (lang) => Boolean(VOICES[lang]);
|
||||
|
||||
/** Warm the models the deployment actually expects to use. */
|
||||
export async function warmup(langs = ['ta', 'en']) {
|
||||
export async function warmup(langs = ENABLED) {
|
||||
try {
|
||||
await getSTT();
|
||||
for (const l of langs) await getTTS(l);
|
||||
@@ -111,6 +129,7 @@ export async function warmup(langs = ['ta', 'en']) {
|
||||
export function speechStatus() {
|
||||
return {
|
||||
stt_model: STT_MODEL,
|
||||
english_only_stt: sttIsEnglishOnly(),
|
||||
tts_voices: Object.fromEntries(Object.entries(VOICES).map(([k, v]) => [k, v.id])),
|
||||
loaded: [...cache.keys()],
|
||||
languages: LANGUAGES,
|
||||
|
||||
Reference in New Issue
Block a user