WeLe Agentic AI with Docker deployment

This commit is contained in:
2026-08-28 08:50:08 +05:30
parent 586059b14f
commit 6bb0ef25ca
34 changed files with 2271 additions and 1343 deletions
+169
View File
@@ -0,0 +1,169 @@
// ============================================
// Speech pipeline — transcribe() and synthesize().
//
// Whisper is multilingual and can identify the spoken language, so "auto"
// costs nothing extra: detection and transcription are the same forward pass.
// That matters for a WeLe agent who switches between Tamil and English inside
// one shift and should never have to touch a language menu.
// ============================================
import { Tensor } from '@huggingface/transformers';
import { getSTT, getTTS, supportsTTS } from './models.js';
import logger from '../utils/logger.js';
export { LANGUAGES, warmup, speechStatus } from './models.js';
export { Endpointer, warmupVad } from './vad.js';
const RATE = 16000;
/** Languages we can both hear and speak. */
const SPOKEN = new Set(['ta', 'en']);
// Below this, trust the caller's preference over the detector. Short or noisy
// utterances — and code-mixed "Tanglish" especially — can land either side.
const DETECT_CONFIDENCE = Number(process.env.DETECT_CONFIDENCE ?? 0.6);
let detectIds = null;
/**
* Identify the spoken language in ONE decoder step.
*
* Passing no `language` to the pipeline does NOT auto-detect — Transformers.js
* logs "No language specified - defaulting to English" and transcribes Tamil
* as English, producing nonsense. Whisper does emit a language token right
* after <|startoftranscript|>, so we read that distribution directly. Measured
* ~700 ms, and 0.998 / 1.000 confidence on clean Tamil / English.
*
* Reuses the pipeline's own model and processor, so nothing loads twice.
*/
async function detectLanguage(audio) {
const stt = await getSTT();
const tok = stt.tokenizer;
if (!detectIds) {
const id = (t) => tok.encode(t, { add_special_tokens: false })[0];
detectIds = { sot: id('<|startoftranscript|>'), langs: [...SPOKEN].map((c) => ({ code: c, id: id(`<|${c}|>`) })) };
}
const inputs = await stt.processor(audio);
const out = await stt.model({
...inputs,
decoder_input_ids: new Tensor('int64', BigInt64Array.from([BigInt(detectIds.sot)]), [1, 1]),
});
const { dims, data } = out.logits;
const row = data.slice((dims[1] - 1) * dims[2], dims[1] * dims[2]);
const scores = detectIds.langs.map((l) => Number(row[l.id]));
const max = Math.max(...scores);
const exp = scores.map((v) => Math.exp(v - max));
const sum = exp.reduce((a, b) => a + b, 0);
const probs = exp.map((v) => v / sum);
const best = probs.indexOf(Math.max(...probs));
return { lang: detectIds.langs[best].code, confidence: probs[best] };
}
/**
* @param {Float32Array} audio mono @16 kHz in [-1, 1]
* @param {string} lang 'auto' | 'ta' | 'en'
* @param {string} prefer used when detection is unusable
*/
export async function transcribe(audio, lang = 'auto', prefer = 'ta') {
if (!audio || audio.length < RATE / 5) { // under 200 ms
return { text: '', lang: prefer, note: 'too short' };
}
const stt = await getSTT();
const t0 = Date.now();
// Whisper must always be told a language — it never detects on its own here.
let used = lang;
let detected = null;
let confidence = null;
if (lang === 'auto') {
try {
const d = await detectLanguage(audio);
detected = d.lang;
confidence = d.confidence;
used = d.confidence >= DETECT_CONFIDENCE ? d.lang : prefer;
if (used !== d.lang) {
logger.info(`language ID unsure (${d.lang} @ ${d.confidence.toFixed(2)}) — using preferred ${prefer}`);
}
} catch (e) {
logger.warn(`language ID failed (${e.message}) — using preferred ${prefer}`);
used = prefer;
}
}
const result = await stt(audio, { task: 'transcribe', language: used, return_timestamps: false });
return finish(result, used, audio, t0, detected, confidence);
}
function finish(result, used, audio, t0, detected, confidence) {
const text = (result?.text || '').trim();
const ms = Date.now() - t0;
const audioMs = Math.round((audio.length / RATE) * 1000);
logger.info(`🎤 STT ${used}${detected && detected !== used ? ` (heard ${detected})` : ''}: ${audioMs}ms → ${ms}ms → ${JSON.stringify(text.slice(0, 70))}`);
return {
text, lang: used, detected: detected || null,
confidence: confidence == null ? null : Number(confidence.toFixed(3)),
ms, audio_ms: audioMs,
};
}
/**
* Synthesise one piece of text.
* @returns {Promise<{audio: Float32Array, sampling_rate: number}>}
*/
export async function synthesize(text, lang = 'ta') {
const clean = (text || '').trim();
if (!clean) return null;
const use = supportsTTS(lang) ? lang : 'en';
const tts = await getTTS(use);
const t0 = Date.now();
const out = await tts(clean);
const ms = Date.now() - t0;
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
logger.debug(`🔈 TTS ${use}: ${clean.length} chars → ${ms}ms for ${audioMs.toFixed(0)}ms (RTF ${(ms / audioMs).toFixed(2)}x)`);
return { audio: out.audio, sampling_rate: out.sampling_rate };
}
/**
* Split into speakable pieces. Short prompts reach audio sooner, and a sentence
* boundary is a clean place to be interrupted.
*/
export function sentences(text, max = 200) {
const out = [];
for (const raw of String(text || '').split(/(?<=[.!?।])\s+/)) {
let s = raw.trim();
if (!s) continue;
while (s.length > max) {
const cut = s.lastIndexOf(' ', max);
out.push(s.slice(0, cut > 0 ? cut : max).trim());
s = s.slice(cut > 0 ? cut : max).trim();
}
if (s) out.push(s);
}
return out;
}
/**
* Strip block markdown before speaking — tables and code read terribly aloud,
* and the visual blocks are already on screen.
*/
export function speakable(markdown = '') {
return markdown
.replace(/```[\s\S]*?```/g, ' ')
.replace(/^\s*\|.*\|\s*$/gm, ' ')
.replace(/^\s*[-*]\s+/gm, '')
.replace(/^#{1,6}\s*/gm, '')
.replace(/\*\*([^*]+)\*\*/g, '$1')
.replace(/`([^`]+)`/g, '$1')
.replace(/\[([^\]]+)\]\([^)]+\)/g, '$1')
.replace(/₹\s?([\d,.]+)/g, 'rupees $1')
.replace(/\s{2,}/g, ' ')
.trim();
}