WeLe Agentic AI with Docker deployment
This commit is contained in:
@@ -0,0 +1,169 @@
|
||||
// ============================================
|
||||
// Speech pipeline — transcribe() and synthesize().
|
||||
//
|
||||
// Whisper is multilingual and can identify the spoken language, so "auto"
|
||||
// costs nothing extra: detection and transcription are the same forward pass.
|
||||
// That matters for a WeLe agent who switches between Tamil and English inside
|
||||
// one shift and should never have to touch a language menu.
|
||||
// ============================================
|
||||
import { Tensor } from '@huggingface/transformers';
|
||||
import { getSTT, getTTS, supportsTTS } from './models.js';
|
||||
import logger from '../utils/logger.js';
|
||||
|
||||
export { LANGUAGES, warmup, speechStatus } from './models.js';
|
||||
export { Endpointer, warmupVad } from './vad.js';
|
||||
|
||||
const RATE = 16000;
|
||||
|
||||
/** Languages we can both hear and speak. */
|
||||
const SPOKEN = new Set(['ta', 'en']);
|
||||
|
||||
// Below this, trust the caller's preference over the detector. Short or noisy
|
||||
// utterances — and code-mixed "Tanglish" especially — can land either side.
|
||||
const DETECT_CONFIDENCE = Number(process.env.DETECT_CONFIDENCE ?? 0.6);
|
||||
|
||||
let detectIds = null;
|
||||
|
||||
/**
|
||||
* Identify the spoken language in ONE decoder step.
|
||||
*
|
||||
* Passing no `language` to the pipeline does NOT auto-detect — Transformers.js
|
||||
* logs "No language specified - defaulting to English" and transcribes Tamil
|
||||
* as English, producing nonsense. Whisper does emit a language token right
|
||||
* after <|startoftranscript|>, so we read that distribution directly. Measured
|
||||
* ~700 ms, and 0.998 / 1.000 confidence on clean Tamil / English.
|
||||
*
|
||||
* Reuses the pipeline's own model and processor, so nothing loads twice.
|
||||
*/
|
||||
async function detectLanguage(audio) {
|
||||
const stt = await getSTT();
|
||||
const tok = stt.tokenizer;
|
||||
|
||||
if (!detectIds) {
|
||||
const id = (t) => tok.encode(t, { add_special_tokens: false })[0];
|
||||
detectIds = { sot: id('<|startoftranscript|>'), langs: [...SPOKEN].map((c) => ({ code: c, id: id(`<|${c}|>`) })) };
|
||||
}
|
||||
|
||||
const inputs = await stt.processor(audio);
|
||||
const out = await stt.model({
|
||||
...inputs,
|
||||
decoder_input_ids: new Tensor('int64', BigInt64Array.from([BigInt(detectIds.sot)]), [1, 1]),
|
||||
});
|
||||
|
||||
const { dims, data } = out.logits;
|
||||
const row = data.slice((dims[1] - 1) * dims[2], dims[1] * dims[2]);
|
||||
const scores = detectIds.langs.map((l) => Number(row[l.id]));
|
||||
const max = Math.max(...scores);
|
||||
const exp = scores.map((v) => Math.exp(v - max));
|
||||
const sum = exp.reduce((a, b) => a + b, 0);
|
||||
const probs = exp.map((v) => v / sum);
|
||||
const best = probs.indexOf(Math.max(...probs));
|
||||
|
||||
return { lang: detectIds.langs[best].code, confidence: probs[best] };
|
||||
}
|
||||
|
||||
/**
|
||||
* @param {Float32Array} audio mono @16 kHz in [-1, 1]
|
||||
* @param {string} lang 'auto' | 'ta' | 'en'
|
||||
* @param {string} prefer used when detection is unusable
|
||||
*/
|
||||
export async function transcribe(audio, lang = 'auto', prefer = 'ta') {
|
||||
if (!audio || audio.length < RATE / 5) { // under 200 ms
|
||||
return { text: '', lang: prefer, note: 'too short' };
|
||||
}
|
||||
|
||||
const stt = await getSTT();
|
||||
const t0 = Date.now();
|
||||
|
||||
// Whisper must always be told a language — it never detects on its own here.
|
||||
let used = lang;
|
||||
let detected = null;
|
||||
let confidence = null;
|
||||
|
||||
if (lang === 'auto') {
|
||||
try {
|
||||
const d = await detectLanguage(audio);
|
||||
detected = d.lang;
|
||||
confidence = d.confidence;
|
||||
used = d.confidence >= DETECT_CONFIDENCE ? d.lang : prefer;
|
||||
if (used !== d.lang) {
|
||||
logger.info(`language ID unsure (${d.lang} @ ${d.confidence.toFixed(2)}) — using preferred ${prefer}`);
|
||||
}
|
||||
} catch (e) {
|
||||
logger.warn(`language ID failed (${e.message}) — using preferred ${prefer}`);
|
||||
used = prefer;
|
||||
}
|
||||
}
|
||||
|
||||
const result = await stt(audio, { task: 'transcribe', language: used, return_timestamps: false });
|
||||
return finish(result, used, audio, t0, detected, confidence);
|
||||
}
|
||||
|
||||
function finish(result, used, audio, t0, detected, confidence) {
|
||||
const text = (result?.text || '').trim();
|
||||
const ms = Date.now() - t0;
|
||||
const audioMs = Math.round((audio.length / RATE) * 1000);
|
||||
logger.info(`🎤 STT ${used}${detected && detected !== used ? ` (heard ${detected})` : ''}: ${audioMs}ms → ${ms}ms → ${JSON.stringify(text.slice(0, 70))}`);
|
||||
return {
|
||||
text, lang: used, detected: detected || null,
|
||||
confidence: confidence == null ? null : Number(confidence.toFixed(3)),
|
||||
ms, audio_ms: audioMs,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Synthesise one piece of text.
|
||||
* @returns {Promise<{audio: Float32Array, sampling_rate: number}>}
|
||||
*/
|
||||
export async function synthesize(text, lang = 'ta') {
|
||||
const clean = (text || '').trim();
|
||||
if (!clean) return null;
|
||||
|
||||
const use = supportsTTS(lang) ? lang : 'en';
|
||||
const tts = await getTTS(use);
|
||||
|
||||
const t0 = Date.now();
|
||||
const out = await tts(clean);
|
||||
const ms = Date.now() - t0;
|
||||
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
|
||||
logger.debug(`🔈 TTS ${use}: ${clean.length} chars → ${ms}ms for ${audioMs.toFixed(0)}ms (RTF ${(ms / audioMs).toFixed(2)}x)`);
|
||||
|
||||
return { audio: out.audio, sampling_rate: out.sampling_rate };
|
||||
}
|
||||
|
||||
/**
|
||||
* Split into speakable pieces. Short prompts reach audio sooner, and a sentence
|
||||
* boundary is a clean place to be interrupted.
|
||||
*/
|
||||
export function sentences(text, max = 200) {
|
||||
const out = [];
|
||||
for (const raw of String(text || '').split(/(?<=[.!?।])\s+/)) {
|
||||
let s = raw.trim();
|
||||
if (!s) continue;
|
||||
while (s.length > max) {
|
||||
const cut = s.lastIndexOf(' ', max);
|
||||
out.push(s.slice(0, cut > 0 ? cut : max).trim());
|
||||
s = s.slice(cut > 0 ? cut : max).trim();
|
||||
}
|
||||
if (s) out.push(s);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Strip block markdown before speaking — tables and code read terribly aloud,
|
||||
* and the visual blocks are already on screen.
|
||||
*/
|
||||
export function speakable(markdown = '') {
|
||||
return markdown
|
||||
.replace(/```[\s\S]*?```/g, ' ')
|
||||
.replace(/^\s*\|.*\|\s*$/gm, ' ')
|
||||
.replace(/^\s*[-*]\s+/gm, '')
|
||||
.replace(/^#{1,6}\s*/gm, '')
|
||||
.replace(/\*\*([^*]+)\*\*/g, '$1')
|
||||
.replace(/`([^`]+)`/g, '$1')
|
||||
.replace(/\[([^\]]+)\]\([^)]+\)/g, '$1')
|
||||
.replace(/₹\s?([\d,.]+)/g, 'rupees $1')
|
||||
.replace(/\s{2,}/g, ' ')
|
||||
.trim();
|
||||
}
|
||||
Reference in New Issue
Block a user