// ============================================ // Speech pipeline — transcribe() and synthesize(). // // Whisper is multilingual and can identify the spoken language, so "auto" // costs nothing extra: detection and transcription are the same forward pass. // That matters for a WeLe agent who switches between Tamil and English inside // one shift and should never have to touch a language menu. // ============================================ import { Tensor } from '@huggingface/transformers'; import { getSTT, getTTS, supportsTTS, sttIsEnglishOnly, defaultLanguage } from './models.js'; import logger from '../utils/logger.js'; export { LANGUAGES, warmup, speechStatus, defaultLanguage } from './models.js'; export { Endpointer, warmupVad } from './vad.js'; const RATE = 16000; /** Languages we can both hear and speak. */ const SPOKEN = new Set(['ta', 'en']); // Below this, trust the caller's preference over the detector. Short or noisy // utterances — and code-mixed "Tanglish" especially — can land either side. const DETECT_CONFIDENCE = Number(process.env.DETECT_CONFIDENCE ?? 0.6); let detectIds = null; /** * Identify the spoken language in ONE decoder step. * * Passing no `language` to the pipeline does NOT auto-detect — Transformers.js * logs "No language specified - defaulting to English" and transcribes Tamil * as English, producing nonsense. Whisper does emit a language token right * after <|startoftranscript|>, so we read that distribution directly. Measured * ~700 ms, and 0.998 / 1.000 confidence on clean Tamil / English. * * Reuses the pipeline's own model and processor, so nothing loads twice. */ async function detectLanguage(audio) { const stt = await getSTT(); const tok = stt.tokenizer; if (!detectIds) { const id = (t) => tok.encode(t, { add_special_tokens: false })[0]; detectIds = { sot: id('<|startoftranscript|>'), langs: [...SPOKEN].map((c) => ({ code: c, id: id(`<|${c}|>`) })) }; } const inputs = await stt.processor(audio); const out = await stt.model({ ...inputs, decoder_input_ids: new Tensor('int64', BigInt64Array.from([BigInt(detectIds.sot)]), [1, 1]), }); const { dims, data } = out.logits; const row = data.slice((dims[1] - 1) * dims[2], dims[1] * dims[2]); const scores = detectIds.langs.map((l) => Number(row[l.id])); const max = Math.max(...scores); const exp = scores.map((v) => Math.exp(v - max)); const sum = exp.reduce((a, b) => a + b, 0); const probs = exp.map((v) => v / sum); const best = probs.indexOf(Math.max(...probs)); return { lang: detectIds.langs[best].code, confidence: probs[best] }; } /** * @param {Float32Array} audio mono @16 kHz in [-1, 1] * @param {string} lang 'auto' | 'ta' | 'en' * @param {string} prefer used when detection is unusable */ export async function transcribe(audio, lang = 'auto', prefer = defaultLanguage()) { if (!audio || audio.length < RATE / 5) { // under 200 ms return { text: '', lang: prefer, note: 'too short' }; } const stt = await getSTT(); const t0 = Date.now(); // Whisper must always be told a language — it never detects on its own here. let used = lang; let detected = null; let confidence = null; // An English-only checkpoint has no language tokens to read, and passing // `language` to it is rejected — so "auto" simply means English there. if (lang === 'auto' && sttIsEnglishOnly()) { used = 'en'; } else if (lang === 'auto') { try { const d = await detectLanguage(audio); detected = d.lang; confidence = d.confidence; used = d.confidence >= DETECT_CONFIDENCE ? d.lang : prefer; if (used !== d.lang) { logger.info(`language ID unsure (${d.lang} @ ${d.confidence.toFixed(2)}) — using preferred ${prefer}`); } } catch (e) { logger.warn(`language ID failed (${e.message}) — using preferred ${prefer}`); used = prefer; } } // An English-only checkpoint rejects BOTH `task` and `language` — it has no // other mode to select. Multilingual builds require the language, since they // silently default to English otherwise. const opts = { return_timestamps: false }; if (!sttIsEnglishOnly()) { opts.task = 'transcribe'; opts.language = used; } const result = await stt(audio, opts); return finish(result, used, audio, t0, detected, confidence); } function finish(result, used, audio, t0, detected, confidence) { const text = (result?.text || '').trim(); const ms = Date.now() - t0; const audioMs = Math.round((audio.length / RATE) * 1000); logger.info(`🎤 STT ${used}${detected && detected !== used ? ` (heard ${detected})` : ''}: ${audioMs}ms → ${ms}ms → ${JSON.stringify(text.slice(0, 70))}`); return { text, lang: used, detected: detected || null, confidence: confidence == null ? null : Number(confidence.toFixed(3)), ms, audio_ms: audioMs, }; } /** * Synthesise one piece of text. * @returns {Promise<{audio: Float32Array, sampling_rate: number}>} */ export async function synthesize(text, lang = defaultLanguage()) { const clean = (text || '').trim(); if (!clean) return null; const use = supportsTTS(lang) ? lang : defaultLanguage(); const tts = await getTTS(use); const t0 = Date.now(); const out = await tts(clean); const ms = Date.now() - t0; const audioMs = (out.audio.length / out.sampling_rate) * 1000; logger.debug(`🔈 TTS ${use}: ${clean.length} chars → ${ms}ms for ${audioMs.toFixed(0)}ms (RTF ${(ms / audioMs).toFixed(2)}x)`); return { audio: out.audio, sampling_rate: out.sampling_rate }; } /** * Split into speakable pieces. Short prompts reach audio sooner, and a sentence * boundary is a clean place to be interrupted. */ export function sentences(text, max = 200) { const out = []; for (const raw of String(text || '').split(/(?<=[.!?।])\s+/)) { let s = raw.trim(); if (!s) continue; while (s.length > max) { const cut = s.lastIndexOf(' ', max); out.push(s.slice(0, cut > 0 ? cut : max).trim()); s = s.slice(cut > 0 ? cut : max).trim(); } if (s) out.push(s); } return out; } /** * Strip block markdown before speaking — tables and code read terribly aloud, * and the visual blocks are already on screen. */ export function speakable(markdown = '') { return markdown .replace(/```[\s\S]*?```/g, ' ') .replace(/^\s*\|.*\|\s*$/gm, ' ') .replace(/^\s*[-*]\s+/gm, '') .replace(/^#{1,6}\s*/gm, '') .replace(/\*\*([^*]+)\*\*/g, '$1') .replace(/`([^`]+)`/g, '$1') .replace(/\[([^\]]+)\]\([^)]+\)/g, '$1') .replace(/₹\s?([\d,.]+)/g, 'rupees $1') .replace(/\s{2,}/g, ' ') .trim(); }