182 lines
6.6 KiB
JavaScript
182 lines
6.6 KiB
JavaScript
// ============================================
|
|
// Speech pipeline — transcribe() and synthesize().
|
|
//
|
|
// Whisper is multilingual and can identify the spoken language, so "auto"
|
|
// costs nothing extra: detection and transcription are the same forward pass.
|
|
// That matters for a WeLe agent who switches between Tamil and English inside
|
|
// one shift and should never have to touch a language menu.
|
|
// ============================================
|
|
import { Tensor } from '@huggingface/transformers';
|
|
import { getSTT, getTTS, supportsTTS, sttIsEnglishOnly, defaultLanguage } from './models.js';
|
|
import logger from '../utils/logger.js';
|
|
|
|
export { LANGUAGES, warmup, speechStatus, defaultLanguage } from './models.js';
|
|
export { Endpointer, warmupVad } from './vad.js';
|
|
|
|
const RATE = 16000;
|
|
|
|
/** Languages we can both hear and speak. */
|
|
const SPOKEN = new Set(['ta', 'en']);
|
|
|
|
// Below this, trust the caller's preference over the detector. Short or noisy
|
|
// utterances — and code-mixed "Tanglish" especially — can land either side.
|
|
const DETECT_CONFIDENCE = Number(process.env.DETECT_CONFIDENCE ?? 0.6);
|
|
|
|
let detectIds = null;
|
|
|
|
/**
|
|
* Identify the spoken language in ONE decoder step.
|
|
*
|
|
* Passing no `language` to the pipeline does NOT auto-detect — Transformers.js
|
|
* logs "No language specified - defaulting to English" and transcribes Tamil
|
|
* as English, producing nonsense. Whisper does emit a language token right
|
|
* after <|startoftranscript|>, so we read that distribution directly. Measured
|
|
* ~700 ms, and 0.998 / 1.000 confidence on clean Tamil / English.
|
|
*
|
|
* Reuses the pipeline's own model and processor, so nothing loads twice.
|
|
*/
|
|
async function detectLanguage(audio) {
|
|
const stt = await getSTT();
|
|
const tok = stt.tokenizer;
|
|
|
|
if (!detectIds) {
|
|
const id = (t) => tok.encode(t, { add_special_tokens: false })[0];
|
|
detectIds = { sot: id('<|startoftranscript|>'), langs: [...SPOKEN].map((c) => ({ code: c, id: id(`<|${c}|>`) })) };
|
|
}
|
|
|
|
const inputs = await stt.processor(audio);
|
|
const out = await stt.model({
|
|
...inputs,
|
|
decoder_input_ids: new Tensor('int64', BigInt64Array.from([BigInt(detectIds.sot)]), [1, 1]),
|
|
});
|
|
|
|
const { dims, data } = out.logits;
|
|
const row = data.slice((dims[1] - 1) * dims[2], dims[1] * dims[2]);
|
|
const scores = detectIds.langs.map((l) => Number(row[l.id]));
|
|
const max = Math.max(...scores);
|
|
const exp = scores.map((v) => Math.exp(v - max));
|
|
const sum = exp.reduce((a, b) => a + b, 0);
|
|
const probs = exp.map((v) => v / sum);
|
|
const best = probs.indexOf(Math.max(...probs));
|
|
|
|
return { lang: detectIds.langs[best].code, confidence: probs[best] };
|
|
}
|
|
|
|
/**
|
|
* @param {Float32Array} audio mono @16 kHz in [-1, 1]
|
|
* @param {string} lang 'auto' | 'ta' | 'en'
|
|
* @param {string} prefer used when detection is unusable
|
|
*/
|
|
export async function transcribe(audio, lang = 'auto', prefer = defaultLanguage()) {
|
|
if (!audio || audio.length < RATE / 5) { // under 200 ms
|
|
return { text: '', lang: prefer, note: 'too short' };
|
|
}
|
|
|
|
const stt = await getSTT();
|
|
const t0 = Date.now();
|
|
|
|
// Whisper must always be told a language — it never detects on its own here.
|
|
let used = lang;
|
|
let detected = null;
|
|
let confidence = null;
|
|
|
|
// An English-only checkpoint has no language tokens to read, and passing
|
|
// `language` to it is rejected — so "auto" simply means English there.
|
|
if (lang === 'auto' && sttIsEnglishOnly()) {
|
|
used = 'en';
|
|
} else if (lang === 'auto') {
|
|
try {
|
|
const d = await detectLanguage(audio);
|
|
detected = d.lang;
|
|
confidence = d.confidence;
|
|
used = d.confidence >= DETECT_CONFIDENCE ? d.lang : prefer;
|
|
if (used !== d.lang) {
|
|
logger.info(`language ID unsure (${d.lang} @ ${d.confidence.toFixed(2)}) — using preferred ${prefer}`);
|
|
}
|
|
} catch (e) {
|
|
logger.warn(`language ID failed (${e.message}) — using preferred ${prefer}`);
|
|
used = prefer;
|
|
}
|
|
}
|
|
|
|
// An English-only checkpoint rejects BOTH `task` and `language` — it has no
|
|
// other mode to select. Multilingual builds require the language, since they
|
|
// silently default to English otherwise.
|
|
const opts = { return_timestamps: false };
|
|
if (!sttIsEnglishOnly()) {
|
|
opts.task = 'transcribe';
|
|
opts.language = used;
|
|
}
|
|
const result = await stt(audio, opts);
|
|
return finish(result, used, audio, t0, detected, confidence);
|
|
}
|
|
|
|
function finish(result, used, audio, t0, detected, confidence) {
|
|
const text = (result?.text || '').trim();
|
|
const ms = Date.now() - t0;
|
|
const audioMs = Math.round((audio.length / RATE) * 1000);
|
|
logger.info(`🎤 STT ${used}${detected && detected !== used ? ` (heard ${detected})` : ''}: ${audioMs}ms → ${ms}ms → ${JSON.stringify(text.slice(0, 70))}`);
|
|
return {
|
|
text, lang: used, detected: detected || null,
|
|
confidence: confidence == null ? null : Number(confidence.toFixed(3)),
|
|
ms, audio_ms: audioMs,
|
|
};
|
|
}
|
|
|
|
/**
|
|
* Synthesise one piece of text.
|
|
* @returns {Promise<{audio: Float32Array, sampling_rate: number}>}
|
|
*/
|
|
export async function synthesize(text, lang = defaultLanguage()) {
|
|
const clean = (text || '').trim();
|
|
if (!clean) return null;
|
|
|
|
const use = supportsTTS(lang) ? lang : defaultLanguage();
|
|
const tts = await getTTS(use);
|
|
|
|
const t0 = Date.now();
|
|
const out = await tts(clean);
|
|
const ms = Date.now() - t0;
|
|
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
|
|
logger.debug(`🔈 TTS ${use}: ${clean.length} chars → ${ms}ms for ${audioMs.toFixed(0)}ms (RTF ${(ms / audioMs).toFixed(2)}x)`);
|
|
|
|
return { audio: out.audio, sampling_rate: out.sampling_rate };
|
|
}
|
|
|
|
/**
|
|
* Split into speakable pieces. Short prompts reach audio sooner, and a sentence
|
|
* boundary is a clean place to be interrupted.
|
|
*/
|
|
export function sentences(text, max = 200) {
|
|
const out = [];
|
|
for (const raw of String(text || '').split(/(?<=[.!?।])\s+/)) {
|
|
let s = raw.trim();
|
|
if (!s) continue;
|
|
while (s.length > max) {
|
|
const cut = s.lastIndexOf(' ', max);
|
|
out.push(s.slice(0, cut > 0 ? cut : max).trim());
|
|
s = s.slice(cut > 0 ? cut : max).trim();
|
|
}
|
|
if (s) out.push(s);
|
|
}
|
|
return out;
|
|
}
|
|
|
|
/**
|
|
* Strip block markdown before speaking — tables and code read terribly aloud,
|
|
* and the visual blocks are already on screen.
|
|
*/
|
|
export function speakable(markdown = '') {
|
|
return markdown
|
|
.replace(/```[\s\S]*?```/g, ' ')
|
|
.replace(/^\s*\|.*\|\s*$/gm, ' ')
|
|
.replace(/^\s*[-*]\s+/gm, '')
|
|
.replace(/^#{1,6}\s*/gm, '')
|
|
.replace(/\*\*([^*]+)\*\*/g, '$1')
|
|
.replace(/`([^`]+)`/g, '$1')
|
|
.replace(/\[([^\]]+)\]\([^)]+\)/g, '$1')
|
|
.replace(/₹\s?([\d,.]+)/g, 'rupees $1')
|
|
.replace(/\s{2,}/g, ' ')
|
|
.trim();
|
|
}
|