/* End-to-end check of the in-process speech pipeline: TTS → VAD → STT. Synthesising a sentence and feeding that audio back through the endpointer and recogniser exercises every stage with real speech, which a noise buffer cannot do — silence never opens a VAD turn. */ import { synthesize, transcribe, sentences, speakable, Endpointer } from '../src/speech/index.js'; const say = (m) => console.log(m); // ── 1. TTS both languages ─────────────────────────────────────────────────── const CASES = [ ['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'], ['en', 'There are three thousand four hundred and twenty seven new leads.'], ]; const rendered = {}; for (const [lang, text] of CASES) { const t0 = Date.now(); await synthesize(text, lang); // warm const t1 = Date.now(); const out = await synthesize(text, lang); const ms = Date.now() - t1; const audioMs = (out.audio.length / out.sampling_rate) * 1000; rendered[lang] = out; say(`TTS ${lang} warm ${((t1 - t0) / 1000).toFixed(1)}s | ${ms}ms for ${audioMs.toFixed(0)}ms ` + `@${out.sampling_rate}Hz | RTF ${(ms / audioMs).toFixed(2)}x`); } // ── 2. VAD: does synthesised speech open and close a turn? ────────────────── function resample(audio, from, to) { if (from === to) return audio; const ratio = from / to; const out = new Float32Array(Math.floor(audio.length / ratio)); for (let i = 0; i < out.length; i++) { const p = i * ratio; const a = Math.floor(p); out[i] = audio[a] + (audio[Math.min(a + 1, audio.length - 1)] - audio[a]) * (p - a); } return out; } const ep = new Endpointer(); const speech = resample(rendered.en.audio, rendered.en.sampling_rate, 16000); // Speech, then a second of silence so the endpointer closes the turn. const withTail = new Float32Array(speech.length + 16000); withTail.set(speech); let started = false; let captured = null; for (let i = 0; i < withTail.length; i += 640) { // 40 ms chunks, as the browser sends const { utterances, started: s } = await ep.push(withTail.subarray(i, Math.min(i + 640, withTail.length))); if (s) started = true; if (utterances.length) { captured = utterances[0]; break; } } say(`VAD speech detected: ${started ? 'yes' : 'NO'} | turn closed: ${captured ? 'yes' : 'NO'}` + (captured ? ` | captured ${(captured.length / 16000).toFixed(2)}s` : '')); // ── 3. STT on that captured audio ─────────────────────────────────────────── if (captured) { for (const lang of ['en', 'auto']) { const r = await transcribe(captured, lang, 'en'); say(`STT ${lang.padEnd(4)} ${r.ms}ms → lang=${r.lang}${r.detected ? ` (heard ${r.detected})` : ''} → ${JSON.stringify(r.text.slice(0, 70))}`); } } // ── 4. Text shaping ───────────────────────────────────────────────────────── const md = '## Leads\n\n**3,427** in `new_lead`.\n\n| a | b |\n|---|---|\n| 1 | 2 |\n\n- Only 12% contacted.\nNext step is triage.'; say(`\nspeakable: ${JSON.stringify(speakable(md))}`); say(`sentences: ${JSON.stringify(sentences(speakable(md)))}`); say(`\nRSS ${(process.memoryUsage().rss / 1e9).toFixed(2)} GB`); process.exit(0);