73 lines
3.5 KiB
JavaScript
73 lines
3.5 KiB
JavaScript
/* End-to-end check of the in-process speech pipeline: TTS → VAD → STT.
|
|
|
|
Synthesising a sentence and feeding that audio back through the endpointer
|
|
and recogniser exercises every stage with real speech, which a noise buffer
|
|
cannot do — silence never opens a VAD turn.
|
|
*/
|
|
import { synthesize, transcribe, sentences, speakable, Endpointer } from '../src/speech/index.js';
|
|
|
|
const say = (m) => console.log(m);
|
|
|
|
// ── 1. TTS both languages ───────────────────────────────────────────────────
|
|
const CASES = [
|
|
['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'],
|
|
['en', 'There are three thousand four hundred and twenty seven new leads.'],
|
|
];
|
|
|
|
const rendered = {};
|
|
for (const [lang, text] of CASES) {
|
|
const t0 = Date.now();
|
|
await synthesize(text, lang); // warm
|
|
const t1 = Date.now();
|
|
const out = await synthesize(text, lang);
|
|
const ms = Date.now() - t1;
|
|
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
|
|
rendered[lang] = out;
|
|
say(`TTS ${lang} warm ${((t1 - t0) / 1000).toFixed(1)}s | ${ms}ms for ${audioMs.toFixed(0)}ms `
|
|
+ `@${out.sampling_rate}Hz | RTF ${(ms / audioMs).toFixed(2)}x`);
|
|
}
|
|
|
|
// ── 2. VAD: does synthesised speech open and close a turn? ──────────────────
|
|
function resample(audio, from, to) {
|
|
if (from === to) return audio;
|
|
const ratio = from / to;
|
|
const out = new Float32Array(Math.floor(audio.length / ratio));
|
|
for (let i = 0; i < out.length; i++) {
|
|
const p = i * ratio;
|
|
const a = Math.floor(p);
|
|
out[i] = audio[a] + (audio[Math.min(a + 1, audio.length - 1)] - audio[a]) * (p - a);
|
|
}
|
|
return out;
|
|
}
|
|
|
|
const ep = new Endpointer();
|
|
const speech = resample(rendered.en.audio, rendered.en.sampling_rate, 16000);
|
|
// Speech, then a second of silence so the endpointer closes the turn.
|
|
const withTail = new Float32Array(speech.length + 16000);
|
|
withTail.set(speech);
|
|
|
|
let started = false;
|
|
let captured = null;
|
|
for (let i = 0; i < withTail.length; i += 640) { // 40 ms chunks, as the browser sends
|
|
const { utterances, started: s } = await ep.push(withTail.subarray(i, Math.min(i + 640, withTail.length)));
|
|
if (s) started = true;
|
|
if (utterances.length) { captured = utterances[0]; break; }
|
|
}
|
|
say(`VAD speech detected: ${started ? 'yes' : 'NO'} | turn closed: ${captured ? 'yes' : 'NO'}`
|
|
+ (captured ? ` | captured ${(captured.length / 16000).toFixed(2)}s` : ''));
|
|
|
|
// ── 3. STT on that captured audio ───────────────────────────────────────────
|
|
if (captured) {
|
|
for (const lang of ['en', 'auto']) {
|
|
const r = await transcribe(captured, lang, 'en');
|
|
say(`STT ${lang.padEnd(4)} ${r.ms}ms → lang=${r.lang}${r.detected ? ` (heard ${r.detected})` : ''} → ${JSON.stringify(r.text.slice(0, 70))}`);
|
|
}
|
|
}
|
|
|
|
// ── 4. Text shaping ─────────────────────────────────────────────────────────
|
|
const md = '## Leads\n\n**3,427** in `new_lead`.\n\n| a | b |\n|---|---|\n| 1 | 2 |\n\n- Only 12% contacted.\nNext step is triage.';
|
|
say(`\nspeakable: ${JSON.stringify(speakable(md))}`);
|
|
say(`sentences: ${JSON.stringify(sentences(speakable(md)))}`);
|
|
say(`\nRSS ${(process.memoryUsage().rss / 1e9).toFixed(2)} GB`);
|
|
process.exit(0);
|