WeLe Agentic AI with Docker deployment
This commit is contained in:
@@ -0,0 +1,72 @@
|
||||
/* End-to-end check of the in-process speech pipeline: TTS → VAD → STT.
|
||||
|
||||
Synthesising a sentence and feeding that audio back through the endpointer
|
||||
and recogniser exercises every stage with real speech, which a noise buffer
|
||||
cannot do — silence never opens a VAD turn.
|
||||
*/
|
||||
import { synthesize, transcribe, sentences, speakable, Endpointer } from '../src/speech/index.js';
|
||||
|
||||
const say = (m) => console.log(m);
|
||||
|
||||
// ── 1. TTS both languages ───────────────────────────────────────────────────
|
||||
const CASES = [
|
||||
['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'],
|
||||
['en', 'There are three thousand four hundred and twenty seven new leads.'],
|
||||
];
|
||||
|
||||
const rendered = {};
|
||||
for (const [lang, text] of CASES) {
|
||||
const t0 = Date.now();
|
||||
await synthesize(text, lang); // warm
|
||||
const t1 = Date.now();
|
||||
const out = await synthesize(text, lang);
|
||||
const ms = Date.now() - t1;
|
||||
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
|
||||
rendered[lang] = out;
|
||||
say(`TTS ${lang} warm ${((t1 - t0) / 1000).toFixed(1)}s | ${ms}ms for ${audioMs.toFixed(0)}ms `
|
||||
+ `@${out.sampling_rate}Hz | RTF ${(ms / audioMs).toFixed(2)}x`);
|
||||
}
|
||||
|
||||
// ── 2. VAD: does synthesised speech open and close a turn? ──────────────────
|
||||
function resample(audio, from, to) {
|
||||
if (from === to) return audio;
|
||||
const ratio = from / to;
|
||||
const out = new Float32Array(Math.floor(audio.length / ratio));
|
||||
for (let i = 0; i < out.length; i++) {
|
||||
const p = i * ratio;
|
||||
const a = Math.floor(p);
|
||||
out[i] = audio[a] + (audio[Math.min(a + 1, audio.length - 1)] - audio[a]) * (p - a);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
const ep = new Endpointer();
|
||||
const speech = resample(rendered.en.audio, rendered.en.sampling_rate, 16000);
|
||||
// Speech, then a second of silence so the endpointer closes the turn.
|
||||
const withTail = new Float32Array(speech.length + 16000);
|
||||
withTail.set(speech);
|
||||
|
||||
let started = false;
|
||||
let captured = null;
|
||||
for (let i = 0; i < withTail.length; i += 640) { // 40 ms chunks, as the browser sends
|
||||
const { utterances, started: s } = await ep.push(withTail.subarray(i, Math.min(i + 640, withTail.length)));
|
||||
if (s) started = true;
|
||||
if (utterances.length) { captured = utterances[0]; break; }
|
||||
}
|
||||
say(`VAD speech detected: ${started ? 'yes' : 'NO'} | turn closed: ${captured ? 'yes' : 'NO'}`
|
||||
+ (captured ? ` | captured ${(captured.length / 16000).toFixed(2)}s` : ''));
|
||||
|
||||
// ── 3. STT on that captured audio ───────────────────────────────────────────
|
||||
if (captured) {
|
||||
for (const lang of ['en', 'auto']) {
|
||||
const r = await transcribe(captured, lang, 'en');
|
||||
say(`STT ${lang.padEnd(4)} ${r.ms}ms → lang=${r.lang}${r.detected ? ` (heard ${r.detected})` : ''} → ${JSON.stringify(r.text.slice(0, 70))}`);
|
||||
}
|
||||
}
|
||||
|
||||
// ── 4. Text shaping ─────────────────────────────────────────────────────────
|
||||
const md = '## Leads\n\n**3,427** in `new_lead`.\n\n| a | b |\n|---|---|\n| 1 | 2 |\n\n- Only 12% contacted.\nNext step is triage.';
|
||||
say(`\nspeakable: ${JSON.stringify(speakable(md))}`);
|
||||
say(`sentences: ${JSON.stringify(sentences(speakable(md)))}`);
|
||||
say(`\nRSS ${(process.memoryUsage().rss / 1e9).toFixed(2)} GB`);
|
||||
process.exit(0);
|
||||
Reference in New Issue
Block a user