WeLe Agentic AI with Docker deployment

This commit is contained in:
2026-08-28 08:50:08 +05:30
parent 586059b14f
commit 6bb0ef25ca
34 changed files with 2271 additions and 1343 deletions
+72
View File
@@ -0,0 +1,72 @@
/* End-to-end check of the in-process speech pipeline: TTS → VAD → STT.
Synthesising a sentence and feeding that audio back through the endpointer
and recogniser exercises every stage with real speech, which a noise buffer
cannot do — silence never opens a VAD turn.
*/
import { synthesize, transcribe, sentences, speakable, Endpointer } from '../src/speech/index.js';
const say = (m) => console.log(m);
// ── 1. TTS both languages ───────────────────────────────────────────────────
const CASES = [
['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'],
['en', 'There are three thousand four hundred and twenty seven new leads.'],
];
const rendered = {};
for (const [lang, text] of CASES) {
const t0 = Date.now();
await synthesize(text, lang); // warm
const t1 = Date.now();
const out = await synthesize(text, lang);
const ms = Date.now() - t1;
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
rendered[lang] = out;
say(`TTS ${lang} warm ${((t1 - t0) / 1000).toFixed(1)}s | ${ms}ms for ${audioMs.toFixed(0)}ms `
+ `@${out.sampling_rate}Hz | RTF ${(ms / audioMs).toFixed(2)}x`);
}
// ── 2. VAD: does synthesised speech open and close a turn? ──────────────────
function resample(audio, from, to) {
if (from === to) return audio;
const ratio = from / to;
const out = new Float32Array(Math.floor(audio.length / ratio));
for (let i = 0; i < out.length; i++) {
const p = i * ratio;
const a = Math.floor(p);
out[i] = audio[a] + (audio[Math.min(a + 1, audio.length - 1)] - audio[a]) * (p - a);
}
return out;
}
const ep = new Endpointer();
const speech = resample(rendered.en.audio, rendered.en.sampling_rate, 16000);
// Speech, then a second of silence so the endpointer closes the turn.
const withTail = new Float32Array(speech.length + 16000);
withTail.set(speech);
let started = false;
let captured = null;
for (let i = 0; i < withTail.length; i += 640) { // 40 ms chunks, as the browser sends
const { utterances, started: s } = await ep.push(withTail.subarray(i, Math.min(i + 640, withTail.length)));
if (s) started = true;
if (utterances.length) { captured = utterances[0]; break; }
}
say(`VAD speech detected: ${started ? 'yes' : 'NO'} | turn closed: ${captured ? 'yes' : 'NO'}`
+ (captured ? ` | captured ${(captured.length / 16000).toFixed(2)}s` : ''));
// ── 3. STT on that captured audio ───────────────────────────────────────────
if (captured) {
for (const lang of ['en', 'auto']) {
const r = await transcribe(captured, lang, 'en');
say(`STT ${lang.padEnd(4)} ${r.ms}ms → lang=${r.lang}${r.detected ? ` (heard ${r.detected})` : ''} → ${JSON.stringify(r.text.slice(0, 70))}`);
}
}
// ── 4. Text shaping ─────────────────────────────────────────────────────────
const md = '## Leads\n\n**3,427** in `new_lead`.\n\n| a | b |\n|---|---|\n| 1 | 2 |\n\n- Only 12% contacted.\nNext step is triage.';
say(`\nspeakable: ${JSON.stringify(speakable(md))}`);
say(`sentences: ${JSON.stringify(sentences(speakable(md)))}`);
say(`\nRSS ${(process.memoryUsage().rss / 1e9).toFixed(2)} GB`);
process.exit(0);