WeLe Agentic AI with Docker deployment

This commit is contained in:
2026-08-28 08:50:08 +05:30
parent 586059b14f
commit 6bb0ef25ca
34 changed files with 2271 additions and 1343 deletions
+66
View File
@@ -0,0 +1,66 @@
/* Generate the tokenizer.json that Transformers.js needs for the exported
Tamil VITS model.
`save_pretrained` does not emit one: VitsTokenizer is a "slow" tokenizer with
no fast counterpart, so Python writes vocab.json + tokenizer_config.json and
nothing else. Transformers.js only reads tokenizer.json, so we synthesise it
from the exported vocab, mirroring the structure of the working English
model (Xenova/mms-tts-eng) exactly.
The four normalizer steps, in order:
1. Lowercase — no-op for Tamil, matters for embedded Latin/digits
2. Replace — drop every character outside the vocab
3. Strip — trim surrounding whitespace
4. Replace — insert the blank token between every character,
which is what `add_blank: true` means for VITS.
Omit this and the audio comes out garbled.
*/
import fs from 'node:fs';
import path from 'node:path';
const DIR = 'assets/tts/mms-tts-tam';
const vocab = JSON.parse(fs.readFileSync(path.join(DIR, 'vocab.json'), 'utf8'));
const cfg = JSON.parse(fs.readFileSync(path.join(DIR, 'tokenizer_config.json'), 'utf8'));
// The blank/pad token is whichever character maps to id 0.
const blank = Object.keys(vocab).find((k) => vocab[k] === 0);
const unk = cfg.unk_token ?? '<unk>';
const unkId = vocab[unk] ?? Object.keys(vocab).length;
// Character class of everything we keep. Escape the regex metacharacters that
// are still special inside a negated class.
const escaped = Object.keys(vocab)
.filter((c) => c !== unk)
.map((c) => (']\\^-'.includes(c) ? '\\' + c : c))
.join('');
const tokenizer = {
version: '1.0',
truncation: null,
padding: null,
added_tokens: [{
id: unkId, content: unk,
single_word: false, lstrip: false, rstrip: false, normalized: false, special: true,
}],
normalizer: {
type: 'Sequence',
normalizers: [
{ type: 'Lowercase' },
{ type: 'Replace', pattern: { Regex: `[^${escaped}]` }, content: '' },
{ type: 'Strip', strip_left: true, strip_right: true },
...(cfg.add_blank ? [{ type: 'Replace', pattern: { Regex: '(?=.)|(?<!^)$' }, content: blank }] : []),
],
},
pre_tokenizer: { type: 'Split', pattern: { Regex: '' }, behavior: 'Isolated', invert: false },
post_processor: null,
decoder: null,
model: { vocab },
};
fs.writeFileSync(path.join(DIR, 'tokenizer.json'), JSON.stringify(tokenizer, null, 1));
console.log(`wrote ${DIR}/tokenizer.json`);
console.log(` vocab ${Object.keys(vocab).length} tokens`);
console.log(` blank token ${JSON.stringify(blank)} (id 0)`);
console.log(` unk ${JSON.stringify(unk)} (id ${unkId})`);
console.log(` add_blank ${cfg.add_blank}`);
+99
View File
@@ -0,0 +1,99 @@
"""One-time export of facebook/mms-tts-tam to ONNX.
No public ONNX build of Tamil MMS-TTS exists, so we make one. This runs ONCE on
a workstation; the committed artefact is what ships. The service itself is pure
JavaScript and never needs Python or this script.
The output must match the contract Transformers.js expects for VITS, taken from
the working English model (Xenova/mms-tts-eng):
inputs : input_ids, attention_mask
outputs: waveform, spectrogram
Layout produced (mirrors the HF repo so Transformers.js can load the folder):
assets/tts/mms-tts-tam/
config.json, tokenizer.json, vocab.json, …
onnx/model.onnx fp32
onnx/model_quantized.onnx int8 ← what we ship
"""
from __future__ import annotations
import json
import shutil
import sys
from pathlib import Path
import torch
from transformers import AutoTokenizer, VitsModel
MODEL = "facebook/mms-tts-tam"
OUT = Path("assets/tts/mms-tts-tam")
ONNX_DIR = OUT / "onnx"
ONNX_DIR.mkdir(parents=True, exist_ok=True)
print(f"loading {MODEL} …")
model = VitsModel.from_pretrained(MODEL).eval()
tok = AutoTokenizer.from_pretrained(MODEL)
# VITS has a stochastic duration predictor. Exporting with noise left on bakes
# RandomNormalLike nodes into the graph, which is fine and keeps prosody
# natural — but seed it so this export is reproducible.
torch.manual_seed(0)
class Exportable(torch.nn.Module):
"""Return only (waveform, spectrogram), in that order — the JS side indexes
outputs by name, but a plain tuple keeps the exported graph simple."""
def __init__(self, m: VitsModel) -> None:
super().__init__()
self.m = m
def forward(self, input_ids: torch.Tensor, attention_mask: torch.Tensor):
out = self.m(input_ids=input_ids, attention_mask=attention_mask)
return out.waveform, out.spectrogram
sample = tok("வணக்கம், இது ஒரு சோதனை.", return_tensors="pt")
fp32 = ONNX_DIR / "model.onnx"
print("exporting to ONNX …")
torch.onnx.export(
Exportable(model),
(sample["input_ids"], sample["attention_mask"]),
str(fp32),
input_names=["input_ids", "attention_mask"],
output_names=["waveform", "spectrogram"],
dynamic_axes={
"input_ids": {0: "batch", 1: "sequence"},
"attention_mask": {0: "batch", 1: "sequence"},
"waveform": {0: "batch", 1: "samples"},
"spectrogram": {0: "batch", 2: "frames"},
},
opset_version=17,
do_constant_folding=True,
)
print(f" fp32: {fp32.stat().st_size / 1e6:.1f} MB")
# ── int8 ────────────────────────────────────────────────────────────────────
try:
from onnxruntime.quantization import QuantType, quantize_dynamic
q = ONNX_DIR / "model_quantized.onnx"
quantize_dynamic(str(fp32), str(q), weight_type=QuantType.QUInt8)
print(f" int8: {q.stat().st_size / 1e6:.1f} MB")
except Exception as e: # noqa: BLE001
print(f" quantisation skipped: {e}")
# ── tokenizer + config, so the folder loads standalone ──────────────────────
tok.save_pretrained(OUT)
model.config.to_json_file(OUT / "config.json")
# Transformers.js reads this to pick a default dtype.
(OUT / "quantize_config.json").write_text(json.dumps({"per_channel": False, "reduce_range": False}, indent=2))
print("\nwrote:")
for p in sorted(OUT.rglob("*")):
if p.is_file():
print(f" {p.relative_to(OUT)} ({p.stat().st_size / 1e6:.2f} MB)")
+21
View File
@@ -0,0 +1,21 @@
/* Does auto mode now route Tamil to Tamil instead of silently using English? */
import { synthesize, transcribe } from '../src/speech/index.js';
const resample = (a, from, to) => {
const r = from / to, out = new Float32Array(Math.floor(a.length / r));
for (let i = 0; i < out.length; i++) { const p = i * r, k = Math.floor(p); out[i] = a[k] + (a[Math.min(k + 1, a.length - 1)] - a[k]) * (p - k); }
return out;
};
for (const [lang, text] of [
['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'],
['en', 'How many new leads did we receive today?'],
]) {
const spoken = await synthesize(text, lang);
const audio = resample(spoken.audio, spoken.sampling_rate, 16000);
const r = await transcribe(audio, 'auto', 'ta');
const ok = r.lang === lang ? 'PASS' : 'FAIL';
console.log(`${ok} spoke ${lang} → routed ${r.lang} (detected ${r.detected} @ ${r.confidence}) ${r.ms}ms`);
console.log(` ${JSON.stringify(r.text.slice(0, 80))}`);
}
process.exit(0);
+62
View File
@@ -0,0 +1,62 @@
/* Can we get real language detection out of Whisper in Transformers.js? */
import { AutoProcessor, WhisperForConditionalGeneration, Tensor, env } from '@huggingface/transformers';
import { synthesize } from '../src/speech/index.js';
env.cacheDir = './.transformers-cache';
const MODEL = 'onnx-community/whisper-base';
const processor = await AutoProcessor.from_pretrained(MODEL);
const model = await WhisperForConditionalGeneration.from_pretrained(MODEL, { dtype: 'q8' });
const tok = processor.tokenizer;
// Whisper emits one language token right after <|startoftranscript|>. Reading
// that distribution is a single decoder step — far cheaper than transcribing
// twice to see which language "looks better".
const id = (t) => tok.encode(t, { add_special_tokens: false })[0];
const sot = id('<|startoftranscript|>');
const CANDIDATES = ['en', 'ta'];
const langIds = CANDIDATES.map((c) => id(`<|${c}|>`));
console.log('sot:', sot, '| language token ids:', JSON.stringify(Object.fromEntries(CANDIDATES.map((c, i) => [c, langIds[i]]))));
function resample(a, from, to) {
const r = from / to;
const out = new Float32Array(Math.floor(a.length / r));
for (let i = 0; i < out.length; i++) {
const p = i * r, k = Math.floor(p);
out[i] = a[k] + (a[Math.min(k + 1, a.length - 1)] - a[k]) * (p - k);
}
return out;
}
for (const [lang, text] of [
['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'],
['en', 'There are three thousand four hundred and twenty seven new leads.'],
]) {
const spoken = await synthesize(text, lang);
const audio = resample(spoken.audio, spoken.sampling_rate, 16000);
const inputs = await processor(audio);
const t0 = Date.now();
const out = await model({
...inputs,
decoder_input_ids: new Tensor('int64', BigInt64Array.from([BigInt(sot)]), [1, 1]),
});
const ms = Date.now() - t0;
const logits = out.logits;
const last = logits.dims[1] - 1;
const vocab = logits.dims[2];
const row = logits.data.slice(last * vocab, (last + 1) * vocab);
const scores = langIds.map((id) => Number(row[id]));
const max = Math.max(...scores);
const exp = scores.map((s) => Math.exp(s - max));
const sum = exp.reduce((a, b) => a + b, 0);
const probs = exp.map((e) => e / sum);
const best = probs.indexOf(Math.max(...probs));
console.log(`spoken ${lang} → detected ${CANDIDATES[best]} `
+ `(${CANDIDATES.map((c, i) => `${c} ${probs[i].toFixed(3)}`).join(', ')}) in ${ms}ms `
+ `${CANDIDATES[best] === lang ? '✅' : '❌'}`);
}
process.exit(0);
+44
View File
@@ -0,0 +1,44 @@
/* Do Whisper (STT) and MMS-TTS (TTS) actually run in Node on CPU? */
import { pipeline, env } from '@huggingface/transformers';
env.cacheDir = './.transformers-cache';
const t = (t0) => `${((performance.now() - t0) / 1000).toFixed(1)}s`;
// ── TTS: MMS-TTS Tamil (VITS, 36M, feed-forward) ────────────────────────────
console.log('[1/2] loading MMS-TTS Tamil…');
let t0 = performance.now();
const tts = await pipeline('text-to-speech', 'Xenova/mms-tts-eng', { dtype: 'fp32' });
console.log(` loaded in ${t(t0)}`);
const TA = 'There are three thousand four hundred leads in the new lead stage.';
await tts(TA); // warm
for (const [label, text] of [['short', TA], ['long', TA + ' ' + TA + ' ' + TA]]) {
t0 = performance.now();
const out = await tts(text);
const ms = performance.now() - t0;
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
console.log(
` ${label.padEnd(5)} ${String(text.length).padStart(3)} chars → ${ms.toFixed(0)}ms `
+ `for ${audioMs.toFixed(0)}ms audio @ ${out.sampling_rate}Hz → RTF ${(ms / audioMs).toFixed(2)}x`,
);
}
// ── STT: Whisper (multilingual — Tamil, Hindi, English + detection) ─────────
console.log('\n[2/2] loading Whisper base…');
t0 = performance.now();
const stt = await pipeline('automatic-speech-recognition', 'onnx-community/whisper-base', { dtype: 'q8' });
console.log(` loaded in ${t(t0)}`);
// 4 s of quiet noise — proves the graph runs and times it.
const audio = Float32Array.from({ length: 16000 * 4 }, () => (Math.random() - 0.5) * 0.02);
t0 = performance.now();
const r = await stt(audio, { language: 'ta', task: 'transcribe' });
console.log(` 4000ms audio → ${(performance.now() - t0).toFixed(0)}ms → ${JSON.stringify(r.text).slice(0, 60)}`);
t0 = performance.now();
const r2 = await stt(audio, { language: 'en', task: 'transcribe' });
console.log(` english pass → ${(performance.now() - t0).toFixed(0)}ms → ${JSON.stringify(r2.text).slice(0, 60)}`);
console.log(`\nRSS ${(process.memoryUsage().rss / 1e9).toFixed(2)} GB`);
process.exit(0);
+55
View File
@@ -0,0 +1,55 @@
/* Does the locally-exported Tamil ONNX load and speak through Transformers.js? */
import { pipeline, env } from '@huggingface/transformers';
import fs from 'node:fs';
// Load from the local folder, not the Hub.
env.allowRemoteModels = false;
env.localModelPath = './assets/tts';
for (const dtype of ['q8', 'fp32']) {
try {
const t0 = performance.now();
const tts = await pipeline('text-to-speech', 'mms-tts-tam', { dtype });
const load = performance.now() - t0;
const TEXT = 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.';
await tts(TEXT); // warm
const t1 = performance.now();
const out = await tts(TEXT);
const ms = performance.now() - t1;
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
console.log(
`${dtype.padEnd(5)} load ${(load / 1000).toFixed(1)}s | `
+ `${ms.toFixed(0)}ms for ${audioMs.toFixed(0)}ms audio @ ${out.sampling_rate}Hz | `
+ `RTF ${(ms / audioMs).toFixed(2)}x`,
);
// Non-silent output is the real proof the graph is wired correctly.
const peak = out.audio.reduce((m, v) => Math.max(m, Math.abs(v)), 0);
console.log(` samples ${out.audio.length}, peak amplitude ${peak.toFixed(3)} ${peak > 0.01 ? '✅ audible' : '⚠️ SILENT'}`);
if (dtype === 'q8') {
const wav = toWav(out.audio, out.sampling_rate);
fs.writeFileSync('scripts/tamil-sample.wav', wav);
console.log(' wrote scripts/tamil-sample.wav — play it to judge quality');
}
} catch (e) {
console.log(`${dtype.padEnd(5)} FAILED: ${e.message.slice(0, 160)}`);
}
}
function toWav(samples, rate) {
const buf = Buffer.alloc(44 + samples.length * 2);
buf.write('RIFF', 0); buf.writeUInt32LE(36 + samples.length * 2, 4); buf.write('WAVE', 8);
buf.write('fmt ', 12); buf.writeUInt32LE(16, 16); buf.writeUInt16LE(1, 20); buf.writeUInt16LE(1, 22);
buf.writeUInt32LE(rate, 24); buf.writeUInt32LE(rate * 2, 28); buf.writeUInt16LE(2, 32); buf.writeUInt16LE(16, 34);
buf.write('data', 36); buf.writeUInt32LE(samples.length * 2, 40);
for (let i = 0; i < samples.length; i++) {
const s = Math.max(-1, Math.min(1, samples[i]));
buf.writeInt16LE(s < 0 ? s * 0x8000 : s * 0x7fff, 44 + i * 2);
}
return buf;
}
process.exit(0);
+72
View File
@@ -0,0 +1,72 @@
/* End-to-end check of the in-process speech pipeline: TTS → VAD → STT.
Synthesising a sentence and feeding that audio back through the endpointer
and recogniser exercises every stage with real speech, which a noise buffer
cannot do — silence never opens a VAD turn.
*/
import { synthesize, transcribe, sentences, speakable, Endpointer } from '../src/speech/index.js';
const say = (m) => console.log(m);
// ── 1. TTS both languages ───────────────────────────────────────────────────
const CASES = [
['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'],
['en', 'There are three thousand four hundred and twenty seven new leads.'],
];
const rendered = {};
for (const [lang, text] of CASES) {
const t0 = Date.now();
await synthesize(text, lang); // warm
const t1 = Date.now();
const out = await synthesize(text, lang);
const ms = Date.now() - t1;
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
rendered[lang] = out;
say(`TTS ${lang} warm ${((t1 - t0) / 1000).toFixed(1)}s | ${ms}ms for ${audioMs.toFixed(0)}ms `
+ `@${out.sampling_rate}Hz | RTF ${(ms / audioMs).toFixed(2)}x`);
}
// ── 2. VAD: does synthesised speech open and close a turn? ──────────────────
function resample(audio, from, to) {
if (from === to) return audio;
const ratio = from / to;
const out = new Float32Array(Math.floor(audio.length / ratio));
for (let i = 0; i < out.length; i++) {
const p = i * ratio;
const a = Math.floor(p);
out[i] = audio[a] + (audio[Math.min(a + 1, audio.length - 1)] - audio[a]) * (p - a);
}
return out;
}
const ep = new Endpointer();
const speech = resample(rendered.en.audio, rendered.en.sampling_rate, 16000);
// Speech, then a second of silence so the endpointer closes the turn.
const withTail = new Float32Array(speech.length + 16000);
withTail.set(speech);
let started = false;
let captured = null;
for (let i = 0; i < withTail.length; i += 640) { // 40 ms chunks, as the browser sends
const { utterances, started: s } = await ep.push(withTail.subarray(i, Math.min(i + 640, withTail.length)));
if (s) started = true;
if (utterances.length) { captured = utterances[0]; break; }
}
say(`VAD speech detected: ${started ? 'yes' : 'NO'} | turn closed: ${captured ? 'yes' : 'NO'}`
+ (captured ? ` | captured ${(captured.length / 16000).toFixed(2)}s` : ''));
// ── 3. STT on that captured audio ───────────────────────────────────────────
if (captured) {
for (const lang of ['en', 'auto']) {
const r = await transcribe(captured, lang, 'en');
say(`STT ${lang.padEnd(4)} ${r.ms}ms → lang=${r.lang}${r.detected ? ` (heard ${r.detected})` : ''} → ${JSON.stringify(r.text.slice(0, 70))}`);
}
}
// ── 4. Text shaping ─────────────────────────────────────────────────────────
const md = '## Leads\n\n**3,427** in `new_lead`.\n\n| a | b |\n|---|---|\n| 1 | 2 |\n\n- Only 12% contacted.\nNext step is triage.';
say(`\nspeakable: ${JSON.stringify(speakable(md))}`);
say(`sentences: ${JSON.stringify(sentences(speakable(md)))}`);
say(`\nRSS ${(process.memoryUsage().rss / 1e9).toFixed(2)} GB`);
process.exit(0);