WeLe Agentic AI with Docker deployment
This commit is contained in:
@@ -0,0 +1,66 @@
|
||||
/* Generate the tokenizer.json that Transformers.js needs for the exported
|
||||
Tamil VITS model.
|
||||
|
||||
`save_pretrained` does not emit one: VitsTokenizer is a "slow" tokenizer with
|
||||
no fast counterpart, so Python writes vocab.json + tokenizer_config.json and
|
||||
nothing else. Transformers.js only reads tokenizer.json, so we synthesise it
|
||||
from the exported vocab, mirroring the structure of the working English
|
||||
model (Xenova/mms-tts-eng) exactly.
|
||||
|
||||
The four normalizer steps, in order:
|
||||
1. Lowercase — no-op for Tamil, matters for embedded Latin/digits
|
||||
2. Replace — drop every character outside the vocab
|
||||
3. Strip — trim surrounding whitespace
|
||||
4. Replace — insert the blank token between every character,
|
||||
which is what `add_blank: true` means for VITS.
|
||||
Omit this and the audio comes out garbled.
|
||||
*/
|
||||
import fs from 'node:fs';
|
||||
import path from 'node:path';
|
||||
|
||||
const DIR = 'assets/tts/mms-tts-tam';
|
||||
const vocab = JSON.parse(fs.readFileSync(path.join(DIR, 'vocab.json'), 'utf8'));
|
||||
const cfg = JSON.parse(fs.readFileSync(path.join(DIR, 'tokenizer_config.json'), 'utf8'));
|
||||
|
||||
// The blank/pad token is whichever character maps to id 0.
|
||||
const blank = Object.keys(vocab).find((k) => vocab[k] === 0);
|
||||
const unk = cfg.unk_token ?? '<unk>';
|
||||
const unkId = vocab[unk] ?? Object.keys(vocab).length;
|
||||
|
||||
// Character class of everything we keep. Escape the regex metacharacters that
|
||||
// are still special inside a negated class.
|
||||
const escaped = Object.keys(vocab)
|
||||
.filter((c) => c !== unk)
|
||||
.map((c) => (']\\^-'.includes(c) ? '\\' + c : c))
|
||||
.join('');
|
||||
|
||||
const tokenizer = {
|
||||
version: '1.0',
|
||||
truncation: null,
|
||||
padding: null,
|
||||
added_tokens: [{
|
||||
id: unkId, content: unk,
|
||||
single_word: false, lstrip: false, rstrip: false, normalized: false, special: true,
|
||||
}],
|
||||
normalizer: {
|
||||
type: 'Sequence',
|
||||
normalizers: [
|
||||
{ type: 'Lowercase' },
|
||||
{ type: 'Replace', pattern: { Regex: `[^${escaped}]` }, content: '' },
|
||||
{ type: 'Strip', strip_left: true, strip_right: true },
|
||||
...(cfg.add_blank ? [{ type: 'Replace', pattern: { Regex: '(?=.)|(?<!^)$' }, content: blank }] : []),
|
||||
],
|
||||
},
|
||||
pre_tokenizer: { type: 'Split', pattern: { Regex: '' }, behavior: 'Isolated', invert: false },
|
||||
post_processor: null,
|
||||
decoder: null,
|
||||
model: { vocab },
|
||||
};
|
||||
|
||||
fs.writeFileSync(path.join(DIR, 'tokenizer.json'), JSON.stringify(tokenizer, null, 1));
|
||||
|
||||
console.log(`wrote ${DIR}/tokenizer.json`);
|
||||
console.log(` vocab ${Object.keys(vocab).length} tokens`);
|
||||
console.log(` blank token ${JSON.stringify(blank)} (id 0)`);
|
||||
console.log(` unk ${JSON.stringify(unk)} (id ${unkId})`);
|
||||
console.log(` add_blank ${cfg.add_blank}`);
|
||||
@@ -0,0 +1,99 @@
|
||||
"""One-time export of facebook/mms-tts-tam to ONNX.
|
||||
|
||||
No public ONNX build of Tamil MMS-TTS exists, so we make one. This runs ONCE on
|
||||
a workstation; the committed artefact is what ships. The service itself is pure
|
||||
JavaScript and never needs Python or this script.
|
||||
|
||||
The output must match the contract Transformers.js expects for VITS, taken from
|
||||
the working English model (Xenova/mms-tts-eng):
|
||||
|
||||
inputs : input_ids, attention_mask
|
||||
outputs: waveform, spectrogram
|
||||
|
||||
Layout produced (mirrors the HF repo so Transformers.js can load the folder):
|
||||
|
||||
assets/tts/mms-tts-tam/
|
||||
config.json, tokenizer.json, vocab.json, …
|
||||
onnx/model.onnx fp32
|
||||
onnx/model_quantized.onnx int8 ← what we ship
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import shutil
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import torch
|
||||
from transformers import AutoTokenizer, VitsModel
|
||||
|
||||
MODEL = "facebook/mms-tts-tam"
|
||||
OUT = Path("assets/tts/mms-tts-tam")
|
||||
ONNX_DIR = OUT / "onnx"
|
||||
ONNX_DIR.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
print(f"loading {MODEL} …")
|
||||
model = VitsModel.from_pretrained(MODEL).eval()
|
||||
tok = AutoTokenizer.from_pretrained(MODEL)
|
||||
|
||||
# VITS has a stochastic duration predictor. Exporting with noise left on bakes
|
||||
# RandomNormalLike nodes into the graph, which is fine and keeps prosody
|
||||
# natural — but seed it so this export is reproducible.
|
||||
torch.manual_seed(0)
|
||||
|
||||
|
||||
class Exportable(torch.nn.Module):
|
||||
"""Return only (waveform, spectrogram), in that order — the JS side indexes
|
||||
outputs by name, but a plain tuple keeps the exported graph simple."""
|
||||
|
||||
def __init__(self, m: VitsModel) -> None:
|
||||
super().__init__()
|
||||
self.m = m
|
||||
|
||||
def forward(self, input_ids: torch.Tensor, attention_mask: torch.Tensor):
|
||||
out = self.m(input_ids=input_ids, attention_mask=attention_mask)
|
||||
return out.waveform, out.spectrogram
|
||||
|
||||
|
||||
sample = tok("வணக்கம், இது ஒரு சோதனை.", return_tensors="pt")
|
||||
fp32 = ONNX_DIR / "model.onnx"
|
||||
|
||||
print("exporting to ONNX …")
|
||||
torch.onnx.export(
|
||||
Exportable(model),
|
||||
(sample["input_ids"], sample["attention_mask"]),
|
||||
str(fp32),
|
||||
input_names=["input_ids", "attention_mask"],
|
||||
output_names=["waveform", "spectrogram"],
|
||||
dynamic_axes={
|
||||
"input_ids": {0: "batch", 1: "sequence"},
|
||||
"attention_mask": {0: "batch", 1: "sequence"},
|
||||
"waveform": {0: "batch", 1: "samples"},
|
||||
"spectrogram": {0: "batch", 2: "frames"},
|
||||
},
|
||||
opset_version=17,
|
||||
do_constant_folding=True,
|
||||
)
|
||||
print(f" fp32: {fp32.stat().st_size / 1e6:.1f} MB")
|
||||
|
||||
# ── int8 ────────────────────────────────────────────────────────────────────
|
||||
try:
|
||||
from onnxruntime.quantization import QuantType, quantize_dynamic
|
||||
|
||||
q = ONNX_DIR / "model_quantized.onnx"
|
||||
quantize_dynamic(str(fp32), str(q), weight_type=QuantType.QUInt8)
|
||||
print(f" int8: {q.stat().st_size / 1e6:.1f} MB")
|
||||
except Exception as e: # noqa: BLE001
|
||||
print(f" quantisation skipped: {e}")
|
||||
|
||||
# ── tokenizer + config, so the folder loads standalone ──────────────────────
|
||||
tok.save_pretrained(OUT)
|
||||
model.config.to_json_file(OUT / "config.json")
|
||||
|
||||
# Transformers.js reads this to pick a default dtype.
|
||||
(OUT / "quantize_config.json").write_text(json.dumps({"per_channel": False, "reduce_range": False}, indent=2))
|
||||
|
||||
print("\nwrote:")
|
||||
for p in sorted(OUT.rglob("*")):
|
||||
if p.is_file():
|
||||
print(f" {p.relative_to(OUT)} ({p.stat().st_size / 1e6:.2f} MB)")
|
||||
@@ -0,0 +1,21 @@
|
||||
/* Does auto mode now route Tamil to Tamil instead of silently using English? */
|
||||
import { synthesize, transcribe } from '../src/speech/index.js';
|
||||
|
||||
const resample = (a, from, to) => {
|
||||
const r = from / to, out = new Float32Array(Math.floor(a.length / r));
|
||||
for (let i = 0; i < out.length; i++) { const p = i * r, k = Math.floor(p); out[i] = a[k] + (a[Math.min(k + 1, a.length - 1)] - a[k]) * (p - k); }
|
||||
return out;
|
||||
};
|
||||
|
||||
for (const [lang, text] of [
|
||||
['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'],
|
||||
['en', 'How many new leads did we receive today?'],
|
||||
]) {
|
||||
const spoken = await synthesize(text, lang);
|
||||
const audio = resample(spoken.audio, spoken.sampling_rate, 16000);
|
||||
const r = await transcribe(audio, 'auto', 'ta');
|
||||
const ok = r.lang === lang ? 'PASS' : 'FAIL';
|
||||
console.log(`${ok} spoke ${lang} → routed ${r.lang} (detected ${r.detected} @ ${r.confidence}) ${r.ms}ms`);
|
||||
console.log(` ${JSON.stringify(r.text.slice(0, 80))}`);
|
||||
}
|
||||
process.exit(0);
|
||||
@@ -0,0 +1,62 @@
|
||||
/* Can we get real language detection out of Whisper in Transformers.js? */
|
||||
import { AutoProcessor, WhisperForConditionalGeneration, Tensor, env } from '@huggingface/transformers';
|
||||
import { synthesize } from '../src/speech/index.js';
|
||||
|
||||
env.cacheDir = './.transformers-cache';
|
||||
const MODEL = 'onnx-community/whisper-base';
|
||||
|
||||
const processor = await AutoProcessor.from_pretrained(MODEL);
|
||||
const model = await WhisperForConditionalGeneration.from_pretrained(MODEL, { dtype: 'q8' });
|
||||
const tok = processor.tokenizer;
|
||||
|
||||
// Whisper emits one language token right after <|startoftranscript|>. Reading
|
||||
// that distribution is a single decoder step — far cheaper than transcribing
|
||||
// twice to see which language "looks better".
|
||||
const id = (t) => tok.encode(t, { add_special_tokens: false })[0];
|
||||
const sot = id('<|startoftranscript|>');
|
||||
const CANDIDATES = ['en', 'ta'];
|
||||
const langIds = CANDIDATES.map((c) => id(`<|${c}|>`));
|
||||
console.log('sot:', sot, '| language token ids:', JSON.stringify(Object.fromEntries(CANDIDATES.map((c, i) => [c, langIds[i]]))));
|
||||
|
||||
function resample(a, from, to) {
|
||||
const r = from / to;
|
||||
const out = new Float32Array(Math.floor(a.length / r));
|
||||
for (let i = 0; i < out.length; i++) {
|
||||
const p = i * r, k = Math.floor(p);
|
||||
out[i] = a[k] + (a[Math.min(k + 1, a.length - 1)] - a[k]) * (p - k);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
for (const [lang, text] of [
|
||||
['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'],
|
||||
['en', 'There are three thousand four hundred and twenty seven new leads.'],
|
||||
]) {
|
||||
const spoken = await synthesize(text, lang);
|
||||
const audio = resample(spoken.audio, spoken.sampling_rate, 16000);
|
||||
|
||||
const inputs = await processor(audio);
|
||||
const t0 = Date.now();
|
||||
const out = await model({
|
||||
...inputs,
|
||||
decoder_input_ids: new Tensor('int64', BigInt64Array.from([BigInt(sot)]), [1, 1]),
|
||||
});
|
||||
const ms = Date.now() - t0;
|
||||
|
||||
const logits = out.logits;
|
||||
const last = logits.dims[1] - 1;
|
||||
const vocab = logits.dims[2];
|
||||
const row = logits.data.slice(last * vocab, (last + 1) * vocab);
|
||||
|
||||
const scores = langIds.map((id) => Number(row[id]));
|
||||
const max = Math.max(...scores);
|
||||
const exp = scores.map((s) => Math.exp(s - max));
|
||||
const sum = exp.reduce((a, b) => a + b, 0);
|
||||
const probs = exp.map((e) => e / sum);
|
||||
const best = probs.indexOf(Math.max(...probs));
|
||||
|
||||
console.log(`spoken ${lang} → detected ${CANDIDATES[best]} `
|
||||
+ `(${CANDIDATES.map((c, i) => `${c} ${probs[i].toFixed(3)}`).join(', ')}) in ${ms}ms `
|
||||
+ `${CANDIDATES[best] === lang ? '✅' : '❌'}`);
|
||||
}
|
||||
process.exit(0);
|
||||
@@ -0,0 +1,44 @@
|
||||
/* Do Whisper (STT) and MMS-TTS (TTS) actually run in Node on CPU? */
|
||||
import { pipeline, env } from '@huggingface/transformers';
|
||||
|
||||
env.cacheDir = './.transformers-cache';
|
||||
|
||||
const t = (t0) => `${((performance.now() - t0) / 1000).toFixed(1)}s`;
|
||||
|
||||
// ── TTS: MMS-TTS Tamil (VITS, 36M, feed-forward) ────────────────────────────
|
||||
console.log('[1/2] loading MMS-TTS Tamil…');
|
||||
let t0 = performance.now();
|
||||
const tts = await pipeline('text-to-speech', 'Xenova/mms-tts-eng', { dtype: 'fp32' });
|
||||
console.log(` loaded in ${t(t0)}`);
|
||||
|
||||
const TA = 'There are three thousand four hundred leads in the new lead stage.';
|
||||
await tts(TA); // warm
|
||||
for (const [label, text] of [['short', TA], ['long', TA + ' ' + TA + ' ' + TA]]) {
|
||||
t0 = performance.now();
|
||||
const out = await tts(text);
|
||||
const ms = performance.now() - t0;
|
||||
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
|
||||
console.log(
|
||||
` ${label.padEnd(5)} ${String(text.length).padStart(3)} chars → ${ms.toFixed(0)}ms `
|
||||
+ `for ${audioMs.toFixed(0)}ms audio @ ${out.sampling_rate}Hz → RTF ${(ms / audioMs).toFixed(2)}x`,
|
||||
);
|
||||
}
|
||||
|
||||
// ── STT: Whisper (multilingual — Tamil, Hindi, English + detection) ─────────
|
||||
console.log('\n[2/2] loading Whisper base…');
|
||||
t0 = performance.now();
|
||||
const stt = await pipeline('automatic-speech-recognition', 'onnx-community/whisper-base', { dtype: 'q8' });
|
||||
console.log(` loaded in ${t(t0)}`);
|
||||
|
||||
// 4 s of quiet noise — proves the graph runs and times it.
|
||||
const audio = Float32Array.from({ length: 16000 * 4 }, () => (Math.random() - 0.5) * 0.02);
|
||||
t0 = performance.now();
|
||||
const r = await stt(audio, { language: 'ta', task: 'transcribe' });
|
||||
console.log(` 4000ms audio → ${(performance.now() - t0).toFixed(0)}ms → ${JSON.stringify(r.text).slice(0, 60)}`);
|
||||
|
||||
t0 = performance.now();
|
||||
const r2 = await stt(audio, { language: 'en', task: 'transcribe' });
|
||||
console.log(` english pass → ${(performance.now() - t0).toFixed(0)}ms → ${JSON.stringify(r2.text).slice(0, 60)}`);
|
||||
|
||||
console.log(`\nRSS ${(process.memoryUsage().rss / 1e9).toFixed(2)} GB`);
|
||||
process.exit(0);
|
||||
@@ -0,0 +1,55 @@
|
||||
/* Does the locally-exported Tamil ONNX load and speak through Transformers.js? */
|
||||
import { pipeline, env } from '@huggingface/transformers';
|
||||
import fs from 'node:fs';
|
||||
|
||||
// Load from the local folder, not the Hub.
|
||||
env.allowRemoteModels = false;
|
||||
env.localModelPath = './assets/tts';
|
||||
|
||||
for (const dtype of ['q8', 'fp32']) {
|
||||
try {
|
||||
const t0 = performance.now();
|
||||
const tts = await pipeline('text-to-speech', 'mms-tts-tam', { dtype });
|
||||
const load = performance.now() - t0;
|
||||
|
||||
const TEXT = 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.';
|
||||
await tts(TEXT); // warm
|
||||
|
||||
const t1 = performance.now();
|
||||
const out = await tts(TEXT);
|
||||
const ms = performance.now() - t1;
|
||||
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
|
||||
|
||||
console.log(
|
||||
`${dtype.padEnd(5)} load ${(load / 1000).toFixed(1)}s | `
|
||||
+ `${ms.toFixed(0)}ms for ${audioMs.toFixed(0)}ms audio @ ${out.sampling_rate}Hz | `
|
||||
+ `RTF ${(ms / audioMs).toFixed(2)}x`,
|
||||
);
|
||||
|
||||
// Non-silent output is the real proof the graph is wired correctly.
|
||||
const peak = out.audio.reduce((m, v) => Math.max(m, Math.abs(v)), 0);
|
||||
console.log(` samples ${out.audio.length}, peak amplitude ${peak.toFixed(3)} ${peak > 0.01 ? '✅ audible' : '⚠️ SILENT'}`);
|
||||
|
||||
if (dtype === 'q8') {
|
||||
const wav = toWav(out.audio, out.sampling_rate);
|
||||
fs.writeFileSync('scripts/tamil-sample.wav', wav);
|
||||
console.log(' wrote scripts/tamil-sample.wav — play it to judge quality');
|
||||
}
|
||||
} catch (e) {
|
||||
console.log(`${dtype.padEnd(5)} FAILED: ${e.message.slice(0, 160)}`);
|
||||
}
|
||||
}
|
||||
|
||||
function toWav(samples, rate) {
|
||||
const buf = Buffer.alloc(44 + samples.length * 2);
|
||||
buf.write('RIFF', 0); buf.writeUInt32LE(36 + samples.length * 2, 4); buf.write('WAVE', 8);
|
||||
buf.write('fmt ', 12); buf.writeUInt32LE(16, 16); buf.writeUInt16LE(1, 20); buf.writeUInt16LE(1, 22);
|
||||
buf.writeUInt32LE(rate, 24); buf.writeUInt32LE(rate * 2, 28); buf.writeUInt16LE(2, 32); buf.writeUInt16LE(16, 34);
|
||||
buf.write('data', 36); buf.writeUInt32LE(samples.length * 2, 40);
|
||||
for (let i = 0; i < samples.length; i++) {
|
||||
const s = Math.max(-1, Math.min(1, samples[i]));
|
||||
buf.writeInt16LE(s < 0 ? s * 0x8000 : s * 0x7fff, 44 + i * 2);
|
||||
}
|
||||
return buf;
|
||||
}
|
||||
process.exit(0);
|
||||
@@ -0,0 +1,72 @@
|
||||
/* End-to-end check of the in-process speech pipeline: TTS → VAD → STT.
|
||||
|
||||
Synthesising a sentence and feeding that audio back through the endpointer
|
||||
and recogniser exercises every stage with real speech, which a noise buffer
|
||||
cannot do — silence never opens a VAD turn.
|
||||
*/
|
||||
import { synthesize, transcribe, sentences, speakable, Endpointer } from '../src/speech/index.js';
|
||||
|
||||
const say = (m) => console.log(m);
|
||||
|
||||
// ── 1. TTS both languages ───────────────────────────────────────────────────
|
||||
const CASES = [
|
||||
['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'],
|
||||
['en', 'There are three thousand four hundred and twenty seven new leads.'],
|
||||
];
|
||||
|
||||
const rendered = {};
|
||||
for (const [lang, text] of CASES) {
|
||||
const t0 = Date.now();
|
||||
await synthesize(text, lang); // warm
|
||||
const t1 = Date.now();
|
||||
const out = await synthesize(text, lang);
|
||||
const ms = Date.now() - t1;
|
||||
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
|
||||
rendered[lang] = out;
|
||||
say(`TTS ${lang} warm ${((t1 - t0) / 1000).toFixed(1)}s | ${ms}ms for ${audioMs.toFixed(0)}ms `
|
||||
+ `@${out.sampling_rate}Hz | RTF ${(ms / audioMs).toFixed(2)}x`);
|
||||
}
|
||||
|
||||
// ── 2. VAD: does synthesised speech open and close a turn? ──────────────────
|
||||
function resample(audio, from, to) {
|
||||
if (from === to) return audio;
|
||||
const ratio = from / to;
|
||||
const out = new Float32Array(Math.floor(audio.length / ratio));
|
||||
for (let i = 0; i < out.length; i++) {
|
||||
const p = i * ratio;
|
||||
const a = Math.floor(p);
|
||||
out[i] = audio[a] + (audio[Math.min(a + 1, audio.length - 1)] - audio[a]) * (p - a);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
const ep = new Endpointer();
|
||||
const speech = resample(rendered.en.audio, rendered.en.sampling_rate, 16000);
|
||||
// Speech, then a second of silence so the endpointer closes the turn.
|
||||
const withTail = new Float32Array(speech.length + 16000);
|
||||
withTail.set(speech);
|
||||
|
||||
let started = false;
|
||||
let captured = null;
|
||||
for (let i = 0; i < withTail.length; i += 640) { // 40 ms chunks, as the browser sends
|
||||
const { utterances, started: s } = await ep.push(withTail.subarray(i, Math.min(i + 640, withTail.length)));
|
||||
if (s) started = true;
|
||||
if (utterances.length) { captured = utterances[0]; break; }
|
||||
}
|
||||
say(`VAD speech detected: ${started ? 'yes' : 'NO'} | turn closed: ${captured ? 'yes' : 'NO'}`
|
||||
+ (captured ? ` | captured ${(captured.length / 16000).toFixed(2)}s` : ''));
|
||||
|
||||
// ── 3. STT on that captured audio ───────────────────────────────────────────
|
||||
if (captured) {
|
||||
for (const lang of ['en', 'auto']) {
|
||||
const r = await transcribe(captured, lang, 'en');
|
||||
say(`STT ${lang.padEnd(4)} ${r.ms}ms → lang=${r.lang}${r.detected ? ` (heard ${r.detected})` : ''} → ${JSON.stringify(r.text.slice(0, 70))}`);
|
||||
}
|
||||
}
|
||||
|
||||
// ── 4. Text shaping ─────────────────────────────────────────────────────────
|
||||
const md = '## Leads\n\n**3,427** in `new_lead`.\n\n| a | b |\n|---|---|\n| 1 | 2 |\n\n- Only 12% contacted.\nNext step is triage.';
|
||||
say(`\nspeakable: ${JSON.stringify(speakable(md))}`);
|
||||
say(`sentences: ${JSON.stringify(sentences(speakable(md)))}`);
|
||||
say(`\nRSS ${(process.memoryUsage().rss / 1e9).toFixed(2)} GB`);
|
||||
process.exit(0);
|
||||
Reference in New Issue
Block a user