WeLe Agentic AI with Docker deployment

This commit is contained in:
2026-08-28 08:50:08 +05:30
parent 586059b14f
commit 6bb0ef25ca
34 changed files with 2271 additions and 1343 deletions
+15
View File
@@ -55,3 +55,18 @@ RATE_LIMIT_MAX=40
# --- Artifacts --- # --- Artifacts ---
ARTIFACT_DIR=./storage/artifacts ARTIFACT_DIR=./storage/artifacts
ARTIFACT_TTL_HOURS=72 ARTIFACT_TTL_HOURS=72
# --- Voice (speech-to-speech, CPU, in-process) ---
# Models are ONNX via Transformers.js — no GPU, no Python, no second service.
# STT onnx-community/whisper-base Tamil + English + language detection
# TTS assets/tts/mms-tts-tam exported locally; no public ONNX exists
# TTS Xenova/mms-tts-eng from the Hub
SPEECH_WARMUP=false
STT_MODEL=onnx-community/whisper-base
# Below this confidence, the user's preferred language beats the detector.
DETECT_CONFIDENCE=0.6
# Endpointing
VAD_SILENCE_MS=700
VAD_MIN_SPEECH_MS=250
VAD_PREFIX_MS=300
+3 -4
View File
@@ -10,7 +10,6 @@
.env.* .env.*
!.env.example !.env.example
!**/*.env.example !**/*.env.example
voice-service/.env
*.pem *.pem
*.key *.key
*.p12 *.p12
@@ -28,7 +27,6 @@ pnpm-debug.log*
*.tsbuildinfo *.tsbuildinfo
# ── Python (voice service) ────────────────────────────────────────────────── # ── Python (voice service) ──────────────────────────────────────────────────
voice-service/.venv/
.venv/ .venv/
/venv/ /venv/
/env/ /env/
@@ -50,11 +48,13 @@ __pycache__/
# redistribute models the licence does not allow us to redistribute. # redistribute models the licence does not allow us to redistribute.
# #
# Leading slashes matter: an unanchored `models/` also matches # Leading slashes matter: an unanchored `models/` also matches
# `src/data/models/` — the CRM read-models — which silently kept them out of # `src/data/models/
.transformers-cache/` — the CRM read-models — which silently kept them out of
# the repo and made the container crash with ERR_MODULE_NOT_FOUND. # the repo and made the container crash with ERR_MODULE_NOT_FOUND.
/.cache/ /.cache/
/huggingface/ /huggingface/
/models/ /models/
.transformers-cache/
*.onnx *.onnx
*.safetensors *.safetensors
*.ckpt *.ckpt
@@ -75,7 +75,6 @@ storage/artifacts/*
*.wav *.wav
*.mp3 *.mp3
*.flac *.flac
!voice-service/app/assets/*.wav
# ── Editors / OS ──────────────────────────────────────────────────────────── # ── Editors / OS ────────────────────────────────────────────────────────────
.vscode/* .vscode/*
+3
View File
@@ -0,0 +1,3 @@
{
"<unk>": 58
}
+82
View File
@@ -0,0 +1,82 @@
{
"activation_dropout": 0.1,
"architectures": [
"VitsModel"
],
"attention_dropout": 0.1,
"depth_separable_channels": 2,
"depth_separable_num_layers": 3,
"dtype": "float32",
"duration_predictor_dropout": 0.5,
"duration_predictor_filter_channels": 256,
"duration_predictor_flow_bins": 10,
"duration_predictor_kernel_size": 3,
"duration_predictor_num_flows": 4,
"duration_predictor_tail_bound": 5.0,
"ffn_dim": 768,
"ffn_kernel_size": 3,
"flow_size": 192,
"hidden_act": "relu",
"hidden_dropout": 0.1,
"hidden_size": 192,
"initializer_range": 0.02,
"layer_norm_eps": 1e-05,
"layerdrop": 0.1,
"leaky_relu_slope": 0.1,
"model_type": "vits",
"noise_scale": 0.667,
"noise_scale_duration": 0.8,
"num_attention_heads": 2,
"num_hidden_layers": 6,
"num_speakers": 1,
"posterior_encoder_num_wavenet_layers": 16,
"prior_encoder_num_flows": 4,
"prior_encoder_num_wavenet_layers": 4,
"resblock_dilation_sizes": [
[
1,
3,
5
],
[
1,
3,
5
],
[
1,
3,
5
]
],
"resblock_kernel_sizes": [
3,
7,
11
],
"sampling_rate": 16000,
"speaker_embedding_size": 0,
"speaking_rate": 1.0,
"spectrogram_bins": 513,
"transformers_version": "4.57.3",
"upsample_initial_channel": 512,
"upsample_kernel_sizes": [
16,
16,
4,
4
],
"upsample_rates": [
8,
8,
2,
2
],
"use_bias": true,
"use_stochastic_duration_prediction": true,
"vocab_size": 58,
"wavenet_dilation_rate": 1,
"wavenet_dropout": 0.0,
"wavenet_kernel_size": 5,
"window_size": 4
}
@@ -0,0 +1,4 @@
{
"per_channel": false,
"reduce_range": false
}
@@ -0,0 +1,4 @@
{
"pad_token": "3",
"unk_token": "<unk>"
}
+115
View File
@@ -0,0 +1,115 @@
{
"version": "1.0",
"truncation": null,
"padding": null,
"added_tokens": [
{
"id": 58,
"content": "<unk>",
"single_word": false,
"lstrip": false,
"rstrip": false,
"normalized": false,
"special": true
}
],
"normalizer": {
"type": "Sequence",
"normalizers": [
{
"type": "Lowercase"
},
{
"type": "Replace",
"pattern": {
"Regex": "[^012345679 '_aஅஆஇஈஉஊஎஏஐஒஓகஙசஜஞடணதநனபமயரறலளழவஷஸஹாிீுூெேைொோௌ்]"
},
"content": ""
},
{
"type": "Strip",
"strip_left": true,
"strip_right": true
},
{
"type": "Replace",
"pattern": {
"Regex": "(?=.)|(?<!^)$"
},
"content": "3"
}
]
},
"pre_tokenizer": {
"type": "Split",
"pattern": {
"Regex": ""
},
"behavior": "Isolated",
"invert": false
},
"post_processor": null,
"decoder": null,
"model": {
"vocab": {
"0": 47,
"1": 44,
"2": 23,
"3": 0,
"4": 54,
"5": 57,
"6": 36,
"7": 14,
"9": 31,
" ": 7,
"'": 13,
"_": 4,
"a": 15,
"அ": 1,
"ஆ": 45,
"இ": 38,
"ஈ": 2,
"உ": 3,
"ஊ": 11,
"எ": 37,
"ஏ": 16,
"ஐ": 52,
"ஒ": 27,
"ஓ": 49,
"க": 6,
"ங": 50,
"ச": 30,
"ஜ": 53,
"ஞ": 29,
"ட": 22,
"ண": 48,
"த": 41,
"ந": 5,
"ன": 35,
"ப": 46,
"ம": 26,
"ய": 39,
"ர": 25,
"ற": 28,
"ல": 21,
"ள": 43,
"ழ": 24,
"வ": 17,
"ஷ": 55,
"ஸ": 33,
"ஹ": 19,
"ா": 9,
"ி": 32,
"ீ": 12,
"ு": 51,
"ூ": 20,
"ெ": 10,
"ே": 8,
"ை": 34,
"ொ": 56,
"ோ": 42,
"ௌ": 40,
"்": 18
}
}
}
@@ -0,0 +1,31 @@
{
"add_blank": true,
"added_tokens_decoder": {
"0": {
"content": "3",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"58": {
"content": "<unk>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
}
},
"clean_up_tokenization_spaces": true,
"extra_special_tokens": {},
"is_uroman": false,
"language": "tam",
"model_max_length": 1000000000000000019884624838656,
"normalize": true,
"pad_token": "3",
"phonemize": false,
"tokenizer_class": "VitsTokenizer",
"unk_token": "<unk>"
}
+60
View File
@@ -0,0 +1,60 @@
{
" ": 7,
"'": 13,
"0": 47,
"1": 44,
"2": 23,
"3": 0,
"4": 54,
"5": 57,
"6": 36,
"7": 14,
"9": 31,
"_": 4,
"a": 15,
"அ": 1,
"ஆ": 45,
"இ": 38,
"ஈ": 2,
"உ": 3,
"ஊ": 11,
"எ": 37,
"ஏ": 16,
"ஐ": 52,
"ஒ": 27,
"ஓ": 49,
"க": 6,
"ங": 50,
"ச": 30,
"ஜ": 53,
"ஞ": 29,
"ட": 22,
"ண": 48,
"த": 41,
"ந": 5,
"ன": 35,
"ப": 46,
"ம": 26,
"ய": 39,
"ர": 25,
"ற": 28,
"ல": 21,
"ள": 43,
"ழ": 24,
"வ": 17,
"ஷ": 55,
"ஸ": 33,
"ஹ": 19,
"ா": 9,
"ி": 32,
"ீ": 12,
"ு": 51,
"ூ": 20,
"ெ": 10,
"ே": 8,
"ை": 34,
"ொ": 56,
"ோ": 42,
"ௌ": 40,
"்": 18
}
+955
View File
File diff suppressed because it is too large Load Diff
+3 -2
View File
@@ -7,12 +7,12 @@
"scripts": { "scripts": {
"start": "node src/server.js", "start": "node src/server.js",
"dev": "node --watch src/server.js", "dev": "node --watch src/server.js",
"smoke": "node scripts/smoke.js", "smoke": "node scripts/smoke.js"
"voice": "voice-service/.venv/Scripts/python.exe -m app.server"
}, },
"author": "WeLe EdTech", "author": "WeLe EdTech",
"license": "ISC", "license": "ISC",
"dependencies": { "dependencies": {
"@huggingface/transformers": "^4.2.0",
"@langchain/core": "^1.1.18", "@langchain/core": "^1.1.18",
"@langchain/langgraph": "^1.1.0", "@langchain/langgraph": "^1.1.0",
"@langchain/openai": "^1.5.10", "@langchain/openai": "^1.5.10",
@@ -27,6 +27,7 @@
"ioredis": "^5.10.1", "ioredis": "^5.10.1",
"jsonwebtoken": "^9.0.2", "jsonwebtoken": "^9.0.2",
"mongoose": "^9.2.3", "mongoose": "^9.2.3",
"onnxruntime-node": "^1.24.3",
"pdfkit": "^0.17.2", "pdfkit": "^0.17.2",
"pptxgenjs": "^4.0.1", "pptxgenjs": "^4.0.1",
"uuid": "^13.0.0", "uuid": "^13.0.0",
+66
View File
@@ -0,0 +1,66 @@
/* Generate the tokenizer.json that Transformers.js needs for the exported
Tamil VITS model.
`save_pretrained` does not emit one: VitsTokenizer is a "slow" tokenizer with
no fast counterpart, so Python writes vocab.json + tokenizer_config.json and
nothing else. Transformers.js only reads tokenizer.json, so we synthesise it
from the exported vocab, mirroring the structure of the working English
model (Xenova/mms-tts-eng) exactly.
The four normalizer steps, in order:
1. Lowercase — no-op for Tamil, matters for embedded Latin/digits
2. Replace — drop every character outside the vocab
3. Strip — trim surrounding whitespace
4. Replace — insert the blank token between every character,
which is what `add_blank: true` means for VITS.
Omit this and the audio comes out garbled.
*/
import fs from 'node:fs';
import path from 'node:path';
const DIR = 'assets/tts/mms-tts-tam';
const vocab = JSON.parse(fs.readFileSync(path.join(DIR, 'vocab.json'), 'utf8'));
const cfg = JSON.parse(fs.readFileSync(path.join(DIR, 'tokenizer_config.json'), 'utf8'));
// The blank/pad token is whichever character maps to id 0.
const blank = Object.keys(vocab).find((k) => vocab[k] === 0);
const unk = cfg.unk_token ?? '<unk>';
const unkId = vocab[unk] ?? Object.keys(vocab).length;
// Character class of everything we keep. Escape the regex metacharacters that
// are still special inside a negated class.
const escaped = Object.keys(vocab)
.filter((c) => c !== unk)
.map((c) => (']\\^-'.includes(c) ? '\\' + c : c))
.join('');
const tokenizer = {
version: '1.0',
truncation: null,
padding: null,
added_tokens: [{
id: unkId, content: unk,
single_word: false, lstrip: false, rstrip: false, normalized: false, special: true,
}],
normalizer: {
type: 'Sequence',
normalizers: [
{ type: 'Lowercase' },
{ type: 'Replace', pattern: { Regex: `[^${escaped}]` }, content: '' },
{ type: 'Strip', strip_left: true, strip_right: true },
...(cfg.add_blank ? [{ type: 'Replace', pattern: { Regex: '(?=.)|(?<!^)$' }, content: blank }] : []),
],
},
pre_tokenizer: { type: 'Split', pattern: { Regex: '' }, behavior: 'Isolated', invert: false },
post_processor: null,
decoder: null,
model: { vocab },
};
fs.writeFileSync(path.join(DIR, 'tokenizer.json'), JSON.stringify(tokenizer, null, 1));
console.log(`wrote ${DIR}/tokenizer.json`);
console.log(` vocab ${Object.keys(vocab).length} tokens`);
console.log(` blank token ${JSON.stringify(blank)} (id 0)`);
console.log(` unk ${JSON.stringify(unk)} (id ${unkId})`);
console.log(` add_blank ${cfg.add_blank}`);
+99
View File
@@ -0,0 +1,99 @@
"""One-time export of facebook/mms-tts-tam to ONNX.
No public ONNX build of Tamil MMS-TTS exists, so we make one. This runs ONCE on
a workstation; the committed artefact is what ships. The service itself is pure
JavaScript and never needs Python or this script.
The output must match the contract Transformers.js expects for VITS, taken from
the working English model (Xenova/mms-tts-eng):
inputs : input_ids, attention_mask
outputs: waveform, spectrogram
Layout produced (mirrors the HF repo so Transformers.js can load the folder):
assets/tts/mms-tts-tam/
config.json, tokenizer.json, vocab.json, …
onnx/model.onnx fp32
onnx/model_quantized.onnx int8 ← what we ship
"""
from __future__ import annotations
import json
import shutil
import sys
from pathlib import Path
import torch
from transformers import AutoTokenizer, VitsModel
MODEL = "facebook/mms-tts-tam"
OUT = Path("assets/tts/mms-tts-tam")
ONNX_DIR = OUT / "onnx"
ONNX_DIR.mkdir(parents=True, exist_ok=True)
print(f"loading {MODEL} …")
model = VitsModel.from_pretrained(MODEL).eval()
tok = AutoTokenizer.from_pretrained(MODEL)
# VITS has a stochastic duration predictor. Exporting with noise left on bakes
# RandomNormalLike nodes into the graph, which is fine and keeps prosody
# natural — but seed it so this export is reproducible.
torch.manual_seed(0)
class Exportable(torch.nn.Module):
"""Return only (waveform, spectrogram), in that order — the JS side indexes
outputs by name, but a plain tuple keeps the exported graph simple."""
def __init__(self, m: VitsModel) -> None:
super().__init__()
self.m = m
def forward(self, input_ids: torch.Tensor, attention_mask: torch.Tensor):
out = self.m(input_ids=input_ids, attention_mask=attention_mask)
return out.waveform, out.spectrogram
sample = tok("வணக்கம், இது ஒரு சோதனை.", return_tensors="pt")
fp32 = ONNX_DIR / "model.onnx"
print("exporting to ONNX …")
torch.onnx.export(
Exportable(model),
(sample["input_ids"], sample["attention_mask"]),
str(fp32),
input_names=["input_ids", "attention_mask"],
output_names=["waveform", "spectrogram"],
dynamic_axes={
"input_ids": {0: "batch", 1: "sequence"},
"attention_mask": {0: "batch", 1: "sequence"},
"waveform": {0: "batch", 1: "samples"},
"spectrogram": {0: "batch", 2: "frames"},
},
opset_version=17,
do_constant_folding=True,
)
print(f" fp32: {fp32.stat().st_size / 1e6:.1f} MB")
# ── int8 ────────────────────────────────────────────────────────────────────
try:
from onnxruntime.quantization import QuantType, quantize_dynamic
q = ONNX_DIR / "model_quantized.onnx"
quantize_dynamic(str(fp32), str(q), weight_type=QuantType.QUInt8)
print(f" int8: {q.stat().st_size / 1e6:.1f} MB")
except Exception as e: # noqa: BLE001
print(f" quantisation skipped: {e}")
# ── tokenizer + config, so the folder loads standalone ──────────────────────
tok.save_pretrained(OUT)
model.config.to_json_file(OUT / "config.json")
# Transformers.js reads this to pick a default dtype.
(OUT / "quantize_config.json").write_text(json.dumps({"per_channel": False, "reduce_range": False}, indent=2))
print("\nwrote:")
for p in sorted(OUT.rglob("*")):
if p.is_file():
print(f" {p.relative_to(OUT)} ({p.stat().st_size / 1e6:.2f} MB)")
+21
View File
@@ -0,0 +1,21 @@
/* Does auto mode now route Tamil to Tamil instead of silently using English? */
import { synthesize, transcribe } from '../src/speech/index.js';
const resample = (a, from, to) => {
const r = from / to, out = new Float32Array(Math.floor(a.length / r));
for (let i = 0; i < out.length; i++) { const p = i * r, k = Math.floor(p); out[i] = a[k] + (a[Math.min(k + 1, a.length - 1)] - a[k]) * (p - k); }
return out;
};
for (const [lang, text] of [
['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'],
['en', 'How many new leads did we receive today?'],
]) {
const spoken = await synthesize(text, lang);
const audio = resample(spoken.audio, spoken.sampling_rate, 16000);
const r = await transcribe(audio, 'auto', 'ta');
const ok = r.lang === lang ? 'PASS' : 'FAIL';
console.log(`${ok} spoke ${lang} → routed ${r.lang} (detected ${r.detected} @ ${r.confidence}) ${r.ms}ms`);
console.log(` ${JSON.stringify(r.text.slice(0, 80))}`);
}
process.exit(0);
+62
View File
@@ -0,0 +1,62 @@
/* Can we get real language detection out of Whisper in Transformers.js? */
import { AutoProcessor, WhisperForConditionalGeneration, Tensor, env } from '@huggingface/transformers';
import { synthesize } from '../src/speech/index.js';
env.cacheDir = './.transformers-cache';
const MODEL = 'onnx-community/whisper-base';
const processor = await AutoProcessor.from_pretrained(MODEL);
const model = await WhisperForConditionalGeneration.from_pretrained(MODEL, { dtype: 'q8' });
const tok = processor.tokenizer;
// Whisper emits one language token right after <|startoftranscript|>. Reading
// that distribution is a single decoder step — far cheaper than transcribing
// twice to see which language "looks better".
const id = (t) => tok.encode(t, { add_special_tokens: false })[0];
const sot = id('<|startoftranscript|>');
const CANDIDATES = ['en', 'ta'];
const langIds = CANDIDATES.map((c) => id(`<|${c}|>`));
console.log('sot:', sot, '| language token ids:', JSON.stringify(Object.fromEntries(CANDIDATES.map((c, i) => [c, langIds[i]]))));
function resample(a, from, to) {
const r = from / to;
const out = new Float32Array(Math.floor(a.length / r));
for (let i = 0; i < out.length; i++) {
const p = i * r, k = Math.floor(p);
out[i] = a[k] + (a[Math.min(k + 1, a.length - 1)] - a[k]) * (p - k);
}
return out;
}
for (const [lang, text] of [
['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'],
['en', 'There are three thousand four hundred and twenty seven new leads.'],
]) {
const spoken = await synthesize(text, lang);
const audio = resample(spoken.audio, spoken.sampling_rate, 16000);
const inputs = await processor(audio);
const t0 = Date.now();
const out = await model({
...inputs,
decoder_input_ids: new Tensor('int64', BigInt64Array.from([BigInt(sot)]), [1, 1]),
});
const ms = Date.now() - t0;
const logits = out.logits;
const last = logits.dims[1] - 1;
const vocab = logits.dims[2];
const row = logits.data.slice(last * vocab, (last + 1) * vocab);
const scores = langIds.map((id) => Number(row[id]));
const max = Math.max(...scores);
const exp = scores.map((s) => Math.exp(s - max));
const sum = exp.reduce((a, b) => a + b, 0);
const probs = exp.map((e) => e / sum);
const best = probs.indexOf(Math.max(...probs));
console.log(`spoken ${lang} → detected ${CANDIDATES[best]} `
+ `(${CANDIDATES.map((c, i) => `${c} ${probs[i].toFixed(3)}`).join(', ')}) in ${ms}ms `
+ `${CANDIDATES[best] === lang ? '✅' : '❌'}`);
}
process.exit(0);
+44
View File
@@ -0,0 +1,44 @@
/* Do Whisper (STT) and MMS-TTS (TTS) actually run in Node on CPU? */
import { pipeline, env } from '@huggingface/transformers';
env.cacheDir = './.transformers-cache';
const t = (t0) => `${((performance.now() - t0) / 1000).toFixed(1)}s`;
// ── TTS: MMS-TTS Tamil (VITS, 36M, feed-forward) ────────────────────────────
console.log('[1/2] loading MMS-TTS Tamil…');
let t0 = performance.now();
const tts = await pipeline('text-to-speech', 'Xenova/mms-tts-eng', { dtype: 'fp32' });
console.log(` loaded in ${t(t0)}`);
const TA = 'There are three thousand four hundred leads in the new lead stage.';
await tts(TA); // warm
for (const [label, text] of [['short', TA], ['long', TA + ' ' + TA + ' ' + TA]]) {
t0 = performance.now();
const out = await tts(text);
const ms = performance.now() - t0;
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
console.log(
` ${label.padEnd(5)} ${String(text.length).padStart(3)} chars → ${ms.toFixed(0)}ms `
+ `for ${audioMs.toFixed(0)}ms audio @ ${out.sampling_rate}Hz → RTF ${(ms / audioMs).toFixed(2)}x`,
);
}
// ── STT: Whisper (multilingual — Tamil, Hindi, English + detection) ─────────
console.log('\n[2/2] loading Whisper base…');
t0 = performance.now();
const stt = await pipeline('automatic-speech-recognition', 'onnx-community/whisper-base', { dtype: 'q8' });
console.log(` loaded in ${t(t0)}`);
// 4 s of quiet noise — proves the graph runs and times it.
const audio = Float32Array.from({ length: 16000 * 4 }, () => (Math.random() - 0.5) * 0.02);
t0 = performance.now();
const r = await stt(audio, { language: 'ta', task: 'transcribe' });
console.log(` 4000ms audio → ${(performance.now() - t0).toFixed(0)}ms → ${JSON.stringify(r.text).slice(0, 60)}`);
t0 = performance.now();
const r2 = await stt(audio, { language: 'en', task: 'transcribe' });
console.log(` english pass → ${(performance.now() - t0).toFixed(0)}ms → ${JSON.stringify(r2.text).slice(0, 60)}`);
console.log(`\nRSS ${(process.memoryUsage().rss / 1e9).toFixed(2)} GB`);
process.exit(0);
+55
View File
@@ -0,0 +1,55 @@
/* Does the locally-exported Tamil ONNX load and speak through Transformers.js? */
import { pipeline, env } from '@huggingface/transformers';
import fs from 'node:fs';
// Load from the local folder, not the Hub.
env.allowRemoteModels = false;
env.localModelPath = './assets/tts';
for (const dtype of ['q8', 'fp32']) {
try {
const t0 = performance.now();
const tts = await pipeline('text-to-speech', 'mms-tts-tam', { dtype });
const load = performance.now() - t0;
const TEXT = 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.';
await tts(TEXT); // warm
const t1 = performance.now();
const out = await tts(TEXT);
const ms = performance.now() - t1;
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
console.log(
`${dtype.padEnd(5)} load ${(load / 1000).toFixed(1)}s | `
+ `${ms.toFixed(0)}ms for ${audioMs.toFixed(0)}ms audio @ ${out.sampling_rate}Hz | `
+ `RTF ${(ms / audioMs).toFixed(2)}x`,
);
// Non-silent output is the real proof the graph is wired correctly.
const peak = out.audio.reduce((m, v) => Math.max(m, Math.abs(v)), 0);
console.log(` samples ${out.audio.length}, peak amplitude ${peak.toFixed(3)} ${peak > 0.01 ? '✅ audible' : '⚠️ SILENT'}`);
if (dtype === 'q8') {
const wav = toWav(out.audio, out.sampling_rate);
fs.writeFileSync('scripts/tamil-sample.wav', wav);
console.log(' wrote scripts/tamil-sample.wav — play it to judge quality');
}
} catch (e) {
console.log(`${dtype.padEnd(5)} FAILED: ${e.message.slice(0, 160)}`);
}
}
function toWav(samples, rate) {
const buf = Buffer.alloc(44 + samples.length * 2);
buf.write('RIFF', 0); buf.writeUInt32LE(36 + samples.length * 2, 4); buf.write('WAVE', 8);
buf.write('fmt ', 12); buf.writeUInt32LE(16, 16); buf.writeUInt16LE(1, 20); buf.writeUInt16LE(1, 22);
buf.writeUInt32LE(rate, 24); buf.writeUInt32LE(rate * 2, 28); buf.writeUInt16LE(2, 32); buf.writeUInt16LE(16, 34);
buf.write('data', 36); buf.writeUInt32LE(samples.length * 2, 40);
for (let i = 0; i < samples.length; i++) {
const s = Math.max(-1, Math.min(1, samples[i]));
buf.writeInt16LE(s < 0 ? s * 0x8000 : s * 0x7fff, 44 + i * 2);
}
return buf;
}
process.exit(0);
+72
View File
@@ -0,0 +1,72 @@
/* End-to-end check of the in-process speech pipeline: TTS → VAD → STT.
Synthesising a sentence and feeding that audio back through the endpointer
and recogniser exercises every stage with real speech, which a noise buffer
cannot do — silence never opens a VAD turn.
*/
import { synthesize, transcribe, sentences, speakable, Endpointer } from '../src/speech/index.js';
const say = (m) => console.log(m);
// ── 1. TTS both languages ───────────────────────────────────────────────────
const CASES = [
['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'],
['en', 'There are three thousand four hundred and twenty seven new leads.'],
];
const rendered = {};
for (const [lang, text] of CASES) {
const t0 = Date.now();
await synthesize(text, lang); // warm
const t1 = Date.now();
const out = await synthesize(text, lang);
const ms = Date.now() - t1;
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
rendered[lang] = out;
say(`TTS ${lang} warm ${((t1 - t0) / 1000).toFixed(1)}s | ${ms}ms for ${audioMs.toFixed(0)}ms `
+ `@${out.sampling_rate}Hz | RTF ${(ms / audioMs).toFixed(2)}x`);
}
// ── 2. VAD: does synthesised speech open and close a turn? ──────────────────
function resample(audio, from, to) {
if (from === to) return audio;
const ratio = from / to;
const out = new Float32Array(Math.floor(audio.length / ratio));
for (let i = 0; i < out.length; i++) {
const p = i * ratio;
const a = Math.floor(p);
out[i] = audio[a] + (audio[Math.min(a + 1, audio.length - 1)] - audio[a]) * (p - a);
}
return out;
}
const ep = new Endpointer();
const speech = resample(rendered.en.audio, rendered.en.sampling_rate, 16000);
// Speech, then a second of silence so the endpointer closes the turn.
const withTail = new Float32Array(speech.length + 16000);
withTail.set(speech);
let started = false;
let captured = null;
for (let i = 0; i < withTail.length; i += 640) { // 40 ms chunks, as the browser sends
const { utterances, started: s } = await ep.push(withTail.subarray(i, Math.min(i + 640, withTail.length)));
if (s) started = true;
if (utterances.length) { captured = utterances[0]; break; }
}
say(`VAD speech detected: ${started ? 'yes' : 'NO'} | turn closed: ${captured ? 'yes' : 'NO'}`
+ (captured ? ` | captured ${(captured.length / 16000).toFixed(2)}s` : ''));
// ── 3. STT on that captured audio ───────────────────────────────────────────
if (captured) {
for (const lang of ['en', 'auto']) {
const r = await transcribe(captured, lang, 'en');
say(`STT ${lang.padEnd(4)} ${r.ms}ms → lang=${r.lang}${r.detected ? ` (heard ${r.detected})` : ''} → ${JSON.stringify(r.text.slice(0, 70))}`);
}
}
// ── 4. Text shaping ─────────────────────────────────────────────────────────
const md = '## Leads\n\n**3,427** in `new_lead`.\n\n| a | b |\n|---|---|\n| 1 | 2 |\n\n- Only 12% contacted.\nNext step is triage.';
say(`\nspeakable: ${JSON.stringify(speakable(md))}`);
say(`sentences: ${JSON.stringify(sentences(speakable(md)))}`);
say(`\nRSS ${(process.memoryUsage().rss / 1e9).toFixed(2)} GB`);
process.exit(0);
+122 -169
View File
@@ -1,104 +1,59 @@
// ============================================ // ============================================
// Voice channel — a WebSocket bridge between the browser and the GPU service. // Voice channel — speech in, speech out, in this same Node process.
// //
// Voice is a *channel*, not a parallel product: a spoken question runs through // Voice is a *channel*, not a parallel product: a spoken question runs through
// the same graph, guardrails and agents as a typed one. Only the transport and // the same graph, guardrails and agents as a typed one. Only the transport and
// the presentation differ, which is why this file contains no CRM logic. // the presentation differ, which is why this file contains no CRM logic.
// //
// browser ──audio──► this ──audio──► python(:4100) ──STT──► transcript // browser ──PCM16──► Endpointer ──► transcribe() ──► runTurn(graph)
// │ │ // │ │
// └──────────── runTurn(graph) ◄───────────┘ // browser ◄──float32──── synthesize() ◄── sentences ◄─────┘
// │
// browser ◄──audio── this ◄──audio── python(TTS) ◄──sentences───┘
// //
// The hard problem is not transport, it is that a turn takes 17–46 s. Silence // The hard problem is not audio, it is that a turn takes 17–46 s and that is
// for that long feels broken, so the bridge speaks immediately, narrates what // dead silence in voice. So this speaks an acknowledgement within ~1 s,
// the agents are doing, and starts reading the answer at the first sentence // narrates each agent delegation aloud, then reads the answer sentence by
// rather than waiting for the last. // sentence as it is composed.
// ============================================ // ============================================
import { WebSocketServer, WebSocket } from 'ws'; import { WebSocketServer, WebSocket } from 'ws';
import { randomUUID } from 'node:crypto';
import { principalFromToken } from './auth.js'; import { principalFromToken } from './auth.js';
import { runTurn } from '../orchestration/runner.js'; import { runTurn } from '../orchestration/runner.js';
import {
Endpointer, transcribe, synthesize, sentences, speakable, LANGUAGES,
} from '../speech/index.js';
import config from '../config/index.js'; import config from '../config/index.js';
import logger from '../utils/logger.js'; import logger from '../utils/logger.js';
const VOICE_URL = process.env.VOICE_SERVICE_URL || 'ws://127.0.0.1:4100/ws/voice'; /** Spoken filler, said the instant a question lands. */
/** Spoken filler, per language. Said the instant a question lands. */
const ACK = { const ACK = {
ta: ['பார்க்கிறேன்...', 'ஒரு நிமிடம், பார்க்கிறேன்.'], ta: ['பார்க்கிறேன்.', 'ஒரு நிமிடம், பார்க்கிறேன்.'],
hi: ['देखता हूँ...', 'एक मिनट, देख रहा हूँ।'],
te: ['చూస్తున్నాను...'],
kn: ['ನೋಡುತ್ತಿದ್ದೇನೆ...'],
ml: ['നോക്കുന്നു...'],
mr: ['बघतो...'],
bn: ['দেখছি...'],
en: ['Let me check.', 'One moment, checking now.'], en: ['Let me check.', 'One moment, checking now.'],
}; };
/** Progress narration, kept short — it is spoken over the user's waiting time. */ /** Progress narration — spoken over the user's waiting time, so keep it short. */
const NARRATE = { const NARRATE = {
ta: { lead: 'லீட் விவரங்களைப் பார்க்கிறேன்.', analytics: 'புள்ளிவிவரங்களைச் சரிபார்க்கிறேன்.', conversation: 'உரையாடல்களைப் பார்க்கிறேன்.', default: 'தரவைச் சரிபார்க்கிறேன்.' }, ta: { lead: 'லீட் விவரங்களைப் பார்க்கிறேன்.', analytics: 'புள்ளிவிவரங்களைச் சரிபார்க்கிறேன்.', conversation: 'உரையாடல்களைப் பார்க்கிறேன்.', default: 'தரவைச் சரிபார்க்கிறேன்.' },
hi: { lead: 'लीड्स देख रहा हूँ।', analytics: 'आँकड़े देख रहा हूँ।', conversation: 'बातचीत देख रहा हूँ।', default: 'डेटा देख रहा हूँ।' },
en: { lead: 'Checking the leads.', analytics: 'Pulling the numbers.', conversation: 'Looking at the conversations.', default: 'Checking the data.' }, en: { lead: 'Checking the leads.', analytics: 'Pulling the numbers.', conversation: 'Looking at the conversations.', default: 'Checking the data.' },
}; };
const pick = (arr) => arr[Math.floor(Math.random() * arr.length)]; const NOTHING = { ta: 'பதில் கிடைக்கவில்லை.', en: 'I could not find an answer for that.' };
const OOPS = { ta: 'மன்னிக்கவும், ஒரு பிழை ஏற்பட்டது.', en: 'Sorry, something went wrong.' };
function ackFor(lang) { const pick = (a) => a[Math.floor(Math.random() * a.length)];
return pick(ACK[lang] || ACK.en); const ackFor = (l) => pick(ACK[l] || ACK.en);
} const narrateFor = (l, agent) => (NARRATE[l] || NARRATE.en)[agent] || (NARRATE[l] || NARRATE.en).default;
function narrationFor(lang, agent) { class VoiceSession {
const set = NARRATE[lang] || NARRATE.en;
return set[agent] || set.default;
}
/**
* Strip block-oriented markdown before speaking. Tables and code read terribly
* aloud, and the visual blocks are already on screen.
*/
export function speakable(markdown = '') {
return markdown
.replace(/```[\s\S]*?```/g, ' ')
.replace(/^\s*\|.*\|\s*$/gm, ' ') // table rows
.replace(/^\s*[-*]\s+/gm, '') // bullets
.replace(/^#{1,6}\s*/gm, '') // headings
.replace(/\*\*([^*]+)\*\*/g, '$1')
.replace(/`([^`]+)`/g, '$1')
.replace(/\[([^\]]+)\]\([^)]+\)/g, '$1')
.replace(/₹\s?([\d,.]+)/g, 'rupees $1')
.replace(/\s{2,}/g, ' ')
.trim();
}
/** Split into sentences so speech can start before the answer is finished. */
export function sentences(text, max = 240) {
const out = [];
for (const raw of text.split(/(?<=[.!?।])\s+/)) {
let s = raw.trim();
if (!s) continue;
while (s.length > max) {
const cut = s.lastIndexOf(' ', max);
out.push(s.slice(0, cut > 0 ? cut : max).trim());
s = s.slice(cut > 0 ? cut : max).trim();
}
if (s) out.push(s);
}
return out;
}
class VoiceBridge {
constructor(client, user) { constructor(client, user) {
this.client = client; this.client = client;
this.user = user; this.user = user;
this.lang = 'auto'; // what the user chose this.lang = 'auto'; // what the user selected
this.replyLang = 'ta'; // what the last utterance actually was this.replyLang = 'ta'; // what the last utterance actually was
this.prefer = 'ta'; // tiebreak when detection is unusable
this.sessionId = `voice:${Date.now().toString(36)}:${Math.random().toString(36).slice(2, 8)}`; this.sessionId = `voice:${Date.now().toString(36)}:${Math.random().toString(36).slice(2, 8)}`;
this.gpu = null; this.endpointer = new Endpointer();
this.busy = false; this.busy = false;
this.abort = null; this.abort = null;
this.speakSeq = 0; // rising token; stale synthesis is discarded
this.narrated = new Set(); this.narrated = new Set();
} }
@@ -106,109 +61,104 @@ class VoiceBridge {
if (this.client.readyState === WebSocket.OPEN) this.client.send(JSON.stringify(obj)); if (this.client.readyState === WebSocket.OPEN) this.client.send(JSON.stringify(obj));
} }
toGpu(obj) { sendAudio(buf) {
if (this.gpu?.readyState === WebSocket.OPEN) this.gpu.send(JSON.stringify(obj)); if (this.client.readyState === WebSocket.OPEN) this.client.send(buf, { binary: true });
} }
async connect() { // ── microphone ───────────────────────────────────────────────────────────
this.gpu = new WebSocket(VOICE_URL); async onAudio(data) {
// Browser sends 16 kHz mono PCM16; the models want float32 in [-1, 1].
const pcm16 = new Int16Array(data.buffer, data.byteOffset, Math.floor(data.byteLength / 2));
const pcm = new Float32Array(pcm16.length);
for (let i = 0; i < pcm16.length; i++) pcm[i] = pcm16[i] / 32768;
this.gpu.on('open', () => { let result;
logger.info(`🎙️ voice session ${this.sessionId} → GPU service`); try {
this.toGpu({ type: 'config', lang: this.lang }); result = await this.endpointer.push(pcm);
}); } catch (e) {
logger.error(`VAD failed: ${e.message}`);
this.gpu.on('message', (data, isBinary) => { this.send({ type: 'error', message: 'Voice input failed to initialise. Check the server logs.' });
// TTS audio: pass straight through, no re-encoding.
if (isBinary) {
if (this.client.readyState === WebSocket.OPEN) this.client.send(data, { binary: true });
return; return;
} }
let msg;
try { msg = JSON.parse(data.toString()); } catch { return; }
this.onGpuMessage(msg);
});
this.gpu.on('error', (err) => { if (result.started) {
logger.error(`voice GPU service: ${err.message}`); // Barge-in: the user talking wins immediately. Bumping the token drops
this.send({ type: 'error', message: 'The voice service is not reachable. Start it with: npm run voice' }); // any in-flight synthesis rather than letting it arrive late.
}); this.speakSeq++;
this.gpu.on('close', () => {
this.send({ type: 'voice_service_closed' });
this.client.close();
});
}
onGpuMessage(msg) {
switch (msg.type) {
case 'ready':
this.send({ type: 'ready', session_id: this.sessionId, languages: msg.languages, sample_rate_out: msg.sample_rate_out });
break;
case 'speech_start':
// The user started talking — the GPU service already stopped speaking.
// Tell the browser to dump whatever is still in its playback buffer,
// and abandon any answer still being composed.
this.send({ type: 'barge_in' });
this.abort?.abort(); this.abort?.abort();
break; this.send({ type: 'barge_in' });
}
case 'transcript': for (const utterance of result.utterances) {
// Answer in the language the person actually spoke, not the menu await this.handleUtterance(utterance);
// setting — that is the whole point of auto mode. }
if (msg.lang) this.replyLang = msg.lang; }
this.send({ type: 'transcript', text: msg.text, lang: msg.lang, detected: msg.detected, confidence: msg.confidence, ms: msg.ms });
this.handleQuestion(msg.text);
break;
case 'transcript_empty': async handleUtterance(audio) {
let heard;
try {
heard = await transcribe(audio, this.lang, this.prefer);
} catch (e) {
logger.error(`STT failed: ${e.message}`);
this.send({ type: 'error', message: 'Could not transcribe that. Try again.' });
return;
}
if (!heard.text) {
this.send({ type: 'heard_nothing' }); this.send({ type: 'heard_nothing' });
break; return;
case 'audio_start':
case 'audio_end':
case 'error':
this.send(msg);
break;
default:
break;
}
} }
speak(text, id = randomUUID()) { this.replyLang = heard.lang;
this.send({ type: 'transcript', text: heard.text, lang: heard.lang, detected: heard.detected, ms: heard.ms });
await this.answer(heard.text);
}
// ── speaking ─────────────────────────────────────────────────────────────
/** Synthesise and stream one piece, unless a newer turn has superseded it. */
async say(text, seq) {
const clean = speakable(text); const clean = speakable(text);
if (clean) this.toGpu({ type: 'speak', text: clean, id, lang: this.replyLang }); if (!clean || seq !== this.speakSeq) return;
try {
const out = await synthesize(clean, this.replyLang);
if (!out || seq !== this.speakSeq) return; // interrupted while generating
this.send({ type: 'audio_start', sample_rate: out.sampling_rate });
// Float32 straight down the socket — the playback worklet takes it as-is.
this.sendAudio(Buffer.from(out.audio.buffer, out.audio.byteOffset, out.audio.byteLength));
this.send({ type: 'audio_end' });
} catch (e) {
logger.error(`TTS failed: ${e.message}`);
}
} }
async handleQuestion(text) { async answer(question) {
if (!text?.trim()) return;
if (this.busy) return; // one turn at a time if (this.busy) return; // one turn at a time
this.busy = true; this.busy = true;
this.narrated.clear(); this.narrated.clear();
this.abort = new AbortController(); this.abort = new AbortController();
const seq = ++this.speakSeq;
// 1. Answer the silence immediately. This is the whole trick: the pipeline // Answer the silence immediately. The pipeline still takes 17–46 s, but
// still takes 17–46 s, but the user hears a response in ~1 s. // the user hears a response in about a second.
this.speak(ackFor(this.replyLang)); this.say(ackFor(this.replyLang), seq);
this.send({ type: 'thinking' }); this.send({ type: 'thinking' });
try { try {
const result = await runTurn({ const result = await runTurn({
sessionId: this.sessionId, sessionId: this.sessionId,
message: text, message: question,
user: this.user, user: this.user,
channel: 'crm_chat', // voice users are staff; full tool access channel: 'crm_chat', // voice users are staff
signal: this.abort.signal, signal: this.abort.signal,
onEvent: (ev) => { onEvent: (ev) => {
this.send(ev); this.send(ev);
// 2. Narrate delegations — but only once per agent, or it chatters. // Narrate delegations, once per agent, or it chatters.
if (ev.type === 'step' && ev.kind === 'delegate') { if (ev.type === 'step' && ev.kind === 'delegate') {
const agent = String(ev.label || '').toLowerCase().split(' ')[0]; const agent = String(ev.label || '').toLowerCase().split(' ')[0];
if (!this.narrated.has(agent)) { if (!this.narrated.has(agent)) {
this.narrated.add(agent); this.narrated.add(agent);
this.speak(narrationFor(this.replyLang, agent)); this.say(narrateFor(this.replyLang, agent), seq);
} }
} }
}, },
@@ -216,23 +166,24 @@ class VoiceBridge {
this.send({ type: 'result', blocks: result.blocks, usage: result.usage }); this.send({ type: 'result', blocks: result.blocks, usage: result.usage });
// 3. Read the answer. Sentence at a time so speech starts sooner and can const answer = (result.blocks || [])
// be cut cleanly if the user interrupts. .filter((b) => b.type === 'text').map((b) => b.markdown).join(' ') || result.answer || '';
const answer = result.blocks?.filter((b) => b.type === 'text').map((b) => b.markdown).join(' ')
|| result.answer || '';
const parts = sentences(speakable(answer)); const parts = sentences(speakable(answer));
if (!parts.length) { if (!parts.length) {
this.speak(this.replyLang === 'ta' ? 'பதில் கிடைக்கவில்லை.' : 'I could not find an answer for that.'); await this.say(NOTHING[this.replyLang] || NOTHING.en, seq);
} else { } else {
// Sequential on purpose: parallel synthesis would race to the socket
// and play the answer out of order.
for (const part of parts) { for (const part of parts) {
if (this.abort.signal.aborted) break; if (seq !== this.speakSeq || this.abort.signal.aborted) break;
this.speak(part); await this.say(part, seq);
} }
} }
} catch (err) { } catch (err) {
if (err?.name !== 'AbortError') { if (err?.name !== 'AbortError') {
logger.error(`voice turn failed: ${err.message}`); logger.error(`voice turn failed: ${err.message}`);
this.speak(this.replyLang === 'ta' ? 'மன்னிக்கவும், ஒரு பிழை ஏற்பட்டது.' : 'Sorry, something went wrong.'); await this.say(OOPS[this.replyLang] || OOPS.en, seq);
} }
} finally { } finally {
this.busy = false; this.busy = false;
@@ -240,32 +191,33 @@ class VoiceBridge {
} }
} }
onClientMessage(data, isBinary) { // ── control ──────────────────────────────────────────────────────────────
if (isBinary) { onMessage(data, isBinary) {
if (this.gpu?.readyState === WebSocket.OPEN) this.gpu.send(data, { binary: true }); if (isBinary) return this.onAudio(data);
return;
}
let msg; let msg;
try { msg = JSON.parse(data.toString()); } catch { return; } try { msg = JSON.parse(data.toString()); } catch { return undefined; }
if (msg.type === 'config' && msg.lang) { if (msg.type === 'config' && msg.lang) {
this.lang = msg.lang; this.lang = msg.lang;
if (msg.lang !== 'auto') this.replyLang = msg.lang; if (msg.lang !== 'auto') this.replyLang = this.prefer = msg.lang;
this.toGpu({ type: 'config', lang: msg.lang }); else if (msg.prefer) this.prefer = msg.prefer;
this.send({ type: 'config_ok', lang: msg.lang }); this.endpointer.reset();
this.send({ type: 'config_ok', lang: this.lang, prefer: this.prefer });
} else if (msg.type === 'cancel') { } else if (msg.type === 'cancel') {
this.speakSeq++;
this.abort?.abort(); this.abort?.abort();
this.toGpu({ type: 'cancel' }); this.send({ type: 'cancelled' });
} else if (msg.type === 'text') { } else if (msg.type === 'text' && msg.text) {
// Typed question while in voice mode — answered aloud like a spoken one.
this.send({ type: 'transcript', text: msg.text, lang: this.replyLang, typed: true }); this.send({ type: 'transcript', text: msg.text, lang: this.replyLang, typed: true });
this.handleQuestion(msg.text); this.answer(msg.text);
} }
return undefined;
} }
close() { close() {
this.speakSeq++;
this.abort?.abort(); this.abort?.abort();
try { this.gpu?.close(); } catch { /* already gone */ }
} }
} }
@@ -278,9 +230,8 @@ export function attachVoice(server) {
if (url.pathname !== '/api/agent/voice') return; // leave other upgrades alone if (url.pathname !== '/api/agent/voice') return; // leave other upgrades alone
// Browsers cannot set headers on a WebSocket, so the CRM token arrives as // Browsers cannot set headers on a WebSocket, so the CRM token arrives as
// a query parameter. It is the same token and the same verification. // a query parameter. Same token, same verification as every other route.
const token = url.searchParams.get('token'); const user = await principalFromToken(url.searchParams.get('token')).catch(() => null);
const user = await principalFromToken(token).catch(() => null);
if (!user) { if (!user) {
socket.write('HTTP/1.1 401 Unauthorized\r\n\r\n'); socket.write('HTTP/1.1 401 Unauthorized\r\n\r\n');
socket.destroy(); socket.destroy();
@@ -288,14 +239,16 @@ export function attachVoice(server) {
} }
wss.handleUpgrade(req, socket, head, (client) => { wss.handleUpgrade(req, socket, head, (client) => {
const bridge = new VoiceBridge(client, user); const session = new VoiceSession(client, user);
bridge.connect(); logger.info(`🎙️ voice session ${session.sessionId} (${user.name})`);
client.on('message', (d, bin) => bridge.onClientMessage(d, bin)); session.send({ type: 'ready', session_id: session.sessionId, languages: LANGUAGES });
client.on('close', () => bridge.close());
client.on('error', () => bridge.close()); client.on('message', (d, bin) => session.onMessage(d, bin));
client.on('close', () => session.close());
client.on('error', () => session.close());
}); });
}); });
logger.info(` voice =ws://localhost:${config.port}/api/agent/voice → ${VOICE_URL}`); logger.info(` voice =ws://localhost:${config.port}/api/agent/voice (in-process, CPU)`);
return wss; return wss;
} }
+7
View File
@@ -17,6 +17,7 @@ import { ensureDir, sweep } from './output/artifactStore.js';
import crmApi from './tools/http/crmApi.js'; import crmApi from './tools/http/crmApi.js';
import { describeChains } from './orchestration/llm.js'; import { describeChains } from './orchestration/llm.js';
import { attachVoice } from './gateway/voice.js'; import { attachVoice } from './gateway/voice.js';
import { warmup as warmSpeech, speechStatus } from './speech/index.js';
const app = express(); const app = express();
@@ -40,6 +41,7 @@ app.get('/health', async (_req, res) => {
redis: redisOk ? redisMode() : 'unavailable', redis: redisOk ? redisMode() : 'unavailable',
crm_api: crm.reachable ? 'reachable' : `unreachable (${crm.error || crm.status})`, crm_api: crm.reachable ? 'reachable' : `unreachable (${crm.error || crm.status})`,
models: describeChains(), models: describeChains(),
speech: speechStatus(),
uptime_s: Math.round(process.uptime()), uptime_s: Math.round(process.uptime()),
}); });
}); });
@@ -83,6 +85,11 @@ async function start() {
// second origin and the CRM token works unchanged. // second origin and the CRM token works unchanged.
attachVoice(server); attachVoice(server);
// Speech models load lazily on the first voice turn (~10 s). Set
// SPEECH_WARMUP=true to pay that at boot instead — worth it in production,
// wasteful in development where most restarts never use voice.
if (process.env.SPEECH_WARMUP === 'true') warmSpeech(['ta', 'en']);
const shutdown = (sig) => { const shutdown = (sig) => {
logger.info(`${sig} — shutting down`); logger.info(`${sig} — shutting down`);
server.close(() => process.exit(0)); server.close(() => process.exit(0));
+169
View File
@@ -0,0 +1,169 @@
// ============================================
// Speech pipeline — transcribe() and synthesize().
//
// Whisper is multilingual and can identify the spoken language, so "auto"
// costs nothing extra: detection and transcription are the same forward pass.
// That matters for a WeLe agent who switches between Tamil and English inside
// one shift and should never have to touch a language menu.
// ============================================
import { Tensor } from '@huggingface/transformers';
import { getSTT, getTTS, supportsTTS } from './models.js';
import logger from '../utils/logger.js';
export { LANGUAGES, warmup, speechStatus } from './models.js';
export { Endpointer, warmupVad } from './vad.js';
const RATE = 16000;
/** Languages we can both hear and speak. */
const SPOKEN = new Set(['ta', 'en']);
// Below this, trust the caller's preference over the detector. Short or noisy
// utterances — and code-mixed "Tanglish" especially — can land either side.
const DETECT_CONFIDENCE = Number(process.env.DETECT_CONFIDENCE ?? 0.6);
let detectIds = null;
/**
* Identify the spoken language in ONE decoder step.
*
* Passing no `language` to the pipeline does NOT auto-detect — Transformers.js
* logs "No language specified - defaulting to English" and transcribes Tamil
* as English, producing nonsense. Whisper does emit a language token right
* after <|startoftranscript|>, so we read that distribution directly. Measured
* ~700 ms, and 0.998 / 1.000 confidence on clean Tamil / English.
*
* Reuses the pipeline's own model and processor, so nothing loads twice.
*/
async function detectLanguage(audio) {
const stt = await getSTT();
const tok = stt.tokenizer;
if (!detectIds) {
const id = (t) => tok.encode(t, { add_special_tokens: false })[0];
detectIds = { sot: id('<|startoftranscript|>'), langs: [...SPOKEN].map((c) => ({ code: c, id: id(`<|${c}|>`) })) };
}
const inputs = await stt.processor(audio);
const out = await stt.model({
...inputs,
decoder_input_ids: new Tensor('int64', BigInt64Array.from([BigInt(detectIds.sot)]), [1, 1]),
});
const { dims, data } = out.logits;
const row = data.slice((dims[1] - 1) * dims[2], dims[1] * dims[2]);
const scores = detectIds.langs.map((l) => Number(row[l.id]));
const max = Math.max(...scores);
const exp = scores.map((v) => Math.exp(v - max));
const sum = exp.reduce((a, b) => a + b, 0);
const probs = exp.map((v) => v / sum);
const best = probs.indexOf(Math.max(...probs));
return { lang: detectIds.langs[best].code, confidence: probs[best] };
}
/**
* @param {Float32Array} audio mono @16 kHz in [-1, 1]
* @param {string} lang 'auto' | 'ta' | 'en'
* @param {string} prefer used when detection is unusable
*/
export async function transcribe(audio, lang = 'auto', prefer = 'ta') {
if (!audio || audio.length < RATE / 5) { // under 200 ms
return { text: '', lang: prefer, note: 'too short' };
}
const stt = await getSTT();
const t0 = Date.now();
// Whisper must always be told a language — it never detects on its own here.
let used = lang;
let detected = null;
let confidence = null;
if (lang === 'auto') {
try {
const d = await detectLanguage(audio);
detected = d.lang;
confidence = d.confidence;
used = d.confidence >= DETECT_CONFIDENCE ? d.lang : prefer;
if (used !== d.lang) {
logger.info(`language ID unsure (${d.lang} @ ${d.confidence.toFixed(2)}) — using preferred ${prefer}`);
}
} catch (e) {
logger.warn(`language ID failed (${e.message}) — using preferred ${prefer}`);
used = prefer;
}
}
const result = await stt(audio, { task: 'transcribe', language: used, return_timestamps: false });
return finish(result, used, audio, t0, detected, confidence);
}
function finish(result, used, audio, t0, detected, confidence) {
const text = (result?.text || '').trim();
const ms = Date.now() - t0;
const audioMs = Math.round((audio.length / RATE) * 1000);
logger.info(`🎤 STT ${used}${detected && detected !== used ? ` (heard ${detected})` : ''}: ${audioMs}ms → ${ms}ms → ${JSON.stringify(text.slice(0, 70))}`);
return {
text, lang: used, detected: detected || null,
confidence: confidence == null ? null : Number(confidence.toFixed(3)),
ms, audio_ms: audioMs,
};
}
/**
* Synthesise one piece of text.
* @returns {Promise<{audio: Float32Array, sampling_rate: number}>}
*/
export async function synthesize(text, lang = 'ta') {
const clean = (text || '').trim();
if (!clean) return null;
const use = supportsTTS(lang) ? lang : 'en';
const tts = await getTTS(use);
const t0 = Date.now();
const out = await tts(clean);
const ms = Date.now() - t0;
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
logger.debug(`🔈 TTS ${use}: ${clean.length} chars → ${ms}ms for ${audioMs.toFixed(0)}ms (RTF ${(ms / audioMs).toFixed(2)}x)`);
return { audio: out.audio, sampling_rate: out.sampling_rate };
}
/**
* Split into speakable pieces. Short prompts reach audio sooner, and a sentence
* boundary is a clean place to be interrupted.
*/
export function sentences(text, max = 200) {
const out = [];
for (const raw of String(text || '').split(/(?<=[.!?।])\s+/)) {
let s = raw.trim();
if (!s) continue;
while (s.length > max) {
const cut = s.lastIndexOf(' ', max);
out.push(s.slice(0, cut > 0 ? cut : max).trim());
s = s.slice(cut > 0 ? cut : max).trim();
}
if (s) out.push(s);
}
return out;
}
/**
* Strip block markdown before speaking — tables and code read terribly aloud,
* and the visual blocks are already on screen.
*/
export function speakable(markdown = '') {
return markdown
.replace(/```[\s\S]*?```/g, ' ')
.replace(/^\s*\|.*\|\s*$/gm, ' ')
.replace(/^\s*[-*]\s+/gm, '')
.replace(/^#{1,6}\s*/gm, '')
.replace(/\*\*([^*]+)\*\*/g, '$1')
.replace(/`([^`]+)`/g, '$1')
.replace(/\[([^\]]+)\]\([^)]+\)/g, '$1')
.replace(/₹\s?([\d,.]+)/g, 'rupees $1')
.replace(/\s{2,}/g, ' ')
.trim();
}
+118
View File
@@ -0,0 +1,118 @@
// ============================================
// Speech models — all ONNX, all CPU, all in this Node process.
//
// There is no GPU and no Python. That is the whole point: the AWS host has
// neither, and a second service was one more thing to deploy and keep alive.
//
// VAD Silero 2 MB endpointing
// STT Whisper base ~80 MB Tamil + English + language detection
// TTS MMS-TTS VITS ~114 MB per language, feed-forward
//
// Measured on an i7-10850H, CPU only:
// TTS RTF 0.28x (3.5x faster than realtime)
// STT ~1.2 s for 4 s of audio
//
// Two findings worth keeping:
//
// * VITS is feed-forward. The earlier Parler-TTS attempt was autoregressive
// and ran at RTF ~5x — i.e. 5x SLOWER than realtime — which is why voice was
// unusable even on a GPU. Architecture mattered far more than hardware here.
//
// * int8 is a trap for a model this small: dynamic quantisation made TTS 5.7x
// SLOWER than fp32 (RTF 1.67x vs 0.28x) because the quantise/dequantise
// overhead dominates. We ship fp32 deliberately.
// ============================================
import path from 'node:path';
import { fileURLToPath } from 'node:url';
import { pipeline, env } from '@huggingface/transformers';
import logger from '../utils/logger.js';
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..');
// Hub downloads are cached here so a container restart does not re-fetch.
env.cacheDir = process.env.SPEECH_CACHE_DIR || path.join(ROOT, '.transformers-cache');
// Tamil is loaded from a folder we exported ourselves — no public ONNX build
// of mms-tts-tam exists. See scripts/export-tamil-tts.py.
env.localModelPath = path.join(ROOT, 'assets/tts');
const STT_MODEL = process.env.STT_MODEL || 'onnx-community/whisper-base';
/** TTS voice per language. Tamil is local; English comes from the Hub. */
const VOICES = {
ta: { id: 'mms-tts-tam', local: true },
en: { id: 'Xenova/mms-tts-eng', local: false },
};
export const LANGUAGES = [
{ code: 'auto', label: 'Auto-detect', native: 'Auto' },
{ code: 'ta', label: 'Tamil', native: 'தமிழ்' },
{ code: 'en', label: 'English', native: 'English' },
];
const cache = new Map();
let sttPromise = null;
/**
* Models load on first use, not at boot. A CRM restart should not wait ~10 s
* for speech models that most sessions never touch.
*/
async function loadOnce(key, build) {
if (!cache.has(key)) {
const t0 = Date.now();
cache.set(key, build().then((m) => {
logger.info(`🔊 loaded ${key} in ${((Date.now() - t0) / 1000).toFixed(1)}s`);
return m;
}).catch((e) => {
cache.delete(key); // let the next attempt retry
throw e;
}));
}
return cache.get(key);
}
export async function getSTT() {
if (!sttPromise) {
sttPromise = loadOnce(STT_MODEL, () =>
// q8 is the right call for Whisper — unlike VITS it is big enough that
// quantisation is a clear win.
pipeline('automatic-speech-recognition', STT_MODEL, { dtype: 'q8' }),
).catch((e) => { sttPromise = null; throw e; });
}
return sttPromise;
}
export async function getTTS(lang = 'ta') {
const voice = VOICES[lang] || VOICES.ta;
const prev = env.allowRemoteModels;
try {
// Local folders must not be looked up on the Hub, and vice versa.
env.allowRemoteModels = !voice.local;
return await loadOnce(`tts:${voice.id}`, () =>
pipeline('text-to-speech', voice.id, { dtype: 'fp32' }),
);
} finally {
env.allowRemoteModels = prev;
}
}
export const supportsTTS = (lang) => Boolean(VOICES[lang]);
/** Warm the models the deployment actually expects to use. */
export async function warmup(langs = ['ta', 'en']) {
try {
await getSTT();
for (const l of langs) await getTTS(l);
logger.info('🔊 speech models warm');
} catch (e) {
logger.warn(`speech warmup failed (will retry on first use): ${e.message}`);
}
}
export function speechStatus() {
return {
stt_model: STT_MODEL,
tts_voices: Object.fromEntries(Object.entries(VOICES).map(([k, v]) => [k, v.id])),
loaded: [...cache.keys()],
languages: LANGUAGES,
};
}
+152
View File
@@ -0,0 +1,152 @@
// ============================================
// Endpointing — Silero VAD via onnxruntime-node.
//
// Deciding turn boundaries on the server rather than in the browser keeps the
// rule in one place for every future channel (a phone bridge has no
// AudioWorklet), and gives the server the signal it needs for barge-in: it has
// to know the user started talking while the assistant was still speaking.
// ============================================
import path from 'node:path';
import { fileURLToPath } from 'node:url';
import fs from 'node:fs/promises';
import ort from 'onnxruntime-node';
import logger from '../utils/logger.js';
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..');
const MODEL_URL = 'https://huggingface.co/onnx-community/silero-vad/resolve/main/onnx/model.onnx';
const MODEL_PATH = path.join(process.env.SPEECH_CACHE_DIR || path.join(ROOT, '.transformers-cache'), 'silero-vad.onnx');
// Silero wants exactly 512 samples at 16 kHz (32 ms). The browser sends 40 ms
// chunks, so audio is buffered and drained in exact frames rather than forcing
// the client to match.
const FRAME = 512;
const RATE = 16000;
const FRAME_MS = (FRAME / RATE) * 1000;
let sessionPromise = null;
async function getSession() {
if (sessionPromise) return sessionPromise;
sessionPromise = (async () => {
try {
await fs.access(MODEL_PATH);
} catch {
logger.info('⬇️ fetching Silero VAD (2 MB)…');
const res = await fetch(MODEL_URL);
if (!res.ok) throw new Error(`VAD download failed: ${res.status}`);
await fs.mkdir(path.dirname(MODEL_PATH), { recursive: true });
await fs.writeFile(MODEL_PATH, Buffer.from(await res.arrayBuffer()));
}
const s = await ort.InferenceSession.create(MODEL_PATH);
logger.info('🎚️ Silero VAD ready');
return s;
})().catch((e) => { sessionPromise = null; throw e; });
return sessionPromise;
}
export const vadOptions = {
threshold: Number(process.env.VAD_THRESHOLD ?? 0.5),
// Trailing silence that ends a turn. Too short truncates someone who pauses
// mid-sentence; too long makes the assistant feel sluggish.
silenceMs: Number(process.env.VAD_SILENCE_MS ?? 700),
// Ignore blips, so a cough or a door does not open a turn.
minSpeechMs: Number(process.env.VAD_MIN_SPEECH_MS ?? 250),
// Audio kept from BEFORE detection, so word onsets are not clipped.
prefixMs: Number(process.env.VAD_PREFIX_MS ?? 300),
maxUtteranceMs: Number(process.env.VAD_MAX_UTTERANCE_MS ?? 30000),
};
/** Streaming endpointer. One instance per connection. */
export class Endpointer {
constructor(opts = {}) {
this.o = { ...vadOptions, ...opts };
this.pending = new Float32Array(0);
this.prefixFrames = Math.max(1, Math.round(this.o.prefixMs / FRAME_MS));
this.reset();
}
reset() {
this.speaking = false;
this.speechMs = 0;
this.silenceMs = 0;
this.buffer = [];
this.prefix = [];
// Silero is recurrent: this 2x1x128 state carries across frames and must
// be reset between turns or the model stays biased by the last utterance.
this.state = new ort.Tensor('float32', new Float32Array(2 * 1 * 128), [2, 1, 128]);
this.pending = new Float32Array(0);
}
/**
* Feed float32 mono @16k.
* @returns {Promise<{utterances: Float32Array[], started: boolean}>}
* `started` flips the moment speech begins — that is the barge-in signal.
*/
async push(pcm) {
const session = await getSession();
const merged = new Float32Array(this.pending.length + pcm.length);
merged.set(this.pending);
merged.set(pcm, this.pending.length);
this.pending = merged;
const utterances = [];
let started = false;
let offset = 0;
while (this.pending.length - offset >= FRAME) {
const frame = this.pending.subarray(offset, offset + FRAME);
offset += FRAME;
const out = await session.run({
input: new ort.Tensor('float32', frame, [1, FRAME]),
sr: new ort.Tensor('int64', BigInt64Array.from([BigInt(RATE)]), []),
state: this.state,
});
this.state = out.stateN ?? out.state_n ?? this.state;
const voiced = out.output.data[0] >= this.o.threshold;
if (!this.speaking) {
this.prefix.push(Float32Array.from(frame));
if (this.prefix.length > this.prefixFrames) this.prefix.shift();
if (voiced) {
this.speechMs += FRAME_MS;
if (this.speechMs >= this.o.minSpeechMs) {
this.speaking = true;
this.silenceMs = 0;
this.buffer = this.prefix; // open the turn with the pre-roll
this.prefix = [];
started = true;
}
} else {
this.speechMs = 0;
}
continue;
}
this.buffer.push(Float32Array.from(frame));
if (voiced) this.silenceMs = 0;
else this.silenceMs += FRAME_MS;
const spokenMs = this.buffer.length * FRAME_MS;
if (this.silenceMs >= this.o.silenceMs || spokenMs >= this.o.maxUtteranceMs) {
utterances.push(concat(this.buffer));
this.reset();
}
}
this.pending = this.pending.slice(offset);
return { utterances, started };
}
}
function concat(frames) {
const total = frames.reduce((n, f) => n + f.length, 0);
const out = new Float32Array(total);
let i = 0;
for (const f of frames) { out.set(f, i); i += f.length; }
return out;
}
export const warmupVad = () => getSession().catch(() => {});
-22
View File
@@ -1,22 +0,0 @@
# Copy to .env and fill in HF_TOKEN.
# The AI4Bharat models are gated: sign in at huggingface.co, accept the terms on
# both model pages, then create a read token at huggingface.co/settings/tokens.
# HuggingFace token — required: the AI4Bharat models are gated repos.
HF_TOKEN=
VOICE_HOST=127.0.0.1
VOICE_PORT=4100
STT_MODEL=ai4bharat/indic-conformer-600m-multilingual
STT_DECODING=ctc
ENGLISH_MODEL=openai/whisper-small
TTS_MODEL=ai4bharat/indic-parler-tts
VOICE_DEFAULT_LANG=ta
PRELOAD_ENGLISH=true
VOICE_WARMUP=true
# Endpointing
VAD_SILENCE_MS=700
VAD_MIN_SPEECH_MS=250
VAD_PREFIX_MS=300
-105
View File
@@ -1,105 +0,0 @@
# WeLe Voice Service
Speech in, speech out. This process holds the GPU models and nothing else — it
has no idea what the CRM is. Orchestration, auth and business logic stay in the
Node service, so **voice is a channel into the same agent**, not a parallel
system with its own brain.
```
browser ──audio──► node :4000 ──audio──► this :4100 ──► IndicConformer / Whisper
│ │
└────────── same graph, agents, ───────────┘
guardrails as text chat
│
browser ◄──audio───── node ◄──audio──── this ◄── Indic Parler-TTS
```
## Models
| Job | Model | Notes |
|---|---|---|
| Endpointing | Silero VAD | 512-sample frames @16 kHz, 300 ms pre-roll |
| Indic ASR | `ai4bharat/indic-conformer-600m-multilingual` | 22 Indian languages, CTC decoding |
| English ASR + language ID | `openai/whisper-small` | multilingual on purpose — the `.en` build cannot identify languages |
| TTS | `ai4bharat/indic-parler-tts` | 21 languages, streaming |
**The AI4Bharat repos are gated.** Access is auto-approved, but the download
needs an authenticated account: sign in to huggingface.co, accept the terms on
both model pages, then put a read token in `.env` as `HF_TOKEN`.
## Why two ASR models
IndicConformer decodes *as* the language you name — it does not detect one, and
English is not among its 22 codes. Whisper covers English and can identify the
spoken language in a single decoder step. So the default mode is `auto`:
```
audio → Whisper mel + 1 decoder step → language ID
├─ "en" → Whisper transcribes (mel already computed — no extra cost)
└─ Indic → IndicConformer with the detected code
```
Below **0.60** confidence the caller's preferred language wins instead of a coin
toss. That matters for Tanglish, where a short code-mixed sentence can honestly
land either side.
## Setup
```bash
python -m venv --system-site-packages .venv # reuses the system torch build
.venv/Scripts/python -m pip install -r requirements.txt
cp .env.example .env # add HF_TOKEN
```
The venv deliberately inherits system site-packages: torch is ~2.5 GB and
already installed with CUDA. Note that `parler-tts` pins `transformers==4.46.1`
**inside the venv only** — the system install is untouched.
## Run
```bash
npm run voice # from the parent directory
# or
.venv/Scripts/python -m app.server
```
First start downloads several GB and warms both models. `GET /health` reports
device, models, sample rate and current VRAM.
## Protocol
One WebSocket at `/ws/voice`, JSON control frames plus binary audio.
| Direction | Message |
|---|---|
| → | binary — 16 kHz mono PCM16 mic frames |
| → | `{"type":"config","lang":"auto","prefer":"ta"}` |
| → | `{"type":"speak","text":"…","id":"…"}` |
| → | `{"type":"cancel"}` — stop speaking now |
| ← | `{"type":"speech_start"}` — VAD opened a turn (drives barge-in) |
| ← | `{"type":"transcript","text":…,"lang":…,"detected":…,"confidence":…}` |
| ← | `{"type":"audio_start","sample_rate":24000}` then binary float32 chunks |
## Tuning
| Env | Default | Effect |
|---|---|---|
| `VAD_SILENCE_MS` | 700 | trailing silence that ends a turn — lower feels snappier, truncates people who pause |
| `VAD_MIN_SPEECH_MS` | 250 | ignores coughs and door slams |
| `VAD_PREFIX_MS` | 300 | audio kept from before detection, so word onsets survive |
| `VOICE_DEFAULT_LANG` | `ta` | tiebreak when language ID is unsure |
| `PRELOAD_ENGLISH` | `true` | set `false` to load Whisper lazily if VRAM is tight |
| `STT_DECODING` | `ctc` | `rnnt` is more accurate but decodes autoregressively |
## VRAM
Roughly 4.7 GB of the 6 GB card with all three models resident. If that proves
too tight, `PRELOAD_ENGLISH=false` defers Whisper (~0.5 GB) until the first
English utterance.
## Scripts
```bash
.venv/Scripts/python probe_access.py # which repos the token can reach
.venv/Scripts/python probe_models.py # load, VRAM, time-to-first-audio
```
-82
View File
@@ -1,82 +0,0 @@
"""Voice service configuration.
Deliberately small: this process does one job — turn audio into text and text
into audio. Everything about *what to say* lives in the Node agentic service.
"""
from __future__ import annotations
import os
import torch
def _int(name: str, default: int) -> int:
try:
return int(os.environ.get(name, default))
except (TypeError, ValueError):
return default
def _flag(name: str, default: bool) -> bool:
return os.environ.get(name, str(default)).lower() in {"1", "true", "yes"}
class Settings:
host: str = os.environ.get("VOICE_HOST", "127.0.0.1")
port: int = _int("VOICE_PORT", 4100)
# ── Models ───────────────────────────────────────────────────────────────
# IndicConformer is a hybrid CTC + RNNT model. CTC decoding is used because
# it is a single forward pass — RNNT is more accurate but decodes
# autoregressively, and in a voice loop the latency costs more than the
# accuracy buys.
stt_model: str = os.environ.get("STT_MODEL", "ai4bharat/indic-conformer-600m-multilingual")
stt_decoding: str = os.environ.get("STT_DECODING", "ctc") # ctc | rnnt
# English is not one of IndicConformer's 22 codes, and it cannot identify
# languages. Whisper covers both — multilingual, not the .en checkpoint,
# because language ID is what makes "auto" work.
english_model: str = os.environ.get("ENGLISH_MODEL", "openai/whisper-small")
preload_english: bool = _flag("PRELOAD_ENGLISH", True)
# Used when auto-detection is not confident enough to overrule the user.
default_lang: str = os.environ.get("VOICE_DEFAULT_LANG", "ta")
tts_model: str = os.environ.get("TTS_MODEL", "ai4bharat/indic-parler-tts")
# ── Device / precision ───────────────────────────────────────────────────
# float16 on CUDA: both models together are ~3 GB in half precision, which
# fits the 6 GB card with room for activations. float32 would not.
device: str = os.environ.get("VOICE_DEVICE", "cuda" if torch.cuda.is_available() else "cpu")
@property
def dtype(self) -> torch.dtype:
return torch.float16 if self.device == "cuda" else torch.float32
# ── Audio ────────────────────────────────────────────────────────────────
sample_rate_in: int = 16000 # what the browser worklet sends
# Indic Parler-TTS emits 44.1 kHz — verified from model.config.sampling_rate,
# not the 24 kHz the upstream Parler-TTS Mini uses. This is only a fallback;
# the real rate is read from the loaded model and sent to the browser, which
# configures its playback worklet from it.
sample_rate_out: int = _int("TTS_SAMPLE_RATE", 44100)
# ── Endpointing (Silero VAD) ─────────────────────────────────────────────
# Silero operates on fixed 512-sample frames at 16 kHz (32 ms).
vad_frame: int = 512
vad_threshold: float = float(os.environ.get("VAD_THRESHOLD", "0.5"))
# How much trailing silence ends a turn. Too short truncates people who
# pause mid-sentence; too long makes the assistant feel sluggish.
vad_silence_ms: int = _int("VAD_SILENCE_MS", 700)
# Ignore blips so a cough or a door does not open a turn.
vad_min_speech_ms: int = _int("VAD_MIN_SPEECH_MS", 250)
# Audio kept from *before* detected speech, so word onsets are not clipped.
vad_prefix_ms: int = _int("VAD_PREFIX_MS", 300)
vad_max_utterance_ms: int = _int("VAD_MAX_UTTERANCE_MS", 30000)
# Warm the models at startup rather than on the first user turn — a cold
# CUDA graph on the first utterance costs several seconds.
warmup: bool = _flag("VOICE_WARMUP", True)
settings = Settings()
-233
View File
@@ -1,233 +0,0 @@
"""Voice service — STT and TTS over one WebSocket.
This process holds the GPU models and nothing else. It has no idea what the
CRM is: it receives audio and returns text, receives text and returns audio.
All orchestration, auth and business logic stay in the Node service, so voice
is just another channel into the same agent rather than a parallel system.
Protocol (ws /ws/voice), JSON control + binary audio:
client → server
binary 16 kHz mono PCM16 mic frames
{"type":"config","lang":"ta"} set the session language
{"type":"speak","text":"…","id":"…"} synthesise
{"type":"cancel"} stop speaking now (barge-in)
{"type":"reset"} clear the endpointer
server → client
{"type":"ready", …}
{"type":"speech_start"} VAD opened a turn → caller ducks TTS
{"type":"transcript","text":…} a finished utterance
{"type":"audio_start","id":…,"sample_rate":24000}
binary float32 mono TTS chunks
{"type":"audio_end","id":…}
"""
from __future__ import annotations
import asyncio
import json
import logging
import time
import numpy as np
from fastapi import FastAPI, WebSocket, WebSocketDisconnect
from .config import settings
from .stt import SUPPORTED, Transcriber
from .tts import Synthesizer, split_sentences
from .vad import Endpointer
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)-5s %(message)s", datefmt="%H:%M:%S")
logger = logging.getLogger("voice")
app = FastAPI(title="WeLe Voice Service")
stt = Transcriber()
tts = Synthesizer()
@app.on_event("startup")
async def _startup() -> None:
t0 = time.perf_counter()
logger.info("loading models on %s…", settings.device)
stt.load()
tts.load()
if settings.warmup:
stt.warmup()
tts.warmup()
logger.info("voice service ready in %.1fs", time.perf_counter() - t0)
@app.get("/health")
async def health() -> dict:
import torch
return {
"ok": True,
"device": settings.device,
"stt_model": settings.stt_model,
"tts_model": settings.tts_model,
"tts_sample_rate": tts.sample_rate,
"languages": SUPPORTED,
"vram_gb": round(torch.cuda.memory_reserved() / 1e9, 2) if settings.device == "cuda" else None,
}
@app.get("/languages")
async def languages() -> dict:
return {"languages": SUPPORTED}
class Session:
"""One browser connection. Owns its endpointer and its speaking state."""
def __init__(self, ws: WebSocket) -> None:
self.ws = ws
# "auto" detects per utterance; `prefer` breaks ties when the detector
# is unsure, which is common on short code-mixed ("Tanglish") speech.
self.lang = "auto"
self.prefer = settings.default_lang
self.endpointer = Endpointer()
self.endpointer.load()
self._speak_task: asyncio.Task | None = None
self._cancel = asyncio.Event()
async def send(self, payload: dict) -> None:
await self.ws.send_text(json.dumps(payload, ensure_ascii=False))
# ── microphone ───────────────────────────────────────────────────────────
async def on_audio(self, raw: bytes) -> None:
pcm = np.frombuffer(raw, dtype=np.int16).astype(np.float32) / 32768.0
loop = asyncio.get_running_loop()
# VAD is a small torch model but still blocking; keep the event loop free.
utterances, started = await loop.run_in_executor(None, self.endpointer.push, pcm)
if started:
# Barge-in: the user talking wins immediately.
await self.stop_speaking()
await self.send({"type": "speech_start"})
for utt in utterances:
request_lang = f"auto:{self.prefer}" if self.lang == "auto" else self.lang
result = await loop.run_in_executor(None, stt.transcribe, utt.audio, request_lang)
if result["text"]:
await self.send({"type": "transcript", **result, "truncated": utt.truncated})
else:
await self.send({"type": "transcript_empty", "reason": result.get("note", "no speech")})
# ── speaking ─────────────────────────────────────────────────────────────
async def speak(self, text: str, msg_id: str, lang: str | None = None, voice: str | None = None) -> None:
await self.stop_speaking()
self._cancel.clear()
self._speak_task = asyncio.create_task(self._speak(text, msg_id, lang or self.lang, voice))
async def _speak(self, text: str, msg_id: str, lang: str, voice: str | None) -> None:
loop = asyncio.get_running_loop()
try:
await self.send({"type": "audio_start", "id": msg_id, "sample_rate": tts.sample_rate})
# Sentence at a time: shorter prompts reach first audio sooner, and
# a boundary is a clean place to stop when interrupted.
for sentence in split_sentences(text):
if self._cancel.is_set():
break
queue: asyncio.Queue = asyncio.Queue(maxsize=32)
def produce() -> None:
try:
for chunk in tts.stream(sentence, lang, voice):
if self._cancel.is_set():
break
asyncio.run_coroutine_threadsafe(queue.put(chunk), loop).result()
finally:
asyncio.run_coroutine_threadsafe(queue.put(None), loop).result()
loop.run_in_executor(None, produce)
while True:
chunk = await queue.get()
if chunk is None:
break
if self._cancel.is_set():
continue # drain, don't send
await self.ws.send_bytes(np.asarray(chunk, dtype=np.float32).tobytes())
await self.send({"type": "audio_end", "id": msg_id, "cancelled": self._cancel.is_set()})
except WebSocketDisconnect:
pass
except Exception as e: # noqa: BLE001
logger.exception("synthesis failed")
try:
await self.send({"type": "error", "where": "tts", "message": str(e)[:200]})
except Exception: # noqa: BLE001
pass
async def stop_speaking(self) -> None:
if self._speak_task and not self._speak_task.done():
self._cancel.set()
try:
await asyncio.wait_for(self._speak_task, timeout=2.0)
except (asyncio.TimeoutError, asyncio.CancelledError):
self._speak_task.cancel()
self._speak_task = None
@app.websocket("/ws/voice")
async def voice(ws: WebSocket) -> None:
await ws.accept()
session = Session(ws)
await session.send({
"type": "ready",
"sample_rate_in": settings.sample_rate_in,
"sample_rate_out": tts.sample_rate,
"languages": SUPPORTED,
})
logger.info("voice session opened")
try:
while True:
msg = await ws.receive()
if msg["type"] == "websocket.disconnect":
break
if (raw := msg.get("bytes")) is not None:
await session.on_audio(raw)
continue
if (text := msg.get("text")) is None:
continue
try:
data = json.loads(text)
except json.JSONDecodeError:
continue
kind = data.get("type")
if kind == "config":
session.lang = data.get("lang", session.lang)
if session.lang != "auto":
session.prefer = session.lang
elif data.get("prefer"):
session.prefer = data["prefer"]
session.endpointer.reset()
await session.send({"type": "config_ok", "lang": session.lang, "prefer": session.prefer})
elif kind == "speak":
await session.speak(data.get("text", ""), data.get("id", ""), data.get("lang"), data.get("voice"))
elif kind == "cancel":
await session.stop_speaking()
await session.send({"type": "cancelled"})
elif kind == "reset":
session.endpointer.reset()
except WebSocketDisconnect:
pass
finally:
await session.stop_speaking()
logger.info("voice session closed")
def main() -> None:
import uvicorn
uvicorn.run(app, host=settings.host, port=settings.port, log_level="info", ws_max_size=16 * 1024 * 1024)
if __name__ == "__main__":
main()
-200
View File
@@ -1,200 +0,0 @@
"""Speech-to-text — AI4Bharat IndicConformer, with an English path and auto routing.
Why two models rather than one:
* **IndicConformer** decodes 22 Indian languages, and decodes *as* the language
you name — it does not detect. Handing it "ta" for English speech produces
Tamil-script nonsense. English is not one of its codes at all.
* **Whisper (multilingual)** covers English well and, usefully, can identify the
spoken language in a single decoder step.
So the default mode is `auto`: Whisper identifies the language from the audio,
English is transcribed by Whisper directly (the mel is already computed, so
this costs nothing extra), and anything Indic is routed to IndicConformer,
which is far stronger on those languages than Whisper is.
Code-mixed speech ("Tanglish") is the awkward case: language ID can land either
side of the fence on a short, mixed utterance. When Whisper is not confident,
the caller's preferred language wins rather than a coin toss — a Tamil-speaking
office gets Tamil, and the occasional English sentence still routes correctly
when it is clearly English.
"""
from __future__ import annotations
import logging
import time
import numpy as np
import torch
from .config import settings
logger = logging.getLogger(__name__)
# The codes IndicConformer accepts.
INDIC_LANGS = {
"as", "bn", "brx", "doi", "gu", "hi", "kn", "kok", "ks", "mai", "ml",
"mni", "mr", "ne", "or", "pa", "sa", "sat", "sd", "ta", "te", "ur",
}
# Offered by the UI. "auto" first: most WeLe agents switch language mid-shift.
SUPPORTED = [
{"code": "auto", "label": "Auto-detect", "native": "Auto"},
{"code": "ta", "label": "Tamil", "native": "தமிழ்"},
{"code": "en", "label": "English", "native": "English"},
{"code": "hi", "label": "Hindi", "native": "हिन्दी"},
{"code": "te", "label": "Telugu", "native": "తెలుగు"},
{"code": "kn", "label": "Kannada", "native": "ಕನ್ನಡ"},
{"code": "ml", "label": "Malayalam", "native": "മലയാളം"},
{"code": "mr", "label": "Marathi", "native": "मराठी"},
{"code": "bn", "label": "Bengali", "native": "বাংলা"},
]
# Below this, trust the user's stated preference over the detector.
DETECT_CONFIDENCE = 0.60
class Transcriber:
def __init__(self) -> None:
self._indic = None
self._whisper = None
self._whisper_proc = None
# ── loading ──────────────────────────────────────────────────────────────
def load(self) -> None:
from transformers import AutoModel
t0 = time.perf_counter()
# float32: the checkpoint ships custom remote code that assumes fp32.
# ~2.4 GB at 600M, which still leaves room for Whisper and the TTS model.
self._indic = AutoModel.from_pretrained(settings.stt_model, trust_remote_code=True)
self._indic.to(settings.device).eval()
logger.info("STT loaded (%s) in %.1fs", settings.stt_model, time.perf_counter() - t0)
if settings.preload_english:
self._load_whisper()
def _load_whisper(self) -> None:
"""English + language ID. Multilingual on purpose — the .en checkpoint
cannot identify languages, which is the whole point of auto mode."""
if self._whisper is not None:
return
from transformers import WhisperForConditionalGeneration, WhisperProcessor
t0 = time.perf_counter()
self._whisper_proc = WhisperProcessor.from_pretrained(settings.english_model)
self._whisper = WhisperForConditionalGeneration.from_pretrained(
settings.english_model, torch_dtype=settings.dtype,
).to(settings.device).eval()
logger.info("English/ID model loaded (%s) in %.1fs", settings.english_model, time.perf_counter() - t0)
# ── inference ────────────────────────────────────────────────────────────
@torch.inference_mode()
def transcribe(self, audio: np.ndarray, lang: str = "auto") -> dict:
"""audio: float32 mono @16 kHz in [-1, 1].
`lang` may be an explicit code, or "auto" / "auto:ta" to detect with a
fallback preference.
"""
t0 = time.perf_counter()
if audio.size < settings.sample_rate_in // 5: # under 200 ms
return {"text": "", "lang": lang, "ms": 0, "note": "too short"}
detected = None
confidence = None
if lang.startswith("auto"):
prefer = lang.split(":", 1)[1] if ":" in lang else settings.default_lang
feats = self._features(audio)
detected, confidence = self._detect(feats)
if confidence is not None and confidence < DETECT_CONFIDENCE:
logger.info("language ID low confidence (%s @ %.2f) — using preferred %s",
detected, confidence, prefer)
use = prefer
elif detected == "en" or detected in INDIC_LANGS:
use = detected
else:
# Whisper reported something we cannot decode (e.g. "nn" on
# noise). Fall back rather than fail.
use = prefer
text = self._english(audio, feats=feats) if use == "en" else self._indic_decode(audio, use)
else:
use = lang
text = self._english(audio) if lang == "en" else self._indic_decode(audio, lang)
ms = int((time.perf_counter() - t0) * 1000)
audio_ms = int(1000 * audio.size / settings.sample_rate_in)
logger.info("STT %s%s: %dms audio → %dms → %r",
use, f" (detected {detected} {confidence:.2f})" if confidence is not None else "",
audio_ms, ms, text[:80])
return {
"text": text.strip(),
"lang": use,
"detected": detected,
"confidence": round(confidence, 3) if confidence is not None else None,
"ms": ms,
"audio_ms": audio_ms,
}
# ── internals ────────────────────────────────────────────────────────────
def _features(self, audio: np.ndarray):
self._load_whisper()
return self._whisper_proc(
audio, sampling_rate=settings.sample_rate_in, return_tensors="pt",
).input_features.to(settings.device, settings.dtype)
def _detect(self, feats) -> tuple[str | None, float | None]:
"""One decoder step: read the language-token distribution."""
try:
tok = self._whisper_proc.tokenizer
sot = tok.convert_tokens_to_ids("<|startoftranscript|>")
start = torch.tensor([[sot]], device=settings.device)
logits = self._whisper(feats, decoder_input_ids=start).logits[:, -1]
lang_ids, codes = [], []
for code in {*INDIC_LANGS, "en"}:
tid = tok.convert_tokens_to_ids(f"<|{code}|>")
# Unknown languages map to the unk id; skip those.
if tid is not None and tid != tok.unk_token_id:
lang_ids.append(tid)
codes.append(code)
if not lang_ids:
return None, None
probs = torch.softmax(logits[0, lang_ids].float(), dim=-1)
best = int(probs.argmax())
return codes[best], float(probs[best])
except Exception as e: # noqa: BLE001 — detection must never break a turn
logger.warning("language ID failed (%s) — falling back to preference", e)
return None, None
def _indic_decode(self, audio: np.ndarray, lang: str) -> str:
if lang not in INDIC_LANGS:
logger.warning("unsupported STT language %r — using %s", lang, settings.default_lang)
lang = settings.default_lang
wav = torch.from_numpy(audio).unsqueeze(0).to(settings.device) # (1, N)
out = self._indic(wav, lang, settings.stt_decoding)
if isinstance(out, (list, tuple)):
return str(out[0]) if out else ""
return str(out)
def _english(self, audio: np.ndarray, feats=None) -> str:
self._load_whisper()
if feats is None:
feats = self._features(audio)
ids = self._whisper.generate(feats, language="en", task="transcribe", max_new_tokens=180)
return self._whisper_proc.batch_decode(ids, skip_special_tokens=True)[0]
def warmup(self) -> None:
"""Silent pass so the first real utterance isn't paying for CUDA init."""
try:
silence = np.zeros(settings.sample_rate_in, dtype=np.float32)
self.transcribe(silence, settings.default_lang)
if settings.preload_english:
self.transcribe(silence, "en")
logger.info("STT warm")
except Exception as e: # noqa: BLE001
logger.warning("STT warmup skipped: %s", e)
-155
View File
@@ -1,155 +0,0 @@
"""Text-to-speech — AI4Bharat Indic Parler-TTS.
Parler is prompted with *two* texts: the words to say, and a natural-language
description of how to say them (speaker, pace, room tone). The description is
what selects a voice — there is no speaker-id argument.
Latency shape: Parler is autoregressive, so a whole paragraph costs whole-
paragraph time before the first sample exists. Two things fix that here:
1. `ParlerTTSStreamer` yields audio while generation continues, so playback
starts after roughly the first `play_steps` frames rather than at the end.
2. The caller sends one *sentence* at a time. Short prompts reach their first
chunk sooner, and a sentence boundary is a natural place for the assistant
to be interrupted.
"""
from __future__ import annotations
import logging
import re
import time
from threading import Thread
from typing import Iterator
import numpy as np
import torch
from .config import settings
logger = logging.getLogger(__name__)
# Voices recommended on the model card, per language.
VOICES = {
"ta": "Jaya", "hi": "Rohit", "te": "Prakash", "kn": "Suresh",
"ml": "Anjali", "mr": "Sanjay", "bn": "Arjun", "en": "Mary",
}
_DESCRIPTION = (
"{speaker} speaks in a warm, clear, professional tone at a natural pace. "
"The recording is very high quality with no background noise."
)
# Split on sentence enders including the Devanagari danda, keeping it simple —
# this only needs to find safe places to cut, not parse language.
_SENTENCE_RX = re.compile(r"(?<=[.!?।॥])\s+")
def split_sentences(text: str, max_chars: int = 220) -> list[str]:
"""Break text into TTS-sized pieces at sentence boundaries where possible."""
out: list[str] = []
for part in _SENTENCE_RX.split(text.strip()):
part = part.strip()
if not part:
continue
while len(part) > max_chars:
cut = part.rfind(" ", 0, max_chars)
if cut <= 0:
cut = max_chars
out.append(part[:cut].strip())
part = part[cut:].strip()
if part:
out.append(part)
return out
class Synthesizer:
def __init__(self) -> None:
self._model = None
self._tok = None
self._desc_tok = None
self.sample_rate = settings.sample_rate_out
def load(self) -> None:
from parler_tts import ParlerTTSForConditionalGeneration
from transformers import AutoTokenizer
t0 = time.perf_counter()
self._model = ParlerTTSForConditionalGeneration.from_pretrained(
settings.tts_model, torch_dtype=settings.dtype,
).to(settings.device).eval()
self._tok = AutoTokenizer.from_pretrained(settings.tts_model)
self._desc_tok = AutoTokenizer.from_pretrained(self._model.config.text_encoder._name_or_path)
self.sample_rate = int(self._model.config.sampling_rate)
logger.info(
"TTS loaded (%s) in %.1fs @ %d Hz", settings.tts_model,
time.perf_counter() - t0, self.sample_rate,
)
def _describe(self, lang: str, voice: str | None) -> str:
return _DESCRIPTION.format(speaker=voice or VOICES.get(lang, "Jaya"))
@torch.inference_mode()
def stream(self, text: str, lang: str = "ta", voice: str | None = None) -> Iterator[np.ndarray]:
"""Yield float32 mono chunks at `self.sample_rate` as they are generated."""
text = (text or "").strip()
if not text:
return
desc = self._desc_tok(self._describe(lang, voice), return_tensors="pt").to(settings.device)
prompt = self._tok(text, return_tensors="pt").to(settings.device)
kwargs = dict(
input_ids=desc.input_ids,
attention_mask=desc.attention_mask,
prompt_input_ids=prompt.input_ids,
prompt_attention_mask=prompt.attention_mask,
)
streamer = self._make_streamer()
if streamer is None:
yield self._generate_blocking(kwargs, text)
return
t0 = time.perf_counter()
# generate() blocks, so it runs on its own thread and the streamer is
# drained here as frames become available.
thread = Thread(target=self._model.generate, kwargs={**kwargs, "streamer": streamer}, daemon=True)
thread.start()
first = True
for chunk in streamer:
if chunk is None or len(chunk) == 0:
continue
audio = chunk.astype(np.float32) if isinstance(chunk, np.ndarray) else chunk.cpu().numpy().astype(np.float32)
if first:
logger.info("TTS first chunk in %dms (%d chars)", int((time.perf_counter() - t0) * 1000), len(text))
first = False
yield audio
thread.join(timeout=1.0)
def _make_streamer(self):
try:
from parler_tts import ParlerTTSStreamer
except ImportError:
logger.warning("ParlerTTSStreamer unavailable — falling back to blocking synthesis")
return None
# play_steps trades first-chunk latency against per-chunk overhead;
# ~0.5 s of audio keeps playback continuous without stalling generation.
frame_rate = getattr(self._model.audio_encoder.config, "frame_rate", 86)
return ParlerTTSStreamer(self._model, device=settings.device, play_steps=int(frame_rate / 2))
@torch.inference_mode()
def _generate_blocking(self, kwargs: dict, text: str) -> np.ndarray:
t0 = time.perf_counter()
gen = self._model.generate(**kwargs)
audio = gen.cpu().numpy().squeeze().astype(np.float32)
logger.info("TTS (blocking) %d chars in %dms", len(text), int((time.perf_counter() - t0) * 1000))
return audio
def warmup(self) -> None:
try:
for _ in self.stream("வணக்கம்", "ta"):
break
logger.info("TTS warm")
except Exception as e: # noqa: BLE001
logger.warning("TTS warmup skipped: %s", e)
-127
View File
@@ -1,127 +0,0 @@
"""Endpointing with Silero VAD.
Turn boundaries are decided here rather than in the browser for two reasons:
the same decision then applies to every future channel (a phone bridge has no
AudioWorklet), and barge-in needs the server to know someone started talking
while the assistant was still speaking.
"""
from __future__ import annotations
import logging
from collections import deque
from dataclasses import dataclass, field
import numpy as np
import torch
from .config import settings
logger = logging.getLogger(__name__)
@dataclass
class Utterance:
audio: np.ndarray # float32 mono @16k, in [-1, 1]
duration_ms: int
truncated: bool = False # hit the max-length guard rather than silence
@dataclass
class VADState:
speaking: bool = False
speech_ms: int = 0
silence_ms: int = 0
buffer: list[np.ndarray] = field(default_factory=list)
class Endpointer:
"""Streaming VAD that emits one Utterance per detected turn.
Silero wants exactly 512 samples at 16 kHz, but the browser sends 40 ms
(640-sample) chunks. Rather than force the client to match, incoming audio
is accumulated and drained in exact frames.
"""
def __init__(self) -> None:
self._model = None
self._pending = np.zeros(0, dtype=np.float32)
self.state = VADState()
# Pre-roll: speech is only *detected* a frame or two in, so without a
# prefix the first phoneme is already gone by the time we start saving.
prefix_frames = max(1, (settings.vad_prefix_ms * settings.sample_rate_in) // (1000 * settings.vad_frame))
self._prefix: deque[np.ndarray] = deque(maxlen=prefix_frames)
def load(self) -> None:
from silero_vad import load_silero_vad
self._model = load_silero_vad()
logger.info("silero VAD loaded")
def reset(self) -> None:
self.state = VADState()
self._pending = np.zeros(0, dtype=np.float32)
self._prefix.clear()
if self._model is not None:
self._model.reset_states()
@property
def is_speaking(self) -> bool:
return self.state.speaking
def push(self, pcm: np.ndarray) -> tuple[list[Utterance], bool]:
"""Feed float32 audio.
Returns (completed utterances, speech_started_this_call). The second
value drives barge-in: the caller cuts TTS playback the moment it flips.
"""
assert self._model is not None, "call load() first"
self._pending = np.concatenate([self._pending, pcm]) if self._pending.size else pcm
frame = settings.vad_frame
frame_ms = int(1000 * frame / settings.sample_rate_in)
done: list[Utterance] = []
started = False
while self._pending.size >= frame:
chunk = self._pending[:frame]
self._pending = self._pending[frame:]
with torch.no_grad():
prob = float(self._model(torch.from_numpy(chunk), settings.sample_rate_in).item())
voiced = prob >= settings.vad_threshold
st = self.state
if not st.speaking:
self._prefix.append(chunk)
if voiced:
st.speech_ms += frame_ms
if st.speech_ms >= settings.vad_min_speech_ms:
# Commit: open the turn with the pre-roll included.
st.speaking = True
st.silence_ms = 0
st.buffer = list(self._prefix)
self._prefix.clear()
started = True
else:
st.speech_ms = 0
continue
# Speaking.
st.buffer.append(chunk)
if voiced:
st.silence_ms = 0
else:
st.silence_ms += frame_ms
spoken_ms = len(st.buffer) * frame_ms
ended = st.silence_ms >= settings.vad_silence_ms
too_long = spoken_ms >= settings.vad_max_utterance_ms
if ended or too_long:
audio = np.concatenate(st.buffer)
done.append(Utterance(audio=audio, duration_ms=spoken_ms, truncated=too_long and not ended))
self.reset()
return done, started
-95
View File
@@ -1,95 +0,0 @@
"""Honest latency benchmark: warm up first, then time repeated runs.
The first CUDA generation pays for kernel autotuning and cache allocation, so a
single cold measurement makes any model look far worse than it is in service.
"""
import logging
import time
from threading import Thread
import numpy as np
import torch
logging.basicConfig(level=logging.INFO, format="%(message)s")
log = logging.getLogger("bench")
DEV = "cuda" if torch.cuda.is_available() else "cpu"
log.info("device=%s gpu=%s", DEV, torch.cuda.get_device_name(0) if DEV == "cuda" else "-")
# ── STT ──────────────────────────────────────────────────────────────────────
from transformers import AutoModel
stt = AutoModel.from_pretrained("ai4bharat/indic-conformer-600m-multilingual", trust_remote_code=True)
stt = stt.to(DEV).eval()
# Where does it actually run? A model wrapping ONNX ignores .to(cuda).
params = list(stt.parameters())
log.info("STT param device: %s (%d tensors)", params[0].device if params else "NO TORCH PARAMS", len(params))
log.info("STT type: %s", type(stt).__name__)
wav = torch.from_numpy((np.random.randn(16000 * 4) * 0.02).astype(np.float32)).unsqueeze(0).to(DEV)
with torch.inference_mode():
stt(wav, "ta", "ctc") # warmup
times = []
for _ in range(3):
t0 = time.perf_counter()
with torch.inference_mode():
stt(wav, "ta", "ctc")
times.append((time.perf_counter() - t0) * 1000)
log.info("STT 4000 ms audio → %.0f / %.0f / %.0f ms (RTF %.2fx)",
*times, (sum(times) / len(times)) / 4000)
# ── TTS ──────────────────────────────────────────────────────────────────────
from parler_tts import ParlerTTSForConditionalGeneration, ParlerTTSStreamer
from transformers import AutoTokenizer
dtype = torch.float16 if DEV == "cuda" else torch.float32
tts = ParlerTTSForConditionalGeneration.from_pretrained("ai4bharat/indic-parler-tts", torch_dtype=dtype).to(DEV).eval()
tok = AutoTokenizer.from_pretrained("ai4bharat/indic-parler-tts")
dtok = AutoTokenizer.from_pretrained(tts.config.text_encoder._name_or_path)
SR = tts.config.sampling_rate
log.info("TTS sampling_rate=%d frame_rate=%s", SR, getattr(tts.audio_encoder.config, "frame_rate", "?"))
desc = "Jaya speaks in a warm, clear, professional tone at a natural pace. The recording is very high quality with no background noise."
d = dtok(desc, return_tensors="pt").to(DEV)
def run(text, stream=True):
p = tok(text, return_tensors="pt").to(DEV)
kw = dict(input_ids=d.input_ids, attention_mask=d.attention_mask,
prompt_input_ids=p.input_ids, prompt_attention_mask=p.attention_mask)
t0 = time.perf_counter()
if stream:
fr = int(getattr(tts.audio_encoder.config, "frame_rate", 86) / 2)
s = ParlerTTSStreamer(tts, device=DEV, play_steps=fr)
Thread(target=tts.generate, kwargs={**kw, "streamer": s}, daemon=True).start()
first, n = None, 0
for c in s:
if c is None or len(c) == 0:
continue
if first is None:
first = (time.perf_counter() - t0) * 1000
n += len(c)
else:
with torch.inference_mode():
g = tts.generate(**kw)
n = g.shape[-1]
first = None
total = (time.perf_counter() - t0) * 1000
return first, total, 1000 * n / SR
SHORT = "மூவாயிரம் நானூறு லீட்கள் உள்ளன."
LONG = "புதிய லீட்கள் மூவாயிரம் நானூற்று இருபத்தேழு. இதில் எழுபத்தாறு சதவீதம் இன்னும் தொடர்பு கொள்ளப்படவில்லை."
run(SHORT) # warmup
log.info("")
for label, text in (("short", SHORT), ("long", LONG)):
first, total, audio = run(text)
log.info("TTS %-5s %2d chars → first %.0f ms | total %.0f ms | audio %.0f ms | RTF %.2fx",
label, len(text), first or -1, total, audio, total / max(audio, 1))
if DEV == "cuda":
log.info("\nVRAM peak reserved: %.2f GB of %.1f GB",
torch.cuda.max_memory_reserved() / 1e9,
torch.cuda.get_device_properties(0).total_memory / 1e9)
-47
View File
@@ -1,47 +0,0 @@
"""Which repos can we actually DOWNLOAD from?
`model_info` succeeds on a gated repo you have not been granted, so it is not a
usable test. Fetching a real file is.
"""
import os
from huggingface_hub import hf_hub_download
CANDIDATES = [
("ai4bharat/indic-conformer-600m-multilingual", "Indic ASR (22 languages)"),
("ai4bharat/indic-parler-tts", "Indic TTS (21 languages)"),
("openai/whisper-small", "English ASR + language ID"),
("ai4bharat/indic-parler-tts-pretrained", "TTS base (fallback)"),
("ai4bharat/indicconformer_stt_ta_hybrid_rnnt_large", "Tamil-only ASR (fallback)"),
]
token = os.environ.get("HF_TOKEN") or os.environ.get("HUGGING_FACE_HUB_TOKEN")
print(f"token present: {bool(token)}\n")
print(f"{'repo':52} {'what it is':30} status")
print("-" * 108)
blocked = []
for repo, what in CANDIDATES:
try:
hf_hub_download(repo_id=repo, filename="config.json", token=token)
print(f"{repo:52} {what:30} DOWNLOADABLE")
except Exception as e: # noqa: BLE001
msg = str(e)
if "not in the authorized list" in msg or "403" in msg:
print(f"{repo:52} {what:30} NEEDS ACCESS — click 'Agree' on the model page")
blocked.append(repo)
elif "401" in msg or "restricted" in msg:
print(f"{repo:52} {what:30} NOT AUTHENTICATED")
blocked.append(repo)
elif "404" in msg or "EntryNotFound" in msg:
# No config.json at the root, but the repo itself is reachable.
print(f"{repo:52} {what:30} reachable (no config.json)")
else:
print(f"{repo:52} {what:30} ERROR {msg[:34]}")
if blocked:
print("\nGrant access here (sign in, click 'Agree and access repository'):")
for repo in blocked:
print(f" https://huggingface.co/{repo}")
else:
print("\nAll required models are downloadable.")
-82
View File
@@ -1,82 +0,0 @@
"""De-risk before building around these models: do they load, fit, and run fast enough?"""
import logging
import time
import numpy as np
import torch
logging.basicConfig(level=logging.INFO, format="%(message)s")
log = logging.getLogger("probe")
DEV = "cuda" if torch.cuda.is_available() else "cpu"
def vram(tag):
if DEV == "cuda":
log.info(" VRAM %-10s alloc %.2f GB | reserved %.2f GB", tag,
torch.cuda.memory_allocated() / 1e9, torch.cuda.memory_reserved() / 1e9)
log.info("device=%s", DEV)
if DEV == "cuda":
log.info("gpu=%s total=%.1f GB", torch.cuda.get_device_name(0),
torch.cuda.get_device_properties(0).total_memory / 1e9)
# ── STT ──────────────────────────────────────────────────────────────────────
log.info("\n[1/2] loading IndicConformer…")
t0 = time.perf_counter()
from transformers import AutoModel
stt = AutoModel.from_pretrained("ai4bharat/indic-conformer-600m-multilingual", trust_remote_code=True)
stt = stt.to(DEV).eval()
log.info(" loaded in %.1fs", time.perf_counter() - t0)
vram("after STT")
# 3 s of quiet noise — we only care that a forward pass runs and how long it takes.
wav = torch.from_numpy((np.random.randn(16000 * 3) * 0.01).astype(np.float32)).unsqueeze(0).to(DEV)
for i in range(2):
t0 = time.perf_counter()
with torch.inference_mode():
out = stt(wav, "ta", "ctc")
log.info(" pass %d: %.0f ms -> %r", i + 1, (time.perf_counter() - t0) * 1000, str(out)[:60])
# ── TTS ──────────────────────────────────────────────────────────────────────
log.info("\n[2/2] loading Indic Parler-TTS…")
t0 = time.perf_counter()
from parler_tts import ParlerTTSForConditionalGeneration, ParlerTTSStreamer
from transformers import AutoTokenizer
dtype = torch.float16 if DEV == "cuda" else torch.float32
tts = ParlerTTSForConditionalGeneration.from_pretrained("ai4bharat/indic-parler-tts", torch_dtype=dtype).to(DEV).eval()
tok = AutoTokenizer.from_pretrained("ai4bharat/indic-parler-tts")
dtok = AutoTokenizer.from_pretrained(tts.config.text_encoder._name_or_path)
log.info(" loaded in %.1fs sr=%d", time.perf_counter() - t0, tts.config.sampling_rate)
vram("after TTS")
desc = "Jaya speaks in a warm, clear, professional tone at a natural pace. The recording is very high quality with no background noise."
prompt = "உங்கள் புதிய லீட்கள் மூன்று ஆயிரம் நானூறு."
d = dtok(desc, return_tensors="pt").to(DEV)
p = tok(prompt, return_tensors="pt").to(DEV)
kw = dict(input_ids=d.input_ids, attention_mask=d.attention_mask,
prompt_input_ids=p.input_ids, prompt_attention_mask=p.attention_mask)
# Streaming: what the user actually experiences is time-to-first-audio.
frame_rate = getattr(tts.audio_encoder.config, "frame_rate", 86)
streamer = ParlerTTSStreamer(tts, device=DEV, play_steps=int(frame_rate / 2))
from threading import Thread
t0 = time.perf_counter()
Thread(target=tts.generate, kwargs={**kw, "streamer": streamer}, daemon=True).start()
first_ms, total = None, 0
for chunk in streamer:
if chunk is None or len(chunk) == 0:
continue
if first_ms is None:
first_ms = (time.perf_counter() - t0) * 1000
total += len(chunk)
gen_ms = (time.perf_counter() - t0) * 1000
audio_ms = 1000 * total / tts.config.sampling_rate
log.info(" time to FIRST audio : %.0f ms", first_ms or -1)
log.info(" full generation : %.0f ms for %.0f ms of audio", gen_ms, audio_ms)
log.info(" realtime factor : %.2fx (<1 means faster than realtime)", gen_ms / max(audio_ms, 1))
vram("peak")
if DEV == "cuda":
log.info(" peak reserved: %.2f GB", torch.cuda.max_memory_reserved() / 1e9)
-11
View File
@@ -1,11 +0,0 @@
# Torch / transformers come from the system site-packages (torch 2.5.1+cu121).
# Only what the voice pipeline adds on top lives here.
fastapi>=0.115
uvicorn[standard]>=0.30
websockets>=12.0
soundfile>=0.13
numpy>=1.26
scipy>=1.10
sentencepiece>=0.2
# Parler-TTS is not published on PyPI; the Indic model needs this fork-compatible package.
git+https://github.com/huggingface/parler-tts.git