WeLe Agentic AI with Docker deployment
This commit is contained in:
@@ -55,3 +55,18 @@ RATE_LIMIT_MAX=40
|
|||||||
# --- Artifacts ---
|
# --- Artifacts ---
|
||||||
ARTIFACT_DIR=./storage/artifacts
|
ARTIFACT_DIR=./storage/artifacts
|
||||||
ARTIFACT_TTL_HOURS=72
|
ARTIFACT_TTL_HOURS=72
|
||||||
|
|
||||||
|
# --- Voice (speech-to-speech, CPU, in-process) ---
|
||||||
|
# Models are ONNX via Transformers.js — no GPU, no Python, no second service.
|
||||||
|
# STT onnx-community/whisper-base Tamil + English + language detection
|
||||||
|
# TTS assets/tts/mms-tts-tam exported locally; no public ONNX exists
|
||||||
|
# TTS Xenova/mms-tts-eng from the Hub
|
||||||
|
SPEECH_WARMUP=false
|
||||||
|
STT_MODEL=onnx-community/whisper-base
|
||||||
|
# Below this confidence, the user's preferred language beats the detector.
|
||||||
|
DETECT_CONFIDENCE=0.6
|
||||||
|
|
||||||
|
# Endpointing
|
||||||
|
VAD_SILENCE_MS=700
|
||||||
|
VAD_MIN_SPEECH_MS=250
|
||||||
|
VAD_PREFIX_MS=300
|
||||||
|
|||||||
+3
-4
@@ -10,7 +10,6 @@
|
|||||||
.env.*
|
.env.*
|
||||||
!.env.example
|
!.env.example
|
||||||
!**/*.env.example
|
!**/*.env.example
|
||||||
voice-service/.env
|
|
||||||
*.pem
|
*.pem
|
||||||
*.key
|
*.key
|
||||||
*.p12
|
*.p12
|
||||||
@@ -28,7 +27,6 @@ pnpm-debug.log*
|
|||||||
*.tsbuildinfo
|
*.tsbuildinfo
|
||||||
|
|
||||||
# ── Python (voice service) ──────────────────────────────────────────────────
|
# ── Python (voice service) ──────────────────────────────────────────────────
|
||||||
voice-service/.venv/
|
|
||||||
.venv/
|
.venv/
|
||||||
/venv/
|
/venv/
|
||||||
/env/
|
/env/
|
||||||
@@ -50,11 +48,13 @@ __pycache__/
|
|||||||
# redistribute models the licence does not allow us to redistribute.
|
# redistribute models the licence does not allow us to redistribute.
|
||||||
#
|
#
|
||||||
# Leading slashes matter: an unanchored `models/` also matches
|
# Leading slashes matter: an unanchored `models/` also matches
|
||||||
# `src/data/models/` — the CRM read-models — which silently kept them out of
|
# `src/data/models/
|
||||||
|
.transformers-cache/` — the CRM read-models — which silently kept them out of
|
||||||
# the repo and made the container crash with ERR_MODULE_NOT_FOUND.
|
# the repo and made the container crash with ERR_MODULE_NOT_FOUND.
|
||||||
/.cache/
|
/.cache/
|
||||||
/huggingface/
|
/huggingface/
|
||||||
/models/
|
/models/
|
||||||
|
.transformers-cache/
|
||||||
*.onnx
|
*.onnx
|
||||||
*.safetensors
|
*.safetensors
|
||||||
*.ckpt
|
*.ckpt
|
||||||
@@ -75,7 +75,6 @@ storage/artifacts/*
|
|||||||
*.wav
|
*.wav
|
||||||
*.mp3
|
*.mp3
|
||||||
*.flac
|
*.flac
|
||||||
!voice-service/app/assets/*.wav
|
|
||||||
|
|
||||||
# ── Editors / OS ────────────────────────────────────────────────────────────
|
# ── Editors / OS ────────────────────────────────────────────────────────────
|
||||||
.vscode/*
|
.vscode/*
|
||||||
|
|||||||
@@ -0,0 +1,3 @@
|
|||||||
|
{
|
||||||
|
"<unk>": 58
|
||||||
|
}
|
||||||
@@ -0,0 +1,82 @@
|
|||||||
|
{
|
||||||
|
"activation_dropout": 0.1,
|
||||||
|
"architectures": [
|
||||||
|
"VitsModel"
|
||||||
|
],
|
||||||
|
"attention_dropout": 0.1,
|
||||||
|
"depth_separable_channels": 2,
|
||||||
|
"depth_separable_num_layers": 3,
|
||||||
|
"dtype": "float32",
|
||||||
|
"duration_predictor_dropout": 0.5,
|
||||||
|
"duration_predictor_filter_channels": 256,
|
||||||
|
"duration_predictor_flow_bins": 10,
|
||||||
|
"duration_predictor_kernel_size": 3,
|
||||||
|
"duration_predictor_num_flows": 4,
|
||||||
|
"duration_predictor_tail_bound": 5.0,
|
||||||
|
"ffn_dim": 768,
|
||||||
|
"ffn_kernel_size": 3,
|
||||||
|
"flow_size": 192,
|
||||||
|
"hidden_act": "relu",
|
||||||
|
"hidden_dropout": 0.1,
|
||||||
|
"hidden_size": 192,
|
||||||
|
"initializer_range": 0.02,
|
||||||
|
"layer_norm_eps": 1e-05,
|
||||||
|
"layerdrop": 0.1,
|
||||||
|
"leaky_relu_slope": 0.1,
|
||||||
|
"model_type": "vits",
|
||||||
|
"noise_scale": 0.667,
|
||||||
|
"noise_scale_duration": 0.8,
|
||||||
|
"num_attention_heads": 2,
|
||||||
|
"num_hidden_layers": 6,
|
||||||
|
"num_speakers": 1,
|
||||||
|
"posterior_encoder_num_wavenet_layers": 16,
|
||||||
|
"prior_encoder_num_flows": 4,
|
||||||
|
"prior_encoder_num_wavenet_layers": 4,
|
||||||
|
"resblock_dilation_sizes": [
|
||||||
|
[
|
||||||
|
1,
|
||||||
|
3,
|
||||||
|
5
|
||||||
|
],
|
||||||
|
[
|
||||||
|
1,
|
||||||
|
3,
|
||||||
|
5
|
||||||
|
],
|
||||||
|
[
|
||||||
|
1,
|
||||||
|
3,
|
||||||
|
5
|
||||||
|
]
|
||||||
|
],
|
||||||
|
"resblock_kernel_sizes": [
|
||||||
|
3,
|
||||||
|
7,
|
||||||
|
11
|
||||||
|
],
|
||||||
|
"sampling_rate": 16000,
|
||||||
|
"speaker_embedding_size": 0,
|
||||||
|
"speaking_rate": 1.0,
|
||||||
|
"spectrogram_bins": 513,
|
||||||
|
"transformers_version": "4.57.3",
|
||||||
|
"upsample_initial_channel": 512,
|
||||||
|
"upsample_kernel_sizes": [
|
||||||
|
16,
|
||||||
|
16,
|
||||||
|
4,
|
||||||
|
4
|
||||||
|
],
|
||||||
|
"upsample_rates": [
|
||||||
|
8,
|
||||||
|
8,
|
||||||
|
2,
|
||||||
|
2
|
||||||
|
],
|
||||||
|
"use_bias": true,
|
||||||
|
"use_stochastic_duration_prediction": true,
|
||||||
|
"vocab_size": 58,
|
||||||
|
"wavenet_dilation_rate": 1,
|
||||||
|
"wavenet_dropout": 0.0,
|
||||||
|
"wavenet_kernel_size": 5,
|
||||||
|
"window_size": 4
|
||||||
|
}
|
||||||
@@ -0,0 +1,4 @@
|
|||||||
|
{
|
||||||
|
"per_channel": false,
|
||||||
|
"reduce_range": false
|
||||||
|
}
|
||||||
@@ -0,0 +1,4 @@
|
|||||||
|
{
|
||||||
|
"pad_token": "3",
|
||||||
|
"unk_token": "<unk>"
|
||||||
|
}
|
||||||
@@ -0,0 +1,115 @@
|
|||||||
|
{
|
||||||
|
"version": "1.0",
|
||||||
|
"truncation": null,
|
||||||
|
"padding": null,
|
||||||
|
"added_tokens": [
|
||||||
|
{
|
||||||
|
"id": 58,
|
||||||
|
"content": "<unk>",
|
||||||
|
"single_word": false,
|
||||||
|
"lstrip": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"special": true
|
||||||
|
}
|
||||||
|
],
|
||||||
|
"normalizer": {
|
||||||
|
"type": "Sequence",
|
||||||
|
"normalizers": [
|
||||||
|
{
|
||||||
|
"type": "Lowercase"
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"type": "Replace",
|
||||||
|
"pattern": {
|
||||||
|
"Regex": "[^012345679 '_aஅஆஇஈஉஊஎஏஐஒஓகஙசஜஞடணதநனபமயரறலளழவஷஸஹாிீுூெேைொோௌ்]"
|
||||||
|
},
|
||||||
|
"content": ""
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"type": "Strip",
|
||||||
|
"strip_left": true,
|
||||||
|
"strip_right": true
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"type": "Replace",
|
||||||
|
"pattern": {
|
||||||
|
"Regex": "(?=.)|(?<!^)$"
|
||||||
|
},
|
||||||
|
"content": "3"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
},
|
||||||
|
"pre_tokenizer": {
|
||||||
|
"type": "Split",
|
||||||
|
"pattern": {
|
||||||
|
"Regex": ""
|
||||||
|
},
|
||||||
|
"behavior": "Isolated",
|
||||||
|
"invert": false
|
||||||
|
},
|
||||||
|
"post_processor": null,
|
||||||
|
"decoder": null,
|
||||||
|
"model": {
|
||||||
|
"vocab": {
|
||||||
|
"0": 47,
|
||||||
|
"1": 44,
|
||||||
|
"2": 23,
|
||||||
|
"3": 0,
|
||||||
|
"4": 54,
|
||||||
|
"5": 57,
|
||||||
|
"6": 36,
|
||||||
|
"7": 14,
|
||||||
|
"9": 31,
|
||||||
|
" ": 7,
|
||||||
|
"'": 13,
|
||||||
|
"_": 4,
|
||||||
|
"a": 15,
|
||||||
|
"அ": 1,
|
||||||
|
"ஆ": 45,
|
||||||
|
"இ": 38,
|
||||||
|
"ஈ": 2,
|
||||||
|
"உ": 3,
|
||||||
|
"ஊ": 11,
|
||||||
|
"எ": 37,
|
||||||
|
"ஏ": 16,
|
||||||
|
"ஐ": 52,
|
||||||
|
"ஒ": 27,
|
||||||
|
"ஓ": 49,
|
||||||
|
"க": 6,
|
||||||
|
"ங": 50,
|
||||||
|
"ச": 30,
|
||||||
|
"ஜ": 53,
|
||||||
|
"ஞ": 29,
|
||||||
|
"ட": 22,
|
||||||
|
"ண": 48,
|
||||||
|
"த": 41,
|
||||||
|
"ந": 5,
|
||||||
|
"ன": 35,
|
||||||
|
"ப": 46,
|
||||||
|
"ம": 26,
|
||||||
|
"ய": 39,
|
||||||
|
"ர": 25,
|
||||||
|
"ற": 28,
|
||||||
|
"ல": 21,
|
||||||
|
"ள": 43,
|
||||||
|
"ழ": 24,
|
||||||
|
"வ": 17,
|
||||||
|
"ஷ": 55,
|
||||||
|
"ஸ": 33,
|
||||||
|
"ஹ": 19,
|
||||||
|
"ா": 9,
|
||||||
|
"ி": 32,
|
||||||
|
"ீ": 12,
|
||||||
|
"ு": 51,
|
||||||
|
"ூ": 20,
|
||||||
|
"ெ": 10,
|
||||||
|
"ே": 8,
|
||||||
|
"ை": 34,
|
||||||
|
"ொ": 56,
|
||||||
|
"ோ": 42,
|
||||||
|
"ௌ": 40,
|
||||||
|
"்": 18
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -0,0 +1,31 @@
|
|||||||
|
{
|
||||||
|
"add_blank": true,
|
||||||
|
"added_tokens_decoder": {
|
||||||
|
"0": {
|
||||||
|
"content": "3",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"58": {
|
||||||
|
"content": "<unk>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"clean_up_tokenization_spaces": true,
|
||||||
|
"extra_special_tokens": {},
|
||||||
|
"is_uroman": false,
|
||||||
|
"language": "tam",
|
||||||
|
"model_max_length": 1000000000000000019884624838656,
|
||||||
|
"normalize": true,
|
||||||
|
"pad_token": "3",
|
||||||
|
"phonemize": false,
|
||||||
|
"tokenizer_class": "VitsTokenizer",
|
||||||
|
"unk_token": "<unk>"
|
||||||
|
}
|
||||||
@@ -0,0 +1,60 @@
|
|||||||
|
{
|
||||||
|
" ": 7,
|
||||||
|
"'": 13,
|
||||||
|
"0": 47,
|
||||||
|
"1": 44,
|
||||||
|
"2": 23,
|
||||||
|
"3": 0,
|
||||||
|
"4": 54,
|
||||||
|
"5": 57,
|
||||||
|
"6": 36,
|
||||||
|
"7": 14,
|
||||||
|
"9": 31,
|
||||||
|
"_": 4,
|
||||||
|
"a": 15,
|
||||||
|
"அ": 1,
|
||||||
|
"ஆ": 45,
|
||||||
|
"இ": 38,
|
||||||
|
"ஈ": 2,
|
||||||
|
"உ": 3,
|
||||||
|
"ஊ": 11,
|
||||||
|
"எ": 37,
|
||||||
|
"ஏ": 16,
|
||||||
|
"ஐ": 52,
|
||||||
|
"ஒ": 27,
|
||||||
|
"ஓ": 49,
|
||||||
|
"க": 6,
|
||||||
|
"ங": 50,
|
||||||
|
"ச": 30,
|
||||||
|
"ஜ": 53,
|
||||||
|
"ஞ": 29,
|
||||||
|
"ட": 22,
|
||||||
|
"ண": 48,
|
||||||
|
"த": 41,
|
||||||
|
"ந": 5,
|
||||||
|
"ன": 35,
|
||||||
|
"ப": 46,
|
||||||
|
"ம": 26,
|
||||||
|
"ய": 39,
|
||||||
|
"ர": 25,
|
||||||
|
"ற": 28,
|
||||||
|
"ல": 21,
|
||||||
|
"ள": 43,
|
||||||
|
"ழ": 24,
|
||||||
|
"வ": 17,
|
||||||
|
"ஷ": 55,
|
||||||
|
"ஸ": 33,
|
||||||
|
"ஹ": 19,
|
||||||
|
"ா": 9,
|
||||||
|
"ி": 32,
|
||||||
|
"ீ": 12,
|
||||||
|
"ு": 51,
|
||||||
|
"ூ": 20,
|
||||||
|
"ெ": 10,
|
||||||
|
"ே": 8,
|
||||||
|
"ை": 34,
|
||||||
|
"ொ": 56,
|
||||||
|
"ோ": 42,
|
||||||
|
"ௌ": 40,
|
||||||
|
"்": 18
|
||||||
|
}
|
||||||
Generated
+955
File diff suppressed because it is too large
Load Diff
+3
-2
@@ -7,12 +7,12 @@
|
|||||||
"scripts": {
|
"scripts": {
|
||||||
"start": "node src/server.js",
|
"start": "node src/server.js",
|
||||||
"dev": "node --watch src/server.js",
|
"dev": "node --watch src/server.js",
|
||||||
"smoke": "node scripts/smoke.js",
|
"smoke": "node scripts/smoke.js"
|
||||||
"voice": "voice-service/.venv/Scripts/python.exe -m app.server"
|
|
||||||
},
|
},
|
||||||
"author": "WeLe EdTech",
|
"author": "WeLe EdTech",
|
||||||
"license": "ISC",
|
"license": "ISC",
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
|
"@huggingface/transformers": "^4.2.0",
|
||||||
"@langchain/core": "^1.1.18",
|
"@langchain/core": "^1.1.18",
|
||||||
"@langchain/langgraph": "^1.1.0",
|
"@langchain/langgraph": "^1.1.0",
|
||||||
"@langchain/openai": "^1.5.10",
|
"@langchain/openai": "^1.5.10",
|
||||||
@@ -27,6 +27,7 @@
|
|||||||
"ioredis": "^5.10.1",
|
"ioredis": "^5.10.1",
|
||||||
"jsonwebtoken": "^9.0.2",
|
"jsonwebtoken": "^9.0.2",
|
||||||
"mongoose": "^9.2.3",
|
"mongoose": "^9.2.3",
|
||||||
|
"onnxruntime-node": "^1.24.3",
|
||||||
"pdfkit": "^0.17.2",
|
"pdfkit": "^0.17.2",
|
||||||
"pptxgenjs": "^4.0.1",
|
"pptxgenjs": "^4.0.1",
|
||||||
"uuid": "^13.0.0",
|
"uuid": "^13.0.0",
|
||||||
|
|||||||
@@ -0,0 +1,66 @@
|
|||||||
|
/* Generate the tokenizer.json that Transformers.js needs for the exported
|
||||||
|
Tamil VITS model.
|
||||||
|
|
||||||
|
`save_pretrained` does not emit one: VitsTokenizer is a "slow" tokenizer with
|
||||||
|
no fast counterpart, so Python writes vocab.json + tokenizer_config.json and
|
||||||
|
nothing else. Transformers.js only reads tokenizer.json, so we synthesise it
|
||||||
|
from the exported vocab, mirroring the structure of the working English
|
||||||
|
model (Xenova/mms-tts-eng) exactly.
|
||||||
|
|
||||||
|
The four normalizer steps, in order:
|
||||||
|
1. Lowercase — no-op for Tamil, matters for embedded Latin/digits
|
||||||
|
2. Replace — drop every character outside the vocab
|
||||||
|
3. Strip — trim surrounding whitespace
|
||||||
|
4. Replace — insert the blank token between every character,
|
||||||
|
which is what `add_blank: true` means for VITS.
|
||||||
|
Omit this and the audio comes out garbled.
|
||||||
|
*/
|
||||||
|
import fs from 'node:fs';
|
||||||
|
import path from 'node:path';
|
||||||
|
|
||||||
|
const DIR = 'assets/tts/mms-tts-tam';
|
||||||
|
const vocab = JSON.parse(fs.readFileSync(path.join(DIR, 'vocab.json'), 'utf8'));
|
||||||
|
const cfg = JSON.parse(fs.readFileSync(path.join(DIR, 'tokenizer_config.json'), 'utf8'));
|
||||||
|
|
||||||
|
// The blank/pad token is whichever character maps to id 0.
|
||||||
|
const blank = Object.keys(vocab).find((k) => vocab[k] === 0);
|
||||||
|
const unk = cfg.unk_token ?? '<unk>';
|
||||||
|
const unkId = vocab[unk] ?? Object.keys(vocab).length;
|
||||||
|
|
||||||
|
// Character class of everything we keep. Escape the regex metacharacters that
|
||||||
|
// are still special inside a negated class.
|
||||||
|
const escaped = Object.keys(vocab)
|
||||||
|
.filter((c) => c !== unk)
|
||||||
|
.map((c) => (']\\^-'.includes(c) ? '\\' + c : c))
|
||||||
|
.join('');
|
||||||
|
|
||||||
|
const tokenizer = {
|
||||||
|
version: '1.0',
|
||||||
|
truncation: null,
|
||||||
|
padding: null,
|
||||||
|
added_tokens: [{
|
||||||
|
id: unkId, content: unk,
|
||||||
|
single_word: false, lstrip: false, rstrip: false, normalized: false, special: true,
|
||||||
|
}],
|
||||||
|
normalizer: {
|
||||||
|
type: 'Sequence',
|
||||||
|
normalizers: [
|
||||||
|
{ type: 'Lowercase' },
|
||||||
|
{ type: 'Replace', pattern: { Regex: `[^${escaped}]` }, content: '' },
|
||||||
|
{ type: 'Strip', strip_left: true, strip_right: true },
|
||||||
|
...(cfg.add_blank ? [{ type: 'Replace', pattern: { Regex: '(?=.)|(?<!^)$' }, content: blank }] : []),
|
||||||
|
],
|
||||||
|
},
|
||||||
|
pre_tokenizer: { type: 'Split', pattern: { Regex: '' }, behavior: 'Isolated', invert: false },
|
||||||
|
post_processor: null,
|
||||||
|
decoder: null,
|
||||||
|
model: { vocab },
|
||||||
|
};
|
||||||
|
|
||||||
|
fs.writeFileSync(path.join(DIR, 'tokenizer.json'), JSON.stringify(tokenizer, null, 1));
|
||||||
|
|
||||||
|
console.log(`wrote ${DIR}/tokenizer.json`);
|
||||||
|
console.log(` vocab ${Object.keys(vocab).length} tokens`);
|
||||||
|
console.log(` blank token ${JSON.stringify(blank)} (id 0)`);
|
||||||
|
console.log(` unk ${JSON.stringify(unk)} (id ${unkId})`);
|
||||||
|
console.log(` add_blank ${cfg.add_blank}`);
|
||||||
@@ -0,0 +1,99 @@
|
|||||||
|
"""One-time export of facebook/mms-tts-tam to ONNX.
|
||||||
|
|
||||||
|
No public ONNX build of Tamil MMS-TTS exists, so we make one. This runs ONCE on
|
||||||
|
a workstation; the committed artefact is what ships. The service itself is pure
|
||||||
|
JavaScript and never needs Python or this script.
|
||||||
|
|
||||||
|
The output must match the contract Transformers.js expects for VITS, taken from
|
||||||
|
the working English model (Xenova/mms-tts-eng):
|
||||||
|
|
||||||
|
inputs : input_ids, attention_mask
|
||||||
|
outputs: waveform, spectrogram
|
||||||
|
|
||||||
|
Layout produced (mirrors the HF repo so Transformers.js can load the folder):
|
||||||
|
|
||||||
|
assets/tts/mms-tts-tam/
|
||||||
|
config.json, tokenizer.json, vocab.json, …
|
||||||
|
onnx/model.onnx fp32
|
||||||
|
onnx/model_quantized.onnx int8 ← what we ship
|
||||||
|
"""
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import shutil
|
||||||
|
import sys
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
import torch
|
||||||
|
from transformers import AutoTokenizer, VitsModel
|
||||||
|
|
||||||
|
MODEL = "facebook/mms-tts-tam"
|
||||||
|
OUT = Path("assets/tts/mms-tts-tam")
|
||||||
|
ONNX_DIR = OUT / "onnx"
|
||||||
|
ONNX_DIR.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
print(f"loading {MODEL} …")
|
||||||
|
model = VitsModel.from_pretrained(MODEL).eval()
|
||||||
|
tok = AutoTokenizer.from_pretrained(MODEL)
|
||||||
|
|
||||||
|
# VITS has a stochastic duration predictor. Exporting with noise left on bakes
|
||||||
|
# RandomNormalLike nodes into the graph, which is fine and keeps prosody
|
||||||
|
# natural — but seed it so this export is reproducible.
|
||||||
|
torch.manual_seed(0)
|
||||||
|
|
||||||
|
|
||||||
|
class Exportable(torch.nn.Module):
|
||||||
|
"""Return only (waveform, spectrogram), in that order — the JS side indexes
|
||||||
|
outputs by name, but a plain tuple keeps the exported graph simple."""
|
||||||
|
|
||||||
|
def __init__(self, m: VitsModel) -> None:
|
||||||
|
super().__init__()
|
||||||
|
self.m = m
|
||||||
|
|
||||||
|
def forward(self, input_ids: torch.Tensor, attention_mask: torch.Tensor):
|
||||||
|
out = self.m(input_ids=input_ids, attention_mask=attention_mask)
|
||||||
|
return out.waveform, out.spectrogram
|
||||||
|
|
||||||
|
|
||||||
|
sample = tok("வணக்கம், இது ஒரு சோதனை.", return_tensors="pt")
|
||||||
|
fp32 = ONNX_DIR / "model.onnx"
|
||||||
|
|
||||||
|
print("exporting to ONNX …")
|
||||||
|
torch.onnx.export(
|
||||||
|
Exportable(model),
|
||||||
|
(sample["input_ids"], sample["attention_mask"]),
|
||||||
|
str(fp32),
|
||||||
|
input_names=["input_ids", "attention_mask"],
|
||||||
|
output_names=["waveform", "spectrogram"],
|
||||||
|
dynamic_axes={
|
||||||
|
"input_ids": {0: "batch", 1: "sequence"},
|
||||||
|
"attention_mask": {0: "batch", 1: "sequence"},
|
||||||
|
"waveform": {0: "batch", 1: "samples"},
|
||||||
|
"spectrogram": {0: "batch", 2: "frames"},
|
||||||
|
},
|
||||||
|
opset_version=17,
|
||||||
|
do_constant_folding=True,
|
||||||
|
)
|
||||||
|
print(f" fp32: {fp32.stat().st_size / 1e6:.1f} MB")
|
||||||
|
|
||||||
|
# ── int8 ────────────────────────────────────────────────────────────────────
|
||||||
|
try:
|
||||||
|
from onnxruntime.quantization import QuantType, quantize_dynamic
|
||||||
|
|
||||||
|
q = ONNX_DIR / "model_quantized.onnx"
|
||||||
|
quantize_dynamic(str(fp32), str(q), weight_type=QuantType.QUInt8)
|
||||||
|
print(f" int8: {q.stat().st_size / 1e6:.1f} MB")
|
||||||
|
except Exception as e: # noqa: BLE001
|
||||||
|
print(f" quantisation skipped: {e}")
|
||||||
|
|
||||||
|
# ── tokenizer + config, so the folder loads standalone ──────────────────────
|
||||||
|
tok.save_pretrained(OUT)
|
||||||
|
model.config.to_json_file(OUT / "config.json")
|
||||||
|
|
||||||
|
# Transformers.js reads this to pick a default dtype.
|
||||||
|
(OUT / "quantize_config.json").write_text(json.dumps({"per_channel": False, "reduce_range": False}, indent=2))
|
||||||
|
|
||||||
|
print("\nwrote:")
|
||||||
|
for p in sorted(OUT.rglob("*")):
|
||||||
|
if p.is_file():
|
||||||
|
print(f" {p.relative_to(OUT)} ({p.stat().st_size / 1e6:.2f} MB)")
|
||||||
@@ -0,0 +1,21 @@
|
|||||||
|
/* Does auto mode now route Tamil to Tamil instead of silently using English? */
|
||||||
|
import { synthesize, transcribe } from '../src/speech/index.js';
|
||||||
|
|
||||||
|
const resample = (a, from, to) => {
|
||||||
|
const r = from / to, out = new Float32Array(Math.floor(a.length / r));
|
||||||
|
for (let i = 0; i < out.length; i++) { const p = i * r, k = Math.floor(p); out[i] = a[k] + (a[Math.min(k + 1, a.length - 1)] - a[k]) * (p - k); }
|
||||||
|
return out;
|
||||||
|
};
|
||||||
|
|
||||||
|
for (const [lang, text] of [
|
||||||
|
['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'],
|
||||||
|
['en', 'How many new leads did we receive today?'],
|
||||||
|
]) {
|
||||||
|
const spoken = await synthesize(text, lang);
|
||||||
|
const audio = resample(spoken.audio, spoken.sampling_rate, 16000);
|
||||||
|
const r = await transcribe(audio, 'auto', 'ta');
|
||||||
|
const ok = r.lang === lang ? 'PASS' : 'FAIL';
|
||||||
|
console.log(`${ok} spoke ${lang} → routed ${r.lang} (detected ${r.detected} @ ${r.confidence}) ${r.ms}ms`);
|
||||||
|
console.log(` ${JSON.stringify(r.text.slice(0, 80))}`);
|
||||||
|
}
|
||||||
|
process.exit(0);
|
||||||
@@ -0,0 +1,62 @@
|
|||||||
|
/* Can we get real language detection out of Whisper in Transformers.js? */
|
||||||
|
import { AutoProcessor, WhisperForConditionalGeneration, Tensor, env } from '@huggingface/transformers';
|
||||||
|
import { synthesize } from '../src/speech/index.js';
|
||||||
|
|
||||||
|
env.cacheDir = './.transformers-cache';
|
||||||
|
const MODEL = 'onnx-community/whisper-base';
|
||||||
|
|
||||||
|
const processor = await AutoProcessor.from_pretrained(MODEL);
|
||||||
|
const model = await WhisperForConditionalGeneration.from_pretrained(MODEL, { dtype: 'q8' });
|
||||||
|
const tok = processor.tokenizer;
|
||||||
|
|
||||||
|
// Whisper emits one language token right after <|startoftranscript|>. Reading
|
||||||
|
// that distribution is a single decoder step — far cheaper than transcribing
|
||||||
|
// twice to see which language "looks better".
|
||||||
|
const id = (t) => tok.encode(t, { add_special_tokens: false })[0];
|
||||||
|
const sot = id('<|startoftranscript|>');
|
||||||
|
const CANDIDATES = ['en', 'ta'];
|
||||||
|
const langIds = CANDIDATES.map((c) => id(`<|${c}|>`));
|
||||||
|
console.log('sot:', sot, '| language token ids:', JSON.stringify(Object.fromEntries(CANDIDATES.map((c, i) => [c, langIds[i]]))));
|
||||||
|
|
||||||
|
function resample(a, from, to) {
|
||||||
|
const r = from / to;
|
||||||
|
const out = new Float32Array(Math.floor(a.length / r));
|
||||||
|
for (let i = 0; i < out.length; i++) {
|
||||||
|
const p = i * r, k = Math.floor(p);
|
||||||
|
out[i] = a[k] + (a[Math.min(k + 1, a.length - 1)] - a[k]) * (p - k);
|
||||||
|
}
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
for (const [lang, text] of [
|
||||||
|
['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'],
|
||||||
|
['en', 'There are three thousand four hundred and twenty seven new leads.'],
|
||||||
|
]) {
|
||||||
|
const spoken = await synthesize(text, lang);
|
||||||
|
const audio = resample(spoken.audio, spoken.sampling_rate, 16000);
|
||||||
|
|
||||||
|
const inputs = await processor(audio);
|
||||||
|
const t0 = Date.now();
|
||||||
|
const out = await model({
|
||||||
|
...inputs,
|
||||||
|
decoder_input_ids: new Tensor('int64', BigInt64Array.from([BigInt(sot)]), [1, 1]),
|
||||||
|
});
|
||||||
|
const ms = Date.now() - t0;
|
||||||
|
|
||||||
|
const logits = out.logits;
|
||||||
|
const last = logits.dims[1] - 1;
|
||||||
|
const vocab = logits.dims[2];
|
||||||
|
const row = logits.data.slice(last * vocab, (last + 1) * vocab);
|
||||||
|
|
||||||
|
const scores = langIds.map((id) => Number(row[id]));
|
||||||
|
const max = Math.max(...scores);
|
||||||
|
const exp = scores.map((s) => Math.exp(s - max));
|
||||||
|
const sum = exp.reduce((a, b) => a + b, 0);
|
||||||
|
const probs = exp.map((e) => e / sum);
|
||||||
|
const best = probs.indexOf(Math.max(...probs));
|
||||||
|
|
||||||
|
console.log(`spoken ${lang} → detected ${CANDIDATES[best]} `
|
||||||
|
+ `(${CANDIDATES.map((c, i) => `${c} ${probs[i].toFixed(3)}`).join(', ')}) in ${ms}ms `
|
||||||
|
+ `${CANDIDATES[best] === lang ? '✅' : '❌'}`);
|
||||||
|
}
|
||||||
|
process.exit(0);
|
||||||
@@ -0,0 +1,44 @@
|
|||||||
|
/* Do Whisper (STT) and MMS-TTS (TTS) actually run in Node on CPU? */
|
||||||
|
import { pipeline, env } from '@huggingface/transformers';
|
||||||
|
|
||||||
|
env.cacheDir = './.transformers-cache';
|
||||||
|
|
||||||
|
const t = (t0) => `${((performance.now() - t0) / 1000).toFixed(1)}s`;
|
||||||
|
|
||||||
|
// ── TTS: MMS-TTS Tamil (VITS, 36M, feed-forward) ────────────────────────────
|
||||||
|
console.log('[1/2] loading MMS-TTS Tamil…');
|
||||||
|
let t0 = performance.now();
|
||||||
|
const tts = await pipeline('text-to-speech', 'Xenova/mms-tts-eng', { dtype: 'fp32' });
|
||||||
|
console.log(` loaded in ${t(t0)}`);
|
||||||
|
|
||||||
|
const TA = 'There are three thousand four hundred leads in the new lead stage.';
|
||||||
|
await tts(TA); // warm
|
||||||
|
for (const [label, text] of [['short', TA], ['long', TA + ' ' + TA + ' ' + TA]]) {
|
||||||
|
t0 = performance.now();
|
||||||
|
const out = await tts(text);
|
||||||
|
const ms = performance.now() - t0;
|
||||||
|
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
|
||||||
|
console.log(
|
||||||
|
` ${label.padEnd(5)} ${String(text.length).padStart(3)} chars → ${ms.toFixed(0)}ms `
|
||||||
|
+ `for ${audioMs.toFixed(0)}ms audio @ ${out.sampling_rate}Hz → RTF ${(ms / audioMs).toFixed(2)}x`,
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── STT: Whisper (multilingual — Tamil, Hindi, English + detection) ─────────
|
||||||
|
console.log('\n[2/2] loading Whisper base…');
|
||||||
|
t0 = performance.now();
|
||||||
|
const stt = await pipeline('automatic-speech-recognition', 'onnx-community/whisper-base', { dtype: 'q8' });
|
||||||
|
console.log(` loaded in ${t(t0)}`);
|
||||||
|
|
||||||
|
// 4 s of quiet noise — proves the graph runs and times it.
|
||||||
|
const audio = Float32Array.from({ length: 16000 * 4 }, () => (Math.random() - 0.5) * 0.02);
|
||||||
|
t0 = performance.now();
|
||||||
|
const r = await stt(audio, { language: 'ta', task: 'transcribe' });
|
||||||
|
console.log(` 4000ms audio → ${(performance.now() - t0).toFixed(0)}ms → ${JSON.stringify(r.text).slice(0, 60)}`);
|
||||||
|
|
||||||
|
t0 = performance.now();
|
||||||
|
const r2 = await stt(audio, { language: 'en', task: 'transcribe' });
|
||||||
|
console.log(` english pass → ${(performance.now() - t0).toFixed(0)}ms → ${JSON.stringify(r2.text).slice(0, 60)}`);
|
||||||
|
|
||||||
|
console.log(`\nRSS ${(process.memoryUsage().rss / 1e9).toFixed(2)} GB`);
|
||||||
|
process.exit(0);
|
||||||
@@ -0,0 +1,55 @@
|
|||||||
|
/* Does the locally-exported Tamil ONNX load and speak through Transformers.js? */
|
||||||
|
import { pipeline, env } from '@huggingface/transformers';
|
||||||
|
import fs from 'node:fs';
|
||||||
|
|
||||||
|
// Load from the local folder, not the Hub.
|
||||||
|
env.allowRemoteModels = false;
|
||||||
|
env.localModelPath = './assets/tts';
|
||||||
|
|
||||||
|
for (const dtype of ['q8', 'fp32']) {
|
||||||
|
try {
|
||||||
|
const t0 = performance.now();
|
||||||
|
const tts = await pipeline('text-to-speech', 'mms-tts-tam', { dtype });
|
||||||
|
const load = performance.now() - t0;
|
||||||
|
|
||||||
|
const TEXT = 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.';
|
||||||
|
await tts(TEXT); // warm
|
||||||
|
|
||||||
|
const t1 = performance.now();
|
||||||
|
const out = await tts(TEXT);
|
||||||
|
const ms = performance.now() - t1;
|
||||||
|
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
|
||||||
|
|
||||||
|
console.log(
|
||||||
|
`${dtype.padEnd(5)} load ${(load / 1000).toFixed(1)}s | `
|
||||||
|
+ `${ms.toFixed(0)}ms for ${audioMs.toFixed(0)}ms audio @ ${out.sampling_rate}Hz | `
|
||||||
|
+ `RTF ${(ms / audioMs).toFixed(2)}x`,
|
||||||
|
);
|
||||||
|
|
||||||
|
// Non-silent output is the real proof the graph is wired correctly.
|
||||||
|
const peak = out.audio.reduce((m, v) => Math.max(m, Math.abs(v)), 0);
|
||||||
|
console.log(` samples ${out.audio.length}, peak amplitude ${peak.toFixed(3)} ${peak > 0.01 ? '✅ audible' : '⚠️ SILENT'}`);
|
||||||
|
|
||||||
|
if (dtype === 'q8') {
|
||||||
|
const wav = toWav(out.audio, out.sampling_rate);
|
||||||
|
fs.writeFileSync('scripts/tamil-sample.wav', wav);
|
||||||
|
console.log(' wrote scripts/tamil-sample.wav — play it to judge quality');
|
||||||
|
}
|
||||||
|
} catch (e) {
|
||||||
|
console.log(`${dtype.padEnd(5)} FAILED: ${e.message.slice(0, 160)}`);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
function toWav(samples, rate) {
|
||||||
|
const buf = Buffer.alloc(44 + samples.length * 2);
|
||||||
|
buf.write('RIFF', 0); buf.writeUInt32LE(36 + samples.length * 2, 4); buf.write('WAVE', 8);
|
||||||
|
buf.write('fmt ', 12); buf.writeUInt32LE(16, 16); buf.writeUInt16LE(1, 20); buf.writeUInt16LE(1, 22);
|
||||||
|
buf.writeUInt32LE(rate, 24); buf.writeUInt32LE(rate * 2, 28); buf.writeUInt16LE(2, 32); buf.writeUInt16LE(16, 34);
|
||||||
|
buf.write('data', 36); buf.writeUInt32LE(samples.length * 2, 40);
|
||||||
|
for (let i = 0; i < samples.length; i++) {
|
||||||
|
const s = Math.max(-1, Math.min(1, samples[i]));
|
||||||
|
buf.writeInt16LE(s < 0 ? s * 0x8000 : s * 0x7fff, 44 + i * 2);
|
||||||
|
}
|
||||||
|
return buf;
|
||||||
|
}
|
||||||
|
process.exit(0);
|
||||||
@@ -0,0 +1,72 @@
|
|||||||
|
/* End-to-end check of the in-process speech pipeline: TTS → VAD → STT.
|
||||||
|
|
||||||
|
Synthesising a sentence and feeding that audio back through the endpointer
|
||||||
|
and recogniser exercises every stage with real speech, which a noise buffer
|
||||||
|
cannot do — silence never opens a VAD turn.
|
||||||
|
*/
|
||||||
|
import { synthesize, transcribe, sentences, speakable, Endpointer } from '../src/speech/index.js';
|
||||||
|
|
||||||
|
const say = (m) => console.log(m);
|
||||||
|
|
||||||
|
// ── 1. TTS both languages ───────────────────────────────────────────────────
|
||||||
|
const CASES = [
|
||||||
|
['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'],
|
||||||
|
['en', 'There are three thousand four hundred and twenty seven new leads.'],
|
||||||
|
];
|
||||||
|
|
||||||
|
const rendered = {};
|
||||||
|
for (const [lang, text] of CASES) {
|
||||||
|
const t0 = Date.now();
|
||||||
|
await synthesize(text, lang); // warm
|
||||||
|
const t1 = Date.now();
|
||||||
|
const out = await synthesize(text, lang);
|
||||||
|
const ms = Date.now() - t1;
|
||||||
|
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
|
||||||
|
rendered[lang] = out;
|
||||||
|
say(`TTS ${lang} warm ${((t1 - t0) / 1000).toFixed(1)}s | ${ms}ms for ${audioMs.toFixed(0)}ms `
|
||||||
|
+ `@${out.sampling_rate}Hz | RTF ${(ms / audioMs).toFixed(2)}x`);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── 2. VAD: does synthesised speech open and close a turn? ──────────────────
|
||||||
|
function resample(audio, from, to) {
|
||||||
|
if (from === to) return audio;
|
||||||
|
const ratio = from / to;
|
||||||
|
const out = new Float32Array(Math.floor(audio.length / ratio));
|
||||||
|
for (let i = 0; i < out.length; i++) {
|
||||||
|
const p = i * ratio;
|
||||||
|
const a = Math.floor(p);
|
||||||
|
out[i] = audio[a] + (audio[Math.min(a + 1, audio.length - 1)] - audio[a]) * (p - a);
|
||||||
|
}
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
const ep = new Endpointer();
|
||||||
|
const speech = resample(rendered.en.audio, rendered.en.sampling_rate, 16000);
|
||||||
|
// Speech, then a second of silence so the endpointer closes the turn.
|
||||||
|
const withTail = new Float32Array(speech.length + 16000);
|
||||||
|
withTail.set(speech);
|
||||||
|
|
||||||
|
let started = false;
|
||||||
|
let captured = null;
|
||||||
|
for (let i = 0; i < withTail.length; i += 640) { // 40 ms chunks, as the browser sends
|
||||||
|
const { utterances, started: s } = await ep.push(withTail.subarray(i, Math.min(i + 640, withTail.length)));
|
||||||
|
if (s) started = true;
|
||||||
|
if (utterances.length) { captured = utterances[0]; break; }
|
||||||
|
}
|
||||||
|
say(`VAD speech detected: ${started ? 'yes' : 'NO'} | turn closed: ${captured ? 'yes' : 'NO'}`
|
||||||
|
+ (captured ? ` | captured ${(captured.length / 16000).toFixed(2)}s` : ''));
|
||||||
|
|
||||||
|
// ── 3. STT on that captured audio ───────────────────────────────────────────
|
||||||
|
if (captured) {
|
||||||
|
for (const lang of ['en', 'auto']) {
|
||||||
|
const r = await transcribe(captured, lang, 'en');
|
||||||
|
say(`STT ${lang.padEnd(4)} ${r.ms}ms → lang=${r.lang}${r.detected ? ` (heard ${r.detected})` : ''} → ${JSON.stringify(r.text.slice(0, 70))}`);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── 4. Text shaping ─────────────────────────────────────────────────────────
|
||||||
|
const md = '## Leads\n\n**3,427** in `new_lead`.\n\n| a | b |\n|---|---|\n| 1 | 2 |\n\n- Only 12% contacted.\nNext step is triage.';
|
||||||
|
say(`\nspeakable: ${JSON.stringify(speakable(md))}`);
|
||||||
|
say(`sentences: ${JSON.stringify(sentences(speakable(md)))}`);
|
||||||
|
say(`\nRSS ${(process.memoryUsage().rss / 1e9).toFixed(2)} GB`);
|
||||||
|
process.exit(0);
|
||||||
+122
-169
@@ -1,104 +1,59 @@
|
|||||||
// ============================================
|
// ============================================
|
||||||
// Voice channel — a WebSocket bridge between the browser and the GPU service.
|
// Voice channel — speech in, speech out, in this same Node process.
|
||||||
//
|
//
|
||||||
// Voice is a *channel*, not a parallel product: a spoken question runs through
|
// Voice is a *channel*, not a parallel product: a spoken question runs through
|
||||||
// the same graph, guardrails and agents as a typed one. Only the transport and
|
// the same graph, guardrails and agents as a typed one. Only the transport and
|
||||||
// the presentation differ, which is why this file contains no CRM logic.
|
// the presentation differ, which is why this file contains no CRM logic.
|
||||||
//
|
//
|
||||||
// browser ──audio──► this ──audio──► python(:4100) ──STT──► transcript
|
// browser ──PCM16──► Endpointer ──► transcribe() ──► runTurn(graph)
|
||||||
// │ │
|
// │ │
|
||||||
// └──────────── runTurn(graph) ◄───────────┘
|
// browser ◄──float32──── synthesize() ◄── sentences ◄─────┘
|
||||||
// │
|
|
||||||
// browser ◄──audio── this ◄──audio── python(TTS) ◄──sentences───┘
|
|
||||||
//
|
//
|
||||||
// The hard problem is not transport, it is that a turn takes 17–46 s. Silence
|
// The hard problem is not audio, it is that a turn takes 17–46 s and that is
|
||||||
// for that long feels broken, so the bridge speaks immediately, narrates what
|
// dead silence in voice. So this speaks an acknowledgement within ~1 s,
|
||||||
// the agents are doing, and starts reading the answer at the first sentence
|
// narrates each agent delegation aloud, then reads the answer sentence by
|
||||||
// rather than waiting for the last.
|
// sentence as it is composed.
|
||||||
// ============================================
|
// ============================================
|
||||||
import { WebSocketServer, WebSocket } from 'ws';
|
import { WebSocketServer, WebSocket } from 'ws';
|
||||||
import { randomUUID } from 'node:crypto';
|
|
||||||
import { principalFromToken } from './auth.js';
|
import { principalFromToken } from './auth.js';
|
||||||
import { runTurn } from '../orchestration/runner.js';
|
import { runTurn } from '../orchestration/runner.js';
|
||||||
|
import {
|
||||||
|
Endpointer, transcribe, synthesize, sentences, speakable, LANGUAGES,
|
||||||
|
} from '../speech/index.js';
|
||||||
import config from '../config/index.js';
|
import config from '../config/index.js';
|
||||||
import logger from '../utils/logger.js';
|
import logger from '../utils/logger.js';
|
||||||
|
|
||||||
const VOICE_URL = process.env.VOICE_SERVICE_URL || 'ws://127.0.0.1:4100/ws/voice';
|
/** Spoken filler, said the instant a question lands. */
|
||||||
|
|
||||||
/** Spoken filler, per language. Said the instant a question lands. */
|
|
||||||
const ACK = {
|
const ACK = {
|
||||||
ta: ['பார்க்கிறேன்...', 'ஒரு நிமிடம், பார்க்கிறேன்.'],
|
ta: ['பார்க்கிறேன்.', 'ஒரு நிமிடம், பார்க்கிறேன்.'],
|
||||||
hi: ['देखता हूँ...', 'एक मिनट, देख रहा हूँ।'],
|
|
||||||
te: ['చూస్తున్నాను...'],
|
|
||||||
kn: ['ನೋಡುತ್ತಿದ್ದೇನೆ...'],
|
|
||||||
ml: ['നോക്കുന്നു...'],
|
|
||||||
mr: ['बघतो...'],
|
|
||||||
bn: ['দেখছি...'],
|
|
||||||
en: ['Let me check.', 'One moment, checking now.'],
|
en: ['Let me check.', 'One moment, checking now.'],
|
||||||
};
|
};
|
||||||
|
|
||||||
/** Progress narration, kept short — it is spoken over the user's waiting time. */
|
/** Progress narration — spoken over the user's waiting time, so keep it short. */
|
||||||
const NARRATE = {
|
const NARRATE = {
|
||||||
ta: { lead: 'லீட் விவரங்களைப் பார்க்கிறேன்.', analytics: 'புள்ளிவிவரங்களைச் சரிபார்க்கிறேன்.', conversation: 'உரையாடல்களைப் பார்க்கிறேன்.', default: 'தரவைச் சரிபார்க்கிறேன்.' },
|
ta: { lead: 'லீட் விவரங்களைப் பார்க்கிறேன்.', analytics: 'புள்ளிவிவரங்களைச் சரிபார்க்கிறேன்.', conversation: 'உரையாடல்களைப் பார்க்கிறேன்.', default: 'தரவைச் சரிபார்க்கிறேன்.' },
|
||||||
hi: { lead: 'लीड्स देख रहा हूँ।', analytics: 'आँकड़े देख रहा हूँ।', conversation: 'बातचीत देख रहा हूँ।', default: 'डेटा देख रहा हूँ।' },
|
|
||||||
en: { lead: 'Checking the leads.', analytics: 'Pulling the numbers.', conversation: 'Looking at the conversations.', default: 'Checking the data.' },
|
en: { lead: 'Checking the leads.', analytics: 'Pulling the numbers.', conversation: 'Looking at the conversations.', default: 'Checking the data.' },
|
||||||
};
|
};
|
||||||
|
|
||||||
const pick = (arr) => arr[Math.floor(Math.random() * arr.length)];
|
const NOTHING = { ta: 'பதில் கிடைக்கவில்லை.', en: 'I could not find an answer for that.' };
|
||||||
|
const OOPS = { ta: 'மன்னிக்கவும், ஒரு பிழை ஏற்பட்டது.', en: 'Sorry, something went wrong.' };
|
||||||
|
|
||||||
function ackFor(lang) {
|
const pick = (a) => a[Math.floor(Math.random() * a.length)];
|
||||||
return pick(ACK[lang] || ACK.en);
|
const ackFor = (l) => pick(ACK[l] || ACK.en);
|
||||||
}
|
const narrateFor = (l, agent) => (NARRATE[l] || NARRATE.en)[agent] || (NARRATE[l] || NARRATE.en).default;
|
||||||
|
|
||||||
function narrationFor(lang, agent) {
|
class VoiceSession {
|
||||||
const set = NARRATE[lang] || NARRATE.en;
|
|
||||||
return set[agent] || set.default;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Strip block-oriented markdown before speaking. Tables and code read terribly
|
|
||||||
* aloud, and the visual blocks are already on screen.
|
|
||||||
*/
|
|
||||||
export function speakable(markdown = '') {
|
|
||||||
return markdown
|
|
||||||
.replace(/```[\s\S]*?```/g, ' ')
|
|
||||||
.replace(/^\s*\|.*\|\s*$/gm, ' ') // table rows
|
|
||||||
.replace(/^\s*[-*]\s+/gm, '') // bullets
|
|
||||||
.replace(/^#{1,6}\s*/gm, '') // headings
|
|
||||||
.replace(/\*\*([^*]+)\*\*/g, '$1')
|
|
||||||
.replace(/`([^`]+)`/g, '$1')
|
|
||||||
.replace(/\[([^\]]+)\]\([^)]+\)/g, '$1')
|
|
||||||
.replace(/₹\s?([\d,.]+)/g, 'rupees $1')
|
|
||||||
.replace(/\s{2,}/g, ' ')
|
|
||||||
.trim();
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Split into sentences so speech can start before the answer is finished. */
|
|
||||||
export function sentences(text, max = 240) {
|
|
||||||
const out = [];
|
|
||||||
for (const raw of text.split(/(?<=[.!?।])\s+/)) {
|
|
||||||
let s = raw.trim();
|
|
||||||
if (!s) continue;
|
|
||||||
while (s.length > max) {
|
|
||||||
const cut = s.lastIndexOf(' ', max);
|
|
||||||
out.push(s.slice(0, cut > 0 ? cut : max).trim());
|
|
||||||
s = s.slice(cut > 0 ? cut : max).trim();
|
|
||||||
}
|
|
||||||
if (s) out.push(s);
|
|
||||||
}
|
|
||||||
return out;
|
|
||||||
}
|
|
||||||
|
|
||||||
class VoiceBridge {
|
|
||||||
constructor(client, user) {
|
constructor(client, user) {
|
||||||
this.client = client;
|
this.client = client;
|
||||||
this.user = user;
|
this.user = user;
|
||||||
this.lang = 'auto'; // what the user chose
|
this.lang = 'auto'; // what the user selected
|
||||||
this.replyLang = 'ta'; // what the last utterance actually was
|
this.replyLang = 'ta'; // what the last utterance actually was
|
||||||
|
this.prefer = 'ta'; // tiebreak when detection is unusable
|
||||||
this.sessionId = `voice:${Date.now().toString(36)}:${Math.random().toString(36).slice(2, 8)}`;
|
this.sessionId = `voice:${Date.now().toString(36)}:${Math.random().toString(36).slice(2, 8)}`;
|
||||||
this.gpu = null;
|
this.endpointer = new Endpointer();
|
||||||
this.busy = false;
|
this.busy = false;
|
||||||
this.abort = null;
|
this.abort = null;
|
||||||
|
this.speakSeq = 0; // rising token; stale synthesis is discarded
|
||||||
this.narrated = new Set();
|
this.narrated = new Set();
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -106,109 +61,104 @@ class VoiceBridge {
|
|||||||
if (this.client.readyState === WebSocket.OPEN) this.client.send(JSON.stringify(obj));
|
if (this.client.readyState === WebSocket.OPEN) this.client.send(JSON.stringify(obj));
|
||||||
}
|
}
|
||||||
|
|
||||||
toGpu(obj) {
|
sendAudio(buf) {
|
||||||
if (this.gpu?.readyState === WebSocket.OPEN) this.gpu.send(JSON.stringify(obj));
|
if (this.client.readyState === WebSocket.OPEN) this.client.send(buf, { binary: true });
|
||||||
}
|
}
|
||||||
|
|
||||||
async connect() {
|
// ── microphone ───────────────────────────────────────────────────────────
|
||||||
this.gpu = new WebSocket(VOICE_URL);
|
async onAudio(data) {
|
||||||
|
// Browser sends 16 kHz mono PCM16; the models want float32 in [-1, 1].
|
||||||
|
const pcm16 = new Int16Array(data.buffer, data.byteOffset, Math.floor(data.byteLength / 2));
|
||||||
|
const pcm = new Float32Array(pcm16.length);
|
||||||
|
for (let i = 0; i < pcm16.length; i++) pcm[i] = pcm16[i] / 32768;
|
||||||
|
|
||||||
this.gpu.on('open', () => {
|
let result;
|
||||||
logger.info(`🎙️ voice session ${this.sessionId} → GPU service`);
|
try {
|
||||||
this.toGpu({ type: 'config', lang: this.lang });
|
result = await this.endpointer.push(pcm);
|
||||||
});
|
} catch (e) {
|
||||||
|
logger.error(`VAD failed: ${e.message}`);
|
||||||
this.gpu.on('message', (data, isBinary) => {
|
this.send({ type: 'error', message: 'Voice input failed to initialise. Check the server logs.' });
|
||||||
// TTS audio: pass straight through, no re-encoding.
|
|
||||||
if (isBinary) {
|
|
||||||
if (this.client.readyState === WebSocket.OPEN) this.client.send(data, { binary: true });
|
|
||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
let msg;
|
|
||||||
try { msg = JSON.parse(data.toString()); } catch { return; }
|
|
||||||
this.onGpuMessage(msg);
|
|
||||||
});
|
|
||||||
|
|
||||||
this.gpu.on('error', (err) => {
|
if (result.started) {
|
||||||
logger.error(`voice GPU service: ${err.message}`);
|
// Barge-in: the user talking wins immediately. Bumping the token drops
|
||||||
this.send({ type: 'error', message: 'The voice service is not reachable. Start it with: npm run voice' });
|
// any in-flight synthesis rather than letting it arrive late.
|
||||||
});
|
this.speakSeq++;
|
||||||
|
|
||||||
this.gpu.on('close', () => {
|
|
||||||
this.send({ type: 'voice_service_closed' });
|
|
||||||
this.client.close();
|
|
||||||
});
|
|
||||||
}
|
|
||||||
|
|
||||||
onGpuMessage(msg) {
|
|
||||||
switch (msg.type) {
|
|
||||||
case 'ready':
|
|
||||||
this.send({ type: 'ready', session_id: this.sessionId, languages: msg.languages, sample_rate_out: msg.sample_rate_out });
|
|
||||||
break;
|
|
||||||
|
|
||||||
case 'speech_start':
|
|
||||||
// The user started talking — the GPU service already stopped speaking.
|
|
||||||
// Tell the browser to dump whatever is still in its playback buffer,
|
|
||||||
// and abandon any answer still being composed.
|
|
||||||
this.send({ type: 'barge_in' });
|
|
||||||
this.abort?.abort();
|
this.abort?.abort();
|
||||||
break;
|
this.send({ type: 'barge_in' });
|
||||||
|
}
|
||||||
|
|
||||||
case 'transcript':
|
for (const utterance of result.utterances) {
|
||||||
// Answer in the language the person actually spoke, not the menu
|
await this.handleUtterance(utterance);
|
||||||
// setting — that is the whole point of auto mode.
|
}
|
||||||
if (msg.lang) this.replyLang = msg.lang;
|
}
|
||||||
this.send({ type: 'transcript', text: msg.text, lang: msg.lang, detected: msg.detected, confidence: msg.confidence, ms: msg.ms });
|
|
||||||
this.handleQuestion(msg.text);
|
|
||||||
break;
|
|
||||||
|
|
||||||
case 'transcript_empty':
|
async handleUtterance(audio) {
|
||||||
|
let heard;
|
||||||
|
try {
|
||||||
|
heard = await transcribe(audio, this.lang, this.prefer);
|
||||||
|
} catch (e) {
|
||||||
|
logger.error(`STT failed: ${e.message}`);
|
||||||
|
this.send({ type: 'error', message: 'Could not transcribe that. Try again.' });
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!heard.text) {
|
||||||
this.send({ type: 'heard_nothing' });
|
this.send({ type: 'heard_nothing' });
|
||||||
break;
|
return;
|
||||||
|
|
||||||
case 'audio_start':
|
|
||||||
case 'audio_end':
|
|
||||||
case 'error':
|
|
||||||
this.send(msg);
|
|
||||||
break;
|
|
||||||
|
|
||||||
default:
|
|
||||||
break;
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
|
|
||||||
speak(text, id = randomUUID()) {
|
this.replyLang = heard.lang;
|
||||||
|
this.send({ type: 'transcript', text: heard.text, lang: heard.lang, detected: heard.detected, ms: heard.ms });
|
||||||
|
await this.answer(heard.text);
|
||||||
|
}
|
||||||
|
|
||||||
|
// ── speaking ─────────────────────────────────────────────────────────────
|
||||||
|
/** Synthesise and stream one piece, unless a newer turn has superseded it. */
|
||||||
|
async say(text, seq) {
|
||||||
const clean = speakable(text);
|
const clean = speakable(text);
|
||||||
if (clean) this.toGpu({ type: 'speak', text: clean, id, lang: this.replyLang });
|
if (!clean || seq !== this.speakSeq) return;
|
||||||
|
try {
|
||||||
|
const out = await synthesize(clean, this.replyLang);
|
||||||
|
if (!out || seq !== this.speakSeq) return; // interrupted while generating
|
||||||
|
|
||||||
|
this.send({ type: 'audio_start', sample_rate: out.sampling_rate });
|
||||||
|
// Float32 straight down the socket — the playback worklet takes it as-is.
|
||||||
|
this.sendAudio(Buffer.from(out.audio.buffer, out.audio.byteOffset, out.audio.byteLength));
|
||||||
|
this.send({ type: 'audio_end' });
|
||||||
|
} catch (e) {
|
||||||
|
logger.error(`TTS failed: ${e.message}`);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
async handleQuestion(text) {
|
async answer(question) {
|
||||||
if (!text?.trim()) return;
|
|
||||||
if (this.busy) return; // one turn at a time
|
if (this.busy) return; // one turn at a time
|
||||||
this.busy = true;
|
this.busy = true;
|
||||||
this.narrated.clear();
|
this.narrated.clear();
|
||||||
this.abort = new AbortController();
|
this.abort = new AbortController();
|
||||||
|
const seq = ++this.speakSeq;
|
||||||
|
|
||||||
// 1. Answer the silence immediately. This is the whole trick: the pipeline
|
// Answer the silence immediately. The pipeline still takes 17–46 s, but
|
||||||
// still takes 17–46 s, but the user hears a response in ~1 s.
|
// the user hears a response in about a second.
|
||||||
this.speak(ackFor(this.replyLang));
|
this.say(ackFor(this.replyLang), seq);
|
||||||
this.send({ type: 'thinking' });
|
this.send({ type: 'thinking' });
|
||||||
|
|
||||||
try {
|
try {
|
||||||
const result = await runTurn({
|
const result = await runTurn({
|
||||||
sessionId: this.sessionId,
|
sessionId: this.sessionId,
|
||||||
message: text,
|
message: question,
|
||||||
user: this.user,
|
user: this.user,
|
||||||
channel: 'crm_chat', // voice users are staff; full tool access
|
channel: 'crm_chat', // voice users are staff
|
||||||
signal: this.abort.signal,
|
signal: this.abort.signal,
|
||||||
onEvent: (ev) => {
|
onEvent: (ev) => {
|
||||||
this.send(ev);
|
this.send(ev);
|
||||||
// 2. Narrate delegations — but only once per agent, or it chatters.
|
// Narrate delegations, once per agent, or it chatters.
|
||||||
if (ev.type === 'step' && ev.kind === 'delegate') {
|
if (ev.type === 'step' && ev.kind === 'delegate') {
|
||||||
const agent = String(ev.label || '').toLowerCase().split(' ')[0];
|
const agent = String(ev.label || '').toLowerCase().split(' ')[0];
|
||||||
if (!this.narrated.has(agent)) {
|
if (!this.narrated.has(agent)) {
|
||||||
this.narrated.add(agent);
|
this.narrated.add(agent);
|
||||||
this.speak(narrationFor(this.replyLang, agent));
|
this.say(narrateFor(this.replyLang, agent), seq);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
@@ -216,23 +166,24 @@ class VoiceBridge {
|
|||||||
|
|
||||||
this.send({ type: 'result', blocks: result.blocks, usage: result.usage });
|
this.send({ type: 'result', blocks: result.blocks, usage: result.usage });
|
||||||
|
|
||||||
// 3. Read the answer. Sentence at a time so speech starts sooner and can
|
const answer = (result.blocks || [])
|
||||||
// be cut cleanly if the user interrupts.
|
.filter((b) => b.type === 'text').map((b) => b.markdown).join(' ') || result.answer || '';
|
||||||
const answer = result.blocks?.filter((b) => b.type === 'text').map((b) => b.markdown).join(' ')
|
|
||||||
|| result.answer || '';
|
|
||||||
const parts = sentences(speakable(answer));
|
const parts = sentences(speakable(answer));
|
||||||
|
|
||||||
if (!parts.length) {
|
if (!parts.length) {
|
||||||
this.speak(this.replyLang === 'ta' ? 'பதில் கிடைக்கவில்லை.' : 'I could not find an answer for that.');
|
await this.say(NOTHING[this.replyLang] || NOTHING.en, seq);
|
||||||
} else {
|
} else {
|
||||||
|
// Sequential on purpose: parallel synthesis would race to the socket
|
||||||
|
// and play the answer out of order.
|
||||||
for (const part of parts) {
|
for (const part of parts) {
|
||||||
if (this.abort.signal.aborted) break;
|
if (seq !== this.speakSeq || this.abort.signal.aborted) break;
|
||||||
this.speak(part);
|
await this.say(part, seq);
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
} catch (err) {
|
} catch (err) {
|
||||||
if (err?.name !== 'AbortError') {
|
if (err?.name !== 'AbortError') {
|
||||||
logger.error(`voice turn failed: ${err.message}`);
|
logger.error(`voice turn failed: ${err.message}`);
|
||||||
this.speak(this.replyLang === 'ta' ? 'மன்னிக்கவும், ஒரு பிழை ஏற்பட்டது.' : 'Sorry, something went wrong.');
|
await this.say(OOPS[this.replyLang] || OOPS.en, seq);
|
||||||
}
|
}
|
||||||
} finally {
|
} finally {
|
||||||
this.busy = false;
|
this.busy = false;
|
||||||
@@ -240,32 +191,33 @@ class VoiceBridge {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
onClientMessage(data, isBinary) {
|
// ── control ──────────────────────────────────────────────────────────────
|
||||||
if (isBinary) {
|
onMessage(data, isBinary) {
|
||||||
if (this.gpu?.readyState === WebSocket.OPEN) this.gpu.send(data, { binary: true });
|
if (isBinary) return this.onAudio(data);
|
||||||
return;
|
|
||||||
}
|
|
||||||
let msg;
|
let msg;
|
||||||
try { msg = JSON.parse(data.toString()); } catch { return; }
|
try { msg = JSON.parse(data.toString()); } catch { return undefined; }
|
||||||
|
|
||||||
if (msg.type === 'config' && msg.lang) {
|
if (msg.type === 'config' && msg.lang) {
|
||||||
this.lang = msg.lang;
|
this.lang = msg.lang;
|
||||||
if (msg.lang !== 'auto') this.replyLang = msg.lang;
|
if (msg.lang !== 'auto') this.replyLang = this.prefer = msg.lang;
|
||||||
this.toGpu({ type: 'config', lang: msg.lang });
|
else if (msg.prefer) this.prefer = msg.prefer;
|
||||||
this.send({ type: 'config_ok', lang: msg.lang });
|
this.endpointer.reset();
|
||||||
|
this.send({ type: 'config_ok', lang: this.lang, prefer: this.prefer });
|
||||||
} else if (msg.type === 'cancel') {
|
} else if (msg.type === 'cancel') {
|
||||||
|
this.speakSeq++;
|
||||||
this.abort?.abort();
|
this.abort?.abort();
|
||||||
this.toGpu({ type: 'cancel' });
|
this.send({ type: 'cancelled' });
|
||||||
} else if (msg.type === 'text') {
|
} else if (msg.type === 'text' && msg.text) {
|
||||||
// Typed question while in voice mode — answered aloud like a spoken one.
|
|
||||||
this.send({ type: 'transcript', text: msg.text, lang: this.replyLang, typed: true });
|
this.send({ type: 'transcript', text: msg.text, lang: this.replyLang, typed: true });
|
||||||
this.handleQuestion(msg.text);
|
this.answer(msg.text);
|
||||||
}
|
}
|
||||||
|
return undefined;
|
||||||
}
|
}
|
||||||
|
|
||||||
close() {
|
close() {
|
||||||
|
this.speakSeq++;
|
||||||
this.abort?.abort();
|
this.abort?.abort();
|
||||||
try { this.gpu?.close(); } catch { /* already gone */ }
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -278,9 +230,8 @@ export function attachVoice(server) {
|
|||||||
if (url.pathname !== '/api/agent/voice') return; // leave other upgrades alone
|
if (url.pathname !== '/api/agent/voice') return; // leave other upgrades alone
|
||||||
|
|
||||||
// Browsers cannot set headers on a WebSocket, so the CRM token arrives as
|
// Browsers cannot set headers on a WebSocket, so the CRM token arrives as
|
||||||
// a query parameter. It is the same token and the same verification.
|
// a query parameter. Same token, same verification as every other route.
|
||||||
const token = url.searchParams.get('token');
|
const user = await principalFromToken(url.searchParams.get('token')).catch(() => null);
|
||||||
const user = await principalFromToken(token).catch(() => null);
|
|
||||||
if (!user) {
|
if (!user) {
|
||||||
socket.write('HTTP/1.1 401 Unauthorized\r\n\r\n');
|
socket.write('HTTP/1.1 401 Unauthorized\r\n\r\n');
|
||||||
socket.destroy();
|
socket.destroy();
|
||||||
@@ -288,14 +239,16 @@ export function attachVoice(server) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
wss.handleUpgrade(req, socket, head, (client) => {
|
wss.handleUpgrade(req, socket, head, (client) => {
|
||||||
const bridge = new VoiceBridge(client, user);
|
const session = new VoiceSession(client, user);
|
||||||
bridge.connect();
|
logger.info(`🎙️ voice session ${session.sessionId} (${user.name})`);
|
||||||
client.on('message', (d, bin) => bridge.onClientMessage(d, bin));
|
session.send({ type: 'ready', session_id: session.sessionId, languages: LANGUAGES });
|
||||||
client.on('close', () => bridge.close());
|
|
||||||
client.on('error', () => bridge.close());
|
client.on('message', (d, bin) => session.onMessage(d, bin));
|
||||||
|
client.on('close', () => session.close());
|
||||||
|
client.on('error', () => session.close());
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
logger.info(` voice =ws://localhost:${config.port}/api/agent/voice → ${VOICE_URL}`);
|
logger.info(` voice =ws://localhost:${config.port}/api/agent/voice (in-process, CPU)`);
|
||||||
return wss;
|
return wss;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -17,6 +17,7 @@ import { ensureDir, sweep } from './output/artifactStore.js';
|
|||||||
import crmApi from './tools/http/crmApi.js';
|
import crmApi from './tools/http/crmApi.js';
|
||||||
import { describeChains } from './orchestration/llm.js';
|
import { describeChains } from './orchestration/llm.js';
|
||||||
import { attachVoice } from './gateway/voice.js';
|
import { attachVoice } from './gateway/voice.js';
|
||||||
|
import { warmup as warmSpeech, speechStatus } from './speech/index.js';
|
||||||
|
|
||||||
const app = express();
|
const app = express();
|
||||||
|
|
||||||
@@ -40,6 +41,7 @@ app.get('/health', async (_req, res) => {
|
|||||||
redis: redisOk ? redisMode() : 'unavailable',
|
redis: redisOk ? redisMode() : 'unavailable',
|
||||||
crm_api: crm.reachable ? 'reachable' : `unreachable (${crm.error || crm.status})`,
|
crm_api: crm.reachable ? 'reachable' : `unreachable (${crm.error || crm.status})`,
|
||||||
models: describeChains(),
|
models: describeChains(),
|
||||||
|
speech: speechStatus(),
|
||||||
uptime_s: Math.round(process.uptime()),
|
uptime_s: Math.round(process.uptime()),
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
@@ -83,6 +85,11 @@ async function start() {
|
|||||||
// second origin and the CRM token works unchanged.
|
// second origin and the CRM token works unchanged.
|
||||||
attachVoice(server);
|
attachVoice(server);
|
||||||
|
|
||||||
|
// Speech models load lazily on the first voice turn (~10 s). Set
|
||||||
|
// SPEECH_WARMUP=true to pay that at boot instead — worth it in production,
|
||||||
|
// wasteful in development where most restarts never use voice.
|
||||||
|
if (process.env.SPEECH_WARMUP === 'true') warmSpeech(['ta', 'en']);
|
||||||
|
|
||||||
const shutdown = (sig) => {
|
const shutdown = (sig) => {
|
||||||
logger.info(`${sig} — shutting down`);
|
logger.info(`${sig} — shutting down`);
|
||||||
server.close(() => process.exit(0));
|
server.close(() => process.exit(0));
|
||||||
|
|||||||
@@ -0,0 +1,169 @@
|
|||||||
|
// ============================================
|
||||||
|
// Speech pipeline — transcribe() and synthesize().
|
||||||
|
//
|
||||||
|
// Whisper is multilingual and can identify the spoken language, so "auto"
|
||||||
|
// costs nothing extra: detection and transcription are the same forward pass.
|
||||||
|
// That matters for a WeLe agent who switches between Tamil and English inside
|
||||||
|
// one shift and should never have to touch a language menu.
|
||||||
|
// ============================================
|
||||||
|
import { Tensor } from '@huggingface/transformers';
|
||||||
|
import { getSTT, getTTS, supportsTTS } from './models.js';
|
||||||
|
import logger from '../utils/logger.js';
|
||||||
|
|
||||||
|
export { LANGUAGES, warmup, speechStatus } from './models.js';
|
||||||
|
export { Endpointer, warmupVad } from './vad.js';
|
||||||
|
|
||||||
|
const RATE = 16000;
|
||||||
|
|
||||||
|
/** Languages we can both hear and speak. */
|
||||||
|
const SPOKEN = new Set(['ta', 'en']);
|
||||||
|
|
||||||
|
// Below this, trust the caller's preference over the detector. Short or noisy
|
||||||
|
// utterances — and code-mixed "Tanglish" especially — can land either side.
|
||||||
|
const DETECT_CONFIDENCE = Number(process.env.DETECT_CONFIDENCE ?? 0.6);
|
||||||
|
|
||||||
|
let detectIds = null;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Identify the spoken language in ONE decoder step.
|
||||||
|
*
|
||||||
|
* Passing no `language` to the pipeline does NOT auto-detect — Transformers.js
|
||||||
|
* logs "No language specified - defaulting to English" and transcribes Tamil
|
||||||
|
* as English, producing nonsense. Whisper does emit a language token right
|
||||||
|
* after <|startoftranscript|>, so we read that distribution directly. Measured
|
||||||
|
* ~700 ms, and 0.998 / 1.000 confidence on clean Tamil / English.
|
||||||
|
*
|
||||||
|
* Reuses the pipeline's own model and processor, so nothing loads twice.
|
||||||
|
*/
|
||||||
|
async function detectLanguage(audio) {
|
||||||
|
const stt = await getSTT();
|
||||||
|
const tok = stt.tokenizer;
|
||||||
|
|
||||||
|
if (!detectIds) {
|
||||||
|
const id = (t) => tok.encode(t, { add_special_tokens: false })[0];
|
||||||
|
detectIds = { sot: id('<|startoftranscript|>'), langs: [...SPOKEN].map((c) => ({ code: c, id: id(`<|${c}|>`) })) };
|
||||||
|
}
|
||||||
|
|
||||||
|
const inputs = await stt.processor(audio);
|
||||||
|
const out = await stt.model({
|
||||||
|
...inputs,
|
||||||
|
decoder_input_ids: new Tensor('int64', BigInt64Array.from([BigInt(detectIds.sot)]), [1, 1]),
|
||||||
|
});
|
||||||
|
|
||||||
|
const { dims, data } = out.logits;
|
||||||
|
const row = data.slice((dims[1] - 1) * dims[2], dims[1] * dims[2]);
|
||||||
|
const scores = detectIds.langs.map((l) => Number(row[l.id]));
|
||||||
|
const max = Math.max(...scores);
|
||||||
|
const exp = scores.map((v) => Math.exp(v - max));
|
||||||
|
const sum = exp.reduce((a, b) => a + b, 0);
|
||||||
|
const probs = exp.map((v) => v / sum);
|
||||||
|
const best = probs.indexOf(Math.max(...probs));
|
||||||
|
|
||||||
|
return { lang: detectIds.langs[best].code, confidence: probs[best] };
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* @param {Float32Array} audio mono @16 kHz in [-1, 1]
|
||||||
|
* @param {string} lang 'auto' | 'ta' | 'en'
|
||||||
|
* @param {string} prefer used when detection is unusable
|
||||||
|
*/
|
||||||
|
export async function transcribe(audio, lang = 'auto', prefer = 'ta') {
|
||||||
|
if (!audio || audio.length < RATE / 5) { // under 200 ms
|
||||||
|
return { text: '', lang: prefer, note: 'too short' };
|
||||||
|
}
|
||||||
|
|
||||||
|
const stt = await getSTT();
|
||||||
|
const t0 = Date.now();
|
||||||
|
|
||||||
|
// Whisper must always be told a language — it never detects on its own here.
|
||||||
|
let used = lang;
|
||||||
|
let detected = null;
|
||||||
|
let confidence = null;
|
||||||
|
|
||||||
|
if (lang === 'auto') {
|
||||||
|
try {
|
||||||
|
const d = await detectLanguage(audio);
|
||||||
|
detected = d.lang;
|
||||||
|
confidence = d.confidence;
|
||||||
|
used = d.confidence >= DETECT_CONFIDENCE ? d.lang : prefer;
|
||||||
|
if (used !== d.lang) {
|
||||||
|
logger.info(`language ID unsure (${d.lang} @ ${d.confidence.toFixed(2)}) — using preferred ${prefer}`);
|
||||||
|
}
|
||||||
|
} catch (e) {
|
||||||
|
logger.warn(`language ID failed (${e.message}) — using preferred ${prefer}`);
|
||||||
|
used = prefer;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
const result = await stt(audio, { task: 'transcribe', language: used, return_timestamps: false });
|
||||||
|
return finish(result, used, audio, t0, detected, confidence);
|
||||||
|
}
|
||||||
|
|
||||||
|
function finish(result, used, audio, t0, detected, confidence) {
|
||||||
|
const text = (result?.text || '').trim();
|
||||||
|
const ms = Date.now() - t0;
|
||||||
|
const audioMs = Math.round((audio.length / RATE) * 1000);
|
||||||
|
logger.info(`🎤 STT ${used}${detected && detected !== used ? ` (heard ${detected})` : ''}: ${audioMs}ms → ${ms}ms → ${JSON.stringify(text.slice(0, 70))}`);
|
||||||
|
return {
|
||||||
|
text, lang: used, detected: detected || null,
|
||||||
|
confidence: confidence == null ? null : Number(confidence.toFixed(3)),
|
||||||
|
ms, audio_ms: audioMs,
|
||||||
|
};
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Synthesise one piece of text.
|
||||||
|
* @returns {Promise<{audio: Float32Array, sampling_rate: number}>}
|
||||||
|
*/
|
||||||
|
export async function synthesize(text, lang = 'ta') {
|
||||||
|
const clean = (text || '').trim();
|
||||||
|
if (!clean) return null;
|
||||||
|
|
||||||
|
const use = supportsTTS(lang) ? lang : 'en';
|
||||||
|
const tts = await getTTS(use);
|
||||||
|
|
||||||
|
const t0 = Date.now();
|
||||||
|
const out = await tts(clean);
|
||||||
|
const ms = Date.now() - t0;
|
||||||
|
const audioMs = (out.audio.length / out.sampling_rate) * 1000;
|
||||||
|
logger.debug(`🔈 TTS ${use}: ${clean.length} chars → ${ms}ms for ${audioMs.toFixed(0)}ms (RTF ${(ms / audioMs).toFixed(2)}x)`);
|
||||||
|
|
||||||
|
return { audio: out.audio, sampling_rate: out.sampling_rate };
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Split into speakable pieces. Short prompts reach audio sooner, and a sentence
|
||||||
|
* boundary is a clean place to be interrupted.
|
||||||
|
*/
|
||||||
|
export function sentences(text, max = 200) {
|
||||||
|
const out = [];
|
||||||
|
for (const raw of String(text || '').split(/(?<=[.!?।])\s+/)) {
|
||||||
|
let s = raw.trim();
|
||||||
|
if (!s) continue;
|
||||||
|
while (s.length > max) {
|
||||||
|
const cut = s.lastIndexOf(' ', max);
|
||||||
|
out.push(s.slice(0, cut > 0 ? cut : max).trim());
|
||||||
|
s = s.slice(cut > 0 ? cut : max).trim();
|
||||||
|
}
|
||||||
|
if (s) out.push(s);
|
||||||
|
}
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Strip block markdown before speaking — tables and code read terribly aloud,
|
||||||
|
* and the visual blocks are already on screen.
|
||||||
|
*/
|
||||||
|
export function speakable(markdown = '') {
|
||||||
|
return markdown
|
||||||
|
.replace(/```[\s\S]*?```/g, ' ')
|
||||||
|
.replace(/^\s*\|.*\|\s*$/gm, ' ')
|
||||||
|
.replace(/^\s*[-*]\s+/gm, '')
|
||||||
|
.replace(/^#{1,6}\s*/gm, '')
|
||||||
|
.replace(/\*\*([^*]+)\*\*/g, '$1')
|
||||||
|
.replace(/`([^`]+)`/g, '$1')
|
||||||
|
.replace(/\[([^\]]+)\]\([^)]+\)/g, '$1')
|
||||||
|
.replace(/₹\s?([\d,.]+)/g, 'rupees $1')
|
||||||
|
.replace(/\s{2,}/g, ' ')
|
||||||
|
.trim();
|
||||||
|
}
|
||||||
@@ -0,0 +1,118 @@
|
|||||||
|
// ============================================
|
||||||
|
// Speech models — all ONNX, all CPU, all in this Node process.
|
||||||
|
//
|
||||||
|
// There is no GPU and no Python. That is the whole point: the AWS host has
|
||||||
|
// neither, and a second service was one more thing to deploy and keep alive.
|
||||||
|
//
|
||||||
|
// VAD Silero 2 MB endpointing
|
||||||
|
// STT Whisper base ~80 MB Tamil + English + language detection
|
||||||
|
// TTS MMS-TTS VITS ~114 MB per language, feed-forward
|
||||||
|
//
|
||||||
|
// Measured on an i7-10850H, CPU only:
|
||||||
|
// TTS RTF 0.28x (3.5x faster than realtime)
|
||||||
|
// STT ~1.2 s for 4 s of audio
|
||||||
|
//
|
||||||
|
// Two findings worth keeping:
|
||||||
|
//
|
||||||
|
// * VITS is feed-forward. The earlier Parler-TTS attempt was autoregressive
|
||||||
|
// and ran at RTF ~5x — i.e. 5x SLOWER than realtime — which is why voice was
|
||||||
|
// unusable even on a GPU. Architecture mattered far more than hardware here.
|
||||||
|
//
|
||||||
|
// * int8 is a trap for a model this small: dynamic quantisation made TTS 5.7x
|
||||||
|
// SLOWER than fp32 (RTF 1.67x vs 0.28x) because the quantise/dequantise
|
||||||
|
// overhead dominates. We ship fp32 deliberately.
|
||||||
|
// ============================================
|
||||||
|
import path from 'node:path';
|
||||||
|
import { fileURLToPath } from 'node:url';
|
||||||
|
import { pipeline, env } from '@huggingface/transformers';
|
||||||
|
import logger from '../utils/logger.js';
|
||||||
|
|
||||||
|
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..');
|
||||||
|
|
||||||
|
// Hub downloads are cached here so a container restart does not re-fetch.
|
||||||
|
env.cacheDir = process.env.SPEECH_CACHE_DIR || path.join(ROOT, '.transformers-cache');
|
||||||
|
// Tamil is loaded from a folder we exported ourselves — no public ONNX build
|
||||||
|
// of mms-tts-tam exists. See scripts/export-tamil-tts.py.
|
||||||
|
env.localModelPath = path.join(ROOT, 'assets/tts');
|
||||||
|
|
||||||
|
const STT_MODEL = process.env.STT_MODEL || 'onnx-community/whisper-base';
|
||||||
|
|
||||||
|
/** TTS voice per language. Tamil is local; English comes from the Hub. */
|
||||||
|
const VOICES = {
|
||||||
|
ta: { id: 'mms-tts-tam', local: true },
|
||||||
|
en: { id: 'Xenova/mms-tts-eng', local: false },
|
||||||
|
};
|
||||||
|
|
||||||
|
export const LANGUAGES = [
|
||||||
|
{ code: 'auto', label: 'Auto-detect', native: 'Auto' },
|
||||||
|
{ code: 'ta', label: 'Tamil', native: 'தமிழ்' },
|
||||||
|
{ code: 'en', label: 'English', native: 'English' },
|
||||||
|
];
|
||||||
|
|
||||||
|
const cache = new Map();
|
||||||
|
let sttPromise = null;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Models load on first use, not at boot. A CRM restart should not wait ~10 s
|
||||||
|
* for speech models that most sessions never touch.
|
||||||
|
*/
|
||||||
|
async function loadOnce(key, build) {
|
||||||
|
if (!cache.has(key)) {
|
||||||
|
const t0 = Date.now();
|
||||||
|
cache.set(key, build().then((m) => {
|
||||||
|
logger.info(`🔊 loaded ${key} in ${((Date.now() - t0) / 1000).toFixed(1)}s`);
|
||||||
|
return m;
|
||||||
|
}).catch((e) => {
|
||||||
|
cache.delete(key); // let the next attempt retry
|
||||||
|
throw e;
|
||||||
|
}));
|
||||||
|
}
|
||||||
|
return cache.get(key);
|
||||||
|
}
|
||||||
|
|
||||||
|
export async function getSTT() {
|
||||||
|
if (!sttPromise) {
|
||||||
|
sttPromise = loadOnce(STT_MODEL, () =>
|
||||||
|
// q8 is the right call for Whisper — unlike VITS it is big enough that
|
||||||
|
// quantisation is a clear win.
|
||||||
|
pipeline('automatic-speech-recognition', STT_MODEL, { dtype: 'q8' }),
|
||||||
|
).catch((e) => { sttPromise = null; throw e; });
|
||||||
|
}
|
||||||
|
return sttPromise;
|
||||||
|
}
|
||||||
|
|
||||||
|
export async function getTTS(lang = 'ta') {
|
||||||
|
const voice = VOICES[lang] || VOICES.ta;
|
||||||
|
const prev = env.allowRemoteModels;
|
||||||
|
try {
|
||||||
|
// Local folders must not be looked up on the Hub, and vice versa.
|
||||||
|
env.allowRemoteModels = !voice.local;
|
||||||
|
return await loadOnce(`tts:${voice.id}`, () =>
|
||||||
|
pipeline('text-to-speech', voice.id, { dtype: 'fp32' }),
|
||||||
|
);
|
||||||
|
} finally {
|
||||||
|
env.allowRemoteModels = prev;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
export const supportsTTS = (lang) => Boolean(VOICES[lang]);
|
||||||
|
|
||||||
|
/** Warm the models the deployment actually expects to use. */
|
||||||
|
export async function warmup(langs = ['ta', 'en']) {
|
||||||
|
try {
|
||||||
|
await getSTT();
|
||||||
|
for (const l of langs) await getTTS(l);
|
||||||
|
logger.info('🔊 speech models warm');
|
||||||
|
} catch (e) {
|
||||||
|
logger.warn(`speech warmup failed (will retry on first use): ${e.message}`);
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
export function speechStatus() {
|
||||||
|
return {
|
||||||
|
stt_model: STT_MODEL,
|
||||||
|
tts_voices: Object.fromEntries(Object.entries(VOICES).map(([k, v]) => [k, v.id])),
|
||||||
|
loaded: [...cache.keys()],
|
||||||
|
languages: LANGUAGES,
|
||||||
|
};
|
||||||
|
}
|
||||||
@@ -0,0 +1,152 @@
|
|||||||
|
// ============================================
|
||||||
|
// Endpointing — Silero VAD via onnxruntime-node.
|
||||||
|
//
|
||||||
|
// Deciding turn boundaries on the server rather than in the browser keeps the
|
||||||
|
// rule in one place for every future channel (a phone bridge has no
|
||||||
|
// AudioWorklet), and gives the server the signal it needs for barge-in: it has
|
||||||
|
// to know the user started talking while the assistant was still speaking.
|
||||||
|
// ============================================
|
||||||
|
import path from 'node:path';
|
||||||
|
import { fileURLToPath } from 'node:url';
|
||||||
|
import fs from 'node:fs/promises';
|
||||||
|
import ort from 'onnxruntime-node';
|
||||||
|
import logger from '../utils/logger.js';
|
||||||
|
|
||||||
|
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..');
|
||||||
|
const MODEL_URL = 'https://huggingface.co/onnx-community/silero-vad/resolve/main/onnx/model.onnx';
|
||||||
|
const MODEL_PATH = path.join(process.env.SPEECH_CACHE_DIR || path.join(ROOT, '.transformers-cache'), 'silero-vad.onnx');
|
||||||
|
|
||||||
|
// Silero wants exactly 512 samples at 16 kHz (32 ms). The browser sends 40 ms
|
||||||
|
// chunks, so audio is buffered and drained in exact frames rather than forcing
|
||||||
|
// the client to match.
|
||||||
|
const FRAME = 512;
|
||||||
|
const RATE = 16000;
|
||||||
|
const FRAME_MS = (FRAME / RATE) * 1000;
|
||||||
|
|
||||||
|
let sessionPromise = null;
|
||||||
|
|
||||||
|
async function getSession() {
|
||||||
|
if (sessionPromise) return sessionPromise;
|
||||||
|
sessionPromise = (async () => {
|
||||||
|
try {
|
||||||
|
await fs.access(MODEL_PATH);
|
||||||
|
} catch {
|
||||||
|
logger.info('⬇️ fetching Silero VAD (2 MB)…');
|
||||||
|
const res = await fetch(MODEL_URL);
|
||||||
|
if (!res.ok) throw new Error(`VAD download failed: ${res.status}`);
|
||||||
|
await fs.mkdir(path.dirname(MODEL_PATH), { recursive: true });
|
||||||
|
await fs.writeFile(MODEL_PATH, Buffer.from(await res.arrayBuffer()));
|
||||||
|
}
|
||||||
|
const s = await ort.InferenceSession.create(MODEL_PATH);
|
||||||
|
logger.info('🎚️ Silero VAD ready');
|
||||||
|
return s;
|
||||||
|
})().catch((e) => { sessionPromise = null; throw e; });
|
||||||
|
return sessionPromise;
|
||||||
|
}
|
||||||
|
|
||||||
|
export const vadOptions = {
|
||||||
|
threshold: Number(process.env.VAD_THRESHOLD ?? 0.5),
|
||||||
|
// Trailing silence that ends a turn. Too short truncates someone who pauses
|
||||||
|
// mid-sentence; too long makes the assistant feel sluggish.
|
||||||
|
silenceMs: Number(process.env.VAD_SILENCE_MS ?? 700),
|
||||||
|
// Ignore blips, so a cough or a door does not open a turn.
|
||||||
|
minSpeechMs: Number(process.env.VAD_MIN_SPEECH_MS ?? 250),
|
||||||
|
// Audio kept from BEFORE detection, so word onsets are not clipped.
|
||||||
|
prefixMs: Number(process.env.VAD_PREFIX_MS ?? 300),
|
||||||
|
maxUtteranceMs: Number(process.env.VAD_MAX_UTTERANCE_MS ?? 30000),
|
||||||
|
};
|
||||||
|
|
||||||
|
/** Streaming endpointer. One instance per connection. */
|
||||||
|
export class Endpointer {
|
||||||
|
constructor(opts = {}) {
|
||||||
|
this.o = { ...vadOptions, ...opts };
|
||||||
|
this.pending = new Float32Array(0);
|
||||||
|
this.prefixFrames = Math.max(1, Math.round(this.o.prefixMs / FRAME_MS));
|
||||||
|
this.reset();
|
||||||
|
}
|
||||||
|
|
||||||
|
reset() {
|
||||||
|
this.speaking = false;
|
||||||
|
this.speechMs = 0;
|
||||||
|
this.silenceMs = 0;
|
||||||
|
this.buffer = [];
|
||||||
|
this.prefix = [];
|
||||||
|
// Silero is recurrent: this 2x1x128 state carries across frames and must
|
||||||
|
// be reset between turns or the model stays biased by the last utterance.
|
||||||
|
this.state = new ort.Tensor('float32', new Float32Array(2 * 1 * 128), [2, 1, 128]);
|
||||||
|
this.pending = new Float32Array(0);
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Feed float32 mono @16k.
|
||||||
|
* @returns {Promise<{utterances: Float32Array[], started: boolean}>}
|
||||||
|
* `started` flips the moment speech begins — that is the barge-in signal.
|
||||||
|
*/
|
||||||
|
async push(pcm) {
|
||||||
|
const session = await getSession();
|
||||||
|
|
||||||
|
const merged = new Float32Array(this.pending.length + pcm.length);
|
||||||
|
merged.set(this.pending);
|
||||||
|
merged.set(pcm, this.pending.length);
|
||||||
|
this.pending = merged;
|
||||||
|
|
||||||
|
const utterances = [];
|
||||||
|
let started = false;
|
||||||
|
let offset = 0;
|
||||||
|
|
||||||
|
while (this.pending.length - offset >= FRAME) {
|
||||||
|
const frame = this.pending.subarray(offset, offset + FRAME);
|
||||||
|
offset += FRAME;
|
||||||
|
|
||||||
|
const out = await session.run({
|
||||||
|
input: new ort.Tensor('float32', frame, [1, FRAME]),
|
||||||
|
sr: new ort.Tensor('int64', BigInt64Array.from([BigInt(RATE)]), []),
|
||||||
|
state: this.state,
|
||||||
|
});
|
||||||
|
this.state = out.stateN ?? out.state_n ?? this.state;
|
||||||
|
const voiced = out.output.data[0] >= this.o.threshold;
|
||||||
|
|
||||||
|
if (!this.speaking) {
|
||||||
|
this.prefix.push(Float32Array.from(frame));
|
||||||
|
if (this.prefix.length > this.prefixFrames) this.prefix.shift();
|
||||||
|
|
||||||
|
if (voiced) {
|
||||||
|
this.speechMs += FRAME_MS;
|
||||||
|
if (this.speechMs >= this.o.minSpeechMs) {
|
||||||
|
this.speaking = true;
|
||||||
|
this.silenceMs = 0;
|
||||||
|
this.buffer = this.prefix; // open the turn with the pre-roll
|
||||||
|
this.prefix = [];
|
||||||
|
started = true;
|
||||||
|
}
|
||||||
|
} else {
|
||||||
|
this.speechMs = 0;
|
||||||
|
}
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
|
||||||
|
this.buffer.push(Float32Array.from(frame));
|
||||||
|
if (voiced) this.silenceMs = 0;
|
||||||
|
else this.silenceMs += FRAME_MS;
|
||||||
|
|
||||||
|
const spokenMs = this.buffer.length * FRAME_MS;
|
||||||
|
if (this.silenceMs >= this.o.silenceMs || spokenMs >= this.o.maxUtteranceMs) {
|
||||||
|
utterances.push(concat(this.buffer));
|
||||||
|
this.reset();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
this.pending = this.pending.slice(offset);
|
||||||
|
return { utterances, started };
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
function concat(frames) {
|
||||||
|
const total = frames.reduce((n, f) => n + f.length, 0);
|
||||||
|
const out = new Float32Array(total);
|
||||||
|
let i = 0;
|
||||||
|
for (const f of frames) { out.set(f, i); i += f.length; }
|
||||||
|
return out;
|
||||||
|
}
|
||||||
|
|
||||||
|
export const warmupVad = () => getSession().catch(() => {});
|
||||||
@@ -1,22 +0,0 @@
|
|||||||
# Copy to .env and fill in HF_TOKEN.
|
|
||||||
# The AI4Bharat models are gated: sign in at huggingface.co, accept the terms on
|
|
||||||
# both model pages, then create a read token at huggingface.co/settings/tokens.
|
|
||||||
# HuggingFace token — required: the AI4Bharat models are gated repos.
|
|
||||||
HF_TOKEN=
|
|
||||||
|
|
||||||
VOICE_HOST=127.0.0.1
|
|
||||||
VOICE_PORT=4100
|
|
||||||
|
|
||||||
STT_MODEL=ai4bharat/indic-conformer-600m-multilingual
|
|
||||||
STT_DECODING=ctc
|
|
||||||
ENGLISH_MODEL=openai/whisper-small
|
|
||||||
TTS_MODEL=ai4bharat/indic-parler-tts
|
|
||||||
|
|
||||||
VOICE_DEFAULT_LANG=ta
|
|
||||||
PRELOAD_ENGLISH=true
|
|
||||||
VOICE_WARMUP=true
|
|
||||||
|
|
||||||
# Endpointing
|
|
||||||
VAD_SILENCE_MS=700
|
|
||||||
VAD_MIN_SPEECH_MS=250
|
|
||||||
VAD_PREFIX_MS=300
|
|
||||||
@@ -1,105 +0,0 @@
|
|||||||
# WeLe Voice Service
|
|
||||||
|
|
||||||
Speech in, speech out. This process holds the GPU models and nothing else — it
|
|
||||||
has no idea what the CRM is. Orchestration, auth and business logic stay in the
|
|
||||||
Node service, so **voice is a channel into the same agent**, not a parallel
|
|
||||||
system with its own brain.
|
|
||||||
|
|
||||||
```
|
|
||||||
browser ──audio──► node :4000 ──audio──► this :4100 ──► IndicConformer / Whisper
|
|
||||||
│ │
|
|
||||||
└────────── same graph, agents, ───────────┘
|
|
||||||
guardrails as text chat
|
|
||||||
│
|
|
||||||
browser ◄──audio───── node ◄──audio──── this ◄── Indic Parler-TTS
|
|
||||||
```
|
|
||||||
|
|
||||||
## Models
|
|
||||||
|
|
||||||
| Job | Model | Notes |
|
|
||||||
|---|---|---|
|
|
||||||
| Endpointing | Silero VAD | 512-sample frames @16 kHz, 300 ms pre-roll |
|
|
||||||
| Indic ASR | `ai4bharat/indic-conformer-600m-multilingual` | 22 Indian languages, CTC decoding |
|
|
||||||
| English ASR + language ID | `openai/whisper-small` | multilingual on purpose — the `.en` build cannot identify languages |
|
|
||||||
| TTS | `ai4bharat/indic-parler-tts` | 21 languages, streaming |
|
|
||||||
|
|
||||||
**The AI4Bharat repos are gated.** Access is auto-approved, but the download
|
|
||||||
needs an authenticated account: sign in to huggingface.co, accept the terms on
|
|
||||||
both model pages, then put a read token in `.env` as `HF_TOKEN`.
|
|
||||||
|
|
||||||
## Why two ASR models
|
|
||||||
|
|
||||||
IndicConformer decodes *as* the language you name — it does not detect one, and
|
|
||||||
English is not among its 22 codes. Whisper covers English and can identify the
|
|
||||||
spoken language in a single decoder step. So the default mode is `auto`:
|
|
||||||
|
|
||||||
```
|
|
||||||
audio → Whisper mel + 1 decoder step → language ID
|
|
||||||
├─ "en" → Whisper transcribes (mel already computed — no extra cost)
|
|
||||||
└─ Indic → IndicConformer with the detected code
|
|
||||||
```
|
|
||||||
|
|
||||||
Below **0.60** confidence the caller's preferred language wins instead of a coin
|
|
||||||
toss. That matters for Tanglish, where a short code-mixed sentence can honestly
|
|
||||||
land either side.
|
|
||||||
|
|
||||||
## Setup
|
|
||||||
|
|
||||||
```bash
|
|
||||||
python -m venv --system-site-packages .venv # reuses the system torch build
|
|
||||||
.venv/Scripts/python -m pip install -r requirements.txt
|
|
||||||
cp .env.example .env # add HF_TOKEN
|
|
||||||
```
|
|
||||||
|
|
||||||
The venv deliberately inherits system site-packages: torch is ~2.5 GB and
|
|
||||||
already installed with CUDA. Note that `parler-tts` pins `transformers==4.46.1`
|
|
||||||
**inside the venv only** — the system install is untouched.
|
|
||||||
|
|
||||||
## Run
|
|
||||||
|
|
||||||
```bash
|
|
||||||
npm run voice # from the parent directory
|
|
||||||
# or
|
|
||||||
.venv/Scripts/python -m app.server
|
|
||||||
```
|
|
||||||
|
|
||||||
First start downloads several GB and warms both models. `GET /health` reports
|
|
||||||
device, models, sample rate and current VRAM.
|
|
||||||
|
|
||||||
## Protocol
|
|
||||||
|
|
||||||
One WebSocket at `/ws/voice`, JSON control frames plus binary audio.
|
|
||||||
|
|
||||||
| Direction | Message |
|
|
||||||
|---|---|
|
|
||||||
| → | binary — 16 kHz mono PCM16 mic frames |
|
|
||||||
| → | `{"type":"config","lang":"auto","prefer":"ta"}` |
|
|
||||||
| → | `{"type":"speak","text":"…","id":"…"}` |
|
|
||||||
| → | `{"type":"cancel"}` — stop speaking now |
|
|
||||||
| ← | `{"type":"speech_start"}` — VAD opened a turn (drives barge-in) |
|
|
||||||
| ← | `{"type":"transcript","text":…,"lang":…,"detected":…,"confidence":…}` |
|
|
||||||
| ← | `{"type":"audio_start","sample_rate":24000}` then binary float32 chunks |
|
|
||||||
|
|
||||||
## Tuning
|
|
||||||
|
|
||||||
| Env | Default | Effect |
|
|
||||||
|---|---|---|
|
|
||||||
| `VAD_SILENCE_MS` | 700 | trailing silence that ends a turn — lower feels snappier, truncates people who pause |
|
|
||||||
| `VAD_MIN_SPEECH_MS` | 250 | ignores coughs and door slams |
|
|
||||||
| `VAD_PREFIX_MS` | 300 | audio kept from before detection, so word onsets survive |
|
|
||||||
| `VOICE_DEFAULT_LANG` | `ta` | tiebreak when language ID is unsure |
|
|
||||||
| `PRELOAD_ENGLISH` | `true` | set `false` to load Whisper lazily if VRAM is tight |
|
|
||||||
| `STT_DECODING` | `ctc` | `rnnt` is more accurate but decodes autoregressively |
|
|
||||||
|
|
||||||
## VRAM
|
|
||||||
|
|
||||||
Roughly 4.7 GB of the 6 GB card with all three models resident. If that proves
|
|
||||||
too tight, `PRELOAD_ENGLISH=false` defers Whisper (~0.5 GB) until the first
|
|
||||||
English utterance.
|
|
||||||
|
|
||||||
## Scripts
|
|
||||||
|
|
||||||
```bash
|
|
||||||
.venv/Scripts/python probe_access.py # which repos the token can reach
|
|
||||||
.venv/Scripts/python probe_models.py # load, VRAM, time-to-first-audio
|
|
||||||
```
|
|
||||||
@@ -1,82 +0,0 @@
|
|||||||
"""Voice service configuration.
|
|
||||||
|
|
||||||
Deliberately small: this process does one job — turn audio into text and text
|
|
||||||
into audio. Everything about *what to say* lives in the Node agentic service.
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import os
|
|
||||||
|
|
||||||
import torch
|
|
||||||
|
|
||||||
|
|
||||||
def _int(name: str, default: int) -> int:
|
|
||||||
try:
|
|
||||||
return int(os.environ.get(name, default))
|
|
||||||
except (TypeError, ValueError):
|
|
||||||
return default
|
|
||||||
|
|
||||||
|
|
||||||
def _flag(name: str, default: bool) -> bool:
|
|
||||||
return os.environ.get(name, str(default)).lower() in {"1", "true", "yes"}
|
|
||||||
|
|
||||||
|
|
||||||
class Settings:
|
|
||||||
host: str = os.environ.get("VOICE_HOST", "127.0.0.1")
|
|
||||||
port: int = _int("VOICE_PORT", 4100)
|
|
||||||
|
|
||||||
# ── Models ───────────────────────────────────────────────────────────────
|
|
||||||
# IndicConformer is a hybrid CTC + RNNT model. CTC decoding is used because
|
|
||||||
# it is a single forward pass — RNNT is more accurate but decodes
|
|
||||||
# autoregressively, and in a voice loop the latency costs more than the
|
|
||||||
# accuracy buys.
|
|
||||||
stt_model: str = os.environ.get("STT_MODEL", "ai4bharat/indic-conformer-600m-multilingual")
|
|
||||||
stt_decoding: str = os.environ.get("STT_DECODING", "ctc") # ctc | rnnt
|
|
||||||
|
|
||||||
# English is not one of IndicConformer's 22 codes, and it cannot identify
|
|
||||||
# languages. Whisper covers both — multilingual, not the .en checkpoint,
|
|
||||||
# because language ID is what makes "auto" work.
|
|
||||||
english_model: str = os.environ.get("ENGLISH_MODEL", "openai/whisper-small")
|
|
||||||
preload_english: bool = _flag("PRELOAD_ENGLISH", True)
|
|
||||||
|
|
||||||
# Used when auto-detection is not confident enough to overrule the user.
|
|
||||||
default_lang: str = os.environ.get("VOICE_DEFAULT_LANG", "ta")
|
|
||||||
|
|
||||||
tts_model: str = os.environ.get("TTS_MODEL", "ai4bharat/indic-parler-tts")
|
|
||||||
|
|
||||||
# ── Device / precision ───────────────────────────────────────────────────
|
|
||||||
# float16 on CUDA: both models together are ~3 GB in half precision, which
|
|
||||||
# fits the 6 GB card with room for activations. float32 would not.
|
|
||||||
device: str = os.environ.get("VOICE_DEVICE", "cuda" if torch.cuda.is_available() else "cpu")
|
|
||||||
|
|
||||||
@property
|
|
||||||
def dtype(self) -> torch.dtype:
|
|
||||||
return torch.float16 if self.device == "cuda" else torch.float32
|
|
||||||
|
|
||||||
# ── Audio ────────────────────────────────────────────────────────────────
|
|
||||||
sample_rate_in: int = 16000 # what the browser worklet sends
|
|
||||||
# Indic Parler-TTS emits 44.1 kHz — verified from model.config.sampling_rate,
|
|
||||||
# not the 24 kHz the upstream Parler-TTS Mini uses. This is only a fallback;
|
|
||||||
# the real rate is read from the loaded model and sent to the browser, which
|
|
||||||
# configures its playback worklet from it.
|
|
||||||
sample_rate_out: int = _int("TTS_SAMPLE_RATE", 44100)
|
|
||||||
|
|
||||||
# ── Endpointing (Silero VAD) ─────────────────────────────────────────────
|
|
||||||
# Silero operates on fixed 512-sample frames at 16 kHz (32 ms).
|
|
||||||
vad_frame: int = 512
|
|
||||||
vad_threshold: float = float(os.environ.get("VAD_THRESHOLD", "0.5"))
|
|
||||||
# How much trailing silence ends a turn. Too short truncates people who
|
|
||||||
# pause mid-sentence; too long makes the assistant feel sluggish.
|
|
||||||
vad_silence_ms: int = _int("VAD_SILENCE_MS", 700)
|
|
||||||
# Ignore blips so a cough or a door does not open a turn.
|
|
||||||
vad_min_speech_ms: int = _int("VAD_MIN_SPEECH_MS", 250)
|
|
||||||
# Audio kept from *before* detected speech, so word onsets are not clipped.
|
|
||||||
vad_prefix_ms: int = _int("VAD_PREFIX_MS", 300)
|
|
||||||
vad_max_utterance_ms: int = _int("VAD_MAX_UTTERANCE_MS", 30000)
|
|
||||||
|
|
||||||
# Warm the models at startup rather than on the first user turn — a cold
|
|
||||||
# CUDA graph on the first utterance costs several seconds.
|
|
||||||
warmup: bool = _flag("VOICE_WARMUP", True)
|
|
||||||
|
|
||||||
|
|
||||||
settings = Settings()
|
|
||||||
@@ -1,233 +0,0 @@
|
|||||||
"""Voice service — STT and TTS over one WebSocket.
|
|
||||||
|
|
||||||
This process holds the GPU models and nothing else. It has no idea what the
|
|
||||||
CRM is: it receives audio and returns text, receives text and returns audio.
|
|
||||||
All orchestration, auth and business logic stay in the Node service, so voice
|
|
||||||
is just another channel into the same agent rather than a parallel system.
|
|
||||||
|
|
||||||
Protocol (ws /ws/voice), JSON control + binary audio:
|
|
||||||
|
|
||||||
client → server
|
|
||||||
binary 16 kHz mono PCM16 mic frames
|
|
||||||
{"type":"config","lang":"ta"} set the session language
|
|
||||||
{"type":"speak","text":"…","id":"…"} synthesise
|
|
||||||
{"type":"cancel"} stop speaking now (barge-in)
|
|
||||||
{"type":"reset"} clear the endpointer
|
|
||||||
|
|
||||||
server → client
|
|
||||||
{"type":"ready", …}
|
|
||||||
{"type":"speech_start"} VAD opened a turn → caller ducks TTS
|
|
||||||
{"type":"transcript","text":…} a finished utterance
|
|
||||||
{"type":"audio_start","id":…,"sample_rate":24000}
|
|
||||||
binary float32 mono TTS chunks
|
|
||||||
{"type":"audio_end","id":…}
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import asyncio
|
|
||||||
import json
|
|
||||||
import logging
|
|
||||||
import time
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
from fastapi import FastAPI, WebSocket, WebSocketDisconnect
|
|
||||||
|
|
||||||
from .config import settings
|
|
||||||
from .stt import SUPPORTED, Transcriber
|
|
||||||
from .tts import Synthesizer, split_sentences
|
|
||||||
from .vad import Endpointer
|
|
||||||
|
|
||||||
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)-5s %(message)s", datefmt="%H:%M:%S")
|
|
||||||
logger = logging.getLogger("voice")
|
|
||||||
|
|
||||||
app = FastAPI(title="WeLe Voice Service")
|
|
||||||
|
|
||||||
stt = Transcriber()
|
|
||||||
tts = Synthesizer()
|
|
||||||
|
|
||||||
|
|
||||||
@app.on_event("startup")
|
|
||||||
async def _startup() -> None:
|
|
||||||
t0 = time.perf_counter()
|
|
||||||
logger.info("loading models on %s…", settings.device)
|
|
||||||
stt.load()
|
|
||||||
tts.load()
|
|
||||||
if settings.warmup:
|
|
||||||
stt.warmup()
|
|
||||||
tts.warmup()
|
|
||||||
logger.info("voice service ready in %.1fs", time.perf_counter() - t0)
|
|
||||||
|
|
||||||
|
|
||||||
@app.get("/health")
|
|
||||||
async def health() -> dict:
|
|
||||||
import torch
|
|
||||||
|
|
||||||
return {
|
|
||||||
"ok": True,
|
|
||||||
"device": settings.device,
|
|
||||||
"stt_model": settings.stt_model,
|
|
||||||
"tts_model": settings.tts_model,
|
|
||||||
"tts_sample_rate": tts.sample_rate,
|
|
||||||
"languages": SUPPORTED,
|
|
||||||
"vram_gb": round(torch.cuda.memory_reserved() / 1e9, 2) if settings.device == "cuda" else None,
|
|
||||||
}
|
|
||||||
|
|
||||||
|
|
||||||
@app.get("/languages")
|
|
||||||
async def languages() -> dict:
|
|
||||||
return {"languages": SUPPORTED}
|
|
||||||
|
|
||||||
|
|
||||||
class Session:
|
|
||||||
"""One browser connection. Owns its endpointer and its speaking state."""
|
|
||||||
|
|
||||||
def __init__(self, ws: WebSocket) -> None:
|
|
||||||
self.ws = ws
|
|
||||||
# "auto" detects per utterance; `prefer` breaks ties when the detector
|
|
||||||
# is unsure, which is common on short code-mixed ("Tanglish") speech.
|
|
||||||
self.lang = "auto"
|
|
||||||
self.prefer = settings.default_lang
|
|
||||||
self.endpointer = Endpointer()
|
|
||||||
self.endpointer.load()
|
|
||||||
self._speak_task: asyncio.Task | None = None
|
|
||||||
self._cancel = asyncio.Event()
|
|
||||||
|
|
||||||
async def send(self, payload: dict) -> None:
|
|
||||||
await self.ws.send_text(json.dumps(payload, ensure_ascii=False))
|
|
||||||
|
|
||||||
# ── microphone ───────────────────────────────────────────────────────────
|
|
||||||
async def on_audio(self, raw: bytes) -> None:
|
|
||||||
pcm = np.frombuffer(raw, dtype=np.int16).astype(np.float32) / 32768.0
|
|
||||||
loop = asyncio.get_running_loop()
|
|
||||||
# VAD is a small torch model but still blocking; keep the event loop free.
|
|
||||||
utterances, started = await loop.run_in_executor(None, self.endpointer.push, pcm)
|
|
||||||
|
|
||||||
if started:
|
|
||||||
# Barge-in: the user talking wins immediately.
|
|
||||||
await self.stop_speaking()
|
|
||||||
await self.send({"type": "speech_start"})
|
|
||||||
|
|
||||||
for utt in utterances:
|
|
||||||
request_lang = f"auto:{self.prefer}" if self.lang == "auto" else self.lang
|
|
||||||
result = await loop.run_in_executor(None, stt.transcribe, utt.audio, request_lang)
|
|
||||||
if result["text"]:
|
|
||||||
await self.send({"type": "transcript", **result, "truncated": utt.truncated})
|
|
||||||
else:
|
|
||||||
await self.send({"type": "transcript_empty", "reason": result.get("note", "no speech")})
|
|
||||||
|
|
||||||
# ── speaking ─────────────────────────────────────────────────────────────
|
|
||||||
async def speak(self, text: str, msg_id: str, lang: str | None = None, voice: str | None = None) -> None:
|
|
||||||
await self.stop_speaking()
|
|
||||||
self._cancel.clear()
|
|
||||||
self._speak_task = asyncio.create_task(self._speak(text, msg_id, lang or self.lang, voice))
|
|
||||||
|
|
||||||
async def _speak(self, text: str, msg_id: str, lang: str, voice: str | None) -> None:
|
|
||||||
loop = asyncio.get_running_loop()
|
|
||||||
try:
|
|
||||||
await self.send({"type": "audio_start", "id": msg_id, "sample_rate": tts.sample_rate})
|
|
||||||
|
|
||||||
# Sentence at a time: shorter prompts reach first audio sooner, and
|
|
||||||
# a boundary is a clean place to stop when interrupted.
|
|
||||||
for sentence in split_sentences(text):
|
|
||||||
if self._cancel.is_set():
|
|
||||||
break
|
|
||||||
queue: asyncio.Queue = asyncio.Queue(maxsize=32)
|
|
||||||
|
|
||||||
def produce() -> None:
|
|
||||||
try:
|
|
||||||
for chunk in tts.stream(sentence, lang, voice):
|
|
||||||
if self._cancel.is_set():
|
|
||||||
break
|
|
||||||
asyncio.run_coroutine_threadsafe(queue.put(chunk), loop).result()
|
|
||||||
finally:
|
|
||||||
asyncio.run_coroutine_threadsafe(queue.put(None), loop).result()
|
|
||||||
|
|
||||||
loop.run_in_executor(None, produce)
|
|
||||||
while True:
|
|
||||||
chunk = await queue.get()
|
|
||||||
if chunk is None:
|
|
||||||
break
|
|
||||||
if self._cancel.is_set():
|
|
||||||
continue # drain, don't send
|
|
||||||
await self.ws.send_bytes(np.asarray(chunk, dtype=np.float32).tobytes())
|
|
||||||
|
|
||||||
await self.send({"type": "audio_end", "id": msg_id, "cancelled": self._cancel.is_set()})
|
|
||||||
except WebSocketDisconnect:
|
|
||||||
pass
|
|
||||||
except Exception as e: # noqa: BLE001
|
|
||||||
logger.exception("synthesis failed")
|
|
||||||
try:
|
|
||||||
await self.send({"type": "error", "where": "tts", "message": str(e)[:200]})
|
|
||||||
except Exception: # noqa: BLE001
|
|
||||||
pass
|
|
||||||
|
|
||||||
async def stop_speaking(self) -> None:
|
|
||||||
if self._speak_task and not self._speak_task.done():
|
|
||||||
self._cancel.set()
|
|
||||||
try:
|
|
||||||
await asyncio.wait_for(self._speak_task, timeout=2.0)
|
|
||||||
except (asyncio.TimeoutError, asyncio.CancelledError):
|
|
||||||
self._speak_task.cancel()
|
|
||||||
self._speak_task = None
|
|
||||||
|
|
||||||
|
|
||||||
@app.websocket("/ws/voice")
|
|
||||||
async def voice(ws: WebSocket) -> None:
|
|
||||||
await ws.accept()
|
|
||||||
session = Session(ws)
|
|
||||||
await session.send({
|
|
||||||
"type": "ready",
|
|
||||||
"sample_rate_in": settings.sample_rate_in,
|
|
||||||
"sample_rate_out": tts.sample_rate,
|
|
||||||
"languages": SUPPORTED,
|
|
||||||
})
|
|
||||||
logger.info("voice session opened")
|
|
||||||
|
|
||||||
try:
|
|
||||||
while True:
|
|
||||||
msg = await ws.receive()
|
|
||||||
if msg["type"] == "websocket.disconnect":
|
|
||||||
break
|
|
||||||
|
|
||||||
if (raw := msg.get("bytes")) is not None:
|
|
||||||
await session.on_audio(raw)
|
|
||||||
continue
|
|
||||||
|
|
||||||
if (text := msg.get("text")) is None:
|
|
||||||
continue
|
|
||||||
try:
|
|
||||||
data = json.loads(text)
|
|
||||||
except json.JSONDecodeError:
|
|
||||||
continue
|
|
||||||
|
|
||||||
kind = data.get("type")
|
|
||||||
if kind == "config":
|
|
||||||
session.lang = data.get("lang", session.lang)
|
|
||||||
if session.lang != "auto":
|
|
||||||
session.prefer = session.lang
|
|
||||||
elif data.get("prefer"):
|
|
||||||
session.prefer = data["prefer"]
|
|
||||||
session.endpointer.reset()
|
|
||||||
await session.send({"type": "config_ok", "lang": session.lang, "prefer": session.prefer})
|
|
||||||
elif kind == "speak":
|
|
||||||
await session.speak(data.get("text", ""), data.get("id", ""), data.get("lang"), data.get("voice"))
|
|
||||||
elif kind == "cancel":
|
|
||||||
await session.stop_speaking()
|
|
||||||
await session.send({"type": "cancelled"})
|
|
||||||
elif kind == "reset":
|
|
||||||
session.endpointer.reset()
|
|
||||||
except WebSocketDisconnect:
|
|
||||||
pass
|
|
||||||
finally:
|
|
||||||
await session.stop_speaking()
|
|
||||||
logger.info("voice session closed")
|
|
||||||
|
|
||||||
|
|
||||||
def main() -> None:
|
|
||||||
import uvicorn
|
|
||||||
|
|
||||||
uvicorn.run(app, host=settings.host, port=settings.port, log_level="info", ws_max_size=16 * 1024 * 1024)
|
|
||||||
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
main()
|
|
||||||
@@ -1,200 +0,0 @@
|
|||||||
"""Speech-to-text — AI4Bharat IndicConformer, with an English path and auto routing.
|
|
||||||
|
|
||||||
Why two models rather than one:
|
|
||||||
|
|
||||||
* **IndicConformer** decodes 22 Indian languages, and decodes *as* the language
|
|
||||||
you name — it does not detect. Handing it "ta" for English speech produces
|
|
||||||
Tamil-script nonsense. English is not one of its codes at all.
|
|
||||||
* **Whisper (multilingual)** covers English well and, usefully, can identify the
|
|
||||||
spoken language in a single decoder step.
|
|
||||||
|
|
||||||
So the default mode is `auto`: Whisper identifies the language from the audio,
|
|
||||||
English is transcribed by Whisper directly (the mel is already computed, so
|
|
||||||
this costs nothing extra), and anything Indic is routed to IndicConformer,
|
|
||||||
which is far stronger on those languages than Whisper is.
|
|
||||||
|
|
||||||
Code-mixed speech ("Tanglish") is the awkward case: language ID can land either
|
|
||||||
side of the fence on a short, mixed utterance. When Whisper is not confident,
|
|
||||||
the caller's preferred language wins rather than a coin toss — a Tamil-speaking
|
|
||||||
office gets Tamil, and the occasional English sentence still routes correctly
|
|
||||||
when it is clearly English.
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import logging
|
|
||||||
import time
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
import torch
|
|
||||||
|
|
||||||
from .config import settings
|
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
# The codes IndicConformer accepts.
|
|
||||||
INDIC_LANGS = {
|
|
||||||
"as", "bn", "brx", "doi", "gu", "hi", "kn", "kok", "ks", "mai", "ml",
|
|
||||||
"mni", "mr", "ne", "or", "pa", "sa", "sat", "sd", "ta", "te", "ur",
|
|
||||||
}
|
|
||||||
|
|
||||||
# Offered by the UI. "auto" first: most WeLe agents switch language mid-shift.
|
|
||||||
SUPPORTED = [
|
|
||||||
{"code": "auto", "label": "Auto-detect", "native": "Auto"},
|
|
||||||
{"code": "ta", "label": "Tamil", "native": "தமிழ்"},
|
|
||||||
{"code": "en", "label": "English", "native": "English"},
|
|
||||||
{"code": "hi", "label": "Hindi", "native": "हिन्दी"},
|
|
||||||
{"code": "te", "label": "Telugu", "native": "తెలుగు"},
|
|
||||||
{"code": "kn", "label": "Kannada", "native": "ಕನ್ನಡ"},
|
|
||||||
{"code": "ml", "label": "Malayalam", "native": "മലയാളം"},
|
|
||||||
{"code": "mr", "label": "Marathi", "native": "मराठी"},
|
|
||||||
{"code": "bn", "label": "Bengali", "native": "বাংলা"},
|
|
||||||
]
|
|
||||||
|
|
||||||
# Below this, trust the user's stated preference over the detector.
|
|
||||||
DETECT_CONFIDENCE = 0.60
|
|
||||||
|
|
||||||
|
|
||||||
class Transcriber:
|
|
||||||
def __init__(self) -> None:
|
|
||||||
self._indic = None
|
|
||||||
self._whisper = None
|
|
||||||
self._whisper_proc = None
|
|
||||||
|
|
||||||
# ── loading ──────────────────────────────────────────────────────────────
|
|
||||||
def load(self) -> None:
|
|
||||||
from transformers import AutoModel
|
|
||||||
|
|
||||||
t0 = time.perf_counter()
|
|
||||||
# float32: the checkpoint ships custom remote code that assumes fp32.
|
|
||||||
# ~2.4 GB at 600M, which still leaves room for Whisper and the TTS model.
|
|
||||||
self._indic = AutoModel.from_pretrained(settings.stt_model, trust_remote_code=True)
|
|
||||||
self._indic.to(settings.device).eval()
|
|
||||||
logger.info("STT loaded (%s) in %.1fs", settings.stt_model, time.perf_counter() - t0)
|
|
||||||
|
|
||||||
if settings.preload_english:
|
|
||||||
self._load_whisper()
|
|
||||||
|
|
||||||
def _load_whisper(self) -> None:
|
|
||||||
"""English + language ID. Multilingual on purpose — the .en checkpoint
|
|
||||||
cannot identify languages, which is the whole point of auto mode."""
|
|
||||||
if self._whisper is not None:
|
|
||||||
return
|
|
||||||
from transformers import WhisperForConditionalGeneration, WhisperProcessor
|
|
||||||
|
|
||||||
t0 = time.perf_counter()
|
|
||||||
self._whisper_proc = WhisperProcessor.from_pretrained(settings.english_model)
|
|
||||||
self._whisper = WhisperForConditionalGeneration.from_pretrained(
|
|
||||||
settings.english_model, torch_dtype=settings.dtype,
|
|
||||||
).to(settings.device).eval()
|
|
||||||
logger.info("English/ID model loaded (%s) in %.1fs", settings.english_model, time.perf_counter() - t0)
|
|
||||||
|
|
||||||
# ── inference ────────────────────────────────────────────────────────────
|
|
||||||
@torch.inference_mode()
|
|
||||||
def transcribe(self, audio: np.ndarray, lang: str = "auto") -> dict:
|
|
||||||
"""audio: float32 mono @16 kHz in [-1, 1].
|
|
||||||
|
|
||||||
`lang` may be an explicit code, or "auto" / "auto:ta" to detect with a
|
|
||||||
fallback preference.
|
|
||||||
"""
|
|
||||||
t0 = time.perf_counter()
|
|
||||||
if audio.size < settings.sample_rate_in // 5: # under 200 ms
|
|
||||||
return {"text": "", "lang": lang, "ms": 0, "note": "too short"}
|
|
||||||
|
|
||||||
detected = None
|
|
||||||
confidence = None
|
|
||||||
|
|
||||||
if lang.startswith("auto"):
|
|
||||||
prefer = lang.split(":", 1)[1] if ":" in lang else settings.default_lang
|
|
||||||
feats = self._features(audio)
|
|
||||||
detected, confidence = self._detect(feats)
|
|
||||||
|
|
||||||
if confidence is not None and confidence < DETECT_CONFIDENCE:
|
|
||||||
logger.info("language ID low confidence (%s @ %.2f) — using preferred %s",
|
|
||||||
detected, confidence, prefer)
|
|
||||||
use = prefer
|
|
||||||
elif detected == "en" or detected in INDIC_LANGS:
|
|
||||||
use = detected
|
|
||||||
else:
|
|
||||||
# Whisper reported something we cannot decode (e.g. "nn" on
|
|
||||||
# noise). Fall back rather than fail.
|
|
||||||
use = prefer
|
|
||||||
|
|
||||||
text = self._english(audio, feats=feats) if use == "en" else self._indic_decode(audio, use)
|
|
||||||
else:
|
|
||||||
use = lang
|
|
||||||
text = self._english(audio) if lang == "en" else self._indic_decode(audio, lang)
|
|
||||||
|
|
||||||
ms = int((time.perf_counter() - t0) * 1000)
|
|
||||||
audio_ms = int(1000 * audio.size / settings.sample_rate_in)
|
|
||||||
logger.info("STT %s%s: %dms audio → %dms → %r",
|
|
||||||
use, f" (detected {detected} {confidence:.2f})" if confidence is not None else "",
|
|
||||||
audio_ms, ms, text[:80])
|
|
||||||
|
|
||||||
return {
|
|
||||||
"text": text.strip(),
|
|
||||||
"lang": use,
|
|
||||||
"detected": detected,
|
|
||||||
"confidence": round(confidence, 3) if confidence is not None else None,
|
|
||||||
"ms": ms,
|
|
||||||
"audio_ms": audio_ms,
|
|
||||||
}
|
|
||||||
|
|
||||||
# ── internals ────────────────────────────────────────────────────────────
|
|
||||||
def _features(self, audio: np.ndarray):
|
|
||||||
self._load_whisper()
|
|
||||||
return self._whisper_proc(
|
|
||||||
audio, sampling_rate=settings.sample_rate_in, return_tensors="pt",
|
|
||||||
).input_features.to(settings.device, settings.dtype)
|
|
||||||
|
|
||||||
def _detect(self, feats) -> tuple[str | None, float | None]:
|
|
||||||
"""One decoder step: read the language-token distribution."""
|
|
||||||
try:
|
|
||||||
tok = self._whisper_proc.tokenizer
|
|
||||||
sot = tok.convert_tokens_to_ids("<|startoftranscript|>")
|
|
||||||
start = torch.tensor([[sot]], device=settings.device)
|
|
||||||
logits = self._whisper(feats, decoder_input_ids=start).logits[:, -1]
|
|
||||||
|
|
||||||
lang_ids, codes = [], []
|
|
||||||
for code in {*INDIC_LANGS, "en"}:
|
|
||||||
tid = tok.convert_tokens_to_ids(f"<|{code}|>")
|
|
||||||
# Unknown languages map to the unk id; skip those.
|
|
||||||
if tid is not None and tid != tok.unk_token_id:
|
|
||||||
lang_ids.append(tid)
|
|
||||||
codes.append(code)
|
|
||||||
if not lang_ids:
|
|
||||||
return None, None
|
|
||||||
|
|
||||||
probs = torch.softmax(logits[0, lang_ids].float(), dim=-1)
|
|
||||||
best = int(probs.argmax())
|
|
||||||
return codes[best], float(probs[best])
|
|
||||||
except Exception as e: # noqa: BLE001 — detection must never break a turn
|
|
||||||
logger.warning("language ID failed (%s) — falling back to preference", e)
|
|
||||||
return None, None
|
|
||||||
|
|
||||||
def _indic_decode(self, audio: np.ndarray, lang: str) -> str:
|
|
||||||
if lang not in INDIC_LANGS:
|
|
||||||
logger.warning("unsupported STT language %r — using %s", lang, settings.default_lang)
|
|
||||||
lang = settings.default_lang
|
|
||||||
wav = torch.from_numpy(audio).unsqueeze(0).to(settings.device) # (1, N)
|
|
||||||
out = self._indic(wav, lang, settings.stt_decoding)
|
|
||||||
if isinstance(out, (list, tuple)):
|
|
||||||
return str(out[0]) if out else ""
|
|
||||||
return str(out)
|
|
||||||
|
|
||||||
def _english(self, audio: np.ndarray, feats=None) -> str:
|
|
||||||
self._load_whisper()
|
|
||||||
if feats is None:
|
|
||||||
feats = self._features(audio)
|
|
||||||
ids = self._whisper.generate(feats, language="en", task="transcribe", max_new_tokens=180)
|
|
||||||
return self._whisper_proc.batch_decode(ids, skip_special_tokens=True)[0]
|
|
||||||
|
|
||||||
def warmup(self) -> None:
|
|
||||||
"""Silent pass so the first real utterance isn't paying for CUDA init."""
|
|
||||||
try:
|
|
||||||
silence = np.zeros(settings.sample_rate_in, dtype=np.float32)
|
|
||||||
self.transcribe(silence, settings.default_lang)
|
|
||||||
if settings.preload_english:
|
|
||||||
self.transcribe(silence, "en")
|
|
||||||
logger.info("STT warm")
|
|
||||||
except Exception as e: # noqa: BLE001
|
|
||||||
logger.warning("STT warmup skipped: %s", e)
|
|
||||||
@@ -1,155 +0,0 @@
|
|||||||
"""Text-to-speech — AI4Bharat Indic Parler-TTS.
|
|
||||||
|
|
||||||
Parler is prompted with *two* texts: the words to say, and a natural-language
|
|
||||||
description of how to say them (speaker, pace, room tone). The description is
|
|
||||||
what selects a voice — there is no speaker-id argument.
|
|
||||||
|
|
||||||
Latency shape: Parler is autoregressive, so a whole paragraph costs whole-
|
|
||||||
paragraph time before the first sample exists. Two things fix that here:
|
|
||||||
|
|
||||||
1. `ParlerTTSStreamer` yields audio while generation continues, so playback
|
|
||||||
starts after roughly the first `play_steps` frames rather than at the end.
|
|
||||||
2. The caller sends one *sentence* at a time. Short prompts reach their first
|
|
||||||
chunk sooner, and a sentence boundary is a natural place for the assistant
|
|
||||||
to be interrupted.
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import logging
|
|
||||||
import re
|
|
||||||
import time
|
|
||||||
from threading import Thread
|
|
||||||
from typing import Iterator
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
import torch
|
|
||||||
|
|
||||||
from .config import settings
|
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
# Voices recommended on the model card, per language.
|
|
||||||
VOICES = {
|
|
||||||
"ta": "Jaya", "hi": "Rohit", "te": "Prakash", "kn": "Suresh",
|
|
||||||
"ml": "Anjali", "mr": "Sanjay", "bn": "Arjun", "en": "Mary",
|
|
||||||
}
|
|
||||||
|
|
||||||
_DESCRIPTION = (
|
|
||||||
"{speaker} speaks in a warm, clear, professional tone at a natural pace. "
|
|
||||||
"The recording is very high quality with no background noise."
|
|
||||||
)
|
|
||||||
|
|
||||||
# Split on sentence enders including the Devanagari danda, keeping it simple —
|
|
||||||
# this only needs to find safe places to cut, not parse language.
|
|
||||||
_SENTENCE_RX = re.compile(r"(?<=[.!?।॥])\s+")
|
|
||||||
|
|
||||||
|
|
||||||
def split_sentences(text: str, max_chars: int = 220) -> list[str]:
|
|
||||||
"""Break text into TTS-sized pieces at sentence boundaries where possible."""
|
|
||||||
out: list[str] = []
|
|
||||||
for part in _SENTENCE_RX.split(text.strip()):
|
|
||||||
part = part.strip()
|
|
||||||
if not part:
|
|
||||||
continue
|
|
||||||
while len(part) > max_chars:
|
|
||||||
cut = part.rfind(" ", 0, max_chars)
|
|
||||||
if cut <= 0:
|
|
||||||
cut = max_chars
|
|
||||||
out.append(part[:cut].strip())
|
|
||||||
part = part[cut:].strip()
|
|
||||||
if part:
|
|
||||||
out.append(part)
|
|
||||||
return out
|
|
||||||
|
|
||||||
|
|
||||||
class Synthesizer:
|
|
||||||
def __init__(self) -> None:
|
|
||||||
self._model = None
|
|
||||||
self._tok = None
|
|
||||||
self._desc_tok = None
|
|
||||||
self.sample_rate = settings.sample_rate_out
|
|
||||||
|
|
||||||
def load(self) -> None:
|
|
||||||
from parler_tts import ParlerTTSForConditionalGeneration
|
|
||||||
from transformers import AutoTokenizer
|
|
||||||
|
|
||||||
t0 = time.perf_counter()
|
|
||||||
self._model = ParlerTTSForConditionalGeneration.from_pretrained(
|
|
||||||
settings.tts_model, torch_dtype=settings.dtype,
|
|
||||||
).to(settings.device).eval()
|
|
||||||
self._tok = AutoTokenizer.from_pretrained(settings.tts_model)
|
|
||||||
self._desc_tok = AutoTokenizer.from_pretrained(self._model.config.text_encoder._name_or_path)
|
|
||||||
self.sample_rate = int(self._model.config.sampling_rate)
|
|
||||||
logger.info(
|
|
||||||
"TTS loaded (%s) in %.1fs @ %d Hz", settings.tts_model,
|
|
||||||
time.perf_counter() - t0, self.sample_rate,
|
|
||||||
)
|
|
||||||
|
|
||||||
def _describe(self, lang: str, voice: str | None) -> str:
|
|
||||||
return _DESCRIPTION.format(speaker=voice or VOICES.get(lang, "Jaya"))
|
|
||||||
|
|
||||||
@torch.inference_mode()
|
|
||||||
def stream(self, text: str, lang: str = "ta", voice: str | None = None) -> Iterator[np.ndarray]:
|
|
||||||
"""Yield float32 mono chunks at `self.sample_rate` as they are generated."""
|
|
||||||
text = (text or "").strip()
|
|
||||||
if not text:
|
|
||||||
return
|
|
||||||
|
|
||||||
desc = self._desc_tok(self._describe(lang, voice), return_tensors="pt").to(settings.device)
|
|
||||||
prompt = self._tok(text, return_tensors="pt").to(settings.device)
|
|
||||||
|
|
||||||
kwargs = dict(
|
|
||||||
input_ids=desc.input_ids,
|
|
||||||
attention_mask=desc.attention_mask,
|
|
||||||
prompt_input_ids=prompt.input_ids,
|
|
||||||
prompt_attention_mask=prompt.attention_mask,
|
|
||||||
)
|
|
||||||
|
|
||||||
streamer = self._make_streamer()
|
|
||||||
if streamer is None:
|
|
||||||
yield self._generate_blocking(kwargs, text)
|
|
||||||
return
|
|
||||||
|
|
||||||
t0 = time.perf_counter()
|
|
||||||
# generate() blocks, so it runs on its own thread and the streamer is
|
|
||||||
# drained here as frames become available.
|
|
||||||
thread = Thread(target=self._model.generate, kwargs={**kwargs, "streamer": streamer}, daemon=True)
|
|
||||||
thread.start()
|
|
||||||
|
|
||||||
first = True
|
|
||||||
for chunk in streamer:
|
|
||||||
if chunk is None or len(chunk) == 0:
|
|
||||||
continue
|
|
||||||
audio = chunk.astype(np.float32) if isinstance(chunk, np.ndarray) else chunk.cpu().numpy().astype(np.float32)
|
|
||||||
if first:
|
|
||||||
logger.info("TTS first chunk in %dms (%d chars)", int((time.perf_counter() - t0) * 1000), len(text))
|
|
||||||
first = False
|
|
||||||
yield audio
|
|
||||||
thread.join(timeout=1.0)
|
|
||||||
|
|
||||||
def _make_streamer(self):
|
|
||||||
try:
|
|
||||||
from parler_tts import ParlerTTSStreamer
|
|
||||||
except ImportError:
|
|
||||||
logger.warning("ParlerTTSStreamer unavailable — falling back to blocking synthesis")
|
|
||||||
return None
|
|
||||||
# play_steps trades first-chunk latency against per-chunk overhead;
|
|
||||||
# ~0.5 s of audio keeps playback continuous without stalling generation.
|
|
||||||
frame_rate = getattr(self._model.audio_encoder.config, "frame_rate", 86)
|
|
||||||
return ParlerTTSStreamer(self._model, device=settings.device, play_steps=int(frame_rate / 2))
|
|
||||||
|
|
||||||
@torch.inference_mode()
|
|
||||||
def _generate_blocking(self, kwargs: dict, text: str) -> np.ndarray:
|
|
||||||
t0 = time.perf_counter()
|
|
||||||
gen = self._model.generate(**kwargs)
|
|
||||||
audio = gen.cpu().numpy().squeeze().astype(np.float32)
|
|
||||||
logger.info("TTS (blocking) %d chars in %dms", len(text), int((time.perf_counter() - t0) * 1000))
|
|
||||||
return audio
|
|
||||||
|
|
||||||
def warmup(self) -> None:
|
|
||||||
try:
|
|
||||||
for _ in self.stream("வணக்கம்", "ta"):
|
|
||||||
break
|
|
||||||
logger.info("TTS warm")
|
|
||||||
except Exception as e: # noqa: BLE001
|
|
||||||
logger.warning("TTS warmup skipped: %s", e)
|
|
||||||
@@ -1,127 +0,0 @@
|
|||||||
"""Endpointing with Silero VAD.
|
|
||||||
|
|
||||||
Turn boundaries are decided here rather than in the browser for two reasons:
|
|
||||||
the same decision then applies to every future channel (a phone bridge has no
|
|
||||||
AudioWorklet), and barge-in needs the server to know someone started talking
|
|
||||||
while the assistant was still speaking.
|
|
||||||
"""
|
|
||||||
from __future__ import annotations
|
|
||||||
|
|
||||||
import logging
|
|
||||||
from collections import deque
|
|
||||||
from dataclasses import dataclass, field
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
import torch
|
|
||||||
|
|
||||||
from .config import settings
|
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
|
||||||
class Utterance:
|
|
||||||
audio: np.ndarray # float32 mono @16k, in [-1, 1]
|
|
||||||
duration_ms: int
|
|
||||||
truncated: bool = False # hit the max-length guard rather than silence
|
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
|
||||||
class VADState:
|
|
||||||
speaking: bool = False
|
|
||||||
speech_ms: int = 0
|
|
||||||
silence_ms: int = 0
|
|
||||||
buffer: list[np.ndarray] = field(default_factory=list)
|
|
||||||
|
|
||||||
|
|
||||||
class Endpointer:
|
|
||||||
"""Streaming VAD that emits one Utterance per detected turn.
|
|
||||||
|
|
||||||
Silero wants exactly 512 samples at 16 kHz, but the browser sends 40 ms
|
|
||||||
(640-sample) chunks. Rather than force the client to match, incoming audio
|
|
||||||
is accumulated and drained in exact frames.
|
|
||||||
"""
|
|
||||||
|
|
||||||
def __init__(self) -> None:
|
|
||||||
self._model = None
|
|
||||||
self._pending = np.zeros(0, dtype=np.float32)
|
|
||||||
self.state = VADState()
|
|
||||||
# Pre-roll: speech is only *detected* a frame or two in, so without a
|
|
||||||
# prefix the first phoneme is already gone by the time we start saving.
|
|
||||||
prefix_frames = max(1, (settings.vad_prefix_ms * settings.sample_rate_in) // (1000 * settings.vad_frame))
|
|
||||||
self._prefix: deque[np.ndarray] = deque(maxlen=prefix_frames)
|
|
||||||
|
|
||||||
def load(self) -> None:
|
|
||||||
from silero_vad import load_silero_vad
|
|
||||||
|
|
||||||
self._model = load_silero_vad()
|
|
||||||
logger.info("silero VAD loaded")
|
|
||||||
|
|
||||||
def reset(self) -> None:
|
|
||||||
self.state = VADState()
|
|
||||||
self._pending = np.zeros(0, dtype=np.float32)
|
|
||||||
self._prefix.clear()
|
|
||||||
if self._model is not None:
|
|
||||||
self._model.reset_states()
|
|
||||||
|
|
||||||
@property
|
|
||||||
def is_speaking(self) -> bool:
|
|
||||||
return self.state.speaking
|
|
||||||
|
|
||||||
def push(self, pcm: np.ndarray) -> tuple[list[Utterance], bool]:
|
|
||||||
"""Feed float32 audio.
|
|
||||||
|
|
||||||
Returns (completed utterances, speech_started_this_call). The second
|
|
||||||
value drives barge-in: the caller cuts TTS playback the moment it flips.
|
|
||||||
"""
|
|
||||||
assert self._model is not None, "call load() first"
|
|
||||||
|
|
||||||
self._pending = np.concatenate([self._pending, pcm]) if self._pending.size else pcm
|
|
||||||
frame = settings.vad_frame
|
|
||||||
frame_ms = int(1000 * frame / settings.sample_rate_in)
|
|
||||||
|
|
||||||
done: list[Utterance] = []
|
|
||||||
started = False
|
|
||||||
|
|
||||||
while self._pending.size >= frame:
|
|
||||||
chunk = self._pending[:frame]
|
|
||||||
self._pending = self._pending[frame:]
|
|
||||||
|
|
||||||
with torch.no_grad():
|
|
||||||
prob = float(self._model(torch.from_numpy(chunk), settings.sample_rate_in).item())
|
|
||||||
|
|
||||||
voiced = prob >= settings.vad_threshold
|
|
||||||
st = self.state
|
|
||||||
|
|
||||||
if not st.speaking:
|
|
||||||
self._prefix.append(chunk)
|
|
||||||
if voiced:
|
|
||||||
st.speech_ms += frame_ms
|
|
||||||
if st.speech_ms >= settings.vad_min_speech_ms:
|
|
||||||
# Commit: open the turn with the pre-roll included.
|
|
||||||
st.speaking = True
|
|
||||||
st.silence_ms = 0
|
|
||||||
st.buffer = list(self._prefix)
|
|
||||||
self._prefix.clear()
|
|
||||||
started = True
|
|
||||||
else:
|
|
||||||
st.speech_ms = 0
|
|
||||||
continue
|
|
||||||
|
|
||||||
# Speaking.
|
|
||||||
st.buffer.append(chunk)
|
|
||||||
if voiced:
|
|
||||||
st.silence_ms = 0
|
|
||||||
else:
|
|
||||||
st.silence_ms += frame_ms
|
|
||||||
|
|
||||||
spoken_ms = len(st.buffer) * frame_ms
|
|
||||||
ended = st.silence_ms >= settings.vad_silence_ms
|
|
||||||
too_long = spoken_ms >= settings.vad_max_utterance_ms
|
|
||||||
|
|
||||||
if ended or too_long:
|
|
||||||
audio = np.concatenate(st.buffer)
|
|
||||||
done.append(Utterance(audio=audio, duration_ms=spoken_ms, truncated=too_long and not ended))
|
|
||||||
self.reset()
|
|
||||||
|
|
||||||
return done, started
|
|
||||||
@@ -1,95 +0,0 @@
|
|||||||
"""Honest latency benchmark: warm up first, then time repeated runs.
|
|
||||||
|
|
||||||
The first CUDA generation pays for kernel autotuning and cache allocation, so a
|
|
||||||
single cold measurement makes any model look far worse than it is in service.
|
|
||||||
"""
|
|
||||||
import logging
|
|
||||||
import time
|
|
||||||
from threading import Thread
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
import torch
|
|
||||||
|
|
||||||
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
|
||||||
log = logging.getLogger("bench")
|
|
||||||
DEV = "cuda" if torch.cuda.is_available() else "cpu"
|
|
||||||
|
|
||||||
log.info("device=%s gpu=%s", DEV, torch.cuda.get_device_name(0) if DEV == "cuda" else "-")
|
|
||||||
|
|
||||||
# ── STT ──────────────────────────────────────────────────────────────────────
|
|
||||||
from transformers import AutoModel
|
|
||||||
|
|
||||||
stt = AutoModel.from_pretrained("ai4bharat/indic-conformer-600m-multilingual", trust_remote_code=True)
|
|
||||||
stt = stt.to(DEV).eval()
|
|
||||||
|
|
||||||
# Where does it actually run? A model wrapping ONNX ignores .to(cuda).
|
|
||||||
params = list(stt.parameters())
|
|
||||||
log.info("STT param device: %s (%d tensors)", params[0].device if params else "NO TORCH PARAMS", len(params))
|
|
||||||
log.info("STT type: %s", type(stt).__name__)
|
|
||||||
|
|
||||||
wav = torch.from_numpy((np.random.randn(16000 * 4) * 0.02).astype(np.float32)).unsqueeze(0).to(DEV)
|
|
||||||
with torch.inference_mode():
|
|
||||||
stt(wav, "ta", "ctc") # warmup
|
|
||||||
times = []
|
|
||||||
for _ in range(3):
|
|
||||||
t0 = time.perf_counter()
|
|
||||||
with torch.inference_mode():
|
|
||||||
stt(wav, "ta", "ctc")
|
|
||||||
times.append((time.perf_counter() - t0) * 1000)
|
|
||||||
log.info("STT 4000 ms audio → %.0f / %.0f / %.0f ms (RTF %.2fx)",
|
|
||||||
*times, (sum(times) / len(times)) / 4000)
|
|
||||||
|
|
||||||
# ── TTS ──────────────────────────────────────────────────────────────────────
|
|
||||||
from parler_tts import ParlerTTSForConditionalGeneration, ParlerTTSStreamer
|
|
||||||
from transformers import AutoTokenizer
|
|
||||||
|
|
||||||
dtype = torch.float16 if DEV == "cuda" else torch.float32
|
|
||||||
tts = ParlerTTSForConditionalGeneration.from_pretrained("ai4bharat/indic-parler-tts", torch_dtype=dtype).to(DEV).eval()
|
|
||||||
tok = AutoTokenizer.from_pretrained("ai4bharat/indic-parler-tts")
|
|
||||||
dtok = AutoTokenizer.from_pretrained(tts.config.text_encoder._name_or_path)
|
|
||||||
SR = tts.config.sampling_rate
|
|
||||||
log.info("TTS sampling_rate=%d frame_rate=%s", SR, getattr(tts.audio_encoder.config, "frame_rate", "?"))
|
|
||||||
|
|
||||||
desc = "Jaya speaks in a warm, clear, professional tone at a natural pace. The recording is very high quality with no background noise."
|
|
||||||
d = dtok(desc, return_tensors="pt").to(DEV)
|
|
||||||
|
|
||||||
|
|
||||||
def run(text, stream=True):
|
|
||||||
p = tok(text, return_tensors="pt").to(DEV)
|
|
||||||
kw = dict(input_ids=d.input_ids, attention_mask=d.attention_mask,
|
|
||||||
prompt_input_ids=p.input_ids, prompt_attention_mask=p.attention_mask)
|
|
||||||
t0 = time.perf_counter()
|
|
||||||
if stream:
|
|
||||||
fr = int(getattr(tts.audio_encoder.config, "frame_rate", 86) / 2)
|
|
||||||
s = ParlerTTSStreamer(tts, device=DEV, play_steps=fr)
|
|
||||||
Thread(target=tts.generate, kwargs={**kw, "streamer": s}, daemon=True).start()
|
|
||||||
first, n = None, 0
|
|
||||||
for c in s:
|
|
||||||
if c is None or len(c) == 0:
|
|
||||||
continue
|
|
||||||
if first is None:
|
|
||||||
first = (time.perf_counter() - t0) * 1000
|
|
||||||
n += len(c)
|
|
||||||
else:
|
|
||||||
with torch.inference_mode():
|
|
||||||
g = tts.generate(**kw)
|
|
||||||
n = g.shape[-1]
|
|
||||||
first = None
|
|
||||||
total = (time.perf_counter() - t0) * 1000
|
|
||||||
return first, total, 1000 * n / SR
|
|
||||||
|
|
||||||
|
|
||||||
SHORT = "மூவாயிரம் நானூறு லீட்கள் உள்ளன."
|
|
||||||
LONG = "புதிய லீட்கள் மூவாயிரம் நானூற்று இருபத்தேழு. இதில் எழுபத்தாறு சதவீதம் இன்னும் தொடர்பு கொள்ளப்படவில்லை."
|
|
||||||
|
|
||||||
run(SHORT) # warmup
|
|
||||||
log.info("")
|
|
||||||
for label, text in (("short", SHORT), ("long", LONG)):
|
|
||||||
first, total, audio = run(text)
|
|
||||||
log.info("TTS %-5s %2d chars → first %.0f ms | total %.0f ms | audio %.0f ms | RTF %.2fx",
|
|
||||||
label, len(text), first or -1, total, audio, total / max(audio, 1))
|
|
||||||
|
|
||||||
if DEV == "cuda":
|
|
||||||
log.info("\nVRAM peak reserved: %.2f GB of %.1f GB",
|
|
||||||
torch.cuda.max_memory_reserved() / 1e9,
|
|
||||||
torch.cuda.get_device_properties(0).total_memory / 1e9)
|
|
||||||
@@ -1,47 +0,0 @@
|
|||||||
"""Which repos can we actually DOWNLOAD from?
|
|
||||||
|
|
||||||
`model_info` succeeds on a gated repo you have not been granted, so it is not a
|
|
||||||
usable test. Fetching a real file is.
|
|
||||||
"""
|
|
||||||
import os
|
|
||||||
|
|
||||||
from huggingface_hub import hf_hub_download
|
|
||||||
|
|
||||||
CANDIDATES = [
|
|
||||||
("ai4bharat/indic-conformer-600m-multilingual", "Indic ASR (22 languages)"),
|
|
||||||
("ai4bharat/indic-parler-tts", "Indic TTS (21 languages)"),
|
|
||||||
("openai/whisper-small", "English ASR + language ID"),
|
|
||||||
("ai4bharat/indic-parler-tts-pretrained", "TTS base (fallback)"),
|
|
||||||
("ai4bharat/indicconformer_stt_ta_hybrid_rnnt_large", "Tamil-only ASR (fallback)"),
|
|
||||||
]
|
|
||||||
|
|
||||||
token = os.environ.get("HF_TOKEN") or os.environ.get("HUGGING_FACE_HUB_TOKEN")
|
|
||||||
print(f"token present: {bool(token)}\n")
|
|
||||||
print(f"{'repo':52} {'what it is':30} status")
|
|
||||||
print("-" * 108)
|
|
||||||
|
|
||||||
blocked = []
|
|
||||||
for repo, what in CANDIDATES:
|
|
||||||
try:
|
|
||||||
hf_hub_download(repo_id=repo, filename="config.json", token=token)
|
|
||||||
print(f"{repo:52} {what:30} DOWNLOADABLE")
|
|
||||||
except Exception as e: # noqa: BLE001
|
|
||||||
msg = str(e)
|
|
||||||
if "not in the authorized list" in msg or "403" in msg:
|
|
||||||
print(f"{repo:52} {what:30} NEEDS ACCESS — click 'Agree' on the model page")
|
|
||||||
blocked.append(repo)
|
|
||||||
elif "401" in msg or "restricted" in msg:
|
|
||||||
print(f"{repo:52} {what:30} NOT AUTHENTICATED")
|
|
||||||
blocked.append(repo)
|
|
||||||
elif "404" in msg or "EntryNotFound" in msg:
|
|
||||||
# No config.json at the root, but the repo itself is reachable.
|
|
||||||
print(f"{repo:52} {what:30} reachable (no config.json)")
|
|
||||||
else:
|
|
||||||
print(f"{repo:52} {what:30} ERROR {msg[:34]}")
|
|
||||||
|
|
||||||
if blocked:
|
|
||||||
print("\nGrant access here (sign in, click 'Agree and access repository'):")
|
|
||||||
for repo in blocked:
|
|
||||||
print(f" https://huggingface.co/{repo}")
|
|
||||||
else:
|
|
||||||
print("\nAll required models are downloadable.")
|
|
||||||
@@ -1,82 +0,0 @@
|
|||||||
"""De-risk before building around these models: do they load, fit, and run fast enough?"""
|
|
||||||
import logging
|
|
||||||
import time
|
|
||||||
|
|
||||||
import numpy as np
|
|
||||||
import torch
|
|
||||||
|
|
||||||
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
|
||||||
log = logging.getLogger("probe")
|
|
||||||
|
|
||||||
DEV = "cuda" if torch.cuda.is_available() else "cpu"
|
|
||||||
def vram(tag):
|
|
||||||
if DEV == "cuda":
|
|
||||||
log.info(" VRAM %-10s alloc %.2f GB | reserved %.2f GB", tag,
|
|
||||||
torch.cuda.memory_allocated() / 1e9, torch.cuda.memory_reserved() / 1e9)
|
|
||||||
|
|
||||||
log.info("device=%s", DEV)
|
|
||||||
if DEV == "cuda":
|
|
||||||
log.info("gpu=%s total=%.1f GB", torch.cuda.get_device_name(0),
|
|
||||||
torch.cuda.get_device_properties(0).total_memory / 1e9)
|
|
||||||
|
|
||||||
# ── STT ──────────────────────────────────────────────────────────────────────
|
|
||||||
log.info("\n[1/2] loading IndicConformer…")
|
|
||||||
t0 = time.perf_counter()
|
|
||||||
from transformers import AutoModel
|
|
||||||
stt = AutoModel.from_pretrained("ai4bharat/indic-conformer-600m-multilingual", trust_remote_code=True)
|
|
||||||
stt = stt.to(DEV).eval()
|
|
||||||
log.info(" loaded in %.1fs", time.perf_counter() - t0)
|
|
||||||
vram("after STT")
|
|
||||||
|
|
||||||
# 3 s of quiet noise — we only care that a forward pass runs and how long it takes.
|
|
||||||
wav = torch.from_numpy((np.random.randn(16000 * 3) * 0.01).astype(np.float32)).unsqueeze(0).to(DEV)
|
|
||||||
for i in range(2):
|
|
||||||
t0 = time.perf_counter()
|
|
||||||
with torch.inference_mode():
|
|
||||||
out = stt(wav, "ta", "ctc")
|
|
||||||
log.info(" pass %d: %.0f ms -> %r", i + 1, (time.perf_counter() - t0) * 1000, str(out)[:60])
|
|
||||||
|
|
||||||
# ── TTS ──────────────────────────────────────────────────────────────────────
|
|
||||||
log.info("\n[2/2] loading Indic Parler-TTS…")
|
|
||||||
t0 = time.perf_counter()
|
|
||||||
from parler_tts import ParlerTTSForConditionalGeneration, ParlerTTSStreamer
|
|
||||||
from transformers import AutoTokenizer
|
|
||||||
|
|
||||||
dtype = torch.float16 if DEV == "cuda" else torch.float32
|
|
||||||
tts = ParlerTTSForConditionalGeneration.from_pretrained("ai4bharat/indic-parler-tts", torch_dtype=dtype).to(DEV).eval()
|
|
||||||
tok = AutoTokenizer.from_pretrained("ai4bharat/indic-parler-tts")
|
|
||||||
dtok = AutoTokenizer.from_pretrained(tts.config.text_encoder._name_or_path)
|
|
||||||
log.info(" loaded in %.1fs sr=%d", time.perf_counter() - t0, tts.config.sampling_rate)
|
|
||||||
vram("after TTS")
|
|
||||||
|
|
||||||
desc = "Jaya speaks in a warm, clear, professional tone at a natural pace. The recording is very high quality with no background noise."
|
|
||||||
prompt = "உங்கள் புதிய லீட்கள் மூன்று ஆயிரம் நானூறு."
|
|
||||||
|
|
||||||
d = dtok(desc, return_tensors="pt").to(DEV)
|
|
||||||
p = tok(prompt, return_tensors="pt").to(DEV)
|
|
||||||
kw = dict(input_ids=d.input_ids, attention_mask=d.attention_mask,
|
|
||||||
prompt_input_ids=p.input_ids, prompt_attention_mask=p.attention_mask)
|
|
||||||
|
|
||||||
# Streaming: what the user actually experiences is time-to-first-audio.
|
|
||||||
frame_rate = getattr(tts.audio_encoder.config, "frame_rate", 86)
|
|
||||||
streamer = ParlerTTSStreamer(tts, device=DEV, play_steps=int(frame_rate / 2))
|
|
||||||
from threading import Thread
|
|
||||||
t0 = time.perf_counter()
|
|
||||||
Thread(target=tts.generate, kwargs={**kw, "streamer": streamer}, daemon=True).start()
|
|
||||||
|
|
||||||
first_ms, total = None, 0
|
|
||||||
for chunk in streamer:
|
|
||||||
if chunk is None or len(chunk) == 0:
|
|
||||||
continue
|
|
||||||
if first_ms is None:
|
|
||||||
first_ms = (time.perf_counter() - t0) * 1000
|
|
||||||
total += len(chunk)
|
|
||||||
gen_ms = (time.perf_counter() - t0) * 1000
|
|
||||||
audio_ms = 1000 * total / tts.config.sampling_rate
|
|
||||||
|
|
||||||
log.info(" time to FIRST audio : %.0f ms", first_ms or -1)
|
|
||||||
log.info(" full generation : %.0f ms for %.0f ms of audio", gen_ms, audio_ms)
|
|
||||||
log.info(" realtime factor : %.2fx (<1 means faster than realtime)", gen_ms / max(audio_ms, 1))
|
|
||||||
vram("peak")
|
|
||||||
if DEV == "cuda":
|
|
||||||
log.info(" peak reserved: %.2f GB", torch.cuda.max_memory_reserved() / 1e9)
|
|
||||||
@@ -1,11 +0,0 @@
|
|||||||
# Torch / transformers come from the system site-packages (torch 2.5.1+cu121).
|
|
||||||
# Only what the voice pipeline adds on top lives here.
|
|
||||||
fastapi>=0.115
|
|
||||||
uvicorn[standard]>=0.30
|
|
||||||
websockets>=12.0
|
|
||||||
soundfile>=0.13
|
|
||||||
numpy>=1.26
|
|
||||||
scipy>=1.10
|
|
||||||
sentencepiece>=0.2
|
|
||||||
# Parler-TTS is not published on PyPI; the Indic model needs this fork-compatible package.
|
|
||||||
git+https://github.com/huggingface/parler-tts.git
|
|
||||||
Reference in New Issue
Block a user