From 6bb0ef25ca3c22683706a661564ec5edc4b03c4f Mon Sep 17 00:00:00 2001 From: thulasiraman S Date: Fri, 28 Aug 2026 08:50:08 +0530 Subject: [PATCH] WeLe Agentic AI with Docker deployment --- .env.example | 15 + .gitignore | 7 +- assets/tts/mms-tts-tam/added_tokens.json | 3 + assets/tts/mms-tts-tam/config.json | 82 ++ assets/tts/mms-tts-tam/quantize_config.json | 4 + .../tts/mms-tts-tam/special_tokens_map.json | 4 + assets/tts/mms-tts-tam/tokenizer.json | 115 +++ assets/tts/mms-tts-tam/tokenizer_config.json | 31 + assets/tts/mms-tts-tam/vocab.json | 60 ++ package-lock.json | 955 ++++++++++++++++++ package.json | 5 +- scripts/build-tamil-tokenizer.mjs | 66 ++ scripts/export-tamil-tts.py | 99 ++ scripts/t-auto.mjs | 21 + scripts/t-detect.mjs | 62 ++ scripts/t-speech.mjs | 44 + scripts/t-tamil-tts.mjs | 55 + scripts/t-voice.mjs | 72 ++ src/gateway/voice.js | 309 +++--- src/server.js | 7 + src/speech/index.js | 169 ++++ src/speech/models.js | 118 +++ src/speech/vad.js | 152 +++ voice-service/.env.example | 22 - voice-service/README.md | 105 -- voice-service/app/config.py | 82 -- voice-service/app/server.py | 233 ----- voice-service/app/stt.py | 200 ---- voice-service/app/tts.py | 155 --- voice-service/app/vad.py | 127 --- voice-service/bench.py | 95 -- voice-service/probe_access.py | 47 - voice-service/probe_models.py | 82 -- voice-service/requirements.txt | 11 - 34 files changed, 2271 insertions(+), 1343 deletions(-) create mode 100644 assets/tts/mms-tts-tam/added_tokens.json create mode 100644 assets/tts/mms-tts-tam/config.json create mode 100644 assets/tts/mms-tts-tam/quantize_config.json create mode 100644 assets/tts/mms-tts-tam/special_tokens_map.json create mode 100644 assets/tts/mms-tts-tam/tokenizer.json create mode 100644 assets/tts/mms-tts-tam/tokenizer_config.json create mode 100644 assets/tts/mms-tts-tam/vocab.json create mode 100644 scripts/build-tamil-tokenizer.mjs create mode 100644 scripts/export-tamil-tts.py create mode 100644 scripts/t-auto.mjs create mode 100644 scripts/t-detect.mjs create mode 100644 scripts/t-speech.mjs create mode 100644 scripts/t-tamil-tts.mjs create mode 100644 scripts/t-voice.mjs create mode 100644 src/speech/index.js create mode 100644 src/speech/models.js create mode 100644 src/speech/vad.js delete mode 100644 voice-service/.env.example delete mode 100644 voice-service/README.md delete mode 100644 voice-service/app/config.py delete mode 100644 voice-service/app/server.py delete mode 100644 voice-service/app/stt.py delete mode 100644 voice-service/app/tts.py delete mode 100644 voice-service/app/vad.py delete mode 100644 voice-service/bench.py delete mode 100644 voice-service/probe_access.py delete mode 100644 voice-service/probe_models.py delete mode 100644 voice-service/requirements.txt diff --git a/.env.example b/.env.example index 48d0ee2..74333d4 100644 --- a/.env.example +++ b/.env.example @@ -55,3 +55,18 @@ RATE_LIMIT_MAX=40 # --- Artifacts --- ARTIFACT_DIR=./storage/artifacts ARTIFACT_TTL_HOURS=72 + +# --- Voice (speech-to-speech, CPU, in-process) --- +# Models are ONNX via Transformers.js — no GPU, no Python, no second service. +# STT onnx-community/whisper-base Tamil + English + language detection +# TTS assets/tts/mms-tts-tam exported locally; no public ONNX exists +# TTS Xenova/mms-tts-eng from the Hub +SPEECH_WARMUP=false +STT_MODEL=onnx-community/whisper-base +# Below this confidence, the user's preferred language beats the detector. +DETECT_CONFIDENCE=0.6 + +# Endpointing +VAD_SILENCE_MS=700 +VAD_MIN_SPEECH_MS=250 +VAD_PREFIX_MS=300 diff --git a/.gitignore b/.gitignore index 68e0ff1..3027d65 100644 --- a/.gitignore +++ b/.gitignore @@ -10,7 +10,6 @@ .env.* !.env.example !**/*.env.example -voice-service/.env *.pem *.key *.p12 @@ -28,7 +27,6 @@ pnpm-debug.log* *.tsbuildinfo # ── Python (voice service) ────────────────────────────────────────────────── -voice-service/.venv/ .venv/ /venv/ /env/ @@ -50,11 +48,13 @@ __pycache__/ # redistribute models the licence does not allow us to redistribute. # # Leading slashes matter: an unanchored `models/` also matches -# `src/data/models/` — the CRM read-models — which silently kept them out of +# `src/data/models/ +.transformers-cache/` — the CRM read-models — which silently kept them out of # the repo and made the container crash with ERR_MODULE_NOT_FOUND. /.cache/ /huggingface/ /models/ +.transformers-cache/ *.onnx *.safetensors *.ckpt @@ -75,7 +75,6 @@ storage/artifacts/* *.wav *.mp3 *.flac -!voice-service/app/assets/*.wav # ── Editors / OS ──────────────────────────────────────────────────────────── .vscode/* diff --git a/assets/tts/mms-tts-tam/added_tokens.json b/assets/tts/mms-tts-tam/added_tokens.json new file mode 100644 index 0000000..76d0ecc --- /dev/null +++ b/assets/tts/mms-tts-tam/added_tokens.json @@ -0,0 +1,3 @@ +{ + "": 58 +} diff --git a/assets/tts/mms-tts-tam/config.json b/assets/tts/mms-tts-tam/config.json new file mode 100644 index 0000000..5bd0d90 --- /dev/null +++ b/assets/tts/mms-tts-tam/config.json @@ -0,0 +1,82 @@ +{ + "activation_dropout": 0.1, + "architectures": [ + "VitsModel" + ], + "attention_dropout": 0.1, + "depth_separable_channels": 2, + "depth_separable_num_layers": 3, + "dtype": "float32", + "duration_predictor_dropout": 0.5, + "duration_predictor_filter_channels": 256, + "duration_predictor_flow_bins": 10, + "duration_predictor_kernel_size": 3, + "duration_predictor_num_flows": 4, + "duration_predictor_tail_bound": 5.0, + "ffn_dim": 768, + "ffn_kernel_size": 3, + "flow_size": 192, + "hidden_act": "relu", + "hidden_dropout": 0.1, + "hidden_size": 192, + "initializer_range": 0.02, + "layer_norm_eps": 1e-05, + "layerdrop": 0.1, + "leaky_relu_slope": 0.1, + "model_type": "vits", + "noise_scale": 0.667, + "noise_scale_duration": 0.8, + "num_attention_heads": 2, + "num_hidden_layers": 6, + "num_speakers": 1, + "posterior_encoder_num_wavenet_layers": 16, + "prior_encoder_num_flows": 4, + "prior_encoder_num_wavenet_layers": 4, + "resblock_dilation_sizes": [ + [ + 1, + 3, + 5 + ], + [ + 1, + 3, + 5 + ], + [ + 1, + 3, + 5 + ] + ], + "resblock_kernel_sizes": [ + 3, + 7, + 11 + ], + "sampling_rate": 16000, + "speaker_embedding_size": 0, + "speaking_rate": 1.0, + "spectrogram_bins": 513, + "transformers_version": "4.57.3", + "upsample_initial_channel": 512, + "upsample_kernel_sizes": [ + 16, + 16, + 4, + 4 + ], + "upsample_rates": [ + 8, + 8, + 2, + 2 + ], + "use_bias": true, + "use_stochastic_duration_prediction": true, + "vocab_size": 58, + "wavenet_dilation_rate": 1, + "wavenet_dropout": 0.0, + "wavenet_kernel_size": 5, + "window_size": 4 +} diff --git a/assets/tts/mms-tts-tam/quantize_config.json b/assets/tts/mms-tts-tam/quantize_config.json new file mode 100644 index 0000000..c3b3fa4 --- /dev/null +++ b/assets/tts/mms-tts-tam/quantize_config.json @@ -0,0 +1,4 @@ +{ + "per_channel": false, + "reduce_range": false +} \ No newline at end of file diff --git a/assets/tts/mms-tts-tam/special_tokens_map.json b/assets/tts/mms-tts-tam/special_tokens_map.json new file mode 100644 index 0000000..adc92cd --- /dev/null +++ b/assets/tts/mms-tts-tam/special_tokens_map.json @@ -0,0 +1,4 @@ +{ + "pad_token": "3", + "unk_token": "" +} diff --git a/assets/tts/mms-tts-tam/tokenizer.json b/assets/tts/mms-tts-tam/tokenizer.json new file mode 100644 index 0000000..c256cc2 --- /dev/null +++ b/assets/tts/mms-tts-tam/tokenizer.json @@ -0,0 +1,115 @@ +{ + "version": "1.0", + "truncation": null, + "padding": null, + "added_tokens": [ + { + "id": 58, + "content": "", + "single_word": false, + "lstrip": false, + "rstrip": false, + "normalized": false, + "special": true + } + ], + "normalizer": { + "type": "Sequence", + "normalizers": [ + { + "type": "Lowercase" + }, + { + "type": "Replace", + "pattern": { + "Regex": "[^012345679 '_aஅஆஇஈஉஊஎஏஐஒஓகஙசஜஞடணதநனபமயரறலளழவஷஸஹாிீுூெேைொோௌ்]" + }, + "content": "" + }, + { + "type": "Strip", + "strip_left": true, + "strip_right": true + }, + { + "type": "Replace", + "pattern": { + "Regex": "(?=.)|(?", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false, + "special": true + } + }, + "clean_up_tokenization_spaces": true, + "extra_special_tokens": {}, + "is_uroman": false, + "language": "tam", + "model_max_length": 1000000000000000019884624838656, + "normalize": true, + "pad_token": "3", + "phonemize": false, + "tokenizer_class": "VitsTokenizer", + "unk_token": "" +} diff --git a/assets/tts/mms-tts-tam/vocab.json b/assets/tts/mms-tts-tam/vocab.json new file mode 100644 index 0000000..086e82e --- /dev/null +++ b/assets/tts/mms-tts-tam/vocab.json @@ -0,0 +1,60 @@ +{ + " ": 7, + "'": 13, + "0": 47, + "1": 44, + "2": 23, + "3": 0, + "4": 54, + "5": 57, + "6": 36, + "7": 14, + "9": 31, + "_": 4, + "a": 15, + "அ": 1, + "ஆ": 45, + "இ": 38, + "ஈ": 2, + "உ": 3, + "ஊ": 11, + "எ": 37, + "ஏ": 16, + "ஐ": 52, + "ஒ": 27, + "ஓ": 49, + "க": 6, + "ங": 50, + "ச": 30, + "ஜ": 53, + "ஞ": 29, + "ட": 22, + "ண": 48, + "த": 41, + "ந": 5, + "ன": 35, + "ப": 46, + "ம": 26, + "ய": 39, + "ர": 25, + "ற": 28, + "ல": 21, + "ள": 43, + "ழ": 24, + "வ": 17, + "ஷ": 55, + "ஸ": 33, + "ஹ": 19, + "ா": 9, + "ி": 32, + "ீ": 12, + "ு": 51, + "ூ": 20, + "ெ": 10, + "ே": 8, + "ை": 34, + "ொ": 56, + "ோ": 42, + "ௌ": 40, + "்": 18 +} diff --git a/package-lock.json b/package-lock.json index c54d600..4a7c47b 100644 --- a/package-lock.json +++ b/package-lock.json @@ -9,6 +9,7 @@ "version": "0.1.0", "license": "ISC", "dependencies": { + "@huggingface/transformers": "^4.2.0", "@langchain/core": "^1.1.18", "@langchain/langgraph": "^1.1.0", "@langchain/openai": "^1.5.10", @@ -23,6 +24,7 @@ "ioredis": "^5.10.1", "jsonwebtoken": "^9.0.2", "mongoose": "^9.2.3", + "onnxruntime-node": "^1.24.3", "pdfkit": "^0.17.2", "pptxgenjs": "^4.0.1", "uuid": "^13.0.0", @@ -57,6 +59,16 @@ "kuler": "^2.0.0" } }, + "node_modules/@emnapi/runtime": { + "version": "1.11.3", + "resolved": "https://registry.npmjs.org/@emnapi/runtime/-/runtime-1.11.3.tgz", + "integrity": "sha512-Xz4Tpyki7XyrpbUK1jR1AhdAdaXyhhY4lZ3neLodmhpuWfy2PAQN5B46sAiU4liOXGLkHypn/qU+jvfWSCYYLA==", + "license": "MIT", + "optional": true, + "dependencies": { + "tslib": "^2.4.0" + } + }, "node_modules/@fast-csv/format": { "version": "4.3.5", "resolved": "https://registry.npmjs.org/@fast-csv/format/-/format-4.3.5.tgz", @@ -98,6 +110,547 @@ "integrity": "sha512-fAtCfv4jJg+ExtXhvCkCqUKZ+4ok/JQk01qDKhL5BDDoS3AxKXhV5/MAVUZyQnSEd2GT92fkgZl0pz0Q0AzcIQ==", "license": "MIT" }, + "node_modules/@huggingface/jinja": { + "version": "0.5.9", + "resolved": "https://registry.npmjs.org/@huggingface/jinja/-/jinja-0.5.9.tgz", + "integrity": "sha512-uWTG+l3VJRsl7EXxYizuL3P+cCPoc3cRqbWWRcQN0FhejRfbdq0RNhCmbY/YDtnTcz9icdLYuLDjsnz4d8JMuw==", + "license": "MIT", + "engines": { + "node": ">=18" + } + }, + "node_modules/@huggingface/tokenizers": { + "version": "0.1.3", + "resolved": "https://registry.npmjs.org/@huggingface/tokenizers/-/tokenizers-0.1.3.tgz", + "integrity": "sha512-8rF/RRT10u+kn7YuUbUg0OF30K8rjTc78aHpxT+qJ1uWSqxT1MHi8+9ltwYfkFYJzT/oS+qw3JVfHtNMGAdqyA==", + "license": "Apache-2.0" + }, + "node_modules/@huggingface/transformers": { + "version": "4.2.0", + "resolved": "https://registry.npmjs.org/@huggingface/transformers/-/transformers-4.2.0.tgz", + "integrity": "sha512-8BRCoBMH0XsWaEIamuR0LrJGAfftgHAfb2Vrffy0VKlSAE/MnUJ5/h/zTfEP3fDIft+nk7TqB8xXEyABGitBjQ==", + "license": "Apache-2.0", + "dependencies": { + "@huggingface/jinja": "^0.5.6", + "@huggingface/tokenizers": "^0.1.3", + "onnxruntime-node": "1.24.3", + "onnxruntime-web": "1.26.0-dev.20260416-b7804b056c", + "sharp": "^0.34.5" + } + }, + "node_modules/@img/colour": { + "version": "1.1.0", + "resolved": "https://registry.npmjs.org/@img/colour/-/colour-1.1.0.tgz", + "integrity": "sha512-Td76q7j57o/tLVdgS746cYARfSyxk8iEfRxewL9h4OMzYhbW4TAcppl0mT4eyqXddh6L/jwoM75mo7ixa/pCeQ==", + "license": "MIT", + "engines": { + "node": ">=18" + } + }, + "node_modules/@img/sharp-darwin-arm64": { + "version": "0.34.5", + "resolved": "https://registry.npmjs.org/@img/sharp-darwin-arm64/-/sharp-darwin-arm64-0.34.5.tgz", + "integrity": "sha512-imtQ3WMJXbMY4fxb/Ndp6HBTNVtWCUI0WdobyheGf5+ad6xX8VIDO8u2xE4qc/fr08CKG/7dDseFtn6M6g/r3w==", + "cpu": [ + "arm64" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + }, + "optionalDependencies": { + "@img/sharp-libvips-darwin-arm64": "1.2.4" + } + }, + "node_modules/@img/sharp-darwin-x64": { + "version": "0.34.5", + "resolved": "https://registry.npmjs.org/@img/sharp-darwin-x64/-/sharp-darwin-x64-0.34.5.tgz", + "integrity": "sha512-YNEFAF/4KQ/PeW0N+r+aVVsoIY0/qxxikF2SWdp+NRkmMB7y9LBZAVqQ4yhGCm/H3H270OSykqmQMKLBhBJDEw==", + "cpu": [ + "x64" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + }, + "optionalDependencies": { + "@img/sharp-libvips-darwin-x64": "1.2.4" + } + }, + "node_modules/@img/sharp-libvips-darwin-arm64": { + "version": "1.2.4", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-darwin-arm64/-/sharp-libvips-darwin-arm64-1.2.4.tgz", + "integrity": "sha512-zqjjo7RatFfFoP0MkQ51jfuFZBnVE2pRiaydKJ1G/rHZvnsrHAOcQALIi9sA5co5xenQdTugCvtb1cuf78Vf4g==", + "cpu": [ + "arm64" + ], + "license": "LGPL-3.0-or-later", + "optional": true, + "os": [ + "darwin" + ], + "funding": { + "url": "https://opencollective.com/libvips" + } + }, + "node_modules/@img/sharp-libvips-darwin-x64": { + "version": "1.2.4", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-darwin-x64/-/sharp-libvips-darwin-x64-1.2.4.tgz", + "integrity": "sha512-1IOd5xfVhlGwX+zXv2N93k0yMONvUlANylbJw1eTah8K/Jtpi15KC+WSiaX/nBmbm2HxRM1gZ0nSdjSsrZbGKg==", + "cpu": [ + "x64" + ], + "license": "LGPL-3.0-or-later", + "optional": true, + "os": [ + "darwin" + ], + "funding": { + "url": "https://opencollective.com/libvips" + } + }, + "node_modules/@img/sharp-libvips-linux-arm": { + "version": "1.2.4", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linux-arm/-/sharp-libvips-linux-arm-1.2.4.tgz", + "integrity": "sha512-bFI7xcKFELdiNCVov8e44Ia4u2byA+l3XtsAj+Q8tfCwO6BQ8iDojYdvoPMqsKDkuoOo+X6HZA0s0q11ANMQ8A==", + "cpu": [ + "arm" + ], + "libc": [ + "glibc" + ], + "license": "LGPL-3.0-or-later", + "optional": true, + "os": [ + "linux" + ], + "funding": { + "url": "https://opencollective.com/libvips" + } + }, + "node_modules/@img/sharp-libvips-linux-arm64": { + "version": "1.2.4", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linux-arm64/-/sharp-libvips-linux-arm64-1.2.4.tgz", + "integrity": "sha512-excjX8DfsIcJ10x1Kzr4RcWe1edC9PquDRRPx3YVCvQv+U5p7Yin2s32ftzikXojb1PIFc/9Mt28/y+iRklkrw==", + "cpu": [ + "arm64" + ], + "libc": [ + "glibc" + ], + "license": "LGPL-3.0-or-later", + "optional": true, + "os": [ + "linux" + ], + "funding": { + "url": "https://opencollective.com/libvips" + } + }, + "node_modules/@img/sharp-libvips-linux-ppc64": { + "version": "1.2.4", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linux-ppc64/-/sharp-libvips-linux-ppc64-1.2.4.tgz", + "integrity": "sha512-FMuvGijLDYG6lW+b/UvyilUWu5Ayu+3r2d1S8notiGCIyYU/76eig1UfMmkZ7vwgOrzKzlQbFSuQfgm7GYUPpA==", + "cpu": [ + "ppc64" + ], + "libc": [ + "glibc" + ], + "license": "LGPL-3.0-or-later", + "optional": true, + "os": [ + "linux" + ], + "funding": { + "url": "https://opencollective.com/libvips" + } + }, + "node_modules/@img/sharp-libvips-linux-riscv64": { + "version": "1.2.4", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linux-riscv64/-/sharp-libvips-linux-riscv64-1.2.4.tgz", + "integrity": "sha512-oVDbcR4zUC0ce82teubSm+x6ETixtKZBh/qbREIOcI3cULzDyb18Sr/Wcyx7NRQeQzOiHTNbZFF1UwPS2scyGA==", + "cpu": [ + "riscv64" + ], + "libc": [ + "glibc" + ], + "license": "LGPL-3.0-or-later", + "optional": true, + "os": [ + "linux" + ], + "funding": { + "url": "https://opencollective.com/libvips" + } + }, + "node_modules/@img/sharp-libvips-linux-s390x": { + "version": "1.2.4", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linux-s390x/-/sharp-libvips-linux-s390x-1.2.4.tgz", + "integrity": "sha512-qmp9VrzgPgMoGZyPvrQHqk02uyjA0/QrTO26Tqk6l4ZV0MPWIW6LTkqOIov+J1yEu7MbFQaDpwdwJKhbJvuRxQ==", + "cpu": [ + "s390x" + ], + "libc": [ + "glibc" + ], + "license": "LGPL-3.0-or-later", + "optional": true, + "os": [ + "linux" + ], + "funding": { + "url": "https://opencollective.com/libvips" + } + }, + "node_modules/@img/sharp-libvips-linux-x64": { + "version": "1.2.4", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linux-x64/-/sharp-libvips-linux-x64-1.2.4.tgz", + "integrity": "sha512-tJxiiLsmHc9Ax1bz3oaOYBURTXGIRDODBqhveVHonrHJ9/+k89qbLl0bcJns+e4t4rvaNBxaEZsFtSfAdquPrw==", + "cpu": [ + "x64" + ], + "libc": [ + "glibc" + ], + "license": "LGPL-3.0-or-later", + "optional": true, + "os": [ + "linux" + ], + "funding": { + "url": "https://opencollective.com/libvips" + } + }, + "node_modules/@img/sharp-libvips-linuxmusl-arm64": { + "version": "1.2.4", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linuxmusl-arm64/-/sharp-libvips-linuxmusl-arm64-1.2.4.tgz", + "integrity": "sha512-FVQHuwx1IIuNow9QAbYUzJ+En8KcVm9Lk5+uGUQJHaZmMECZmOlix9HnH7n1TRkXMS0pGxIJokIVB9SuqZGGXw==", + "cpu": [ + "arm64" + ], + "libc": [ + "musl" + ], + "license": "LGPL-3.0-or-later", + "optional": true, + "os": [ + "linux" + ], + "funding": { + "url": "https://opencollective.com/libvips" + } + }, + "node_modules/@img/sharp-libvips-linuxmusl-x64": { + "version": "1.2.4", + "resolved": "https://registry.npmjs.org/@img/sharp-libvips-linuxmusl-x64/-/sharp-libvips-linuxmusl-x64-1.2.4.tgz", + "integrity": "sha512-+LpyBk7L44ZIXwz/VYfglaX/okxezESc6UxDSoyo2Ks6Jxc4Y7sGjpgU9s4PMgqgjj1gZCylTieNamqA1MF7Dg==", + "cpu": [ + "x64" + ], + "libc": [ + "musl" + ], + "license": "LGPL-3.0-or-later", + "optional": true, + "os": [ + "linux" + ], + "funding": { + "url": "https://opencollective.com/libvips" + } + }, + "node_modules/@img/sharp-linux-arm": { + "version": "0.34.5", + "resolved": "https://registry.npmjs.org/@img/sharp-linux-arm/-/sharp-linux-arm-0.34.5.tgz", + "integrity": "sha512-9dLqsvwtg1uuXBGZKsxem9595+ujv0sJ6Vi8wcTANSFpwV/GONat5eCkzQo/1O6zRIkh0m/8+5BjrRr7jDUSZw==", + "cpu": [ + "arm" + ], + "libc": [ + "glibc" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + }, + "optionalDependencies": { + "@img/sharp-libvips-linux-arm": "1.2.4" + } + }, + "node_modules/@img/sharp-linux-arm64": { + "version": "0.34.5", + "resolved": "https://registry.npmjs.org/@img/sharp-linux-arm64/-/sharp-linux-arm64-0.34.5.tgz", + "integrity": "sha512-bKQzaJRY/bkPOXyKx5EVup7qkaojECG6NLYswgktOZjaXecSAeCWiZwwiFf3/Y+O1HrauiE3FVsGxFg8c24rZg==", + "cpu": [ + "arm64" + ], + "libc": [ + "glibc" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + }, + "optionalDependencies": { + "@img/sharp-libvips-linux-arm64": "1.2.4" + } + }, + "node_modules/@img/sharp-linux-ppc64": { + "version": "0.34.5", + "resolved": "https://registry.npmjs.org/@img/sharp-linux-ppc64/-/sharp-linux-ppc64-0.34.5.tgz", + "integrity": "sha512-7zznwNaqW6YtsfrGGDA6BRkISKAAE1Jo0QdpNYXNMHu2+0dTrPflTLNkpc8l7MUP5M16ZJcUvysVWWrMefZquA==", + "cpu": [ + "ppc64" + ], + "libc": [ + "glibc" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + }, + "optionalDependencies": { + "@img/sharp-libvips-linux-ppc64": "1.2.4" + } + }, + "node_modules/@img/sharp-linux-riscv64": { + "version": "0.34.5", + "resolved": "https://registry.npmjs.org/@img/sharp-linux-riscv64/-/sharp-linux-riscv64-0.34.5.tgz", + "integrity": "sha512-51gJuLPTKa7piYPaVs8GmByo7/U7/7TZOq+cnXJIHZKavIRHAP77e3N2HEl3dgiqdD/w0yUfiJnII77PuDDFdw==", + "cpu": [ + "riscv64" + ], + "libc": [ + "glibc" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + }, + "optionalDependencies": { + "@img/sharp-libvips-linux-riscv64": "1.2.4" + } + }, + "node_modules/@img/sharp-linux-s390x": { + "version": "0.34.5", + "resolved": "https://registry.npmjs.org/@img/sharp-linux-s390x/-/sharp-linux-s390x-0.34.5.tgz", + "integrity": "sha512-nQtCk0PdKfho3eC5MrbQoigJ2gd1CgddUMkabUj+rBevs8tZ2cULOx46E7oyX+04WGfABgIwmMC0VqieTiR4jg==", + "cpu": [ + "s390x" + ], + "libc": [ + "glibc" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + }, + "optionalDependencies": { + "@img/sharp-libvips-linux-s390x": "1.2.4" + } + }, + "node_modules/@img/sharp-linux-x64": { + "version": "0.34.5", + "resolved": "https://registry.npmjs.org/@img/sharp-linux-x64/-/sharp-linux-x64-0.34.5.tgz", + "integrity": "sha512-MEzd8HPKxVxVenwAa+JRPwEC7QFjoPWuS5NZnBt6B3pu7EG2Ge0id1oLHZpPJdn3OQK+BQDiw9zStiHBTJQQQQ==", + "cpu": [ + "x64" + ], + "libc": [ + "glibc" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + }, + "optionalDependencies": { + "@img/sharp-libvips-linux-x64": "1.2.4" + } + }, + "node_modules/@img/sharp-linuxmusl-arm64": { + "version": "0.34.5", + "resolved": "https://registry.npmjs.org/@img/sharp-linuxmusl-arm64/-/sharp-linuxmusl-arm64-0.34.5.tgz", + "integrity": "sha512-fprJR6GtRsMt6Kyfq44IsChVZeGN97gTD331weR1ex1c1rypDEABN6Tm2xa1wE6lYb5DdEnk03NZPqA7Id21yg==", + "cpu": [ + "arm64" + ], + "libc": [ + "musl" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + }, + "optionalDependencies": { + "@img/sharp-libvips-linuxmusl-arm64": "1.2.4" + } + }, + "node_modules/@img/sharp-linuxmusl-x64": { + "version": "0.34.5", + "resolved": "https://registry.npmjs.org/@img/sharp-linuxmusl-x64/-/sharp-linuxmusl-x64-0.34.5.tgz", + "integrity": "sha512-Jg8wNT1MUzIvhBFxViqrEhWDGzqymo3sV7z7ZsaWbZNDLXRJZoRGrjulp60YYtV4wfY8VIKcWidjojlLcWrd8Q==", + "cpu": [ + "x64" + ], + "libc": [ + "musl" + ], + "license": "Apache-2.0", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + }, + "optionalDependencies": { + "@img/sharp-libvips-linuxmusl-x64": "1.2.4" + } + }, + "node_modules/@img/sharp-wasm32": { + "version": "0.34.5", + "resolved": "https://registry.npmjs.org/@img/sharp-wasm32/-/sharp-wasm32-0.34.5.tgz", + "integrity": "sha512-OdWTEiVkY2PHwqkbBI8frFxQQFekHaSSkUIJkwzclWZe64O1X4UlUjqqqLaPbUpMOQk6FBu/HtlGXNblIs0huw==", + "cpu": [ + "wasm32" + ], + "license": "Apache-2.0 AND LGPL-3.0-or-later AND MIT", + "optional": true, + "dependencies": { + "@emnapi/runtime": "^1.7.0" + }, + "engines": { + "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + } + }, + "node_modules/@img/sharp-win32-arm64": { + "version": "0.34.5", + "resolved": "https://registry.npmjs.org/@img/sharp-win32-arm64/-/sharp-win32-arm64-0.34.5.tgz", + "integrity": "sha512-WQ3AgWCWYSb2yt+IG8mnC6Jdk9Whs7O0gxphblsLvdhSpSTtmu69ZG1Gkb6NuvxsNACwiPV6cNSZNzt0KPsw7g==", + "cpu": [ + "arm64" + ], + "license": "Apache-2.0 AND LGPL-3.0-or-later", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + } + }, + "node_modules/@img/sharp-win32-ia32": { + "version": "0.34.5", + "resolved": "https://registry.npmjs.org/@img/sharp-win32-ia32/-/sharp-win32-ia32-0.34.5.tgz", + "integrity": "sha512-FV9m/7NmeCmSHDD5j4+4pNI8Cp3aW+JvLoXcTUo0IqyjSfAZJ8dIUmijx1qaJsIiU+Hosw6xM5KijAWRJCSgNg==", + "cpu": [ + "ia32" + ], + "license": "Apache-2.0 AND LGPL-3.0-or-later", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + } + }, + "node_modules/@img/sharp-win32-x64": { + "version": "0.34.5", + "resolved": "https://registry.npmjs.org/@img/sharp-win32-x64/-/sharp-win32-x64-0.34.5.tgz", + "integrity": "sha512-+29YMsqY2/9eFEiW93eqWnuLcWcufowXewwSNIT6UwZdUUCrM3oFjMWH/Z6/TMmb4hlFenmfAVbpWeup2jryCw==", + "cpu": [ + "x64" + ], + "license": "Apache-2.0 AND LGPL-3.0-or-later", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + } + }, "node_modules/@ioredis/commands": { "version": "1.10.0", "resolved": "https://registry.npmjs.org/@ioredis/commands/-/commands-1.10.0.tgz", @@ -252,6 +805,63 @@ "sparse-bitfield": "^3.0.3" } }, + "node_modules/@protobufjs/aspromise": { + "version": "1.1.2", + "resolved": "https://registry.npmjs.org/@protobufjs/aspromise/-/aspromise-1.1.2.tgz", + "integrity": "sha512-j+gKExEuLmKwvz3OgROXtrJ2UG2x8Ch2YZUxahh+s1F2HZ+wAceUNLkvy6zKCPVRkU++ZWQrdxsUeQXmcg4uoQ==", + "license": "BSD-3-Clause" + }, + "node_modules/@protobufjs/base64": { + "version": "1.1.2", + "resolved": "https://registry.npmjs.org/@protobufjs/base64/-/base64-1.1.2.tgz", + "integrity": "sha512-AZkcAA5vnN/v4PDqKyMR5lx7hZttPDgClv83E//FMNhR2TMcLUhfRUBHCmSl0oi9zMgDDqRUJkSxO3wm85+XLg==", + "license": "BSD-3-Clause" + }, + "node_modules/@protobufjs/codegen": { + "version": "2.0.5", + "resolved": "https://registry.npmjs.org/@protobufjs/codegen/-/codegen-2.0.5.tgz", + "integrity": "sha512-zgXFLzW3Ap33e6d0Wlj4MGIm6Ce8O89n/apUaGNB/jx+hw+ruWEp7EwGUshdLKVRCxZW12fp9r40E1mQrf/34g==", + "license": "BSD-3-Clause" + }, + "node_modules/@protobufjs/eventemitter": { + "version": "1.1.1", + "resolved": "https://registry.npmjs.org/@protobufjs/eventemitter/-/eventemitter-1.1.1.tgz", + "integrity": "sha512-vW1GmwMZNnL+gMRaovlh9yZX74kc+TTU3FObkkurpMaRtBfLP3ldjS9KQWlwZgraRE0+dheEEoAxdzcJQ8eXZg==", + "license": "BSD-3-Clause" + }, + "node_modules/@protobufjs/fetch": { + "version": "1.1.1", + "resolved": "https://registry.npmjs.org/@protobufjs/fetch/-/fetch-1.1.1.tgz", + "integrity": "sha512-GpptLrs57adMSuHi3VNj0mAF8dwh36LMaYF6XyJ6JMWlVsc+t42tm1HSEDmOs3A8fC9yyeisgLhsTVQokOZ0zw==", + "license": "BSD-3-Clause", + "dependencies": { + "@protobufjs/aspromise": "^1.1.1" + } + }, + "node_modules/@protobufjs/float": { + "version": "1.0.2", + "resolved": "https://registry.npmjs.org/@protobufjs/float/-/float-1.0.2.tgz", + "integrity": "sha512-Ddb+kVXlXst9d+R9PfTIxh1EdNkgoRe5tOX6t01f1lYWOvJnSPDBlG241QLzcyPdoNTsblLUdujGSE4RzrTZGQ==", + "license": "BSD-3-Clause" + }, + "node_modules/@protobufjs/path": { + "version": "1.1.2", + "resolved": "https://registry.npmjs.org/@protobufjs/path/-/path-1.1.2.tgz", + "integrity": "sha512-6JOcJ5Tm08dOHAbdR3GrvP+yUUfkjG5ePsHYczMFLq3ZmMkAD98cDgcT2iA1lJ9NVwFd4tH/iSSoe44YWkltEA==", + "license": "BSD-3-Clause" + }, + "node_modules/@protobufjs/pool": { + "version": "1.1.0", + "resolved": "https://registry.npmjs.org/@protobufjs/pool/-/pool-1.1.0.tgz", + "integrity": "sha512-0kELaGSIDBKvcgS4zkjz1PeddatrjYcmMWOlAuAPwAeccUrPHdUqo/J6LiymHHEiJT5NrF1UVwxY14f+fy4WQw==", + "license": "BSD-3-Clause" + }, + "node_modules/@protobufjs/utf8": { + "version": "1.1.2", + "resolved": "https://registry.npmjs.org/@protobufjs/utf8/-/utf8-1.1.2.tgz", + "integrity": "sha512-b1UQwcEZ4yCnMCD8DAL1VlbvBJE9/IX4FTIp7BG1xYpf29SLazLSrqUkj4w7Y5y7cCVP6E5tcqqcI0xemPkHug==", + "license": "BSD-3-Clause" + }, "node_modules/@resvg/resvg-js": { "version": "2.6.2", "resolved": "https://registry.npmjs.org/@resvg/resvg-js/-/resvg-js-2.6.2.tgz", @@ -553,6 +1163,15 @@ "node": ">= 0.6" } }, + "node_modules/adm-zip": { + "version": "0.5.18", + "resolved": "https://registry.npmjs.org/adm-zip/-/adm-zip-0.5.18.tgz", + "integrity": "sha512-ufJnssQGbxzLNS1Ho9bCtX4rQKCCvoVuDLHoJyc3F9dOGDB4BkWs2Ci0kv53lqocAEQ/Cbi+I2XCsNYGqVYqng==", + "license": "MIT", + "engines": { + "node": ">=12.0" + } + }, "node_modules/archiver": { "version": "5.3.2", "resolved": "https://registry.npmjs.org/archiver/-/archiver-5.3.2.tgz", @@ -730,6 +1349,13 @@ "url": "https://opencollective.com/express" } }, + "node_modules/boolean": { + "version": "3.2.0", + "resolved": "https://registry.npmjs.org/boolean/-/boolean-3.2.0.tgz", + "integrity": "sha512-d0II/GO9uf9lfUHH2BQsjxzRJZBdsjgsBiW4BvhWk/3qoKwQFjIDVN19PfX8F2D/r9PCMTtLWjYVCFrpeYUzsw==", + "deprecated": "Package no longer supported. Contact Support at https://www.npmjs.com/support for more info.", + "license": "MIT" + }, "node_modules/brace-expansion": { "version": "1.1.18", "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.18.tgz", @@ -1076,6 +1702,40 @@ } } }, + "node_modules/define-data-property": { + "version": "1.1.4", + "resolved": "https://registry.npmjs.org/define-data-property/-/define-data-property-1.1.4.tgz", + "integrity": "sha512-rBMvIzlpA8v6E+SJZoo++HAYqsLrkg7MSfIinMPFhmkorw7X+dOXVJQs+QT69zGkzMyfDnIMN2Wid1+NbL3T+A==", + "license": "MIT", + "dependencies": { + "es-define-property": "^1.0.0", + "es-errors": "^1.3.0", + "gopd": "^1.0.1" + }, + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "https://github.com/sponsors/ljharb" + } + }, + "node_modules/define-properties": { + "version": "1.2.1", + "resolved": "https://registry.npmjs.org/define-properties/-/define-properties-1.2.1.tgz", + "integrity": "sha512-8QmQKqEASLd5nx0U1B1okLElbUuuttJ/AnYmRXbbbGDWh6uS208EjD4Xqq/I9wK7u0v6O08XhTWnt5XtEbR6Dg==", + "license": "MIT", + "dependencies": { + "define-data-property": "^1.0.1", + "has-property-descriptors": "^1.0.0", + "object-keys": "^1.1.1" + }, + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "https://github.com/sponsors/ljharb" + } + }, "node_modules/denque": { "version": "2.1.0", "resolved": "https://registry.npmjs.org/denque/-/denque-2.1.0.tgz", @@ -1094,6 +1754,21 @@ "node": ">= 0.8" } }, + "node_modules/detect-libc": { + "version": "2.1.2", + "resolved": "https://registry.npmjs.org/detect-libc/-/detect-libc-2.1.2.tgz", + "integrity": "sha512-Btj2BOOO83o3WyH59e8MgXsxEQVcarkUOpEYrubB0urwnN10yQ364rsiByU11nZlqWYZm05i/of7io4mzihBtQ==", + "license": "Apache-2.0", + "engines": { + "node": ">=8" + } + }, + "node_modules/detect-node": { + "version": "2.1.0", + "resolved": "https://registry.npmjs.org/detect-node/-/detect-node-2.1.0.tgz", + "integrity": "sha512-T0NIuQpnTvFDATNuHN5roPwSBG83rFsuO+MXXH9/3N1eFbn4wcPjttvjMLEPWJ0RGUYgQE7cGgS3tNxbqCGM7g==", + "license": "MIT" + }, "node_modules/dfa": { "version": "1.2.0", "resolved": "https://registry.npmjs.org/dfa/-/dfa-1.2.0.tgz", @@ -1251,12 +1926,30 @@ "node": ">= 0.4" } }, + "node_modules/es6-error": { + "version": "4.1.1", + "resolved": "https://registry.npmjs.org/es6-error/-/es6-error-4.1.1.tgz", + "integrity": "sha512-Um/+FxMr9CISWh0bi5Zv0iOD+4cFh5qLeks1qhAopKVAJw3drgKbKySikp7wGhDL0HPeaja0P5ULZrxLkniUVg==", + "license": "MIT" + }, "node_modules/escape-html": { "version": "1.0.3", "resolved": "https://registry.npmjs.org/escape-html/-/escape-html-1.0.3.tgz", "integrity": "sha512-NiSupZ4OeuGwr68lGIeym/ksIZMJodUGOSCZ/FSnTxcrekbvqrgdUxlJOMpijaKZVjAJrWrGs/6Jy8OMuyj9ow==", "license": "MIT" }, + "node_modules/escape-string-regexp": { + "version": "4.0.0", + "resolved": "https://registry.npmjs.org/escape-string-regexp/-/escape-string-regexp-4.0.0.tgz", + "integrity": "sha512-TtpcNJ3XAzx3Gq8sWRzJaVajRs0uVxA2YAkdb1jm2YkPz4G6egUFAyA3n5vtEIZefPk5Wa4UXbKuS5fKkJWdgA==", + "license": "MIT", + "engines": { + "node": ">=10" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, "node_modules/etag": { "version": "1.8.1", "resolved": "https://registry.npmjs.org/etag/-/etag-1.8.1.tgz", @@ -1410,6 +2103,12 @@ "url": "https://opencollective.com/express" } }, + "node_modules/flatbuffers": { + "version": "25.9.23", + "resolved": "https://registry.npmjs.org/flatbuffers/-/flatbuffers-25.9.23.tgz", + "integrity": "sha512-MI1qs7Lo4Syw0EOzUl0xjs2lsoeqFku44KpngfIduHBYvzm8h2+7K8YMQh1JtVVVrUvhLpNwqVi4DERegUJhPQ==", + "license": "Apache-2.0" + }, "node_modules/fn.name": { "version": "1.1.0", "resolved": "https://registry.npmjs.org/fn.name/-/fn.name-1.1.0.tgz", @@ -1546,6 +2245,39 @@ "url": "https://github.com/sponsors/isaacs" } }, + "node_modules/global-agent": { + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/global-agent/-/global-agent-3.0.0.tgz", + "integrity": "sha512-PT6XReJ+D07JvGoxQMkT6qji/jVNfX/h364XHZOWeRzy64sSFr+xJ5OX7LI3b4MPQzdL4H8Y8M0xzPpsVMwA8Q==", + "license": "BSD-3-Clause", + "dependencies": { + "boolean": "^3.0.1", + "es6-error": "^4.1.1", + "matcher": "^3.0.0", + "roarr": "^2.15.3", + "semver": "^7.3.2", + "serialize-error": "^7.0.1" + }, + "engines": { + "node": ">=10.0" + } + }, + "node_modules/globalthis": { + "version": "1.0.4", + "resolved": "https://registry.npmjs.org/globalthis/-/globalthis-1.0.4.tgz", + "integrity": "sha512-DpLKbNU4WylpxJykQujfCcwYWiV/Jhm50Goo0wrVILAv5jOr9d+H+UR3PhSCD2rCCEIg0uc+G+muBTwD54JhDQ==", + "license": "MIT", + "dependencies": { + "define-properties": "^1.2.1", + "gopd": "^1.0.1" + }, + "engines": { + "node": ">= 0.4" + }, + "funding": { + "url": "https://github.com/sponsors/ljharb" + } + }, "node_modules/gopd": { "version": "1.2.0", "resolved": "https://registry.npmjs.org/gopd/-/gopd-1.2.0.tgz", @@ -1564,6 +2296,24 @@ "integrity": "sha512-RbJ5/jmFcNNCcDV5o9eTnBLJ/HszWV0P73bc+Ff4nS/rJj+YaS6IGyiOL0VoBYX+l1Wrl3k63h/KrH+nhJ0XvQ==", "license": "ISC" }, + "node_modules/guid-typescript": { + "version": "1.0.9", + "resolved": "https://registry.npmjs.org/guid-typescript/-/guid-typescript-1.0.9.tgz", + "integrity": "sha512-Y8T4vYhEfwJOTbouREvG+3XDsjr8E3kIr7uf+JZ0BYloFsttiHU0WfvANVsR7TxNUJa/WpCnw/Ino/p+DeBhBQ==", + "license": "ISC" + }, + "node_modules/has-property-descriptors": { + "version": "1.0.2", + "resolved": "https://registry.npmjs.org/has-property-descriptors/-/has-property-descriptors-1.0.2.tgz", + "integrity": "sha512-55JNKuIW+vq4Ke1BjOTjM2YctQIvCT7GFzHwmfZPGo5wnrgkid0YQtnAleFSqumZm4az3n2BS+erby5ipJdgrg==", + "license": "MIT", + "dependencies": { + "es-define-property": "^1.0.0" + }, + "funding": { + "url": "https://github.com/sponsors/ljharb" + } + }, "node_modules/has-symbols": { "version": "1.1.0", "resolved": "https://registry.npmjs.org/has-symbols/-/has-symbols-1.1.0.tgz", @@ -1802,6 +2552,12 @@ "base64-js": "^1.5.1" } }, + "node_modules/json-stringify-safe": { + "version": "5.0.1", + "resolved": "https://registry.npmjs.org/json-stringify-safe/-/json-stringify-safe-5.0.1.tgz", + "integrity": "sha512-ZClg6AaYvamvYEE82d3Iyd3vSSIjQ+odgjaTzRuO3s7toCdFKczob2i0zCh7JE8kWn17yvAWhUVxvqGwUalsRA==", + "license": "ISC" + }, "node_modules/jsonwebtoken": { "version": "9.0.3", "resolved": "https://registry.npmjs.org/jsonwebtoken/-/jsonwebtoken-9.0.3.tgz", @@ -2137,6 +2893,24 @@ "node": ">= 12.0.0" } }, + "node_modules/long": { + "version": "5.3.2", + "resolved": "https://registry.npmjs.org/long/-/long-5.3.2.tgz", + "integrity": "sha512-mNAgZ1GmyNhD7AuqnTG3/VQ26o760+ZYBPKjPvugO8+nLbYfX6TVpJPseBvopbdY+qpZ/lKUnmEc1LeZYS3QAA==", + "license": "Apache-2.0" + }, + "node_modules/matcher": { + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/matcher/-/matcher-3.0.0.tgz", + "integrity": "sha512-OkeDaAZ/bQCxeFAozM55PKcKU0yJMPGifLwV4Qgjitu+5MoAfSQN4lsLJeXZ1b8w0x+/Emda6MZgXS1jvsapng==", + "license": "MIT", + "dependencies": { + "escape-string-regexp": "^4.0.0" + }, + "engines": { + "node": ">=10" + } + }, "node_modules/math-intrinsics": { "version": "1.1.0", "resolved": "https://registry.npmjs.org/math-intrinsics/-/math-intrinsics-1.1.0.tgz", @@ -2432,6 +3206,15 @@ "url": "https://github.com/sponsors/ljharb" } }, + "node_modules/object-keys": { + "version": "1.1.1", + "resolved": "https://registry.npmjs.org/object-keys/-/object-keys-1.1.1.tgz", + "integrity": "sha512-NuAESUOUMrlIXOfHKzD6bpPu3tYt3xvjNdRIQ+FeT0lNb4K8WR70CaDxhuNguS2XG+GjkyMwOzsN5ZktImfhLA==", + "license": "MIT", + "engines": { + "node": ">= 0.4" + } + }, "node_modules/on-finished": { "version": "2.4.1", "resolved": "https://registry.npmjs.org/on-finished/-/on-finished-2.4.1.tgz", @@ -2462,6 +3245,49 @@ "fn.name": "1.x.x" } }, + "node_modules/onnxruntime-common": { + "version": "1.24.3", + "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.24.3.tgz", + "integrity": "sha512-GeuPZO6U/LBJXvwdaqHbuUmoXiEdeCjWi/EG7Y1HNnDwJYuk6WUbNXpF6luSUY8yASul3cmUlLGrCCL1ZgVXqA==", + "license": "MIT" + }, + "node_modules/onnxruntime-node": { + "version": "1.24.3", + "resolved": "https://registry.npmjs.org/onnxruntime-node/-/onnxruntime-node-1.24.3.tgz", + "integrity": "sha512-JH7+czbc8ALA819vlTgcV+Q214/+VjGeBHDjX81+ZCD0PCVCIFGFNtT0V4sXG/1JXypKPgScQcB3ij/hk3YnTg==", + "hasInstallScript": true, + "license": "MIT", + "os": [ + "win32", + "darwin", + "linux" + ], + "dependencies": { + "adm-zip": "^0.5.16", + "global-agent": "^3.0.0", + "onnxruntime-common": "1.24.3" + } + }, + "node_modules/onnxruntime-web": { + "version": "1.26.0-dev.20260416-b7804b056c", + "resolved": "https://registry.npmjs.org/onnxruntime-web/-/onnxruntime-web-1.26.0-dev.20260416-b7804b056c.tgz", + "integrity": "sha512-MD6Ss4GSpQBo6zqoJzyT9LRbKYs7x/JVN23FT24EcEvlqF4VuzPOeH6X38orZPKHQDbprn7K+SBpu0/mj2CQiw==", + "license": "MIT", + "dependencies": { + "flatbuffers": "^25.1.24", + "guid-typescript": "^1.0.9", + "long": "^5.2.3", + "onnxruntime-common": "1.24.0-dev.20251116-b39e144322", + "platform": "^1.3.6", + "protobufjs": "^7.2.4" + } + }, + "node_modules/onnxruntime-web/node_modules/onnxruntime-common": { + "version": "1.24.0-dev.20251116-b39e144322", + "resolved": "https://registry.npmjs.org/onnxruntime-common/-/onnxruntime-common-1.24.0-dev.20251116-b39e144322.tgz", + "integrity": "sha512-BOoomdHYmNRL5r4iQ4bMvsl2t0/hzVQ3OM3PHD0gxeXu1PmggqBv3puZicEUVOA3AtHHYmqZtjMj9FOfGrATTw==", + "license": "MIT" + }, "node_modules/openai": { "version": "7.5.0", "resolved": "https://registry.npmjs.org/openai/-/openai-7.5.0.tgz", @@ -2594,6 +3420,12 @@ "png-js": "^1.0.0" } }, + "node_modules/platform": { + "version": "1.3.6", + "resolved": "https://registry.npmjs.org/platform/-/platform-1.3.6.tgz", + "integrity": "sha512-fnWVljUchTro6RiCFvCXBbNhJc2NijN7oIQxbwsyL0buWJPG85v81ehlHI9fXrJsMNgTofEoWIQeClKpgxFLrg==", + "license": "MIT" + }, "node_modules/png-js": { "version": "1.1.0", "resolved": "https://registry.npmjs.org/png-js/-/png-js-1.1.0.tgz", @@ -2635,6 +3467,29 @@ "integrity": "sha512-3ouUOpQhtgrbOa17J7+uxOTpITYWaGP7/AhoR3+A+/1e9skrzelGi/dXzEYyvbxubEF6Wn2ypscTKiKJFFn1ag==", "license": "MIT" }, + "node_modules/protobufjs": { + "version": "7.6.6", + "resolved": "https://registry.npmjs.org/protobufjs/-/protobufjs-7.6.6.tgz", + "integrity": "sha512-dYDWdjSl5RNb7SgPxGQcRU+GtvP7s2fpkrY0r432PcOIaZ0/rBcxEZnQN67iJhFuQiVw754JDoPruPCNdGsbjg==", + "hasInstallScript": true, + "license": "BSD-3-Clause", + "dependencies": { + "@protobufjs/aspromise": "^1.1.2", + "@protobufjs/base64": "^1.1.2", + "@protobufjs/codegen": "^2.0.5", + "@protobufjs/eventemitter": "^1.1.1", + "@protobufjs/fetch": "^1.1.1", + "@protobufjs/float": "^1.0.2", + "@protobufjs/path": "^1.1.2", + "@protobufjs/pool": "^1.1.0", + "@protobufjs/utf8": "^1.1.1", + "@types/node": ">=13.7.0", + "long": "^5.3.2" + }, + "engines": { + "node": ">=12.0.0" + } + }, "node_modules/proxy-addr": { "version": "2.0.7", "resolved": "https://registry.npmjs.org/proxy-addr/-/proxy-addr-2.0.7.tgz", @@ -2794,6 +3649,23 @@ "rimraf": "bin.js" } }, + "node_modules/roarr": { + "version": "2.15.4", + "resolved": "https://registry.npmjs.org/roarr/-/roarr-2.15.4.tgz", + "integrity": "sha512-CHhPh+UNHD2GTXNYhPWLnU8ONHdI+5DI+4EYIAOaiD63rHeYlZvyh8P+in5999TTSFgUYuKUAjzRI4mdh/p+2A==", + "license": "BSD-3-Clause", + "dependencies": { + "boolean": "^3.0.1", + "detect-node": "^2.0.4", + "globalthis": "^1.0.1", + "json-stringify-safe": "^5.0.1", + "semver-compare": "^1.0.0", + "sprintf-js": "^1.1.2" + }, + "engines": { + "node": ">=8.0" + } + }, "node_modules/router": { "version": "2.2.0", "resolved": "https://registry.npmjs.org/router/-/router-2.2.0.tgz", @@ -2878,6 +3750,12 @@ "node": ">=10" } }, + "node_modules/semver-compare": { + "version": "1.0.0", + "resolved": "https://registry.npmjs.org/semver-compare/-/semver-compare-1.0.0.tgz", + "integrity": "sha512-YM3/ITh2MJ5MtzaM429anh+x2jiLVjqILF4m4oyQB18W7Ggea7BfqdH/wGMK7dDiMghv/6WG7znWMwUDzJiXow==", + "license": "MIT" + }, "node_modules/send": { "version": "1.2.1", "resolved": "https://registry.npmjs.org/send/-/send-1.2.1.tgz", @@ -2904,6 +3782,21 @@ "url": "https://opencollective.com/express" } }, + "node_modules/serialize-error": { + "version": "7.0.1", + "resolved": "https://registry.npmjs.org/serialize-error/-/serialize-error-7.0.1.tgz", + "integrity": "sha512-8I8TjW5KMOKsZQTvoxjuSIa7foAwPWGOts+6o7sgjz41/qMD9VQHEDxi6PBvK2l0MXUmqZyNpUK+T2tQaaElvw==", + "license": "MIT", + "dependencies": { + "type-fest": "^0.13.1" + }, + "engines": { + "node": ">=10" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, "node_modules/serve-static": { "version": "2.2.1", "resolved": "https://registry.npmjs.org/serve-static/-/serve-static-2.2.1.tgz", @@ -2935,6 +3828,50 @@ "integrity": "sha512-E5LDX7Wrp85Kil5bhZv46j8jOeboKq5JMmYM3gVGdGH8xFpPWXUMsNrlODCrkoxMEeNi/XZIwuRvY4XNwYMJpw==", "license": "ISC" }, + "node_modules/sharp": { + "version": "0.34.5", + "resolved": "https://registry.npmjs.org/sharp/-/sharp-0.34.5.tgz", + "integrity": "sha512-Ou9I5Ft9WNcCbXrU9cMgPBcCK8LiwLqcbywW3t4oDV37n1pzpuNLsYiAV8eODnjbtQlSDwZ2cUEeQz4E54Hltg==", + "hasInstallScript": true, + "license": "Apache-2.0", + "dependencies": { + "@img/colour": "^1.0.0", + "detect-libc": "^2.1.2", + "semver": "^7.7.3" + }, + "engines": { + "node": "^18.17.0 || ^20.3.0 || >=21.0.0" + }, + "funding": { + "url": "https://opencollective.com/libvips" + }, + "optionalDependencies": { + "@img/sharp-darwin-arm64": "0.34.5", + "@img/sharp-darwin-x64": "0.34.5", + "@img/sharp-libvips-darwin-arm64": "1.2.4", + "@img/sharp-libvips-darwin-x64": "1.2.4", + "@img/sharp-libvips-linux-arm": "1.2.4", + "@img/sharp-libvips-linux-arm64": "1.2.4", + "@img/sharp-libvips-linux-ppc64": "1.2.4", + "@img/sharp-libvips-linux-riscv64": "1.2.4", + "@img/sharp-libvips-linux-s390x": "1.2.4", + "@img/sharp-libvips-linux-x64": "1.2.4", + "@img/sharp-libvips-linuxmusl-arm64": "1.2.4", + "@img/sharp-libvips-linuxmusl-x64": "1.2.4", + "@img/sharp-linux-arm": "0.34.5", + "@img/sharp-linux-arm64": "0.34.5", + "@img/sharp-linux-ppc64": "0.34.5", + "@img/sharp-linux-riscv64": "0.34.5", + "@img/sharp-linux-s390x": "0.34.5", + "@img/sharp-linux-x64": "0.34.5", + "@img/sharp-linuxmusl-arm64": "0.34.5", + "@img/sharp-linuxmusl-x64": "0.34.5", + "@img/sharp-wasm32": "0.34.5", + "@img/sharp-win32-arm64": "0.34.5", + "@img/sharp-win32-ia32": "0.34.5", + "@img/sharp-win32-x64": "0.34.5" + } + }, "node_modules/side-channel": { "version": "1.1.1", "resolved": "https://registry.npmjs.org/side-channel/-/side-channel-1.1.1.tgz", @@ -3022,6 +3959,12 @@ "memory-pager": "^1.0.2" } }, + "node_modules/sprintf-js": { + "version": "1.1.3", + "resolved": "https://registry.npmjs.org/sprintf-js/-/sprintf-js-1.1.3.tgz", + "integrity": "sha512-Oo+0REFV59/rz3gfJNKQiBlwfHaSESl1pcGyABQsnnIfWOFt6JNj5gCog2U6MLZ//IGYD+nA8nI+mTShREReaA==", + "license": "BSD-3-Clause" + }, "node_modules/stack-trace": { "version": "0.0.10", "resolved": "https://registry.npmjs.org/stack-trace/-/stack-trace-0.0.10.tgz", @@ -3137,6 +4080,18 @@ "integrity": "sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w==", "license": "0BSD" }, + "node_modules/type-fest": { + "version": "0.13.1", + "resolved": "https://registry.npmjs.org/type-fest/-/type-fest-0.13.1.tgz", + "integrity": "sha512-34R7HTnG0XIJcBSn5XhDd7nNFPRcXYRZrBB2O2jdKqYODldSzBAqzsWoZYYvduky73toYS/ESqxPvkDf/F0XMg==", + "license": "(MIT OR CC0-1.0)", + "engines": { + "node": ">=10" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, "node_modules/type-is": { "version": "2.1.0", "resolved": "https://registry.npmjs.org/type-is/-/type-is-2.1.0.tgz", diff --git a/package.json b/package.json index 85082b5..860652d 100644 --- a/package.json +++ b/package.json @@ -7,12 +7,12 @@ "scripts": { "start": "node src/server.js", "dev": "node --watch src/server.js", - "smoke": "node scripts/smoke.js", - "voice": "voice-service/.venv/Scripts/python.exe -m app.server" + "smoke": "node scripts/smoke.js" }, "author": "WeLe EdTech", "license": "ISC", "dependencies": { + "@huggingface/transformers": "^4.2.0", "@langchain/core": "^1.1.18", "@langchain/langgraph": "^1.1.0", "@langchain/openai": "^1.5.10", @@ -27,6 +27,7 @@ "ioredis": "^5.10.1", "jsonwebtoken": "^9.0.2", "mongoose": "^9.2.3", + "onnxruntime-node": "^1.24.3", "pdfkit": "^0.17.2", "pptxgenjs": "^4.0.1", "uuid": "^13.0.0", diff --git a/scripts/build-tamil-tokenizer.mjs b/scripts/build-tamil-tokenizer.mjs new file mode 100644 index 0000000..f183970 --- /dev/null +++ b/scripts/build-tamil-tokenizer.mjs @@ -0,0 +1,66 @@ +/* Generate the tokenizer.json that Transformers.js needs for the exported + Tamil VITS model. + + `save_pretrained` does not emit one: VitsTokenizer is a "slow" tokenizer with + no fast counterpart, so Python writes vocab.json + tokenizer_config.json and + nothing else. Transformers.js only reads tokenizer.json, so we synthesise it + from the exported vocab, mirroring the structure of the working English + model (Xenova/mms-tts-eng) exactly. + + The four normalizer steps, in order: + 1. Lowercase — no-op for Tamil, matters for embedded Latin/digits + 2. Replace — drop every character outside the vocab + 3. Strip — trim surrounding whitespace + 4. Replace — insert the blank token between every character, + which is what `add_blank: true` means for VITS. + Omit this and the audio comes out garbled. +*/ +import fs from 'node:fs'; +import path from 'node:path'; + +const DIR = 'assets/tts/mms-tts-tam'; +const vocab = JSON.parse(fs.readFileSync(path.join(DIR, 'vocab.json'), 'utf8')); +const cfg = JSON.parse(fs.readFileSync(path.join(DIR, 'tokenizer_config.json'), 'utf8')); + +// The blank/pad token is whichever character maps to id 0. +const blank = Object.keys(vocab).find((k) => vocab[k] === 0); +const unk = cfg.unk_token ?? ''; +const unkId = vocab[unk] ?? Object.keys(vocab).length; + +// Character class of everything we keep. Escape the regex metacharacters that +// are still special inside a negated class. +const escaped = Object.keys(vocab) + .filter((c) => c !== unk) + .map((c) => (']\\^-'.includes(c) ? '\\' + c : c)) + .join(''); + +const tokenizer = { + version: '1.0', + truncation: null, + padding: null, + added_tokens: [{ + id: unkId, content: unk, + single_word: false, lstrip: false, rstrip: false, normalized: false, special: true, + }], + normalizer: { + type: 'Sequence', + normalizers: [ + { type: 'Lowercase' }, + { type: 'Replace', pattern: { Regex: `[^${escaped}]` }, content: '' }, + { type: 'Strip', strip_left: true, strip_right: true }, + ...(cfg.add_blank ? [{ type: 'Replace', pattern: { Regex: '(?=.)|(? None: + super().__init__() + self.m = m + + def forward(self, input_ids: torch.Tensor, attention_mask: torch.Tensor): + out = self.m(input_ids=input_ids, attention_mask=attention_mask) + return out.waveform, out.spectrogram + + +sample = tok("வணக்கம், இது ஒரு சோதனை.", return_tensors="pt") +fp32 = ONNX_DIR / "model.onnx" + +print("exporting to ONNX …") +torch.onnx.export( + Exportable(model), + (sample["input_ids"], sample["attention_mask"]), + str(fp32), + input_names=["input_ids", "attention_mask"], + output_names=["waveform", "spectrogram"], + dynamic_axes={ + "input_ids": {0: "batch", 1: "sequence"}, + "attention_mask": {0: "batch", 1: "sequence"}, + "waveform": {0: "batch", 1: "samples"}, + "spectrogram": {0: "batch", 2: "frames"}, + }, + opset_version=17, + do_constant_folding=True, +) +print(f" fp32: {fp32.stat().st_size / 1e6:.1f} MB") + +# ── int8 ──────────────────────────────────────────────────────────────────── +try: + from onnxruntime.quantization import QuantType, quantize_dynamic + + q = ONNX_DIR / "model_quantized.onnx" + quantize_dynamic(str(fp32), str(q), weight_type=QuantType.QUInt8) + print(f" int8: {q.stat().st_size / 1e6:.1f} MB") +except Exception as e: # noqa: BLE001 + print(f" quantisation skipped: {e}") + +# ── tokenizer + config, so the folder loads standalone ────────────────────── +tok.save_pretrained(OUT) +model.config.to_json_file(OUT / "config.json") + +# Transformers.js reads this to pick a default dtype. +(OUT / "quantize_config.json").write_text(json.dumps({"per_channel": False, "reduce_range": False}, indent=2)) + +print("\nwrote:") +for p in sorted(OUT.rglob("*")): + if p.is_file(): + print(f" {p.relative_to(OUT)} ({p.stat().st_size / 1e6:.2f} MB)") diff --git a/scripts/t-auto.mjs b/scripts/t-auto.mjs new file mode 100644 index 0000000..0e633d4 --- /dev/null +++ b/scripts/t-auto.mjs @@ -0,0 +1,21 @@ +/* Does auto mode now route Tamil to Tamil instead of silently using English? */ +import { synthesize, transcribe } from '../src/speech/index.js'; + +const resample = (a, from, to) => { + const r = from / to, out = new Float32Array(Math.floor(a.length / r)); + for (let i = 0; i < out.length; i++) { const p = i * r, k = Math.floor(p); out[i] = a[k] + (a[Math.min(k + 1, a.length - 1)] - a[k]) * (p - k); } + return out; +}; + +for (const [lang, text] of [ + ['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'], + ['en', 'How many new leads did we receive today?'], +]) { + const spoken = await synthesize(text, lang); + const audio = resample(spoken.audio, spoken.sampling_rate, 16000); + const r = await transcribe(audio, 'auto', 'ta'); + const ok = r.lang === lang ? 'PASS' : 'FAIL'; + console.log(`${ok} spoke ${lang} → routed ${r.lang} (detected ${r.detected} @ ${r.confidence}) ${r.ms}ms`); + console.log(` ${JSON.stringify(r.text.slice(0, 80))}`); +} +process.exit(0); diff --git a/scripts/t-detect.mjs b/scripts/t-detect.mjs new file mode 100644 index 0000000..2c0d9c4 --- /dev/null +++ b/scripts/t-detect.mjs @@ -0,0 +1,62 @@ +/* Can we get real language detection out of Whisper in Transformers.js? */ +import { AutoProcessor, WhisperForConditionalGeneration, Tensor, env } from '@huggingface/transformers'; +import { synthesize } from '../src/speech/index.js'; + +env.cacheDir = './.transformers-cache'; +const MODEL = 'onnx-community/whisper-base'; + +const processor = await AutoProcessor.from_pretrained(MODEL); +const model = await WhisperForConditionalGeneration.from_pretrained(MODEL, { dtype: 'q8' }); +const tok = processor.tokenizer; + +// Whisper emits one language token right after <|startoftranscript|>. Reading +// that distribution is a single decoder step — far cheaper than transcribing +// twice to see which language "looks better". +const id = (t) => tok.encode(t, { add_special_tokens: false })[0]; +const sot = id('<|startoftranscript|>'); +const CANDIDATES = ['en', 'ta']; +const langIds = CANDIDATES.map((c) => id(`<|${c}|>`)); +console.log('sot:', sot, '| language token ids:', JSON.stringify(Object.fromEntries(CANDIDATES.map((c, i) => [c, langIds[i]])))); + +function resample(a, from, to) { + const r = from / to; + const out = new Float32Array(Math.floor(a.length / r)); + for (let i = 0; i < out.length; i++) { + const p = i * r, k = Math.floor(p); + out[i] = a[k] + (a[Math.min(k + 1, a.length - 1)] - a[k]) * (p - k); + } + return out; +} + +for (const [lang, text] of [ + ['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'], + ['en', 'There are three thousand four hundred and twenty seven new leads.'], +]) { + const spoken = await synthesize(text, lang); + const audio = resample(spoken.audio, spoken.sampling_rate, 16000); + + const inputs = await processor(audio); + const t0 = Date.now(); + const out = await model({ + ...inputs, + decoder_input_ids: new Tensor('int64', BigInt64Array.from([BigInt(sot)]), [1, 1]), + }); + const ms = Date.now() - t0; + + const logits = out.logits; + const last = logits.dims[1] - 1; + const vocab = logits.dims[2]; + const row = logits.data.slice(last * vocab, (last + 1) * vocab); + + const scores = langIds.map((id) => Number(row[id])); + const max = Math.max(...scores); + const exp = scores.map((s) => Math.exp(s - max)); + const sum = exp.reduce((a, b) => a + b, 0); + const probs = exp.map((e) => e / sum); + const best = probs.indexOf(Math.max(...probs)); + + console.log(`spoken ${lang} → detected ${CANDIDATES[best]} ` + + `(${CANDIDATES.map((c, i) => `${c} ${probs[i].toFixed(3)}`).join(', ')}) in ${ms}ms ` + + `${CANDIDATES[best] === lang ? '✅' : '❌'}`); +} +process.exit(0); diff --git a/scripts/t-speech.mjs b/scripts/t-speech.mjs new file mode 100644 index 0000000..a66fd98 --- /dev/null +++ b/scripts/t-speech.mjs @@ -0,0 +1,44 @@ +/* Do Whisper (STT) and MMS-TTS (TTS) actually run in Node on CPU? */ +import { pipeline, env } from '@huggingface/transformers'; + +env.cacheDir = './.transformers-cache'; + +const t = (t0) => `${((performance.now() - t0) / 1000).toFixed(1)}s`; + +// ── TTS: MMS-TTS Tamil (VITS, 36M, feed-forward) ──────────────────────────── +console.log('[1/2] loading MMS-TTS Tamil…'); +let t0 = performance.now(); +const tts = await pipeline('text-to-speech', 'Xenova/mms-tts-eng', { dtype: 'fp32' }); +console.log(` loaded in ${t(t0)}`); + +const TA = 'There are three thousand four hundred leads in the new lead stage.'; +await tts(TA); // warm +for (const [label, text] of [['short', TA], ['long', TA + ' ' + TA + ' ' + TA]]) { + t0 = performance.now(); + const out = await tts(text); + const ms = performance.now() - t0; + const audioMs = (out.audio.length / out.sampling_rate) * 1000; + console.log( + ` ${label.padEnd(5)} ${String(text.length).padStart(3)} chars → ${ms.toFixed(0)}ms ` + + `for ${audioMs.toFixed(0)}ms audio @ ${out.sampling_rate}Hz → RTF ${(ms / audioMs).toFixed(2)}x`, + ); +} + +// ── STT: Whisper (multilingual — Tamil, Hindi, English + detection) ───────── +console.log('\n[2/2] loading Whisper base…'); +t0 = performance.now(); +const stt = await pipeline('automatic-speech-recognition', 'onnx-community/whisper-base', { dtype: 'q8' }); +console.log(` loaded in ${t(t0)}`); + +// 4 s of quiet noise — proves the graph runs and times it. +const audio = Float32Array.from({ length: 16000 * 4 }, () => (Math.random() - 0.5) * 0.02); +t0 = performance.now(); +const r = await stt(audio, { language: 'ta', task: 'transcribe' }); +console.log(` 4000ms audio → ${(performance.now() - t0).toFixed(0)}ms → ${JSON.stringify(r.text).slice(0, 60)}`); + +t0 = performance.now(); +const r2 = await stt(audio, { language: 'en', task: 'transcribe' }); +console.log(` english pass → ${(performance.now() - t0).toFixed(0)}ms → ${JSON.stringify(r2.text).slice(0, 60)}`); + +console.log(`\nRSS ${(process.memoryUsage().rss / 1e9).toFixed(2)} GB`); +process.exit(0); diff --git a/scripts/t-tamil-tts.mjs b/scripts/t-tamil-tts.mjs new file mode 100644 index 0000000..9d04e45 --- /dev/null +++ b/scripts/t-tamil-tts.mjs @@ -0,0 +1,55 @@ +/* Does the locally-exported Tamil ONNX load and speak through Transformers.js? */ +import { pipeline, env } from '@huggingface/transformers'; +import fs from 'node:fs'; + +// Load from the local folder, not the Hub. +env.allowRemoteModels = false; +env.localModelPath = './assets/tts'; + +for (const dtype of ['q8', 'fp32']) { + try { + const t0 = performance.now(); + const tts = await pipeline('text-to-speech', 'mms-tts-tam', { dtype }); + const load = performance.now() - t0; + + const TEXT = 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'; + await tts(TEXT); // warm + + const t1 = performance.now(); + const out = await tts(TEXT); + const ms = performance.now() - t1; + const audioMs = (out.audio.length / out.sampling_rate) * 1000; + + console.log( + `${dtype.padEnd(5)} load ${(load / 1000).toFixed(1)}s | ` + + `${ms.toFixed(0)}ms for ${audioMs.toFixed(0)}ms audio @ ${out.sampling_rate}Hz | ` + + `RTF ${(ms / audioMs).toFixed(2)}x`, + ); + + // Non-silent output is the real proof the graph is wired correctly. + const peak = out.audio.reduce((m, v) => Math.max(m, Math.abs(v)), 0); + console.log(` samples ${out.audio.length}, peak amplitude ${peak.toFixed(3)} ${peak > 0.01 ? '✅ audible' : '⚠️ SILENT'}`); + + if (dtype === 'q8') { + const wav = toWav(out.audio, out.sampling_rate); + fs.writeFileSync('scripts/tamil-sample.wav', wav); + console.log(' wrote scripts/tamil-sample.wav — play it to judge quality'); + } + } catch (e) { + console.log(`${dtype.padEnd(5)} FAILED: ${e.message.slice(0, 160)}`); + } +} + +function toWav(samples, rate) { + const buf = Buffer.alloc(44 + samples.length * 2); + buf.write('RIFF', 0); buf.writeUInt32LE(36 + samples.length * 2, 4); buf.write('WAVE', 8); + buf.write('fmt ', 12); buf.writeUInt32LE(16, 16); buf.writeUInt16LE(1, 20); buf.writeUInt16LE(1, 22); + buf.writeUInt32LE(rate, 24); buf.writeUInt32LE(rate * 2, 28); buf.writeUInt16LE(2, 32); buf.writeUInt16LE(16, 34); + buf.write('data', 36); buf.writeUInt32LE(samples.length * 2, 40); + for (let i = 0; i < samples.length; i++) { + const s = Math.max(-1, Math.min(1, samples[i])); + buf.writeInt16LE(s < 0 ? s * 0x8000 : s * 0x7fff, 44 + i * 2); + } + return buf; +} +process.exit(0); diff --git a/scripts/t-voice.mjs b/scripts/t-voice.mjs new file mode 100644 index 0000000..5e5e223 --- /dev/null +++ b/scripts/t-voice.mjs @@ -0,0 +1,72 @@ +/* End-to-end check of the in-process speech pipeline: TTS → VAD → STT. + + Synthesising a sentence and feeding that audio back through the endpointer + and recogniser exercises every stage with real speech, which a noise buffer + cannot do — silence never opens a VAD turn. +*/ +import { synthesize, transcribe, sentences, speakable, Endpointer } from '../src/speech/index.js'; + +const say = (m) => console.log(m); + +// ── 1. TTS both languages ─────────────────────────────────────────────────── +const CASES = [ + ['ta', 'மூவாயிரம் நானூற்று இருபத்தேழு புதிய லீட்கள் உள்ளன.'], + ['en', 'There are three thousand four hundred and twenty seven new leads.'], +]; + +const rendered = {}; +for (const [lang, text] of CASES) { + const t0 = Date.now(); + await synthesize(text, lang); // warm + const t1 = Date.now(); + const out = await synthesize(text, lang); + const ms = Date.now() - t1; + const audioMs = (out.audio.length / out.sampling_rate) * 1000; + rendered[lang] = out; + say(`TTS ${lang} warm ${((t1 - t0) / 1000).toFixed(1)}s | ${ms}ms for ${audioMs.toFixed(0)}ms ` + + `@${out.sampling_rate}Hz | RTF ${(ms / audioMs).toFixed(2)}x`); +} + +// ── 2. VAD: does synthesised speech open and close a turn? ────────────────── +function resample(audio, from, to) { + if (from === to) return audio; + const ratio = from / to; + const out = new Float32Array(Math.floor(audio.length / ratio)); + for (let i = 0; i < out.length; i++) { + const p = i * ratio; + const a = Math.floor(p); + out[i] = audio[a] + (audio[Math.min(a + 1, audio.length - 1)] - audio[a]) * (p - a); + } + return out; +} + +const ep = new Endpointer(); +const speech = resample(rendered.en.audio, rendered.en.sampling_rate, 16000); +// Speech, then a second of silence so the endpointer closes the turn. +const withTail = new Float32Array(speech.length + 16000); +withTail.set(speech); + +let started = false; +let captured = null; +for (let i = 0; i < withTail.length; i += 640) { // 40 ms chunks, as the browser sends + const { utterances, started: s } = await ep.push(withTail.subarray(i, Math.min(i + 640, withTail.length))); + if (s) started = true; + if (utterances.length) { captured = utterances[0]; break; } +} +say(`VAD speech detected: ${started ? 'yes' : 'NO'} | turn closed: ${captured ? 'yes' : 'NO'}` + + (captured ? ` | captured ${(captured.length / 16000).toFixed(2)}s` : '')); + +// ── 3. STT on that captured audio ─────────────────────────────────────────── +if (captured) { + for (const lang of ['en', 'auto']) { + const r = await transcribe(captured, lang, 'en'); + say(`STT ${lang.padEnd(4)} ${r.ms}ms → lang=${r.lang}${r.detected ? ` (heard ${r.detected})` : ''} → ${JSON.stringify(r.text.slice(0, 70))}`); + } +} + +// ── 4. Text shaping ───────────────────────────────────────────────────────── +const md = '## Leads\n\n**3,427** in `new_lead`.\n\n| a | b |\n|---|---|\n| 1 | 2 |\n\n- Only 12% contacted.\nNext step is triage.'; +say(`\nspeakable: ${JSON.stringify(speakable(md))}`); +say(`sentences: ${JSON.stringify(sentences(speakable(md)))}`); +say(`\nRSS ${(process.memoryUsage().rss / 1e9).toFixed(2)} GB`); +process.exit(0); diff --git a/src/gateway/voice.js b/src/gateway/voice.js index 72f8be0..cc6ab6a 100644 --- a/src/gateway/voice.js +++ b/src/gateway/voice.js @@ -1,104 +1,59 @@ // ============================================ -// Voice channel — a WebSocket bridge between the browser and the GPU service. +// Voice channel — speech in, speech out, in this same Node process. // // Voice is a *channel*, not a parallel product: a spoken question runs through // the same graph, guardrails and agents as a typed one. Only the transport and // the presentation differ, which is why this file contains no CRM logic. // -// browser ──audio──► this ──audio──► python(:4100) ──STT──► transcript -// │ │ -// └──────────── runTurn(graph) ◄───────────┘ -// │ -// browser ◄──audio── this ◄──audio── python(TTS) ◄──sentences───┘ +// browser ──PCM16──► Endpointer ──► transcribe() ──► runTurn(graph) +// │ │ +// browser ◄──float32──── synthesize() ◄── sentences ◄─────┘ // -// The hard problem is not transport, it is that a turn takes 17–46 s. Silence -// for that long feels broken, so the bridge speaks immediately, narrates what -// the agents are doing, and starts reading the answer at the first sentence -// rather than waiting for the last. +// The hard problem is not audio, it is that a turn takes 17–46 s and that is +// dead silence in voice. So this speaks an acknowledgement within ~1 s, +// narrates each agent delegation aloud, then reads the answer sentence by +// sentence as it is composed. // ============================================ import { WebSocketServer, WebSocket } from 'ws'; -import { randomUUID } from 'node:crypto'; import { principalFromToken } from './auth.js'; import { runTurn } from '../orchestration/runner.js'; +import { + Endpointer, transcribe, synthesize, sentences, speakable, LANGUAGES, +} from '../speech/index.js'; import config from '../config/index.js'; import logger from '../utils/logger.js'; -const VOICE_URL = process.env.VOICE_SERVICE_URL || 'ws://127.0.0.1:4100/ws/voice'; - -/** Spoken filler, per language. Said the instant a question lands. */ +/** Spoken filler, said the instant a question lands. */ const ACK = { - ta: ['பார்க்கிறேன்...', 'ஒரு நிமிடம், பார்க்கிறேன்.'], - hi: ['देखता हूँ...', 'एक मिनट, देख रहा हूँ।'], - te: ['చూస్తున్నాను...'], - kn: ['ನೋಡುತ್ತಿದ್ದೇನೆ...'], - ml: ['നോക്കുന്നു...'], - mr: ['बघतो...'], - bn: ['দেখছি...'], + ta: ['பார்க்கிறேன்.', 'ஒரு நிமிடம், பார்க்கிறேன்.'], en: ['Let me check.', 'One moment, checking now.'], }; -/** Progress narration, kept short — it is spoken over the user's waiting time. */ +/** Progress narration — spoken over the user's waiting time, so keep it short. */ const NARRATE = { ta: { lead: 'லீட் விவரங்களைப் பார்க்கிறேன்.', analytics: 'புள்ளிவிவரங்களைச் சரிபார்க்கிறேன்.', conversation: 'உரையாடல்களைப் பார்க்கிறேன்.', default: 'தரவைச் சரிபார்க்கிறேன்.' }, - hi: { lead: 'लीड्स देख रहा हूँ।', analytics: 'आँकड़े देख रहा हूँ।', conversation: 'बातचीत देख रहा हूँ।', default: 'डेटा देख रहा हूँ।' }, en: { lead: 'Checking the leads.', analytics: 'Pulling the numbers.', conversation: 'Looking at the conversations.', default: 'Checking the data.' }, }; -const pick = (arr) => arr[Math.floor(Math.random() * arr.length)]; +const NOTHING = { ta: 'பதில் கிடைக்கவில்லை.', en: 'I could not find an answer for that.' }; +const OOPS = { ta: 'மன்னிக்கவும், ஒரு பிழை ஏற்பட்டது.', en: 'Sorry, something went wrong.' }; -function ackFor(lang) { - return pick(ACK[lang] || ACK.en); -} +const pick = (a) => a[Math.floor(Math.random() * a.length)]; +const ackFor = (l) => pick(ACK[l] || ACK.en); +const narrateFor = (l, agent) => (NARRATE[l] || NARRATE.en)[agent] || (NARRATE[l] || NARRATE.en).default; -function narrationFor(lang, agent) { - const set = NARRATE[lang] || NARRATE.en; - return set[agent] || set.default; -} - -/** - * Strip block-oriented markdown before speaking. Tables and code read terribly - * aloud, and the visual blocks are already on screen. - */ -export function speakable(markdown = '') { - return markdown - .replace(/```[\s\S]*?```/g, ' ') - .replace(/^\s*\|.*\|\s*$/gm, ' ') // table rows - .replace(/^\s*[-*]\s+/gm, '') // bullets - .replace(/^#{1,6}\s*/gm, '') // headings - .replace(/\*\*([^*]+)\*\*/g, '$1') - .replace(/`([^`]+)`/g, '$1') - .replace(/\[([^\]]+)\]\([^)]+\)/g, '$1') - .replace(/₹\s?([\d,.]+)/g, 'rupees $1') - .replace(/\s{2,}/g, ' ') - .trim(); -} - -/** Split into sentences so speech can start before the answer is finished. */ -export function sentences(text, max = 240) { - const out = []; - for (const raw of text.split(/(?<=[.!?।])\s+/)) { - let s = raw.trim(); - if (!s) continue; - while (s.length > max) { - const cut = s.lastIndexOf(' ', max); - out.push(s.slice(0, cut > 0 ? cut : max).trim()); - s = s.slice(cut > 0 ? cut : max).trim(); - } - if (s) out.push(s); - } - return out; -} - -class VoiceBridge { +class VoiceSession { constructor(client, user) { this.client = client; this.user = user; - this.lang = 'auto'; // what the user chose - this.replyLang = 'ta'; // what the last utterance actually was + this.lang = 'auto'; // what the user selected + this.replyLang = 'ta'; // what the last utterance actually was + this.prefer = 'ta'; // tiebreak when detection is unusable this.sessionId = `voice:${Date.now().toString(36)}:${Math.random().toString(36).slice(2, 8)}`; - this.gpu = null; + this.endpointer = new Endpointer(); this.busy = false; this.abort = null; + this.speakSeq = 0; // rising token; stale synthesis is discarded this.narrated = new Set(); } @@ -106,109 +61,104 @@ class VoiceBridge { if (this.client.readyState === WebSocket.OPEN) this.client.send(JSON.stringify(obj)); } - toGpu(obj) { - if (this.gpu?.readyState === WebSocket.OPEN) this.gpu.send(JSON.stringify(obj)); + sendAudio(buf) { + if (this.client.readyState === WebSocket.OPEN) this.client.send(buf, { binary: true }); } - async connect() { - this.gpu = new WebSocket(VOICE_URL); + // ── microphone ─────────────────────────────────────────────────────────── + async onAudio(data) { + // Browser sends 16 kHz mono PCM16; the models want float32 in [-1, 1]. + const pcm16 = new Int16Array(data.buffer, data.byteOffset, Math.floor(data.byteLength / 2)); + const pcm = new Float32Array(pcm16.length); + for (let i = 0; i < pcm16.length; i++) pcm[i] = pcm16[i] / 32768; - this.gpu.on('open', () => { - logger.info(`🎙️ voice session ${this.sessionId} → GPU service`); - this.toGpu({ type: 'config', lang: this.lang }); - }); + let result; + try { + result = await this.endpointer.push(pcm); + } catch (e) { + logger.error(`VAD failed: ${e.message}`); + this.send({ type: 'error', message: 'Voice input failed to initialise. Check the server logs.' }); + return; + } - this.gpu.on('message', (data, isBinary) => { - // TTS audio: pass straight through, no re-encoding. - if (isBinary) { - if (this.client.readyState === WebSocket.OPEN) this.client.send(data, { binary: true }); - return; - } - let msg; - try { msg = JSON.parse(data.toString()); } catch { return; } - this.onGpuMessage(msg); - }); + if (result.started) { + // Barge-in: the user talking wins immediately. Bumping the token drops + // any in-flight synthesis rather than letting it arrive late. + this.speakSeq++; + this.abort?.abort(); + this.send({ type: 'barge_in' }); + } - this.gpu.on('error', (err) => { - logger.error(`voice GPU service: ${err.message}`); - this.send({ type: 'error', message: 'The voice service is not reachable. Start it with: npm run voice' }); - }); - - this.gpu.on('close', () => { - this.send({ type: 'voice_service_closed' }); - this.client.close(); - }); - } - - onGpuMessage(msg) { - switch (msg.type) { - case 'ready': - this.send({ type: 'ready', session_id: this.sessionId, languages: msg.languages, sample_rate_out: msg.sample_rate_out }); - break; - - case 'speech_start': - // The user started talking — the GPU service already stopped speaking. - // Tell the browser to dump whatever is still in its playback buffer, - // and abandon any answer still being composed. - this.send({ type: 'barge_in' }); - this.abort?.abort(); - break; - - case 'transcript': - // Answer in the language the person actually spoke, not the menu - // setting — that is the whole point of auto mode. - if (msg.lang) this.replyLang = msg.lang; - this.send({ type: 'transcript', text: msg.text, lang: msg.lang, detected: msg.detected, confidence: msg.confidence, ms: msg.ms }); - this.handleQuestion(msg.text); - break; - - case 'transcript_empty': - this.send({ type: 'heard_nothing' }); - break; - - case 'audio_start': - case 'audio_end': - case 'error': - this.send(msg); - break; - - default: - break; + for (const utterance of result.utterances) { + await this.handleUtterance(utterance); } } - speak(text, id = randomUUID()) { - const clean = speakable(text); - if (clean) this.toGpu({ type: 'speak', text: clean, id, lang: this.replyLang }); + async handleUtterance(audio) { + let heard; + try { + heard = await transcribe(audio, this.lang, this.prefer); + } catch (e) { + logger.error(`STT failed: ${e.message}`); + this.send({ type: 'error', message: 'Could not transcribe that. Try again.' }); + return; + } + + if (!heard.text) { + this.send({ type: 'heard_nothing' }); + return; + } + + this.replyLang = heard.lang; + this.send({ type: 'transcript', text: heard.text, lang: heard.lang, detected: heard.detected, ms: heard.ms }); + await this.answer(heard.text); } - async handleQuestion(text) { - if (!text?.trim()) return; - if (this.busy) return; // one turn at a time + // ── speaking ───────────────────────────────────────────────────────────── + /** Synthesise and stream one piece, unless a newer turn has superseded it. */ + async say(text, seq) { + const clean = speakable(text); + if (!clean || seq !== this.speakSeq) return; + try { + const out = await synthesize(clean, this.replyLang); + if (!out || seq !== this.speakSeq) return; // interrupted while generating + + this.send({ type: 'audio_start', sample_rate: out.sampling_rate }); + // Float32 straight down the socket — the playback worklet takes it as-is. + this.sendAudio(Buffer.from(out.audio.buffer, out.audio.byteOffset, out.audio.byteLength)); + this.send({ type: 'audio_end' }); + } catch (e) { + logger.error(`TTS failed: ${e.message}`); + } + } + + async answer(question) { + if (this.busy) return; // one turn at a time this.busy = true; this.narrated.clear(); this.abort = new AbortController(); + const seq = ++this.speakSeq; - // 1. Answer the silence immediately. This is the whole trick: the pipeline - // still takes 17–46 s, but the user hears a response in ~1 s. - this.speak(ackFor(this.replyLang)); + // Answer the silence immediately. The pipeline still takes 17–46 s, but + // the user hears a response in about a second. + this.say(ackFor(this.replyLang), seq); this.send({ type: 'thinking' }); try { const result = await runTurn({ sessionId: this.sessionId, - message: text, + message: question, user: this.user, - channel: 'crm_chat', // voice users are staff; full tool access + channel: 'crm_chat', // voice users are staff signal: this.abort.signal, onEvent: (ev) => { this.send(ev); - // 2. Narrate delegations — but only once per agent, or it chatters. + // Narrate delegations, once per agent, or it chatters. if (ev.type === 'step' && ev.kind === 'delegate') { const agent = String(ev.label || '').toLowerCase().split(' ')[0]; if (!this.narrated.has(agent)) { this.narrated.add(agent); - this.speak(narrationFor(this.replyLang, agent)); + this.say(narrateFor(this.replyLang, agent), seq); } } }, @@ -216,23 +166,24 @@ class VoiceBridge { this.send({ type: 'result', blocks: result.blocks, usage: result.usage }); - // 3. Read the answer. Sentence at a time so speech starts sooner and can - // be cut cleanly if the user interrupts. - const answer = result.blocks?.filter((b) => b.type === 'text').map((b) => b.markdown).join(' ') - || result.answer || ''; + const answer = (result.blocks || []) + .filter((b) => b.type === 'text').map((b) => b.markdown).join(' ') || result.answer || ''; const parts = sentences(speakable(answer)); + if (!parts.length) { - this.speak(this.replyLang === 'ta' ? 'பதில் கிடைக்கவில்லை.' : 'I could not find an answer for that.'); + await this.say(NOTHING[this.replyLang] || NOTHING.en, seq); } else { + // Sequential on purpose: parallel synthesis would race to the socket + // and play the answer out of order. for (const part of parts) { - if (this.abort.signal.aborted) break; - this.speak(part); + if (seq !== this.speakSeq || this.abort.signal.aborted) break; + await this.say(part, seq); } } } catch (err) { if (err?.name !== 'AbortError') { logger.error(`voice turn failed: ${err.message}`); - this.speak(this.replyLang === 'ta' ? 'மன்னிக்கவும், ஒரு பிழை ஏற்பட்டது.' : 'Sorry, something went wrong.'); + await this.say(OOPS[this.replyLang] || OOPS.en, seq); } } finally { this.busy = false; @@ -240,32 +191,33 @@ class VoiceBridge { } } - onClientMessage(data, isBinary) { - if (isBinary) { - if (this.gpu?.readyState === WebSocket.OPEN) this.gpu.send(data, { binary: true }); - return; - } + // ── control ────────────────────────────────────────────────────────────── + onMessage(data, isBinary) { + if (isBinary) return this.onAudio(data); + let msg; - try { msg = JSON.parse(data.toString()); } catch { return; } + try { msg = JSON.parse(data.toString()); } catch { return undefined; } if (msg.type === 'config' && msg.lang) { this.lang = msg.lang; - if (msg.lang !== 'auto') this.replyLang = msg.lang; - this.toGpu({ type: 'config', lang: msg.lang }); - this.send({ type: 'config_ok', lang: msg.lang }); + if (msg.lang !== 'auto') this.replyLang = this.prefer = msg.lang; + else if (msg.prefer) this.prefer = msg.prefer; + this.endpointer.reset(); + this.send({ type: 'config_ok', lang: this.lang, prefer: this.prefer }); } else if (msg.type === 'cancel') { + this.speakSeq++; this.abort?.abort(); - this.toGpu({ type: 'cancel' }); - } else if (msg.type === 'text') { - // Typed question while in voice mode — answered aloud like a spoken one. + this.send({ type: 'cancelled' }); + } else if (msg.type === 'text' && msg.text) { this.send({ type: 'transcript', text: msg.text, lang: this.replyLang, typed: true }); - this.handleQuestion(msg.text); + this.answer(msg.text); } + return undefined; } close() { + this.speakSeq++; this.abort?.abort(); - try { this.gpu?.close(); } catch { /* already gone */ } } } @@ -275,12 +227,11 @@ export function attachVoice(server) { server.on('upgrade', async (req, socket, head) => { const url = new URL(req.url, `http://${req.headers.host}`); - if (url.pathname !== '/api/agent/voice') return; // leave other upgrades alone + if (url.pathname !== '/api/agent/voice') return; // leave other upgrades alone // Browsers cannot set headers on a WebSocket, so the CRM token arrives as - // a query parameter. It is the same token and the same verification. - const token = url.searchParams.get('token'); - const user = await principalFromToken(token).catch(() => null); + // a query parameter. Same token, same verification as every other route. + const user = await principalFromToken(url.searchParams.get('token')).catch(() => null); if (!user) { socket.write('HTTP/1.1 401 Unauthorized\r\n\r\n'); socket.destroy(); @@ -288,14 +239,16 @@ export function attachVoice(server) { } wss.handleUpgrade(req, socket, head, (client) => { - const bridge = new VoiceBridge(client, user); - bridge.connect(); - client.on('message', (d, bin) => bridge.onClientMessage(d, bin)); - client.on('close', () => bridge.close()); - client.on('error', () => bridge.close()); + const session = new VoiceSession(client, user); + logger.info(`🎙️ voice session ${session.sessionId} (${user.name})`); + session.send({ type: 'ready', session_id: session.sessionId, languages: LANGUAGES }); + + client.on('message', (d, bin) => session.onMessage(d, bin)); + client.on('close', () => session.close()); + client.on('error', () => session.close()); }); }); - logger.info(` voice =ws://localhost:${config.port}/api/agent/voice → ${VOICE_URL}`); + logger.info(` voice =ws://localhost:${config.port}/api/agent/voice (in-process, CPU)`); return wss; } diff --git a/src/server.js b/src/server.js index a510791..3f9841b 100644 --- a/src/server.js +++ b/src/server.js @@ -17,6 +17,7 @@ import { ensureDir, sweep } from './output/artifactStore.js'; import crmApi from './tools/http/crmApi.js'; import { describeChains } from './orchestration/llm.js'; import { attachVoice } from './gateway/voice.js'; +import { warmup as warmSpeech, speechStatus } from './speech/index.js'; const app = express(); @@ -40,6 +41,7 @@ app.get('/health', async (_req, res) => { redis: redisOk ? redisMode() : 'unavailable', crm_api: crm.reachable ? 'reachable' : `unreachable (${crm.error || crm.status})`, models: describeChains(), + speech: speechStatus(), uptime_s: Math.round(process.uptime()), }); }); @@ -83,6 +85,11 @@ async function start() { // second origin and the CRM token works unchanged. attachVoice(server); + // Speech models load lazily on the first voice turn (~10 s). Set + // SPEECH_WARMUP=true to pay that at boot instead — worth it in production, + // wasteful in development where most restarts never use voice. + if (process.env.SPEECH_WARMUP === 'true') warmSpeech(['ta', 'en']); + const shutdown = (sig) => { logger.info(`${sig} — shutting down`); server.close(() => process.exit(0)); diff --git a/src/speech/index.js b/src/speech/index.js new file mode 100644 index 0000000..518feab --- /dev/null +++ b/src/speech/index.js @@ -0,0 +1,169 @@ +// ============================================ +// Speech pipeline — transcribe() and synthesize(). +// +// Whisper is multilingual and can identify the spoken language, so "auto" +// costs nothing extra: detection and transcription are the same forward pass. +// That matters for a WeLe agent who switches between Tamil and English inside +// one shift and should never have to touch a language menu. +// ============================================ +import { Tensor } from '@huggingface/transformers'; +import { getSTT, getTTS, supportsTTS } from './models.js'; +import logger from '../utils/logger.js'; + +export { LANGUAGES, warmup, speechStatus } from './models.js'; +export { Endpointer, warmupVad } from './vad.js'; + +const RATE = 16000; + +/** Languages we can both hear and speak. */ +const SPOKEN = new Set(['ta', 'en']); + +// Below this, trust the caller's preference over the detector. Short or noisy +// utterances — and code-mixed "Tanglish" especially — can land either side. +const DETECT_CONFIDENCE = Number(process.env.DETECT_CONFIDENCE ?? 0.6); + +let detectIds = null; + +/** + * Identify the spoken language in ONE decoder step. + * + * Passing no `language` to the pipeline does NOT auto-detect — Transformers.js + * logs "No language specified - defaulting to English" and transcribes Tamil + * as English, producing nonsense. Whisper does emit a language token right + * after <|startoftranscript|>, so we read that distribution directly. Measured + * ~700 ms, and 0.998 / 1.000 confidence on clean Tamil / English. + * + * Reuses the pipeline's own model and processor, so nothing loads twice. + */ +async function detectLanguage(audio) { + const stt = await getSTT(); + const tok = stt.tokenizer; + + if (!detectIds) { + const id = (t) => tok.encode(t, { add_special_tokens: false })[0]; + detectIds = { sot: id('<|startoftranscript|>'), langs: [...SPOKEN].map((c) => ({ code: c, id: id(`<|${c}|>`) })) }; + } + + const inputs = await stt.processor(audio); + const out = await stt.model({ + ...inputs, + decoder_input_ids: new Tensor('int64', BigInt64Array.from([BigInt(detectIds.sot)]), [1, 1]), + }); + + const { dims, data } = out.logits; + const row = data.slice((dims[1] - 1) * dims[2], dims[1] * dims[2]); + const scores = detectIds.langs.map((l) => Number(row[l.id])); + const max = Math.max(...scores); + const exp = scores.map((v) => Math.exp(v - max)); + const sum = exp.reduce((a, b) => a + b, 0); + const probs = exp.map((v) => v / sum); + const best = probs.indexOf(Math.max(...probs)); + + return { lang: detectIds.langs[best].code, confidence: probs[best] }; +} + +/** + * @param {Float32Array} audio mono @16 kHz in [-1, 1] + * @param {string} lang 'auto' | 'ta' | 'en' + * @param {string} prefer used when detection is unusable + */ +export async function transcribe(audio, lang = 'auto', prefer = 'ta') { + if (!audio || audio.length < RATE / 5) { // under 200 ms + return { text: '', lang: prefer, note: 'too short' }; + } + + const stt = await getSTT(); + const t0 = Date.now(); + + // Whisper must always be told a language — it never detects on its own here. + let used = lang; + let detected = null; + let confidence = null; + + if (lang === 'auto') { + try { + const d = await detectLanguage(audio); + detected = d.lang; + confidence = d.confidence; + used = d.confidence >= DETECT_CONFIDENCE ? d.lang : prefer; + if (used !== d.lang) { + logger.info(`language ID unsure (${d.lang} @ ${d.confidence.toFixed(2)}) — using preferred ${prefer}`); + } + } catch (e) { + logger.warn(`language ID failed (${e.message}) — using preferred ${prefer}`); + used = prefer; + } + } + + const result = await stt(audio, { task: 'transcribe', language: used, return_timestamps: false }); + return finish(result, used, audio, t0, detected, confidence); +} + +function finish(result, used, audio, t0, detected, confidence) { + const text = (result?.text || '').trim(); + const ms = Date.now() - t0; + const audioMs = Math.round((audio.length / RATE) * 1000); + logger.info(`🎤 STT ${used}${detected && detected !== used ? ` (heard ${detected})` : ''}: ${audioMs}ms → ${ms}ms → ${JSON.stringify(text.slice(0, 70))}`); + return { + text, lang: used, detected: detected || null, + confidence: confidence == null ? null : Number(confidence.toFixed(3)), + ms, audio_ms: audioMs, + }; +} + +/** + * Synthesise one piece of text. + * @returns {Promise<{audio: Float32Array, sampling_rate: number}>} + */ +export async function synthesize(text, lang = 'ta') { + const clean = (text || '').trim(); + if (!clean) return null; + + const use = supportsTTS(lang) ? lang : 'en'; + const tts = await getTTS(use); + + const t0 = Date.now(); + const out = await tts(clean); + const ms = Date.now() - t0; + const audioMs = (out.audio.length / out.sampling_rate) * 1000; + logger.debug(`🔈 TTS ${use}: ${clean.length} chars → ${ms}ms for ${audioMs.toFixed(0)}ms (RTF ${(ms / audioMs).toFixed(2)}x)`); + + return { audio: out.audio, sampling_rate: out.sampling_rate }; +} + +/** + * Split into speakable pieces. Short prompts reach audio sooner, and a sentence + * boundary is a clean place to be interrupted. + */ +export function sentences(text, max = 200) { + const out = []; + for (const raw of String(text || '').split(/(?<=[.!?।])\s+/)) { + let s = raw.trim(); + if (!s) continue; + while (s.length > max) { + const cut = s.lastIndexOf(' ', max); + out.push(s.slice(0, cut > 0 ? cut : max).trim()); + s = s.slice(cut > 0 ? cut : max).trim(); + } + if (s) out.push(s); + } + return out; +} + +/** + * Strip block markdown before speaking — tables and code read terribly aloud, + * and the visual blocks are already on screen. + */ +export function speakable(markdown = '') { + return markdown + .replace(/```[\s\S]*?```/g, ' ') + .replace(/^\s*\|.*\|\s*$/gm, ' ') + .replace(/^\s*[-*]\s+/gm, '') + .replace(/^#{1,6}\s*/gm, '') + .replace(/\*\*([^*]+)\*\*/g, '$1') + .replace(/`([^`]+)`/g, '$1') + .replace(/\[([^\]]+)\]\([^)]+\)/g, '$1') + .replace(/₹\s?([\d,.]+)/g, 'rupees $1') + .replace(/\s{2,}/g, ' ') + .trim(); +} diff --git a/src/speech/models.js b/src/speech/models.js new file mode 100644 index 0000000..f35ec40 --- /dev/null +++ b/src/speech/models.js @@ -0,0 +1,118 @@ +// ============================================ +// Speech models — all ONNX, all CPU, all in this Node process. +// +// There is no GPU and no Python. That is the whole point: the AWS host has +// neither, and a second service was one more thing to deploy and keep alive. +// +// VAD Silero 2 MB endpointing +// STT Whisper base ~80 MB Tamil + English + language detection +// TTS MMS-TTS VITS ~114 MB per language, feed-forward +// +// Measured on an i7-10850H, CPU only: +// TTS RTF 0.28x (3.5x faster than realtime) +// STT ~1.2 s for 4 s of audio +// +// Two findings worth keeping: +// +// * VITS is feed-forward. The earlier Parler-TTS attempt was autoregressive +// and ran at RTF ~5x — i.e. 5x SLOWER than realtime — which is why voice was +// unusable even on a GPU. Architecture mattered far more than hardware here. +// +// * int8 is a trap for a model this small: dynamic quantisation made TTS 5.7x +// SLOWER than fp32 (RTF 1.67x vs 0.28x) because the quantise/dequantise +// overhead dominates. We ship fp32 deliberately. +// ============================================ +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import { pipeline, env } from '@huggingface/transformers'; +import logger from '../utils/logger.js'; + +const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..'); + +// Hub downloads are cached here so a container restart does not re-fetch. +env.cacheDir = process.env.SPEECH_CACHE_DIR || path.join(ROOT, '.transformers-cache'); +// Tamil is loaded from a folder we exported ourselves — no public ONNX build +// of mms-tts-tam exists. See scripts/export-tamil-tts.py. +env.localModelPath = path.join(ROOT, 'assets/tts'); + +const STT_MODEL = process.env.STT_MODEL || 'onnx-community/whisper-base'; + +/** TTS voice per language. Tamil is local; English comes from the Hub. */ +const VOICES = { + ta: { id: 'mms-tts-tam', local: true }, + en: { id: 'Xenova/mms-tts-eng', local: false }, +}; + +export const LANGUAGES = [ + { code: 'auto', label: 'Auto-detect', native: 'Auto' }, + { code: 'ta', label: 'Tamil', native: 'தமிழ்' }, + { code: 'en', label: 'English', native: 'English' }, +]; + +const cache = new Map(); +let sttPromise = null; + +/** + * Models load on first use, not at boot. A CRM restart should not wait ~10 s + * for speech models that most sessions never touch. + */ +async function loadOnce(key, build) { + if (!cache.has(key)) { + const t0 = Date.now(); + cache.set(key, build().then((m) => { + logger.info(`🔊 loaded ${key} in ${((Date.now() - t0) / 1000).toFixed(1)}s`); + return m; + }).catch((e) => { + cache.delete(key); // let the next attempt retry + throw e; + })); + } + return cache.get(key); +} + +export async function getSTT() { + if (!sttPromise) { + sttPromise = loadOnce(STT_MODEL, () => + // q8 is the right call for Whisper — unlike VITS it is big enough that + // quantisation is a clear win. + pipeline('automatic-speech-recognition', STT_MODEL, { dtype: 'q8' }), + ).catch((e) => { sttPromise = null; throw e; }); + } + return sttPromise; +} + +export async function getTTS(lang = 'ta') { + const voice = VOICES[lang] || VOICES.ta; + const prev = env.allowRemoteModels; + try { + // Local folders must not be looked up on the Hub, and vice versa. + env.allowRemoteModels = !voice.local; + return await loadOnce(`tts:${voice.id}`, () => + pipeline('text-to-speech', voice.id, { dtype: 'fp32' }), + ); + } finally { + env.allowRemoteModels = prev; + } +} + +export const supportsTTS = (lang) => Boolean(VOICES[lang]); + +/** Warm the models the deployment actually expects to use. */ +export async function warmup(langs = ['ta', 'en']) { + try { + await getSTT(); + for (const l of langs) await getTTS(l); + logger.info('🔊 speech models warm'); + } catch (e) { + logger.warn(`speech warmup failed (will retry on first use): ${e.message}`); + } +} + +export function speechStatus() { + return { + stt_model: STT_MODEL, + tts_voices: Object.fromEntries(Object.entries(VOICES).map(([k, v]) => [k, v.id])), + loaded: [...cache.keys()], + languages: LANGUAGES, + }; +} diff --git a/src/speech/vad.js b/src/speech/vad.js new file mode 100644 index 0000000..1f8077a --- /dev/null +++ b/src/speech/vad.js @@ -0,0 +1,152 @@ +// ============================================ +// Endpointing — Silero VAD via onnxruntime-node. +// +// Deciding turn boundaries on the server rather than in the browser keeps the +// rule in one place for every future channel (a phone bridge has no +// AudioWorklet), and gives the server the signal it needs for barge-in: it has +// to know the user started talking while the assistant was still speaking. +// ============================================ +import path from 'node:path'; +import { fileURLToPath } from 'node:url'; +import fs from 'node:fs/promises'; +import ort from 'onnxruntime-node'; +import logger from '../utils/logger.js'; + +const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), '../..'); +const MODEL_URL = 'https://huggingface.co/onnx-community/silero-vad/resolve/main/onnx/model.onnx'; +const MODEL_PATH = path.join(process.env.SPEECH_CACHE_DIR || path.join(ROOT, '.transformers-cache'), 'silero-vad.onnx'); + +// Silero wants exactly 512 samples at 16 kHz (32 ms). The browser sends 40 ms +// chunks, so audio is buffered and drained in exact frames rather than forcing +// the client to match. +const FRAME = 512; +const RATE = 16000; +const FRAME_MS = (FRAME / RATE) * 1000; + +let sessionPromise = null; + +async function getSession() { + if (sessionPromise) return sessionPromise; + sessionPromise = (async () => { + try { + await fs.access(MODEL_PATH); + } catch { + logger.info('⬇️ fetching Silero VAD (2 MB)…'); + const res = await fetch(MODEL_URL); + if (!res.ok) throw new Error(`VAD download failed: ${res.status}`); + await fs.mkdir(path.dirname(MODEL_PATH), { recursive: true }); + await fs.writeFile(MODEL_PATH, Buffer.from(await res.arrayBuffer())); + } + const s = await ort.InferenceSession.create(MODEL_PATH); + logger.info('🎚️ Silero VAD ready'); + return s; + })().catch((e) => { sessionPromise = null; throw e; }); + return sessionPromise; +} + +export const vadOptions = { + threshold: Number(process.env.VAD_THRESHOLD ?? 0.5), + // Trailing silence that ends a turn. Too short truncates someone who pauses + // mid-sentence; too long makes the assistant feel sluggish. + silenceMs: Number(process.env.VAD_SILENCE_MS ?? 700), + // Ignore blips, so a cough or a door does not open a turn. + minSpeechMs: Number(process.env.VAD_MIN_SPEECH_MS ?? 250), + // Audio kept from BEFORE detection, so word onsets are not clipped. + prefixMs: Number(process.env.VAD_PREFIX_MS ?? 300), + maxUtteranceMs: Number(process.env.VAD_MAX_UTTERANCE_MS ?? 30000), +}; + +/** Streaming endpointer. One instance per connection. */ +export class Endpointer { + constructor(opts = {}) { + this.o = { ...vadOptions, ...opts }; + this.pending = new Float32Array(0); + this.prefixFrames = Math.max(1, Math.round(this.o.prefixMs / FRAME_MS)); + this.reset(); + } + + reset() { + this.speaking = false; + this.speechMs = 0; + this.silenceMs = 0; + this.buffer = []; + this.prefix = []; + // Silero is recurrent: this 2x1x128 state carries across frames and must + // be reset between turns or the model stays biased by the last utterance. + this.state = new ort.Tensor('float32', new Float32Array(2 * 1 * 128), [2, 1, 128]); + this.pending = new Float32Array(0); + } + + /** + * Feed float32 mono @16k. + * @returns {Promise<{utterances: Float32Array[], started: boolean}>} + * `started` flips the moment speech begins — that is the barge-in signal. + */ + async push(pcm) { + const session = await getSession(); + + const merged = new Float32Array(this.pending.length + pcm.length); + merged.set(this.pending); + merged.set(pcm, this.pending.length); + this.pending = merged; + + const utterances = []; + let started = false; + let offset = 0; + + while (this.pending.length - offset >= FRAME) { + const frame = this.pending.subarray(offset, offset + FRAME); + offset += FRAME; + + const out = await session.run({ + input: new ort.Tensor('float32', frame, [1, FRAME]), + sr: new ort.Tensor('int64', BigInt64Array.from([BigInt(RATE)]), []), + state: this.state, + }); + this.state = out.stateN ?? out.state_n ?? this.state; + const voiced = out.output.data[0] >= this.o.threshold; + + if (!this.speaking) { + this.prefix.push(Float32Array.from(frame)); + if (this.prefix.length > this.prefixFrames) this.prefix.shift(); + + if (voiced) { + this.speechMs += FRAME_MS; + if (this.speechMs >= this.o.minSpeechMs) { + this.speaking = true; + this.silenceMs = 0; + this.buffer = this.prefix; // open the turn with the pre-roll + this.prefix = []; + started = true; + } + } else { + this.speechMs = 0; + } + continue; + } + + this.buffer.push(Float32Array.from(frame)); + if (voiced) this.silenceMs = 0; + else this.silenceMs += FRAME_MS; + + const spokenMs = this.buffer.length * FRAME_MS; + if (this.silenceMs >= this.o.silenceMs || spokenMs >= this.o.maxUtteranceMs) { + utterances.push(concat(this.buffer)); + this.reset(); + } + } + + this.pending = this.pending.slice(offset); + return { utterances, started }; + } +} + +function concat(frames) { + const total = frames.reduce((n, f) => n + f.length, 0); + const out = new Float32Array(total); + let i = 0; + for (const f of frames) { out.set(f, i); i += f.length; } + return out; +} + +export const warmupVad = () => getSession().catch(() => {}); diff --git a/voice-service/.env.example b/voice-service/.env.example deleted file mode 100644 index d49617d..0000000 --- a/voice-service/.env.example +++ /dev/null @@ -1,22 +0,0 @@ -# Copy to .env and fill in HF_TOKEN. -# The AI4Bharat models are gated: sign in at huggingface.co, accept the terms on -# both model pages, then create a read token at huggingface.co/settings/tokens. -# HuggingFace token — required: the AI4Bharat models are gated repos. -HF_TOKEN= - -VOICE_HOST=127.0.0.1 -VOICE_PORT=4100 - -STT_MODEL=ai4bharat/indic-conformer-600m-multilingual -STT_DECODING=ctc -ENGLISH_MODEL=openai/whisper-small -TTS_MODEL=ai4bharat/indic-parler-tts - -VOICE_DEFAULT_LANG=ta -PRELOAD_ENGLISH=true -VOICE_WARMUP=true - -# Endpointing -VAD_SILENCE_MS=700 -VAD_MIN_SPEECH_MS=250 -VAD_PREFIX_MS=300 diff --git a/voice-service/README.md b/voice-service/README.md deleted file mode 100644 index f221953..0000000 --- a/voice-service/README.md +++ /dev/null @@ -1,105 +0,0 @@ -# WeLe Voice Service - -Speech in, speech out. This process holds the GPU models and nothing else — it -has no idea what the CRM is. Orchestration, auth and business logic stay in the -Node service, so **voice is a channel into the same agent**, not a parallel -system with its own brain. - -``` -browser ──audio──► node :4000 ──audio──► this :4100 ──► IndicConformer / Whisper - │ │ - └────────── same graph, agents, ───────────┘ - guardrails as text chat - │ -browser ◄──audio───── node ◄──audio──── this ◄── Indic Parler-TTS -``` - -## Models - -| Job | Model | Notes | -|---|---|---| -| Endpointing | Silero VAD | 512-sample frames @16 kHz, 300 ms pre-roll | -| Indic ASR | `ai4bharat/indic-conformer-600m-multilingual` | 22 Indian languages, CTC decoding | -| English ASR + language ID | `openai/whisper-small` | multilingual on purpose — the `.en` build cannot identify languages | -| TTS | `ai4bharat/indic-parler-tts` | 21 languages, streaming | - -**The AI4Bharat repos are gated.** Access is auto-approved, but the download -needs an authenticated account: sign in to huggingface.co, accept the terms on -both model pages, then put a read token in `.env` as `HF_TOKEN`. - -## Why two ASR models - -IndicConformer decodes *as* the language you name — it does not detect one, and -English is not among its 22 codes. Whisper covers English and can identify the -spoken language in a single decoder step. So the default mode is `auto`: - -``` -audio → Whisper mel + 1 decoder step → language ID - ├─ "en" → Whisper transcribes (mel already computed — no extra cost) - └─ Indic → IndicConformer with the detected code -``` - -Below **0.60** confidence the caller's preferred language wins instead of a coin -toss. That matters for Tanglish, where a short code-mixed sentence can honestly -land either side. - -## Setup - -```bash -python -m venv --system-site-packages .venv # reuses the system torch build -.venv/Scripts/python -m pip install -r requirements.txt -cp .env.example .env # add HF_TOKEN -``` - -The venv deliberately inherits system site-packages: torch is ~2.5 GB and -already installed with CUDA. Note that `parler-tts` pins `transformers==4.46.1` -**inside the venv only** — the system install is untouched. - -## Run - -```bash -npm run voice # from the parent directory -# or -.venv/Scripts/python -m app.server -``` - -First start downloads several GB and warms both models. `GET /health` reports -device, models, sample rate and current VRAM. - -## Protocol - -One WebSocket at `/ws/voice`, JSON control frames plus binary audio. - -| Direction | Message | -|---|---| -| → | binary — 16 kHz mono PCM16 mic frames | -| → | `{"type":"config","lang":"auto","prefer":"ta"}` | -| → | `{"type":"speak","text":"…","id":"…"}` | -| → | `{"type":"cancel"}` — stop speaking now | -| ← | `{"type":"speech_start"}` — VAD opened a turn (drives barge-in) | -| ← | `{"type":"transcript","text":…,"lang":…,"detected":…,"confidence":…}` | -| ← | `{"type":"audio_start","sample_rate":24000}` then binary float32 chunks | - -## Tuning - -| Env | Default | Effect | -|---|---|---| -| `VAD_SILENCE_MS` | 700 | trailing silence that ends a turn — lower feels snappier, truncates people who pause | -| `VAD_MIN_SPEECH_MS` | 250 | ignores coughs and door slams | -| `VAD_PREFIX_MS` | 300 | audio kept from before detection, so word onsets survive | -| `VOICE_DEFAULT_LANG` | `ta` | tiebreak when language ID is unsure | -| `PRELOAD_ENGLISH` | `true` | set `false` to load Whisper lazily if VRAM is tight | -| `STT_DECODING` | `ctc` | `rnnt` is more accurate but decodes autoregressively | - -## VRAM - -Roughly 4.7 GB of the 6 GB card with all three models resident. If that proves -too tight, `PRELOAD_ENGLISH=false` defers Whisper (~0.5 GB) until the first -English utterance. - -## Scripts - -```bash -.venv/Scripts/python probe_access.py # which repos the token can reach -.venv/Scripts/python probe_models.py # load, VRAM, time-to-first-audio -``` diff --git a/voice-service/app/config.py b/voice-service/app/config.py deleted file mode 100644 index c00cb34..0000000 --- a/voice-service/app/config.py +++ /dev/null @@ -1,82 +0,0 @@ -"""Voice service configuration. - -Deliberately small: this process does one job — turn audio into text and text -into audio. Everything about *what to say* lives in the Node agentic service. -""" -from __future__ import annotations - -import os - -import torch - - -def _int(name: str, default: int) -> int: - try: - return int(os.environ.get(name, default)) - except (TypeError, ValueError): - return default - - -def _flag(name: str, default: bool) -> bool: - return os.environ.get(name, str(default)).lower() in {"1", "true", "yes"} - - -class Settings: - host: str = os.environ.get("VOICE_HOST", "127.0.0.1") - port: int = _int("VOICE_PORT", 4100) - - # ── Models ─────────────────────────────────────────────────────────────── - # IndicConformer is a hybrid CTC + RNNT model. CTC decoding is used because - # it is a single forward pass — RNNT is more accurate but decodes - # autoregressively, and in a voice loop the latency costs more than the - # accuracy buys. - stt_model: str = os.environ.get("STT_MODEL", "ai4bharat/indic-conformer-600m-multilingual") - stt_decoding: str = os.environ.get("STT_DECODING", "ctc") # ctc | rnnt - - # English is not one of IndicConformer's 22 codes, and it cannot identify - # languages. Whisper covers both — multilingual, not the .en checkpoint, - # because language ID is what makes "auto" work. - english_model: str = os.environ.get("ENGLISH_MODEL", "openai/whisper-small") - preload_english: bool = _flag("PRELOAD_ENGLISH", True) - - # Used when auto-detection is not confident enough to overrule the user. - default_lang: str = os.environ.get("VOICE_DEFAULT_LANG", "ta") - - tts_model: str = os.environ.get("TTS_MODEL", "ai4bharat/indic-parler-tts") - - # ── Device / precision ─────────────────────────────────────────────────── - # float16 on CUDA: both models together are ~3 GB in half precision, which - # fits the 6 GB card with room for activations. float32 would not. - device: str = os.environ.get("VOICE_DEVICE", "cuda" if torch.cuda.is_available() else "cpu") - - @property - def dtype(self) -> torch.dtype: - return torch.float16 if self.device == "cuda" else torch.float32 - - # ── Audio ──────────────────────────────────────────────────────────────── - sample_rate_in: int = 16000 # what the browser worklet sends - # Indic Parler-TTS emits 44.1 kHz — verified from model.config.sampling_rate, - # not the 24 kHz the upstream Parler-TTS Mini uses. This is only a fallback; - # the real rate is read from the loaded model and sent to the browser, which - # configures its playback worklet from it. - sample_rate_out: int = _int("TTS_SAMPLE_RATE", 44100) - - # ── Endpointing (Silero VAD) ───────────────────────────────────────────── - # Silero operates on fixed 512-sample frames at 16 kHz (32 ms). - vad_frame: int = 512 - vad_threshold: float = float(os.environ.get("VAD_THRESHOLD", "0.5")) - # How much trailing silence ends a turn. Too short truncates people who - # pause mid-sentence; too long makes the assistant feel sluggish. - vad_silence_ms: int = _int("VAD_SILENCE_MS", 700) - # Ignore blips so a cough or a door does not open a turn. - vad_min_speech_ms: int = _int("VAD_MIN_SPEECH_MS", 250) - # Audio kept from *before* detected speech, so word onsets are not clipped. - vad_prefix_ms: int = _int("VAD_PREFIX_MS", 300) - vad_max_utterance_ms: int = _int("VAD_MAX_UTTERANCE_MS", 30000) - - # Warm the models at startup rather than on the first user turn — a cold - # CUDA graph on the first utterance costs several seconds. - warmup: bool = _flag("VOICE_WARMUP", True) - - -settings = Settings() diff --git a/voice-service/app/server.py b/voice-service/app/server.py deleted file mode 100644 index b8b5137..0000000 --- a/voice-service/app/server.py +++ /dev/null @@ -1,233 +0,0 @@ -"""Voice service — STT and TTS over one WebSocket. - -This process holds the GPU models and nothing else. It has no idea what the -CRM is: it receives audio and returns text, receives text and returns audio. -All orchestration, auth and business logic stay in the Node service, so voice -is just another channel into the same agent rather than a parallel system. - -Protocol (ws /ws/voice), JSON control + binary audio: - - client → server - binary 16 kHz mono PCM16 mic frames - {"type":"config","lang":"ta"} set the session language - {"type":"speak","text":"…","id":"…"} synthesise - {"type":"cancel"} stop speaking now (barge-in) - {"type":"reset"} clear the endpointer - - server → client - {"type":"ready", …} - {"type":"speech_start"} VAD opened a turn → caller ducks TTS - {"type":"transcript","text":…} a finished utterance - {"type":"audio_start","id":…,"sample_rate":24000} - binary float32 mono TTS chunks - {"type":"audio_end","id":…} -""" -from __future__ import annotations - -import asyncio -import json -import logging -import time - -import numpy as np -from fastapi import FastAPI, WebSocket, WebSocketDisconnect - -from .config import settings -from .stt import SUPPORTED, Transcriber -from .tts import Synthesizer, split_sentences -from .vad import Endpointer - -logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)-5s %(message)s", datefmt="%H:%M:%S") -logger = logging.getLogger("voice") - -app = FastAPI(title="WeLe Voice Service") - -stt = Transcriber() -tts = Synthesizer() - - -@app.on_event("startup") -async def _startup() -> None: - t0 = time.perf_counter() - logger.info("loading models on %s…", settings.device) - stt.load() - tts.load() - if settings.warmup: - stt.warmup() - tts.warmup() - logger.info("voice service ready in %.1fs", time.perf_counter() - t0) - - -@app.get("/health") -async def health() -> dict: - import torch - - return { - "ok": True, - "device": settings.device, - "stt_model": settings.stt_model, - "tts_model": settings.tts_model, - "tts_sample_rate": tts.sample_rate, - "languages": SUPPORTED, - "vram_gb": round(torch.cuda.memory_reserved() / 1e9, 2) if settings.device == "cuda" else None, - } - - -@app.get("/languages") -async def languages() -> dict: - return {"languages": SUPPORTED} - - -class Session: - """One browser connection. Owns its endpointer and its speaking state.""" - - def __init__(self, ws: WebSocket) -> None: - self.ws = ws - # "auto" detects per utterance; `prefer` breaks ties when the detector - # is unsure, which is common on short code-mixed ("Tanglish") speech. - self.lang = "auto" - self.prefer = settings.default_lang - self.endpointer = Endpointer() - self.endpointer.load() - self._speak_task: asyncio.Task | None = None - self._cancel = asyncio.Event() - - async def send(self, payload: dict) -> None: - await self.ws.send_text(json.dumps(payload, ensure_ascii=False)) - - # ── microphone ─────────────────────────────────────────────────────────── - async def on_audio(self, raw: bytes) -> None: - pcm = np.frombuffer(raw, dtype=np.int16).astype(np.float32) / 32768.0 - loop = asyncio.get_running_loop() - # VAD is a small torch model but still blocking; keep the event loop free. - utterances, started = await loop.run_in_executor(None, self.endpointer.push, pcm) - - if started: - # Barge-in: the user talking wins immediately. - await self.stop_speaking() - await self.send({"type": "speech_start"}) - - for utt in utterances: - request_lang = f"auto:{self.prefer}" if self.lang == "auto" else self.lang - result = await loop.run_in_executor(None, stt.transcribe, utt.audio, request_lang) - if result["text"]: - await self.send({"type": "transcript", **result, "truncated": utt.truncated}) - else: - await self.send({"type": "transcript_empty", "reason": result.get("note", "no speech")}) - - # ── speaking ───────────────────────────────────────────────────────────── - async def speak(self, text: str, msg_id: str, lang: str | None = None, voice: str | None = None) -> None: - await self.stop_speaking() - self._cancel.clear() - self._speak_task = asyncio.create_task(self._speak(text, msg_id, lang or self.lang, voice)) - - async def _speak(self, text: str, msg_id: str, lang: str, voice: str | None) -> None: - loop = asyncio.get_running_loop() - try: - await self.send({"type": "audio_start", "id": msg_id, "sample_rate": tts.sample_rate}) - - # Sentence at a time: shorter prompts reach first audio sooner, and - # a boundary is a clean place to stop when interrupted. - for sentence in split_sentences(text): - if self._cancel.is_set(): - break - queue: asyncio.Queue = asyncio.Queue(maxsize=32) - - def produce() -> None: - try: - for chunk in tts.stream(sentence, lang, voice): - if self._cancel.is_set(): - break - asyncio.run_coroutine_threadsafe(queue.put(chunk), loop).result() - finally: - asyncio.run_coroutine_threadsafe(queue.put(None), loop).result() - - loop.run_in_executor(None, produce) - while True: - chunk = await queue.get() - if chunk is None: - break - if self._cancel.is_set(): - continue # drain, don't send - await self.ws.send_bytes(np.asarray(chunk, dtype=np.float32).tobytes()) - - await self.send({"type": "audio_end", "id": msg_id, "cancelled": self._cancel.is_set()}) - except WebSocketDisconnect: - pass - except Exception as e: # noqa: BLE001 - logger.exception("synthesis failed") - try: - await self.send({"type": "error", "where": "tts", "message": str(e)[:200]}) - except Exception: # noqa: BLE001 - pass - - async def stop_speaking(self) -> None: - if self._speak_task and not self._speak_task.done(): - self._cancel.set() - try: - await asyncio.wait_for(self._speak_task, timeout=2.0) - except (asyncio.TimeoutError, asyncio.CancelledError): - self._speak_task.cancel() - self._speak_task = None - - -@app.websocket("/ws/voice") -async def voice(ws: WebSocket) -> None: - await ws.accept() - session = Session(ws) - await session.send({ - "type": "ready", - "sample_rate_in": settings.sample_rate_in, - "sample_rate_out": tts.sample_rate, - "languages": SUPPORTED, - }) - logger.info("voice session opened") - - try: - while True: - msg = await ws.receive() - if msg["type"] == "websocket.disconnect": - break - - if (raw := msg.get("bytes")) is not None: - await session.on_audio(raw) - continue - - if (text := msg.get("text")) is None: - continue - try: - data = json.loads(text) - except json.JSONDecodeError: - continue - - kind = data.get("type") - if kind == "config": - session.lang = data.get("lang", session.lang) - if session.lang != "auto": - session.prefer = session.lang - elif data.get("prefer"): - session.prefer = data["prefer"] - session.endpointer.reset() - await session.send({"type": "config_ok", "lang": session.lang, "prefer": session.prefer}) - elif kind == "speak": - await session.speak(data.get("text", ""), data.get("id", ""), data.get("lang"), data.get("voice")) - elif kind == "cancel": - await session.stop_speaking() - await session.send({"type": "cancelled"}) - elif kind == "reset": - session.endpointer.reset() - except WebSocketDisconnect: - pass - finally: - await session.stop_speaking() - logger.info("voice session closed") - - -def main() -> None: - import uvicorn - - uvicorn.run(app, host=settings.host, port=settings.port, log_level="info", ws_max_size=16 * 1024 * 1024) - - -if __name__ == "__main__": - main() diff --git a/voice-service/app/stt.py b/voice-service/app/stt.py deleted file mode 100644 index 0df99f2..0000000 --- a/voice-service/app/stt.py +++ /dev/null @@ -1,200 +0,0 @@ -"""Speech-to-text — AI4Bharat IndicConformer, with an English path and auto routing. - -Why two models rather than one: - -* **IndicConformer** decodes 22 Indian languages, and decodes *as* the language - you name — it does not detect. Handing it "ta" for English speech produces - Tamil-script nonsense. English is not one of its codes at all. -* **Whisper (multilingual)** covers English well and, usefully, can identify the - spoken language in a single decoder step. - -So the default mode is `auto`: Whisper identifies the language from the audio, -English is transcribed by Whisper directly (the mel is already computed, so -this costs nothing extra), and anything Indic is routed to IndicConformer, -which is far stronger on those languages than Whisper is. - -Code-mixed speech ("Tanglish") is the awkward case: language ID can land either -side of the fence on a short, mixed utterance. When Whisper is not confident, -the caller's preferred language wins rather than a coin toss — a Tamil-speaking -office gets Tamil, and the occasional English sentence still routes correctly -when it is clearly English. -""" -from __future__ import annotations - -import logging -import time - -import numpy as np -import torch - -from .config import settings - -logger = logging.getLogger(__name__) - -# The codes IndicConformer accepts. -INDIC_LANGS = { - "as", "bn", "brx", "doi", "gu", "hi", "kn", "kok", "ks", "mai", "ml", - "mni", "mr", "ne", "or", "pa", "sa", "sat", "sd", "ta", "te", "ur", -} - -# Offered by the UI. "auto" first: most WeLe agents switch language mid-shift. -SUPPORTED = [ - {"code": "auto", "label": "Auto-detect", "native": "Auto"}, - {"code": "ta", "label": "Tamil", "native": "தமிழ்"}, - {"code": "en", "label": "English", "native": "English"}, - {"code": "hi", "label": "Hindi", "native": "हिन्दी"}, - {"code": "te", "label": "Telugu", "native": "తెలుగు"}, - {"code": "kn", "label": "Kannada", "native": "ಕನ್ನಡ"}, - {"code": "ml", "label": "Malayalam", "native": "മലയാളം"}, - {"code": "mr", "label": "Marathi", "native": "मराठी"}, - {"code": "bn", "label": "Bengali", "native": "বাংলা"}, -] - -# Below this, trust the user's stated preference over the detector. -DETECT_CONFIDENCE = 0.60 - - -class Transcriber: - def __init__(self) -> None: - self._indic = None - self._whisper = None - self._whisper_proc = None - - # ── loading ────────────────────────────────────────────────────────────── - def load(self) -> None: - from transformers import AutoModel - - t0 = time.perf_counter() - # float32: the checkpoint ships custom remote code that assumes fp32. - # ~2.4 GB at 600M, which still leaves room for Whisper and the TTS model. - self._indic = AutoModel.from_pretrained(settings.stt_model, trust_remote_code=True) - self._indic.to(settings.device).eval() - logger.info("STT loaded (%s) in %.1fs", settings.stt_model, time.perf_counter() - t0) - - if settings.preload_english: - self._load_whisper() - - def _load_whisper(self) -> None: - """English + language ID. Multilingual on purpose — the .en checkpoint - cannot identify languages, which is the whole point of auto mode.""" - if self._whisper is not None: - return - from transformers import WhisperForConditionalGeneration, WhisperProcessor - - t0 = time.perf_counter() - self._whisper_proc = WhisperProcessor.from_pretrained(settings.english_model) - self._whisper = WhisperForConditionalGeneration.from_pretrained( - settings.english_model, torch_dtype=settings.dtype, - ).to(settings.device).eval() - logger.info("English/ID model loaded (%s) in %.1fs", settings.english_model, time.perf_counter() - t0) - - # ── inference ──────────────────────────────────────────────────────────── - @torch.inference_mode() - def transcribe(self, audio: np.ndarray, lang: str = "auto") -> dict: - """audio: float32 mono @16 kHz in [-1, 1]. - - `lang` may be an explicit code, or "auto" / "auto:ta" to detect with a - fallback preference. - """ - t0 = time.perf_counter() - if audio.size < settings.sample_rate_in // 5: # under 200 ms - return {"text": "", "lang": lang, "ms": 0, "note": "too short"} - - detected = None - confidence = None - - if lang.startswith("auto"): - prefer = lang.split(":", 1)[1] if ":" in lang else settings.default_lang - feats = self._features(audio) - detected, confidence = self._detect(feats) - - if confidence is not None and confidence < DETECT_CONFIDENCE: - logger.info("language ID low confidence (%s @ %.2f) — using preferred %s", - detected, confidence, prefer) - use = prefer - elif detected == "en" or detected in INDIC_LANGS: - use = detected - else: - # Whisper reported something we cannot decode (e.g. "nn" on - # noise). Fall back rather than fail. - use = prefer - - text = self._english(audio, feats=feats) if use == "en" else self._indic_decode(audio, use) - else: - use = lang - text = self._english(audio) if lang == "en" else self._indic_decode(audio, lang) - - ms = int((time.perf_counter() - t0) * 1000) - audio_ms = int(1000 * audio.size / settings.sample_rate_in) - logger.info("STT %s%s: %dms audio → %dms → %r", - use, f" (detected {detected} {confidence:.2f})" if confidence is not None else "", - audio_ms, ms, text[:80]) - - return { - "text": text.strip(), - "lang": use, - "detected": detected, - "confidence": round(confidence, 3) if confidence is not None else None, - "ms": ms, - "audio_ms": audio_ms, - } - - # ── internals ──────────────────────────────────────────────────────────── - def _features(self, audio: np.ndarray): - self._load_whisper() - return self._whisper_proc( - audio, sampling_rate=settings.sample_rate_in, return_tensors="pt", - ).input_features.to(settings.device, settings.dtype) - - def _detect(self, feats) -> tuple[str | None, float | None]: - """One decoder step: read the language-token distribution.""" - try: - tok = self._whisper_proc.tokenizer - sot = tok.convert_tokens_to_ids("<|startoftranscript|>") - start = torch.tensor([[sot]], device=settings.device) - logits = self._whisper(feats, decoder_input_ids=start).logits[:, -1] - - lang_ids, codes = [], [] - for code in {*INDIC_LANGS, "en"}: - tid = tok.convert_tokens_to_ids(f"<|{code}|>") - # Unknown languages map to the unk id; skip those. - if tid is not None and tid != tok.unk_token_id: - lang_ids.append(tid) - codes.append(code) - if not lang_ids: - return None, None - - probs = torch.softmax(logits[0, lang_ids].float(), dim=-1) - best = int(probs.argmax()) - return codes[best], float(probs[best]) - except Exception as e: # noqa: BLE001 — detection must never break a turn - logger.warning("language ID failed (%s) — falling back to preference", e) - return None, None - - def _indic_decode(self, audio: np.ndarray, lang: str) -> str: - if lang not in INDIC_LANGS: - logger.warning("unsupported STT language %r — using %s", lang, settings.default_lang) - lang = settings.default_lang - wav = torch.from_numpy(audio).unsqueeze(0).to(settings.device) # (1, N) - out = self._indic(wav, lang, settings.stt_decoding) - if isinstance(out, (list, tuple)): - return str(out[0]) if out else "" - return str(out) - - def _english(self, audio: np.ndarray, feats=None) -> str: - self._load_whisper() - if feats is None: - feats = self._features(audio) - ids = self._whisper.generate(feats, language="en", task="transcribe", max_new_tokens=180) - return self._whisper_proc.batch_decode(ids, skip_special_tokens=True)[0] - - def warmup(self) -> None: - """Silent pass so the first real utterance isn't paying for CUDA init.""" - try: - silence = np.zeros(settings.sample_rate_in, dtype=np.float32) - self.transcribe(silence, settings.default_lang) - if settings.preload_english: - self.transcribe(silence, "en") - logger.info("STT warm") - except Exception as e: # noqa: BLE001 - logger.warning("STT warmup skipped: %s", e) diff --git a/voice-service/app/tts.py b/voice-service/app/tts.py deleted file mode 100644 index 00e2736..0000000 --- a/voice-service/app/tts.py +++ /dev/null @@ -1,155 +0,0 @@ -"""Text-to-speech — AI4Bharat Indic Parler-TTS. - -Parler is prompted with *two* texts: the words to say, and a natural-language -description of how to say them (speaker, pace, room tone). The description is -what selects a voice — there is no speaker-id argument. - -Latency shape: Parler is autoregressive, so a whole paragraph costs whole- -paragraph time before the first sample exists. Two things fix that here: - -1. `ParlerTTSStreamer` yields audio while generation continues, so playback - starts after roughly the first `play_steps` frames rather than at the end. -2. The caller sends one *sentence* at a time. Short prompts reach their first - chunk sooner, and a sentence boundary is a natural place for the assistant - to be interrupted. -""" -from __future__ import annotations - -import logging -import re -import time -from threading import Thread -from typing import Iterator - -import numpy as np -import torch - -from .config import settings - -logger = logging.getLogger(__name__) - -# Voices recommended on the model card, per language. -VOICES = { - "ta": "Jaya", "hi": "Rohit", "te": "Prakash", "kn": "Suresh", - "ml": "Anjali", "mr": "Sanjay", "bn": "Arjun", "en": "Mary", -} - -_DESCRIPTION = ( - "{speaker} speaks in a warm, clear, professional tone at a natural pace. " - "The recording is very high quality with no background noise." -) - -# Split on sentence enders including the Devanagari danda, keeping it simple — -# this only needs to find safe places to cut, not parse language. -_SENTENCE_RX = re.compile(r"(?<=[.!?।॥])\s+") - - -def split_sentences(text: str, max_chars: int = 220) -> list[str]: - """Break text into TTS-sized pieces at sentence boundaries where possible.""" - out: list[str] = [] - for part in _SENTENCE_RX.split(text.strip()): - part = part.strip() - if not part: - continue - while len(part) > max_chars: - cut = part.rfind(" ", 0, max_chars) - if cut <= 0: - cut = max_chars - out.append(part[:cut].strip()) - part = part[cut:].strip() - if part: - out.append(part) - return out - - -class Synthesizer: - def __init__(self) -> None: - self._model = None - self._tok = None - self._desc_tok = None - self.sample_rate = settings.sample_rate_out - - def load(self) -> None: - from parler_tts import ParlerTTSForConditionalGeneration - from transformers import AutoTokenizer - - t0 = time.perf_counter() - self._model = ParlerTTSForConditionalGeneration.from_pretrained( - settings.tts_model, torch_dtype=settings.dtype, - ).to(settings.device).eval() - self._tok = AutoTokenizer.from_pretrained(settings.tts_model) - self._desc_tok = AutoTokenizer.from_pretrained(self._model.config.text_encoder._name_or_path) - self.sample_rate = int(self._model.config.sampling_rate) - logger.info( - "TTS loaded (%s) in %.1fs @ %d Hz", settings.tts_model, - time.perf_counter() - t0, self.sample_rate, - ) - - def _describe(self, lang: str, voice: str | None) -> str: - return _DESCRIPTION.format(speaker=voice or VOICES.get(lang, "Jaya")) - - @torch.inference_mode() - def stream(self, text: str, lang: str = "ta", voice: str | None = None) -> Iterator[np.ndarray]: - """Yield float32 mono chunks at `self.sample_rate` as they are generated.""" - text = (text or "").strip() - if not text: - return - - desc = self._desc_tok(self._describe(lang, voice), return_tensors="pt").to(settings.device) - prompt = self._tok(text, return_tensors="pt").to(settings.device) - - kwargs = dict( - input_ids=desc.input_ids, - attention_mask=desc.attention_mask, - prompt_input_ids=prompt.input_ids, - prompt_attention_mask=prompt.attention_mask, - ) - - streamer = self._make_streamer() - if streamer is None: - yield self._generate_blocking(kwargs, text) - return - - t0 = time.perf_counter() - # generate() blocks, so it runs on its own thread and the streamer is - # drained here as frames become available. - thread = Thread(target=self._model.generate, kwargs={**kwargs, "streamer": streamer}, daemon=True) - thread.start() - - first = True - for chunk in streamer: - if chunk is None or len(chunk) == 0: - continue - audio = chunk.astype(np.float32) if isinstance(chunk, np.ndarray) else chunk.cpu().numpy().astype(np.float32) - if first: - logger.info("TTS first chunk in %dms (%d chars)", int((time.perf_counter() - t0) * 1000), len(text)) - first = False - yield audio - thread.join(timeout=1.0) - - def _make_streamer(self): - try: - from parler_tts import ParlerTTSStreamer - except ImportError: - logger.warning("ParlerTTSStreamer unavailable — falling back to blocking synthesis") - return None - # play_steps trades first-chunk latency against per-chunk overhead; - # ~0.5 s of audio keeps playback continuous without stalling generation. - frame_rate = getattr(self._model.audio_encoder.config, "frame_rate", 86) - return ParlerTTSStreamer(self._model, device=settings.device, play_steps=int(frame_rate / 2)) - - @torch.inference_mode() - def _generate_blocking(self, kwargs: dict, text: str) -> np.ndarray: - t0 = time.perf_counter() - gen = self._model.generate(**kwargs) - audio = gen.cpu().numpy().squeeze().astype(np.float32) - logger.info("TTS (blocking) %d chars in %dms", len(text), int((time.perf_counter() - t0) * 1000)) - return audio - - def warmup(self) -> None: - try: - for _ in self.stream("வணக்கம்", "ta"): - break - logger.info("TTS warm") - except Exception as e: # noqa: BLE001 - logger.warning("TTS warmup skipped: %s", e) diff --git a/voice-service/app/vad.py b/voice-service/app/vad.py deleted file mode 100644 index c9c8ae7..0000000 --- a/voice-service/app/vad.py +++ /dev/null @@ -1,127 +0,0 @@ -"""Endpointing with Silero VAD. - -Turn boundaries are decided here rather than in the browser for two reasons: -the same decision then applies to every future channel (a phone bridge has no -AudioWorklet), and barge-in needs the server to know someone started talking -while the assistant was still speaking. -""" -from __future__ import annotations - -import logging -from collections import deque -from dataclasses import dataclass, field - -import numpy as np -import torch - -from .config import settings - -logger = logging.getLogger(__name__) - - -@dataclass -class Utterance: - audio: np.ndarray # float32 mono @16k, in [-1, 1] - duration_ms: int - truncated: bool = False # hit the max-length guard rather than silence - - -@dataclass -class VADState: - speaking: bool = False - speech_ms: int = 0 - silence_ms: int = 0 - buffer: list[np.ndarray] = field(default_factory=list) - - -class Endpointer: - """Streaming VAD that emits one Utterance per detected turn. - - Silero wants exactly 512 samples at 16 kHz, but the browser sends 40 ms - (640-sample) chunks. Rather than force the client to match, incoming audio - is accumulated and drained in exact frames. - """ - - def __init__(self) -> None: - self._model = None - self._pending = np.zeros(0, dtype=np.float32) - self.state = VADState() - # Pre-roll: speech is only *detected* a frame or two in, so without a - # prefix the first phoneme is already gone by the time we start saving. - prefix_frames = max(1, (settings.vad_prefix_ms * settings.sample_rate_in) // (1000 * settings.vad_frame)) - self._prefix: deque[np.ndarray] = deque(maxlen=prefix_frames) - - def load(self) -> None: - from silero_vad import load_silero_vad - - self._model = load_silero_vad() - logger.info("silero VAD loaded") - - def reset(self) -> None: - self.state = VADState() - self._pending = np.zeros(0, dtype=np.float32) - self._prefix.clear() - if self._model is not None: - self._model.reset_states() - - @property - def is_speaking(self) -> bool: - return self.state.speaking - - def push(self, pcm: np.ndarray) -> tuple[list[Utterance], bool]: - """Feed float32 audio. - - Returns (completed utterances, speech_started_this_call). The second - value drives barge-in: the caller cuts TTS playback the moment it flips. - """ - assert self._model is not None, "call load() first" - - self._pending = np.concatenate([self._pending, pcm]) if self._pending.size else pcm - frame = settings.vad_frame - frame_ms = int(1000 * frame / settings.sample_rate_in) - - done: list[Utterance] = [] - started = False - - while self._pending.size >= frame: - chunk = self._pending[:frame] - self._pending = self._pending[frame:] - - with torch.no_grad(): - prob = float(self._model(torch.from_numpy(chunk), settings.sample_rate_in).item()) - - voiced = prob >= settings.vad_threshold - st = self.state - - if not st.speaking: - self._prefix.append(chunk) - if voiced: - st.speech_ms += frame_ms - if st.speech_ms >= settings.vad_min_speech_ms: - # Commit: open the turn with the pre-roll included. - st.speaking = True - st.silence_ms = 0 - st.buffer = list(self._prefix) - self._prefix.clear() - started = True - else: - st.speech_ms = 0 - continue - - # Speaking. - st.buffer.append(chunk) - if voiced: - st.silence_ms = 0 - else: - st.silence_ms += frame_ms - - spoken_ms = len(st.buffer) * frame_ms - ended = st.silence_ms >= settings.vad_silence_ms - too_long = spoken_ms >= settings.vad_max_utterance_ms - - if ended or too_long: - audio = np.concatenate(st.buffer) - done.append(Utterance(audio=audio, duration_ms=spoken_ms, truncated=too_long and not ended)) - self.reset() - - return done, started diff --git a/voice-service/bench.py b/voice-service/bench.py deleted file mode 100644 index d62c7d7..0000000 --- a/voice-service/bench.py +++ /dev/null @@ -1,95 +0,0 @@ -"""Honest latency benchmark: warm up first, then time repeated runs. - -The first CUDA generation pays for kernel autotuning and cache allocation, so a -single cold measurement makes any model look far worse than it is in service. -""" -import logging -import time -from threading import Thread - -import numpy as np -import torch - -logging.basicConfig(level=logging.INFO, format="%(message)s") -log = logging.getLogger("bench") -DEV = "cuda" if torch.cuda.is_available() else "cpu" - -log.info("device=%s gpu=%s", DEV, torch.cuda.get_device_name(0) if DEV == "cuda" else "-") - -# ── STT ────────────────────────────────────────────────────────────────────── -from transformers import AutoModel - -stt = AutoModel.from_pretrained("ai4bharat/indic-conformer-600m-multilingual", trust_remote_code=True) -stt = stt.to(DEV).eval() - -# Where does it actually run? A model wrapping ONNX ignores .to(cuda). -params = list(stt.parameters()) -log.info("STT param device: %s (%d tensors)", params[0].device if params else "NO TORCH PARAMS", len(params)) -log.info("STT type: %s", type(stt).__name__) - -wav = torch.from_numpy((np.random.randn(16000 * 4) * 0.02).astype(np.float32)).unsqueeze(0).to(DEV) -with torch.inference_mode(): - stt(wav, "ta", "ctc") # warmup -times = [] -for _ in range(3): - t0 = time.perf_counter() - with torch.inference_mode(): - stt(wav, "ta", "ctc") - times.append((time.perf_counter() - t0) * 1000) -log.info("STT 4000 ms audio → %.0f / %.0f / %.0f ms (RTF %.2fx)", - *times, (sum(times) / len(times)) / 4000) - -# ── TTS ────────────────────────────────────────────────────────────────────── -from parler_tts import ParlerTTSForConditionalGeneration, ParlerTTSStreamer -from transformers import AutoTokenizer - -dtype = torch.float16 if DEV == "cuda" else torch.float32 -tts = ParlerTTSForConditionalGeneration.from_pretrained("ai4bharat/indic-parler-tts", torch_dtype=dtype).to(DEV).eval() -tok = AutoTokenizer.from_pretrained("ai4bharat/indic-parler-tts") -dtok = AutoTokenizer.from_pretrained(tts.config.text_encoder._name_or_path) -SR = tts.config.sampling_rate -log.info("TTS sampling_rate=%d frame_rate=%s", SR, getattr(tts.audio_encoder.config, "frame_rate", "?")) - -desc = "Jaya speaks in a warm, clear, professional tone at a natural pace. The recording is very high quality with no background noise." -d = dtok(desc, return_tensors="pt").to(DEV) - - -def run(text, stream=True): - p = tok(text, return_tensors="pt").to(DEV) - kw = dict(input_ids=d.input_ids, attention_mask=d.attention_mask, - prompt_input_ids=p.input_ids, prompt_attention_mask=p.attention_mask) - t0 = time.perf_counter() - if stream: - fr = int(getattr(tts.audio_encoder.config, "frame_rate", 86) / 2) - s = ParlerTTSStreamer(tts, device=DEV, play_steps=fr) - Thread(target=tts.generate, kwargs={**kw, "streamer": s}, daemon=True).start() - first, n = None, 0 - for c in s: - if c is None or len(c) == 0: - continue - if first is None: - first = (time.perf_counter() - t0) * 1000 - n += len(c) - else: - with torch.inference_mode(): - g = tts.generate(**kw) - n = g.shape[-1] - first = None - total = (time.perf_counter() - t0) * 1000 - return first, total, 1000 * n / SR - - -SHORT = "மூவாயிரம் நானூறு லீட்கள் உள்ளன." -LONG = "புதிய லீட்கள் மூவாயிரம் நானூற்று இருபத்தேழு. இதில் எழுபத்தாறு சதவீதம் இன்னும் தொடர்பு கொள்ளப்படவில்லை." - -run(SHORT) # warmup -log.info("") -for label, text in (("short", SHORT), ("long", LONG)): - first, total, audio = run(text) - log.info("TTS %-5s %2d chars → first %.0f ms | total %.0f ms | audio %.0f ms | RTF %.2fx", - label, len(text), first or -1, total, audio, total / max(audio, 1)) - -if DEV == "cuda": - log.info("\nVRAM peak reserved: %.2f GB of %.1f GB", - torch.cuda.max_memory_reserved() / 1e9, - torch.cuda.get_device_properties(0).total_memory / 1e9) diff --git a/voice-service/probe_access.py b/voice-service/probe_access.py deleted file mode 100644 index 37b28b6..0000000 --- a/voice-service/probe_access.py +++ /dev/null @@ -1,47 +0,0 @@ -"""Which repos can we actually DOWNLOAD from? - -`model_info` succeeds on a gated repo you have not been granted, so it is not a -usable test. Fetching a real file is. -""" -import os - -from huggingface_hub import hf_hub_download - -CANDIDATES = [ - ("ai4bharat/indic-conformer-600m-multilingual", "Indic ASR (22 languages)"), - ("ai4bharat/indic-parler-tts", "Indic TTS (21 languages)"), - ("openai/whisper-small", "English ASR + language ID"), - ("ai4bharat/indic-parler-tts-pretrained", "TTS base (fallback)"), - ("ai4bharat/indicconformer_stt_ta_hybrid_rnnt_large", "Tamil-only ASR (fallback)"), -] - -token = os.environ.get("HF_TOKEN") or os.environ.get("HUGGING_FACE_HUB_TOKEN") -print(f"token present: {bool(token)}\n") -print(f"{'repo':52} {'what it is':30} status") -print("-" * 108) - -blocked = [] -for repo, what in CANDIDATES: - try: - hf_hub_download(repo_id=repo, filename="config.json", token=token) - print(f"{repo:52} {what:30} DOWNLOADABLE") - except Exception as e: # noqa: BLE001 - msg = str(e) - if "not in the authorized list" in msg or "403" in msg: - print(f"{repo:52} {what:30} NEEDS ACCESS — click 'Agree' on the model page") - blocked.append(repo) - elif "401" in msg or "restricted" in msg: - print(f"{repo:52} {what:30} NOT AUTHENTICATED") - blocked.append(repo) - elif "404" in msg or "EntryNotFound" in msg: - # No config.json at the root, but the repo itself is reachable. - print(f"{repo:52} {what:30} reachable (no config.json)") - else: - print(f"{repo:52} {what:30} ERROR {msg[:34]}") - -if blocked: - print("\nGrant access here (sign in, click 'Agree and access repository'):") - for repo in blocked: - print(f" https://huggingface.co/{repo}") -else: - print("\nAll required models are downloadable.") diff --git a/voice-service/probe_models.py b/voice-service/probe_models.py deleted file mode 100644 index 118d940..0000000 --- a/voice-service/probe_models.py +++ /dev/null @@ -1,82 +0,0 @@ -"""De-risk before building around these models: do they load, fit, and run fast enough?""" -import logging -import time - -import numpy as np -import torch - -logging.basicConfig(level=logging.INFO, format="%(message)s") -log = logging.getLogger("probe") - -DEV = "cuda" if torch.cuda.is_available() else "cpu" -def vram(tag): - if DEV == "cuda": - log.info(" VRAM %-10s alloc %.2f GB | reserved %.2f GB", tag, - torch.cuda.memory_allocated() / 1e9, torch.cuda.memory_reserved() / 1e9) - -log.info("device=%s", DEV) -if DEV == "cuda": - log.info("gpu=%s total=%.1f GB", torch.cuda.get_device_name(0), - torch.cuda.get_device_properties(0).total_memory / 1e9) - -# ── STT ────────────────────────────────────────────────────────────────────── -log.info("\n[1/2] loading IndicConformer…") -t0 = time.perf_counter() -from transformers import AutoModel -stt = AutoModel.from_pretrained("ai4bharat/indic-conformer-600m-multilingual", trust_remote_code=True) -stt = stt.to(DEV).eval() -log.info(" loaded in %.1fs", time.perf_counter() - t0) -vram("after STT") - -# 3 s of quiet noise — we only care that a forward pass runs and how long it takes. -wav = torch.from_numpy((np.random.randn(16000 * 3) * 0.01).astype(np.float32)).unsqueeze(0).to(DEV) -for i in range(2): - t0 = time.perf_counter() - with torch.inference_mode(): - out = stt(wav, "ta", "ctc") - log.info(" pass %d: %.0f ms -> %r", i + 1, (time.perf_counter() - t0) * 1000, str(out)[:60]) - -# ── TTS ────────────────────────────────────────────────────────────────────── -log.info("\n[2/2] loading Indic Parler-TTS…") -t0 = time.perf_counter() -from parler_tts import ParlerTTSForConditionalGeneration, ParlerTTSStreamer -from transformers import AutoTokenizer - -dtype = torch.float16 if DEV == "cuda" else torch.float32 -tts = ParlerTTSForConditionalGeneration.from_pretrained("ai4bharat/indic-parler-tts", torch_dtype=dtype).to(DEV).eval() -tok = AutoTokenizer.from_pretrained("ai4bharat/indic-parler-tts") -dtok = AutoTokenizer.from_pretrained(tts.config.text_encoder._name_or_path) -log.info(" loaded in %.1fs sr=%d", time.perf_counter() - t0, tts.config.sampling_rate) -vram("after TTS") - -desc = "Jaya speaks in a warm, clear, professional tone at a natural pace. The recording is very high quality with no background noise." -prompt = "உங்கள் புதிய லீட்கள் மூன்று ஆயிரம் நானூறு." - -d = dtok(desc, return_tensors="pt").to(DEV) -p = tok(prompt, return_tensors="pt").to(DEV) -kw = dict(input_ids=d.input_ids, attention_mask=d.attention_mask, - prompt_input_ids=p.input_ids, prompt_attention_mask=p.attention_mask) - -# Streaming: what the user actually experiences is time-to-first-audio. -frame_rate = getattr(tts.audio_encoder.config, "frame_rate", 86) -streamer = ParlerTTSStreamer(tts, device=DEV, play_steps=int(frame_rate / 2)) -from threading import Thread -t0 = time.perf_counter() -Thread(target=tts.generate, kwargs={**kw, "streamer": streamer}, daemon=True).start() - -first_ms, total = None, 0 -for chunk in streamer: - if chunk is None or len(chunk) == 0: - continue - if first_ms is None: - first_ms = (time.perf_counter() - t0) * 1000 - total += len(chunk) -gen_ms = (time.perf_counter() - t0) * 1000 -audio_ms = 1000 * total / tts.config.sampling_rate - -log.info(" time to FIRST audio : %.0f ms", first_ms or -1) -log.info(" full generation : %.0f ms for %.0f ms of audio", gen_ms, audio_ms) -log.info(" realtime factor : %.2fx (<1 means faster than realtime)", gen_ms / max(audio_ms, 1)) -vram("peak") -if DEV == "cuda": - log.info(" peak reserved: %.2f GB", torch.cuda.max_memory_reserved() / 1e9) diff --git a/voice-service/requirements.txt b/voice-service/requirements.txt deleted file mode 100644 index 3355736..0000000 --- a/voice-service/requirements.txt +++ /dev/null @@ -1,11 +0,0 @@ -# Torch / transformers come from the system site-packages (torch 2.5.1+cu121). -# Only what the voice pipeline adds on top lives here. -fastapi>=0.115 -uvicorn[standard]>=0.30 -websockets>=12.0 -soundfile>=0.13 -numpy>=1.26 -scipy>=1.10 -sentencepiece>=0.2 -# Parler-TTS is not published on PyPI; the Indic model needs this fork-compatible package. -git+https://github.com/huggingface/parler-tts.git