WeLe Agentic AI: LangGraph multi-agent CRM assistant with voice

This commit is contained in:
2026-08-28 02:16:03 +05:30
commit 105e58e02a
69 changed files with 11501 additions and 0 deletions
+82
View File
@@ -0,0 +1,82 @@
"""Voice service configuration.
Deliberately small: this process does one job — turn audio into text and text
into audio. Everything about *what to say* lives in the Node agentic service.
"""
from __future__ import annotations
import os
import torch
def _int(name: str, default: int) -> int:
try:
return int(os.environ.get(name, default))
except (TypeError, ValueError):
return default
def _flag(name: str, default: bool) -> bool:
return os.environ.get(name, str(default)).lower() in {"1", "true", "yes"}
class Settings:
host: str = os.environ.get("VOICE_HOST", "127.0.0.1")
port: int = _int("VOICE_PORT", 4100)
# ── Models ───────────────────────────────────────────────────────────────
# IndicConformer is a hybrid CTC + RNNT model. CTC decoding is used because
# it is a single forward pass — RNNT is more accurate but decodes
# autoregressively, and in a voice loop the latency costs more than the
# accuracy buys.
stt_model: str = os.environ.get("STT_MODEL", "ai4bharat/indic-conformer-600m-multilingual")
stt_decoding: str = os.environ.get("STT_DECODING", "ctc") # ctc | rnnt
# English is not one of IndicConformer's 22 codes, and it cannot identify
# languages. Whisper covers both — multilingual, not the .en checkpoint,
# because language ID is what makes "auto" work.
english_model: str = os.environ.get("ENGLISH_MODEL", "openai/whisper-small")
preload_english: bool = _flag("PRELOAD_ENGLISH", True)
# Used when auto-detection is not confident enough to overrule the user.
default_lang: str = os.environ.get("VOICE_DEFAULT_LANG", "ta")
tts_model: str = os.environ.get("TTS_MODEL", "ai4bharat/indic-parler-tts")
# ── Device / precision ───────────────────────────────────────────────────
# float16 on CUDA: both models together are ~3 GB in half precision, which
# fits the 6 GB card with room for activations. float32 would not.
device: str = os.environ.get("VOICE_DEVICE", "cuda" if torch.cuda.is_available() else "cpu")
@property
def dtype(self) -> torch.dtype:
return torch.float16 if self.device == "cuda" else torch.float32
# ── Audio ────────────────────────────────────────────────────────────────
sample_rate_in: int = 16000 # what the browser worklet sends
# Indic Parler-TTS emits 44.1 kHz — verified from model.config.sampling_rate,
# not the 24 kHz the upstream Parler-TTS Mini uses. This is only a fallback;
# the real rate is read from the loaded model and sent to the browser, which
# configures its playback worklet from it.
sample_rate_out: int = _int("TTS_SAMPLE_RATE", 44100)
# ── Endpointing (Silero VAD) ─────────────────────────────────────────────
# Silero operates on fixed 512-sample frames at 16 kHz (32 ms).
vad_frame: int = 512
vad_threshold: float = float(os.environ.get("VAD_THRESHOLD", "0.5"))
# How much trailing silence ends a turn. Too short truncates people who
# pause mid-sentence; too long makes the assistant feel sluggish.
vad_silence_ms: int = _int("VAD_SILENCE_MS", 700)
# Ignore blips so a cough or a door does not open a turn.
vad_min_speech_ms: int = _int("VAD_MIN_SPEECH_MS", 250)
# Audio kept from *before* detected speech, so word onsets are not clipped.
vad_prefix_ms: int = _int("VAD_PREFIX_MS", 300)
vad_max_utterance_ms: int = _int("VAD_MAX_UTTERANCE_MS", 30000)
# Warm the models at startup rather than on the first user turn — a cold
# CUDA graph on the first utterance costs several seconds.
warmup: bool = _flag("VOICE_WARMUP", True)
settings = Settings()