WeLe Agentic AI: LangGraph multi-agent CRM assistant with voice
This commit is contained in:
@@ -0,0 +1,82 @@
|
||||
"""Voice service configuration.
|
||||
|
||||
Deliberately small: this process does one job — turn audio into text and text
|
||||
into audio. Everything about *what to say* lives in the Node agentic service.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
|
||||
import torch
|
||||
|
||||
|
||||
def _int(name: str, default: int) -> int:
|
||||
try:
|
||||
return int(os.environ.get(name, default))
|
||||
except (TypeError, ValueError):
|
||||
return default
|
||||
|
||||
|
||||
def _flag(name: str, default: bool) -> bool:
|
||||
return os.environ.get(name, str(default)).lower() in {"1", "true", "yes"}
|
||||
|
||||
|
||||
class Settings:
|
||||
host: str = os.environ.get("VOICE_HOST", "127.0.0.1")
|
||||
port: int = _int("VOICE_PORT", 4100)
|
||||
|
||||
# ── Models ───────────────────────────────────────────────────────────────
|
||||
# IndicConformer is a hybrid CTC + RNNT model. CTC decoding is used because
|
||||
# it is a single forward pass — RNNT is more accurate but decodes
|
||||
# autoregressively, and in a voice loop the latency costs more than the
|
||||
# accuracy buys.
|
||||
stt_model: str = os.environ.get("STT_MODEL", "ai4bharat/indic-conformer-600m-multilingual")
|
||||
stt_decoding: str = os.environ.get("STT_DECODING", "ctc") # ctc | rnnt
|
||||
|
||||
# English is not one of IndicConformer's 22 codes, and it cannot identify
|
||||
# languages. Whisper covers both — multilingual, not the .en checkpoint,
|
||||
# because language ID is what makes "auto" work.
|
||||
english_model: str = os.environ.get("ENGLISH_MODEL", "openai/whisper-small")
|
||||
preload_english: bool = _flag("PRELOAD_ENGLISH", True)
|
||||
|
||||
# Used when auto-detection is not confident enough to overrule the user.
|
||||
default_lang: str = os.environ.get("VOICE_DEFAULT_LANG", "ta")
|
||||
|
||||
tts_model: str = os.environ.get("TTS_MODEL", "ai4bharat/indic-parler-tts")
|
||||
|
||||
# ── Device / precision ───────────────────────────────────────────────────
|
||||
# float16 on CUDA: both models together are ~3 GB in half precision, which
|
||||
# fits the 6 GB card with room for activations. float32 would not.
|
||||
device: str = os.environ.get("VOICE_DEVICE", "cuda" if torch.cuda.is_available() else "cpu")
|
||||
|
||||
@property
|
||||
def dtype(self) -> torch.dtype:
|
||||
return torch.float16 if self.device == "cuda" else torch.float32
|
||||
|
||||
# ── Audio ────────────────────────────────────────────────────────────────
|
||||
sample_rate_in: int = 16000 # what the browser worklet sends
|
||||
# Indic Parler-TTS emits 44.1 kHz — verified from model.config.sampling_rate,
|
||||
# not the 24 kHz the upstream Parler-TTS Mini uses. This is only a fallback;
|
||||
# the real rate is read from the loaded model and sent to the browser, which
|
||||
# configures its playback worklet from it.
|
||||
sample_rate_out: int = _int("TTS_SAMPLE_RATE", 44100)
|
||||
|
||||
# ── Endpointing (Silero VAD) ─────────────────────────────────────────────
|
||||
# Silero operates on fixed 512-sample frames at 16 kHz (32 ms).
|
||||
vad_frame: int = 512
|
||||
vad_threshold: float = float(os.environ.get("VAD_THRESHOLD", "0.5"))
|
||||
# How much trailing silence ends a turn. Too short truncates people who
|
||||
# pause mid-sentence; too long makes the assistant feel sluggish.
|
||||
vad_silence_ms: int = _int("VAD_SILENCE_MS", 700)
|
||||
# Ignore blips so a cough or a door does not open a turn.
|
||||
vad_min_speech_ms: int = _int("VAD_MIN_SPEECH_MS", 250)
|
||||
# Audio kept from *before* detected speech, so word onsets are not clipped.
|
||||
vad_prefix_ms: int = _int("VAD_PREFIX_MS", 300)
|
||||
vad_max_utterance_ms: int = _int("VAD_MAX_UTTERANCE_MS", 30000)
|
||||
|
||||
# Warm the models at startup rather than on the first user turn — a cold
|
||||
# CUDA graph on the first utterance costs several seconds.
|
||||
warmup: bool = _flag("VOICE_WARMUP", True)
|
||||
|
||||
|
||||
settings = Settings()
|
||||
Reference in New Issue
Block a user