"""Voice service configuration. Deliberately small: this process does one job — turn audio into text and text into audio. Everything about *what to say* lives in the Node agentic service. """ from __future__ import annotations import os import torch def _int(name: str, default: int) -> int: try: return int(os.environ.get(name, default)) except (TypeError, ValueError): return default def _flag(name: str, default: bool) -> bool: return os.environ.get(name, str(default)).lower() in {"1", "true", "yes"} class Settings: host: str = os.environ.get("VOICE_HOST", "127.0.0.1") port: int = _int("VOICE_PORT", 4100) # ── Models ─────────────────────────────────────────────────────────────── # IndicConformer is a hybrid CTC + RNNT model. CTC decoding is used because # it is a single forward pass — RNNT is more accurate but decodes # autoregressively, and in a voice loop the latency costs more than the # accuracy buys. stt_model: str = os.environ.get("STT_MODEL", "ai4bharat/indic-conformer-600m-multilingual") stt_decoding: str = os.environ.get("STT_DECODING", "ctc") # ctc | rnnt # English is not one of IndicConformer's 22 codes, and it cannot identify # languages. Whisper covers both — multilingual, not the .en checkpoint, # because language ID is what makes "auto" work. english_model: str = os.environ.get("ENGLISH_MODEL", "openai/whisper-small") preload_english: bool = _flag("PRELOAD_ENGLISH", True) # Used when auto-detection is not confident enough to overrule the user. default_lang: str = os.environ.get("VOICE_DEFAULT_LANG", "ta") tts_model: str = os.environ.get("TTS_MODEL", "ai4bharat/indic-parler-tts") # ── Device / precision ─────────────────────────────────────────────────── # float16 on CUDA: both models together are ~3 GB in half precision, which # fits the 6 GB card with room for activations. float32 would not. device: str = os.environ.get("VOICE_DEVICE", "cuda" if torch.cuda.is_available() else "cpu") @property def dtype(self) -> torch.dtype: return torch.float16 if self.device == "cuda" else torch.float32 # ── Audio ──────────────────────────────────────────────────────────────── sample_rate_in: int = 16000 # what the browser worklet sends # Indic Parler-TTS emits 44.1 kHz — verified from model.config.sampling_rate, # not the 24 kHz the upstream Parler-TTS Mini uses. This is only a fallback; # the real rate is read from the loaded model and sent to the browser, which # configures its playback worklet from it. sample_rate_out: int = _int("TTS_SAMPLE_RATE", 44100) # ── Endpointing (Silero VAD) ───────────────────────────────────────────── # Silero operates on fixed 512-sample frames at 16 kHz (32 ms). vad_frame: int = 512 vad_threshold: float = float(os.environ.get("VAD_THRESHOLD", "0.5")) # How much trailing silence ends a turn. Too short truncates people who # pause mid-sentence; too long makes the assistant feel sluggish. vad_silence_ms: int = _int("VAD_SILENCE_MS", 700) # Ignore blips so a cough or a door does not open a turn. vad_min_speech_ms: int = _int("VAD_MIN_SPEECH_MS", 250) # Audio kept from *before* detected speech, so word onsets are not clipped. vad_prefix_ms: int = _int("VAD_PREFIX_MS", 300) vad_max_utterance_ms: int = _int("VAD_MAX_UTTERANCE_MS", 30000) # Warm the models at startup rather than on the first user turn — a cold # CUDA graph on the first utterance costs several seconds. warmup: bool = _flag("VOICE_WARMUP", True) settings = Settings()