83 lines
4.0 KiB
Python
83 lines
4.0 KiB
Python
"""Voice service configuration.
|
|
|
|
Deliberately small: this process does one job — turn audio into text and text
|
|
into audio. Everything about *what to say* lives in the Node agentic service.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
|
|
import torch
|
|
|
|
|
|
def _int(name: str, default: int) -> int:
|
|
try:
|
|
return int(os.environ.get(name, default))
|
|
except (TypeError, ValueError):
|
|
return default
|
|
|
|
|
|
def _flag(name: str, default: bool) -> bool:
|
|
return os.environ.get(name, str(default)).lower() in {"1", "true", "yes"}
|
|
|
|
|
|
class Settings:
|
|
host: str = os.environ.get("VOICE_HOST", "127.0.0.1")
|
|
port: int = _int("VOICE_PORT", 4100)
|
|
|
|
# ── Models ───────────────────────────────────────────────────────────────
|
|
# IndicConformer is a hybrid CTC + RNNT model. CTC decoding is used because
|
|
# it is a single forward pass — RNNT is more accurate but decodes
|
|
# autoregressively, and in a voice loop the latency costs more than the
|
|
# accuracy buys.
|
|
stt_model: str = os.environ.get("STT_MODEL", "ai4bharat/indic-conformer-600m-multilingual")
|
|
stt_decoding: str = os.environ.get("STT_DECODING", "ctc") # ctc | rnnt
|
|
|
|
# English is not one of IndicConformer's 22 codes, and it cannot identify
|
|
# languages. Whisper covers both — multilingual, not the .en checkpoint,
|
|
# because language ID is what makes "auto" work.
|
|
english_model: str = os.environ.get("ENGLISH_MODEL", "openai/whisper-small")
|
|
preload_english: bool = _flag("PRELOAD_ENGLISH", True)
|
|
|
|
# Used when auto-detection is not confident enough to overrule the user.
|
|
default_lang: str = os.environ.get("VOICE_DEFAULT_LANG", "ta")
|
|
|
|
tts_model: str = os.environ.get("TTS_MODEL", "ai4bharat/indic-parler-tts")
|
|
|
|
# ── Device / precision ───────────────────────────────────────────────────
|
|
# float16 on CUDA: both models together are ~3 GB in half precision, which
|
|
# fits the 6 GB card with room for activations. float32 would not.
|
|
device: str = os.environ.get("VOICE_DEVICE", "cuda" if torch.cuda.is_available() else "cpu")
|
|
|
|
@property
|
|
def dtype(self) -> torch.dtype:
|
|
return torch.float16 if self.device == "cuda" else torch.float32
|
|
|
|
# ── Audio ────────────────────────────────────────────────────────────────
|
|
sample_rate_in: int = 16000 # what the browser worklet sends
|
|
# Indic Parler-TTS emits 44.1 kHz — verified from model.config.sampling_rate,
|
|
# not the 24 kHz the upstream Parler-TTS Mini uses. This is only a fallback;
|
|
# the real rate is read from the loaded model and sent to the browser, which
|
|
# configures its playback worklet from it.
|
|
sample_rate_out: int = _int("TTS_SAMPLE_RATE", 44100)
|
|
|
|
# ── Endpointing (Silero VAD) ─────────────────────────────────────────────
|
|
# Silero operates on fixed 512-sample frames at 16 kHz (32 ms).
|
|
vad_frame: int = 512
|
|
vad_threshold: float = float(os.environ.get("VAD_THRESHOLD", "0.5"))
|
|
# How much trailing silence ends a turn. Too short truncates people who
|
|
# pause mid-sentence; too long makes the assistant feel sluggish.
|
|
vad_silence_ms: int = _int("VAD_SILENCE_MS", 700)
|
|
# Ignore blips so a cough or a door does not open a turn.
|
|
vad_min_speech_ms: int = _int("VAD_MIN_SPEECH_MS", 250)
|
|
# Audio kept from *before* detected speech, so word onsets are not clipped.
|
|
vad_prefix_ms: int = _int("VAD_PREFIX_MS", 300)
|
|
vad_max_utterance_ms: int = _int("VAD_MAX_UTTERANCE_MS", 30000)
|
|
|
|
# Warm the models at startup rather than on the first user turn — a cold
|
|
# CUDA graph on the first utterance costs several seconds.
|
|
warmup: bool = _flag("VOICE_WARMUP", True)
|
|
|
|
|
|
settings = Settings()
|