WeLe Agentic AI: LangGraph multi-agent CRM assistant with voice
This commit is contained in:
@@ -0,0 +1,95 @@
|
||||
"""Honest latency benchmark: warm up first, then time repeated runs.
|
||||
|
||||
The first CUDA generation pays for kernel autotuning and cache allocation, so a
|
||||
single cold measurement makes any model look far worse than it is in service.
|
||||
"""
|
||||
import logging
|
||||
import time
|
||||
from threading import Thread
|
||||
|
||||
import numpy as np
|
||||
import torch
|
||||
|
||||
logging.basicConfig(level=logging.INFO, format="%(message)s")
|
||||
log = logging.getLogger("bench")
|
||||
DEV = "cuda" if torch.cuda.is_available() else "cpu"
|
||||
|
||||
log.info("device=%s gpu=%s", DEV, torch.cuda.get_device_name(0) if DEV == "cuda" else "-")
|
||||
|
||||
# ── STT ──────────────────────────────────────────────────────────────────────
|
||||
from transformers import AutoModel
|
||||
|
||||
stt = AutoModel.from_pretrained("ai4bharat/indic-conformer-600m-multilingual", trust_remote_code=True)
|
||||
stt = stt.to(DEV).eval()
|
||||
|
||||
# Where does it actually run? A model wrapping ONNX ignores .to(cuda).
|
||||
params = list(stt.parameters())
|
||||
log.info("STT param device: %s (%d tensors)", params[0].device if params else "NO TORCH PARAMS", len(params))
|
||||
log.info("STT type: %s", type(stt).__name__)
|
||||
|
||||
wav = torch.from_numpy((np.random.randn(16000 * 4) * 0.02).astype(np.float32)).unsqueeze(0).to(DEV)
|
||||
with torch.inference_mode():
|
||||
stt(wav, "ta", "ctc") # warmup
|
||||
times = []
|
||||
for _ in range(3):
|
||||
t0 = time.perf_counter()
|
||||
with torch.inference_mode():
|
||||
stt(wav, "ta", "ctc")
|
||||
times.append((time.perf_counter() - t0) * 1000)
|
||||
log.info("STT 4000 ms audio → %.0f / %.0f / %.0f ms (RTF %.2fx)",
|
||||
*times, (sum(times) / len(times)) / 4000)
|
||||
|
||||
# ── TTS ──────────────────────────────────────────────────────────────────────
|
||||
from parler_tts import ParlerTTSForConditionalGeneration, ParlerTTSStreamer
|
||||
from transformers import AutoTokenizer
|
||||
|
||||
dtype = torch.float16 if DEV == "cuda" else torch.float32
|
||||
tts = ParlerTTSForConditionalGeneration.from_pretrained("ai4bharat/indic-parler-tts", torch_dtype=dtype).to(DEV).eval()
|
||||
tok = AutoTokenizer.from_pretrained("ai4bharat/indic-parler-tts")
|
||||
dtok = AutoTokenizer.from_pretrained(tts.config.text_encoder._name_or_path)
|
||||
SR = tts.config.sampling_rate
|
||||
log.info("TTS sampling_rate=%d frame_rate=%s", SR, getattr(tts.audio_encoder.config, "frame_rate", "?"))
|
||||
|
||||
desc = "Jaya speaks in a warm, clear, professional tone at a natural pace. The recording is very high quality with no background noise."
|
||||
d = dtok(desc, return_tensors="pt").to(DEV)
|
||||
|
||||
|
||||
def run(text, stream=True):
|
||||
p = tok(text, return_tensors="pt").to(DEV)
|
||||
kw = dict(input_ids=d.input_ids, attention_mask=d.attention_mask,
|
||||
prompt_input_ids=p.input_ids, prompt_attention_mask=p.attention_mask)
|
||||
t0 = time.perf_counter()
|
||||
if stream:
|
||||
fr = int(getattr(tts.audio_encoder.config, "frame_rate", 86) / 2)
|
||||
s = ParlerTTSStreamer(tts, device=DEV, play_steps=fr)
|
||||
Thread(target=tts.generate, kwargs={**kw, "streamer": s}, daemon=True).start()
|
||||
first, n = None, 0
|
||||
for c in s:
|
||||
if c is None or len(c) == 0:
|
||||
continue
|
||||
if first is None:
|
||||
first = (time.perf_counter() - t0) * 1000
|
||||
n += len(c)
|
||||
else:
|
||||
with torch.inference_mode():
|
||||
g = tts.generate(**kw)
|
||||
n = g.shape[-1]
|
||||
first = None
|
||||
total = (time.perf_counter() - t0) * 1000
|
||||
return first, total, 1000 * n / SR
|
||||
|
||||
|
||||
SHORT = "மூவாயிரம் நானூறு லீட்கள் உள்ளன."
|
||||
LONG = "புதிய லீட்கள் மூவாயிரம் நானூற்று இருபத்தேழு. இதில் எழுபத்தாறு சதவீதம் இன்னும் தொடர்பு கொள்ளப்படவில்லை."
|
||||
|
||||
run(SHORT) # warmup
|
||||
log.info("")
|
||||
for label, text in (("short", SHORT), ("long", LONG)):
|
||||
first, total, audio = run(text)
|
||||
log.info("TTS %-5s %2d chars → first %.0f ms | total %.0f ms | audio %.0f ms | RTF %.2fx",
|
||||
label, len(text), first or -1, total, audio, total / max(audio, 1))
|
||||
|
||||
if DEV == "cuda":
|
||||
log.info("\nVRAM peak reserved: %.2f GB of %.1f GB",
|
||||
torch.cuda.max_memory_reserved() / 1e9,
|
||||
torch.cuda.get_device_properties(0).total_memory / 1e9)
|
||||
Reference in New Issue
Block a user