/* Do Whisper (STT) and MMS-TTS (TTS) actually run in Node on CPU? */ import { pipeline, env } from '@huggingface/transformers'; env.cacheDir = './.transformers-cache'; const t = (t0) => `${((performance.now() - t0) / 1000).toFixed(1)}s`; // ── TTS: MMS-TTS Tamil (VITS, 36M, feed-forward) ──────────────────────────── console.log('[1/2] loading MMS-TTS Tamil…'); let t0 = performance.now(); const tts = await pipeline('text-to-speech', 'Xenova/mms-tts-eng', { dtype: 'fp32' }); console.log(` loaded in ${t(t0)}`); const TA = 'There are three thousand four hundred leads in the new lead stage.'; await tts(TA); // warm for (const [label, text] of [['short', TA], ['long', TA + ' ' + TA + ' ' + TA]]) { t0 = performance.now(); const out = await tts(text); const ms = performance.now() - t0; const audioMs = (out.audio.length / out.sampling_rate) * 1000; console.log( ` ${label.padEnd(5)} ${String(text.length).padStart(3)} chars → ${ms.toFixed(0)}ms ` + `for ${audioMs.toFixed(0)}ms audio @ ${out.sampling_rate}Hz → RTF ${(ms / audioMs).toFixed(2)}x`, ); } // ── STT: Whisper (multilingual — Tamil, Hindi, English + detection) ───────── console.log('\n[2/2] loading Whisper base…'); t0 = performance.now(); const stt = await pipeline('automatic-speech-recognition', 'onnx-community/whisper-base', { dtype: 'q8' }); console.log(` loaded in ${t(t0)}`); // 4 s of quiet noise — proves the graph runs and times it. const audio = Float32Array.from({ length: 16000 * 4 }, () => (Math.random() - 0.5) * 0.02); t0 = performance.now(); const r = await stt(audio, { language: 'ta', task: 'transcribe' }); console.log(` 4000ms audio → ${(performance.now() - t0).toFixed(0)}ms → ${JSON.stringify(r.text).slice(0, 60)}`); t0 = performance.now(); const r2 = await stt(audio, { language: 'en', task: 'transcribe' }); console.log(` english pass → ${(performance.now() - t0).toFixed(0)}ms → ${JSON.stringify(r2.text).slice(0, 60)}`); console.log(`\nRSS ${(process.memoryUsage().rss / 1e9).toFixed(2)} GB`); process.exit(0);