/* Generate the tokenizer.json that Transformers.js needs for the exported Tamil VITS model. `save_pretrained` does not emit one: VitsTokenizer is a "slow" tokenizer with no fast counterpart, so Python writes vocab.json + tokenizer_config.json and nothing else. Transformers.js only reads tokenizer.json, so we synthesise it from the exported vocab, mirroring the structure of the working English model (Xenova/mms-tts-eng) exactly. The four normalizer steps, in order: 1. Lowercase — no-op for Tamil, matters for embedded Latin/digits 2. Replace — drop every character outside the vocab 3. Strip — trim surrounding whitespace 4. Replace — insert the blank token between every character, which is what `add_blank: true` means for VITS. Omit this and the audio comes out garbled. */ import fs from 'node:fs'; import path from 'node:path'; const DIR = 'assets/tts/mms-tts-tam'; const vocab = JSON.parse(fs.readFileSync(path.join(DIR, 'vocab.json'), 'utf8')); const cfg = JSON.parse(fs.readFileSync(path.join(DIR, 'tokenizer_config.json'), 'utf8')); // The blank/pad token is whichever character maps to id 0. const blank = Object.keys(vocab).find((k) => vocab[k] === 0); const unk = cfg.unk_token ?? ''; const unkId = vocab[unk] ?? Object.keys(vocab).length; // Character class of everything we keep. Escape the regex metacharacters that // are still special inside a negated class. const escaped = Object.keys(vocab) .filter((c) => c !== unk) .map((c) => (']\\^-'.includes(c) ? '\\' + c : c)) .join(''); const tokenizer = { version: '1.0', truncation: null, padding: null, added_tokens: [{ id: unkId, content: unk, single_word: false, lstrip: false, rstrip: false, normalized: false, special: true, }], normalizer: { type: 'Sequence', normalizers: [ { type: 'Lowercase' }, { type: 'Replace', pattern: { Regex: `[^${escaped}]` }, content: '' }, { type: 'Strip', strip_left: true, strip_right: true }, ...(cfg.add_blank ? [{ type: 'Replace', pattern: { Regex: '(?=.)|(?