67 lines
2.7 KiB
JavaScript
67 lines
2.7 KiB
JavaScript
/* Generate the tokenizer.json that Transformers.js needs for the exported
|
|
Tamil VITS model.
|
|
|
|
`save_pretrained` does not emit one: VitsTokenizer is a "slow" tokenizer with
|
|
no fast counterpart, so Python writes vocab.json + tokenizer_config.json and
|
|
nothing else. Transformers.js only reads tokenizer.json, so we synthesise it
|
|
from the exported vocab, mirroring the structure of the working English
|
|
model (Xenova/mms-tts-eng) exactly.
|
|
|
|
The four normalizer steps, in order:
|
|
1. Lowercase — no-op for Tamil, matters for embedded Latin/digits
|
|
2. Replace — drop every character outside the vocab
|
|
3. Strip — trim surrounding whitespace
|
|
4. Replace — insert the blank token between every character,
|
|
which is what `add_blank: true` means for VITS.
|
|
Omit this and the audio comes out garbled.
|
|
*/
|
|
import fs from 'node:fs';
|
|
import path from 'node:path';
|
|
|
|
const DIR = 'assets/tts/mms-tts-tam';
|
|
const vocab = JSON.parse(fs.readFileSync(path.join(DIR, 'vocab.json'), 'utf8'));
|
|
const cfg = JSON.parse(fs.readFileSync(path.join(DIR, 'tokenizer_config.json'), 'utf8'));
|
|
|
|
// The blank/pad token is whichever character maps to id 0.
|
|
const blank = Object.keys(vocab).find((k) => vocab[k] === 0);
|
|
const unk = cfg.unk_token ?? '<unk>';
|
|
const unkId = vocab[unk] ?? Object.keys(vocab).length;
|
|
|
|
// Character class of everything we keep. Escape the regex metacharacters that
|
|
// are still special inside a negated class.
|
|
const escaped = Object.keys(vocab)
|
|
.filter((c) => c !== unk)
|
|
.map((c) => (']\\^-'.includes(c) ? '\\' + c : c))
|
|
.join('');
|
|
|
|
const tokenizer = {
|
|
version: '1.0',
|
|
truncation: null,
|
|
padding: null,
|
|
added_tokens: [{
|
|
id: unkId, content: unk,
|
|
single_word: false, lstrip: false, rstrip: false, normalized: false, special: true,
|
|
}],
|
|
normalizer: {
|
|
type: 'Sequence',
|
|
normalizers: [
|
|
{ type: 'Lowercase' },
|
|
{ type: 'Replace', pattern: { Regex: `[^${escaped}]` }, content: '' },
|
|
{ type: 'Strip', strip_left: true, strip_right: true },
|
|
...(cfg.add_blank ? [{ type: 'Replace', pattern: { Regex: '(?=.)|(?<!^)$' }, content: blank }] : []),
|
|
],
|
|
},
|
|
pre_tokenizer: { type: 'Split', pattern: { Regex: '' }, behavior: 'Isolated', invert: false },
|
|
post_processor: null,
|
|
decoder: null,
|
|
model: { vocab },
|
|
};
|
|
|
|
fs.writeFileSync(path.join(DIR, 'tokenizer.json'), JSON.stringify(tokenizer, null, 1));
|
|
|
|
console.log(`wrote ${DIR}/tokenizer.json`);
|
|
console.log(` vocab ${Object.keys(vocab).length} tokens`);
|
|
console.log(` blank token ${JSON.stringify(blank)} (id 0)`);
|
|
console.log(` unk ${JSON.stringify(unk)} (id ${unkId})`);
|
|
console.log(` add_blank ${cfg.add_blank}`);
|