WeLe Agentic AI with Docker deployment
This commit is contained in:
@@ -0,0 +1,66 @@
|
||||
/* Generate the tokenizer.json that Transformers.js needs for the exported
|
||||
Tamil VITS model.
|
||||
|
||||
`save_pretrained` does not emit one: VitsTokenizer is a "slow" tokenizer with
|
||||
no fast counterpart, so Python writes vocab.json + tokenizer_config.json and
|
||||
nothing else. Transformers.js only reads tokenizer.json, so we synthesise it
|
||||
from the exported vocab, mirroring the structure of the working English
|
||||
model (Xenova/mms-tts-eng) exactly.
|
||||
|
||||
The four normalizer steps, in order:
|
||||
1. Lowercase — no-op for Tamil, matters for embedded Latin/digits
|
||||
2. Replace — drop every character outside the vocab
|
||||
3. Strip — trim surrounding whitespace
|
||||
4. Replace — insert the blank token between every character,
|
||||
which is what `add_blank: true` means for VITS.
|
||||
Omit this and the audio comes out garbled.
|
||||
*/
|
||||
import fs from 'node:fs';
|
||||
import path from 'node:path';
|
||||
|
||||
const DIR = 'assets/tts/mms-tts-tam';
|
||||
const vocab = JSON.parse(fs.readFileSync(path.join(DIR, 'vocab.json'), 'utf8'));
|
||||
const cfg = JSON.parse(fs.readFileSync(path.join(DIR, 'tokenizer_config.json'), 'utf8'));
|
||||
|
||||
// The blank/pad token is whichever character maps to id 0.
|
||||
const blank = Object.keys(vocab).find((k) => vocab[k] === 0);
|
||||
const unk = cfg.unk_token ?? '<unk>';
|
||||
const unkId = vocab[unk] ?? Object.keys(vocab).length;
|
||||
|
||||
// Character class of everything we keep. Escape the regex metacharacters that
|
||||
// are still special inside a negated class.
|
||||
const escaped = Object.keys(vocab)
|
||||
.filter((c) => c !== unk)
|
||||
.map((c) => (']\\^-'.includes(c) ? '\\' + c : c))
|
||||
.join('');
|
||||
|
||||
const tokenizer = {
|
||||
version: '1.0',
|
||||
truncation: null,
|
||||
padding: null,
|
||||
added_tokens: [{
|
||||
id: unkId, content: unk,
|
||||
single_word: false, lstrip: false, rstrip: false, normalized: false, special: true,
|
||||
}],
|
||||
normalizer: {
|
||||
type: 'Sequence',
|
||||
normalizers: [
|
||||
{ type: 'Lowercase' },
|
||||
{ type: 'Replace', pattern: { Regex: `[^${escaped}]` }, content: '' },
|
||||
{ type: 'Strip', strip_left: true, strip_right: true },
|
||||
...(cfg.add_blank ? [{ type: 'Replace', pattern: { Regex: '(?=.)|(?<!^)$' }, content: blank }] : []),
|
||||
],
|
||||
},
|
||||
pre_tokenizer: { type: 'Split', pattern: { Regex: '' }, behavior: 'Isolated', invert: false },
|
||||
post_processor: null,
|
||||
decoder: null,
|
||||
model: { vocab },
|
||||
};
|
||||
|
||||
fs.writeFileSync(path.join(DIR, 'tokenizer.json'), JSON.stringify(tokenizer, null, 1));
|
||||
|
||||
console.log(`wrote ${DIR}/tokenizer.json`);
|
||||
console.log(` vocab ${Object.keys(vocab).length} tokens`);
|
||||
console.log(` blank token ${JSON.stringify(blank)} (id 0)`);
|
||||
console.log(` unk ${JSON.stringify(unk)} (id ${unkId})`);
|
||||
console.log(` add_blank ${cfg.add_blank}`);
|
||||
Reference in New Issue
Block a user