WeLe Agentic AI with Docker deployment

This commit is contained in:
2026-08-28 08:50:08 +05:30
parent 586059b14f
commit 6bb0ef25ca
34 changed files with 2271 additions and 1343 deletions
+66
View File
@@ -0,0 +1,66 @@
/* Generate the tokenizer.json that Transformers.js needs for the exported
Tamil VITS model.
`save_pretrained` does not emit one: VitsTokenizer is a "slow" tokenizer with
no fast counterpart, so Python writes vocab.json + tokenizer_config.json and
nothing else. Transformers.js only reads tokenizer.json, so we synthesise it
from the exported vocab, mirroring the structure of the working English
model (Xenova/mms-tts-eng) exactly.
The four normalizer steps, in order:
1. Lowercase — no-op for Tamil, matters for embedded Latin/digits
2. Replace — drop every character outside the vocab
3. Strip — trim surrounding whitespace
4. Replace — insert the blank token between every character,
which is what `add_blank: true` means for VITS.
Omit this and the audio comes out garbled.
*/
import fs from 'node:fs';
import path from 'node:path';
const DIR = 'assets/tts/mms-tts-tam';
const vocab = JSON.parse(fs.readFileSync(path.join(DIR, 'vocab.json'), 'utf8'));
const cfg = JSON.parse(fs.readFileSync(path.join(DIR, 'tokenizer_config.json'), 'utf8'));
// The blank/pad token is whichever character maps to id 0.
const blank = Object.keys(vocab).find((k) => vocab[k] === 0);
const unk = cfg.unk_token ?? '<unk>';
const unkId = vocab[unk] ?? Object.keys(vocab).length;
// Character class of everything we keep. Escape the regex metacharacters that
// are still special inside a negated class.
const escaped = Object.keys(vocab)
.filter((c) => c !== unk)
.map((c) => (']\\^-'.includes(c) ? '\\' + c : c))
.join('');
const tokenizer = {
version: '1.0',
truncation: null,
padding: null,
added_tokens: [{
id: unkId, content: unk,
single_word: false, lstrip: false, rstrip: false, normalized: false, special: true,
}],
normalizer: {
type: 'Sequence',
normalizers: [
{ type: 'Lowercase' },
{ type: 'Replace', pattern: { Regex: `[^${escaped}]` }, content: '' },
{ type: 'Strip', strip_left: true, strip_right: true },
...(cfg.add_blank ? [{ type: 'Replace', pattern: { Regex: '(?=.)|(?<!^)$' }, content: blank }] : []),
],
},
pre_tokenizer: { type: 'Split', pattern: { Regex: '' }, behavior: 'Isolated', invert: false },
post_processor: null,
decoder: null,
model: { vocab },
};
fs.writeFileSync(path.join(DIR, 'tokenizer.json'), JSON.stringify(tokenizer, null, 1));
console.log(`wrote ${DIR}/tokenizer.json`);
console.log(` vocab ${Object.keys(vocab).length} tokens`);
console.log(` blank token ${JSON.stringify(blank)} (id 0)`);
console.log(` unk ${JSON.stringify(unk)} (id ${unkId})`);
console.log(` add_blank ${cfg.add_blank}`);