tokenizers / tokenizers.mjs
qikp's picture
Vendor @huggingface/tokenizers locally from unpkg
2d28d93
Raw
History Blame Contribute Delete
82 kB
// src/utils/data-structures/DictionarySplitter.ts
var DictionarySplitter = class {
/**
* @param dictionary The dictionary of words to use for splitting.
*/
constructor(dictionary) {
this.trie = this._build_trie(dictionary);
}
/**
* Builds a trie from the given dictionary.
* @param dictionary The dictionary of words to build the trie from.
* @returns The root node of the trie.
* @private
*/
_build_trie(dictionary) {
const trie = /* @__PURE__ */ Object.create(null);
for (const word of dictionary) {
let node = trie;
for (let i = 0; i < word.length; ++i) {
const char = word[i];
node = node[char] ??= /* @__PURE__ */ Object.create(null);
}
node.end = word;
}
return trie;
}
/**
* Splits the input text into tokens based on the dictionary.
* @param text The input text to split.
* @returns An array of tokens.
*/
split(text) {
const result = [];
const n = text.length;
let start = 0;
let i = 0;
while (i < n) {
let node = this.trie;
let match = null;
let j = i;
while (j < n && (node = node[text[j]])) {
if (node.end) {
match = node.end;
}
++j;
}
if (match) {
if (i > start) {
result.push(text.slice(start, i));
}
result.push(match);
i += match.length;
start = i;
} else {
++i;
}
}
if (start < n) {
result.push(text.slice(start));
}
return result;
}
};
var DictionarySplitter_default = DictionarySplitter;
// src/core/AddedToken.ts
var AddedToken = class {
/**
* Creates a new instance of AddedToken.
* @param config Added token configuration object.
*/
constructor(config) {
this.content = config.content;
this.id = config.id;
this.single_word = config.single_word ?? false;
this.lstrip = config.lstrip ?? false;
this.rstrip = config.rstrip ?? false;
this.special = config.special ?? false;
this.normalized = config.normalized ?? !this.special;
}
};
var AddedToken_default = AddedToken;
// src/static/constants.ts
var BYTES_TO_UNICODE = (() => {
const bs = [
...Array.from(
{ length: "~".charCodeAt(0) - "!".charCodeAt(0) + 1 },
(_, i) => i + "!".charCodeAt(0)
),
...Array.from(
{ length: "\xAC".charCodeAt(0) - "\xA1".charCodeAt(0) + 1 },
(_, i) => i + "\xA1".charCodeAt(0)
),
...Array.from(
{ length: "\xFF".charCodeAt(0) - "\xAE".charCodeAt(0) + 1 },
(_, i) => i + "\xAE".charCodeAt(0)
)
];
const cs = bs.slice();
let n = 0;
for (let b = 0; b < 256; ++b) {
if (!bs.includes(b)) {
bs.push(b);
cs.push(256 + n);
n += 1;
}
}
const ccs = cs.map((n2) => String.fromCharCode(n2));
return Object.fromEntries(bs.map((b, i) => [b, ccs[i]]));
})();
var reverse_dictionary = (data) => Object.fromEntries(Object.entries(data).map(([key, value]) => [value, key]));
var UNICODE_TO_BYTES = reverse_dictionary(BYTES_TO_UNICODE);
var BLOOM_SPLIT_CHARS = ".,!?\u2026\u3002\uFF0C\u3001\u0964\u06D4\u060C";
var PROBLEMATIC_REGEX_MAP = /* @__PURE__ */ new Map([
// These uses the case insensitive group modifier, which is not supported in JavaScript.
// When parsing the regex, an "Invalid group" error is thrown.
[
"(?i:'s|'t|'re|'ve|'m|'ll|'d)",
"(?:'([sS]|[tT]|[rR][eE]|[vV][eE]|[mM]|[lL][lL]|[dD]))"
],
[
"(?i:[sdmt]|ll|ve|re)",
"(?:[sS]|[dD]|[mM]|[tT]|[lL][lL]|[vV][eE]|[rR][eE])"
],
// JS doesn't support possessive quantifiers (these are used in recent OpenAI tokenizers).
["[^\\r\\n\\p{L}\\p{N}]?+", "[^\\r\\n\\p{L}\\p{N}]?"],
["[^\\s\\p{L}\\p{N}]++", "[^\\s\\p{L}\\p{N}]+"],
// JS doesn't support atomic groups (these are used in AFMoE tokenizers).
["(?>\\p{Nd}{510})", "(?:\\p{Nd}{510})"],
// JS doesn't support stacking quantifiers.
// Uncaught SyntaxError: Invalid regular expression: /\p{Nd}{3}+/u: Nothing to repeat
["\\p{Nd}{3}+", "(?:\\p{Nd}{3})+"],
// \G is an invalid escape in JS, and in most cases is just used as an optimization.
// So, we can safely remove it.
["\\G", ""],
// Used to override the default (invalid) regex of the bloom pretokenizer.
// For more information, see https://github.com/huggingface/transformers.js/issues/94
[` ?[^(\\s|[${BLOOM_SPLIT_CHARS}])]+`, ` ?[^\\s${BLOOM_SPLIT_CHARS}]+`]
]);
var PUNCTUATION_REGEX = "\\p{P}\\u0021-\\u002F\\u003A-\\u0040\\u005B-\\u0060\\u007B-\\u007E";
// src/utils/core.ts
var clean_up_tokenization = (text) => text.replace(/ \./g, ".").replace(/ \?/g, "?").replace(/ \!/g, "!").replace(/ ,/g, ",").replace(/ \' /g, "'").replace(/ n't/g, "n't").replace(/ 'm/g, "'m").replace(/ 's/g, "'s").replace(/ 've/g, "'ve").replace(/ 're/g, "'re");
var create_pattern = (pattern, invert = true) => {
if (pattern.Regex !== void 0) {
let regex = pattern.Regex.replace(/\\([#&~])/g, "$1");
regex = regex.replace(/\\A/g, "^").replace(/\\z/g, "$").replace(/\\Z/g, "(?=\\r?\\n?$)");
for (const [key, value] of PROBLEMATIC_REGEX_MAP) {
regex = regex.replaceAll(key, value);
}
try {
return new RegExp(regex, "gu");
} catch (error) {
if (!(error instanceof SyntaxError) || !error.message.toLowerCase().includes("invalid property name"))
throw error;
let changed = false;
const fixed = regex.replace(/(\\[pP])\{([^}=]+)\}/g, (_, p, n) => {
try {
new RegExp(`\\p{${n}}`, "u");
return `${p}{${n}}`;
} catch {
changed = true;
return `${p}{Script=${n}}`;
}
});
if (!changed) throw error;
try {
return new RegExp(fixed, "gu");
} catch (e) {
throw error;
}
}
} else if (pattern.String !== void 0) {
const escaped = escape_reg_exp(pattern.String);
return new RegExp(invert ? escaped : `(${escaped})`, "gu");
} else {
console.warn("Unknown pattern type:", pattern);
return null;
}
};
var escape_reg_exp = (string) => string.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
var fuse_unk = (arr, tokens_to_ids, unk_token_id) => {
const fused = [];
let i = 0;
while (i < arr.length) {
fused.push(arr[i]);
const token_id = tokens_to_ids.get(arr[i]) ?? unk_token_id;
if (token_id !== unk_token_id) {
++i;
continue;
}
while (++i < arr.length && (tokens_to_ids.get(arr[i]) ?? unk_token_id) === unk_token_id) {
if (tokens_to_ids.get(fused.at(-1)) !== unk_token_id) {
fused[fused.length - 1] += arr[i];
}
}
}
return fused;
};
var is_chinese_char = (cp) => cp >= 19968 && cp <= 40959 || cp >= 13312 && cp <= 19903 || cp >= 131072 && cp <= 173791 || cp >= 173824 && cp <= 177983 || cp >= 177984 && cp <= 178207 || cp >= 178208 && cp <= 183983 || cp >= 63744 && cp <= 64255 || cp >= 194560 && cp <= 195103;
var is_integral_number = (x) => Number.isInteger(x) || typeof x === "bigint";
var len = (s) => {
let length = 0;
for (const c of s) ++length;
return length;
};
var lowercase_and_remove_accents = (text) => remove_accents(text.toLowerCase());
var merge_arrays = (...arrs) => Array.prototype.concat.apply([], arrs);
var object_to_map = (obj) => new Map(Object.entries(obj));
var regex_split = (text, regex) => {
const result = [];
let prev = 0;
for (const match of text.matchAll(regex)) {
const full_match = match[0];
if (prev < match.index) {
result.push(text.slice(prev, match.index));
}
if (full_match.length > 0) {
result.push(full_match);
}
prev = match.index + full_match.length;
}
if (prev < text.length) {
result.push(text.slice(prev));
}
return result;
};
var remove_accents = (text) => text.replace(/\p{M}/gu, "");
var validate_object = (obj, name, required_keys = []) => {
if (!obj || Array.isArray(obj) || typeof obj !== "object") {
return `${name} must be a valid object`;
}
for (const key of required_keys) {
if (!(key in obj)) {
return `${name} must contain a "${key}" property`;
}
}
return null;
};
var whitespace_split = (text) => text.match(/\S+/g) || [];
// src/utils/Callable.ts
var Callable = class {
/**
* Creates a new instance of the Callable class.
*/
constructor() {
const closure = function(...args) {
return closure._call(...args);
};
return Object.setPrototypeOf(closure, new.target.prototype);
}
};
var Callable_default = Callable;
// src/core/Normalizer.ts
var Normalizer = class extends Callable_default {
/**
* @param config The configuration object for the normalizer.
*/
constructor(config) {
super();
this.config = config;
}
/**
* Alias for {@link Normalizer#normalize}.
* @param text The text to normalize.
* @returns The normalized text.
*/
_call(text) {
return this.normalize(text);
}
};
var Normalizer_default = Normalizer;
// src/core/normalizer/BertNormalizer.ts
var BertNormalizer = class extends Normalizer_default {
/**
* Adds whitespace around any CJK (Chinese, Japanese, or Korean) character in the input text.
*
* @param text The input text to tokenize.
* @returns The tokenized text with whitespace added around CJK characters.
*/
tokenize_chinese_chars(text) {
const output = [];
for (let i = 0; i < text.length; ++i) {
const char = text[i];
const cp = char.charCodeAt(0);
if (is_chinese_char(cp)) {
output.push(" ");
output.push(char);
output.push(" ");
} else {
output.push(char);
}
}
return output.join("");
}
/**
* Strips accents from the given text.
* @param text The text to strip accents from.
* @returns The text with accents removed.
*/
strip_accents(text) {
return text.normalize("NFD").replace(/\p{Mn}/gu, "");
}
/**
* Checks whether `char` is a control character.
* @param char The character to check.
* @returns Whether `char` is a control character.
*/
is_control(char) {
switch (char) {
case " ":
case "\n":
case "\r":
return false;
default:
return /^\p{Cc}|\p{Cf}|\p{Co}|\p{Cs}$/u.test(char);
}
}
/**
* Performs invalid character removal and whitespace cleanup on text.
* @param text The text to clean.
* @returns The cleaned text.
*/
clean_text(text) {
const output = [];
for (const char of text) {
const cp = char.charCodeAt(0);
if (cp === 0 || cp === 65533 || this.is_control(char)) {
continue;
}
if (/^\s$/.test(char)) {
output.push(" ");
} else {
output.push(char);
}
}
return output.join("");
}
/**
* Normalizes the given text based on the configuration.
* @param text The text to normalize.
* @returns The normalized text.
*/
normalize(text) {
if (this.config.clean_text) {
text = this.clean_text(text);
}
if (this.config.handle_chinese_chars) {
text = this.tokenize_chinese_chars(text);
}
if (this.config.lowercase) {
text = text.toLowerCase();
if (this.config.strip_accents !== false) {
text = this.strip_accents(text);
}
} else if (this.config.strip_accents) {
text = this.strip_accents(text);
}
return text;
}
};
var BertNormalizer_default = BertNormalizer;
// src/core/normalizer/Precompiled.ts
var Precompiled = class extends Normalizer_default {
/**
* Create a new instance of Precompiled normalizer.
* @param config The configuration object.
*/
constructor(config) {
super(config);
this.charsmap = config.precompiled_charsmap ?? null;
}
/**
* Normalizes the given text by applying the precompiled charsmap.
* @param text The text to normalize.
* @returns The normalized text.
*/
normalize(text) {
text = text.replace(
/[\u0001-\u0008\u000B\u000E-\u001F\u007F\u008F\u009F]/gm,
""
);
text = text.replace(
/[\u0009\u000A\u000C\u000D\u00A0\u1680\u2000-\u200F\u2028\u2029\u202F\u205F\u2581\u3000\uFEFF\uFFFD]/gm,
" "
);
if (text.includes("\uFF5E")) {
const parts = text.split("\uFF5E");
text = parts.map((part) => part.normalize("NFKC")).join("\uFF5E");
} else {
text = text.normalize("NFKC");
}
return text;
}
};
var Precompiled_default = Precompiled;
// src/core/normalizer/Sequence.ts
var Sequence = class extends Normalizer_default {
/**
* Create a new instance of NormalizerSequence.
* @param config The configuration object.
*/
constructor(config) {
super(config);
this.normalizers = (config.normalizers ?? []).map(
(x) => create_normalizer_default(x)
);
}
/**
* Apply a sequence of Normalizers to the input text.
* @param text The text to normalize.
* @returns The normalized text.
*/
normalize(text) {
return this.normalizers.reduce((t, normalizer) => {
return normalizer ? normalizer.normalize(t) : t;
}, text);
}
};
var Sequence_default = Sequence;
// src/core/normalizer/Replace.ts
var Replace = class extends Normalizer_default {
/**
* Normalize the input text by replacing the pattern with the content.
* @param text The input text to be normalized.
* @returns The normalized text after replacing the pattern with the content.
*/
normalize(text) {
const pattern = create_pattern(this.config.pattern ?? {});
return pattern === null ? text : text.replaceAll(pattern, this.config.content ?? "");
}
};
var Replace_default = Replace;
// src/core/normalizer/UnicodeNormalizer.ts
var UnicodeNormalizer = class extends Normalizer_default {
constructor() {
super(...arguments);
/**
* The Unicode normalization form to apply.
* Should be one of: 'NFC', 'NFD', 'NFKC', or 'NFKD'.
*/
this.form = "NFC";
}
/**
* Normalize the input text by applying Unicode normalization.
* @param text The input text to be normalized.
* @returns The normalized text.
*/
normalize(text) {
text = text.normalize(this.form);
return text;
}
};
var UnicodeNormalizer_default = UnicodeNormalizer;
// src/core/normalizer/NFC.ts
var NFC = class extends UnicodeNormalizer_default {
constructor() {
super(...arguments);
this.form = "NFC";
}
};
var NFC_default = NFC;
// src/core/normalizer/NFD.ts
var NFD = class extends UnicodeNormalizer_default {
constructor() {
super(...arguments);
this.form = "NFD";
}
};
var NFD_default = NFD;
// src/core/normalizer/NFKC.ts
var NFKC = class extends UnicodeNormalizer_default {
constructor() {
super(...arguments);
this.form = "NFKC";
}
};
var NFKC_default = NFKC;
// src/core/normalizer/NFKD.ts
var NFKD = class extends UnicodeNormalizer_default {
constructor() {
super(...arguments);
this.form = "NFKD";
}
};
var NFKD_default = NFKD;
// src/core/normalizer/Strip.ts
var Strip = class extends Normalizer_default {
/**
* Strip leading and/or trailing whitespace from the input text.
* @param text The input text.
* @returns The normalized text.
*/
normalize(text) {
if (this.config.strip_left && this.config.strip_right) {
text = text.trim();
} else {
if (this.config.strip_left) {
text = text.trimStart();
}
if (this.config.strip_right) {
text = text.trimEnd();
}
}
return text;
}
};
var Strip_default = Strip;
// src/core/normalizer/StripAccents.ts
var StripAccents = class extends Normalizer_default {
/**
* Remove all accents from the text.
* @param text The input text.
* @returns The normalized text without accents.
*/
normalize(text) {
return remove_accents(text);
}
};
var StripAccents_default = StripAccents;
// src/core/normalizer/Lowercase.ts
var Lowercase = class extends Normalizer_default {
/**
* Lowercases the input string.
* @param {string} text The text to normalize.
* @returns {string} The normalized text.
*/
normalize(text) {
return text.toLowerCase();
}
};
var Lowercase_default = Lowercase;
// src/core/normalizer/Prepend.ts
var Prepend = class extends Normalizer_default {
/**
* Prepends the input string.
* @param text The text to normalize.
* @returns The normalized text.
*/
normalize(text) {
text = this.config.prepend + text;
return text;
}
};
var Prepend_default = Prepend;
// src/core/normalizer/create_normalizer.ts
function create_normalizer(config) {
if (config === null) return null;
switch (config.type) {
case "BertNormalizer":
return new BertNormalizer_default(config);
case "Precompiled":
return new Precompiled_default(config);
case "Sequence":
return new Sequence_default(config);
case "Replace":
return new Replace_default(config);
case "NFC":
return new NFC_default(config);
case "NFD":
return new NFD_default(config);
case "NFKC":
return new NFKC_default(config);
case "NFKD":
return new NFKD_default(config);
case "Strip":
return new Strip_default(config);
case "StripAccents":
return new StripAccents_default(config);
case "Lowercase":
return new Lowercase_default(config);
case "Prepend":
return new Prepend_default(config);
default:
throw new Error(`Unknown Normalizer type: ${config.type}`);
}
}
var create_normalizer_default = create_normalizer;
// src/core/PreTokenizer.ts
var PreTokenizer = class extends Callable_default {
/**
* Tokenizes the given text into pre-tokens.
* @param text The text or array of texts to pre-tokenize.
* @param options Additional options for the pre-tokenization logic.
* @returns An array of pre-tokens.
*/
pre_tokenize(text, options) {
return (Array.isArray(text) ? text.map((x) => this.pre_tokenize_text(x, options)) : this.pre_tokenize_text(text, options)).flat();
}
/**
* Alias for {@link PreTokenizer#pre_tokenize}.
* @param text The text or array of texts to pre-tokenize.
* @param options Additional options for the pre-tokenization logic.
* @returns An array of pre-tokens.
*/
_call(text, options) {
return this.pre_tokenize(text, options);
}
};
var PreTokenizer_default = PreTokenizer;
// src/core/preTokenizer/ByteLevel.ts
var ByteLevel = class extends PreTokenizer_default {
/**
* Creates a new instance of the `ByteLevelPreTokenizer` class.
* @param config The configuration object.
*/
constructor(config) {
super();
this.config = config;
this.add_prefix_space = this.config.add_prefix_space ?? false;
this.trim_offsets = this.config.trim_offsets ?? false;
this.use_regex = this.config.use_regex ?? true;
this.pattern = /'s|'t|'re|'ve|'m|'ll|'d| ?\p{L}+| ?\p{N}+| ?[^\s\p{L}\p{N}]+|\s+(?!\S)|\s+/gu;
this.byte_encoder = BYTES_TO_UNICODE;
this.text_encoder = new TextEncoder();
}
/**
* Tokenizes a single piece of text using byte-level tokenization.
* @param text The text to tokenize.
* @param options Additional options for the pre-tokenization logic.
* @returns An array of tokens.
*/
pre_tokenize_text(text, options) {
if (this.add_prefix_space && !text.startsWith(" ")) {
text = " " + text;
}
const tokens = this.use_regex ? text.match(this.pattern) || [] : [text];
return tokens.map(
(token) => Array.from(
this.text_encoder.encode(token),
(byte) => this.byte_encoder[byte]
).join("")
);
}
};
var ByteLevel_default = ByteLevel;
// src/core/preTokenizer/Whitespace.ts
var Whitespace = class extends PreTokenizer_default {
/**
* Pre-tokenizes the input text by splitting it on word boundaries.
* @param text The text to be pre-tokenized.
* @param options Additional options for the pre-tokenization logic.
* @returns An array of tokens produced by splitting the input text on whitespace.
*/
pre_tokenize_text(text, options) {
return text.match(/\w+|[^\w\s]+/g) || [];
}
};
var Whitespace_default = Whitespace;
// src/core/preTokenizer/Metaspace.ts
var Metaspace = class extends PreTokenizer_default {
/**
* @param config The configuration object for the MetaspacePreTokenizer.
*/
constructor(config) {
super();
this.replacement = config.replacement ?? "\u2581";
this.str_rep = config.str_rep || this.replacement;
this.prepend_scheme = config.prepend_scheme ?? "always";
}
/**
* This method takes a string, replaces spaces with the replacement character,
* adds a prefix space if requested, and returns a new list of tokens.
* @param text The text to pre-tokenize.
* @param options The options for the pre-tokenization.
* @returns A new list of pre-tokenized tokens.
*/
pre_tokenize_text(text, options) {
const { section_index = void 0 } = options ?? {};
let normalized = text.replaceAll(" ", this.str_rep);
if (
// We add a prefix space if:
// (1) The normalized token does not already start with the replacement character.
!normalized.startsWith(this.replacement) && // and (2) either:
// (a) prepend_scheme is 'always'
// (b) prepend_scheme is 'first' and this is the first section
(this.prepend_scheme === "always" || this.prepend_scheme === "first" && section_index === 0)
) {
normalized = this.str_rep + normalized;
}
return [normalized];
}
};
var Metaspace_default = Metaspace;
// src/core/preTokenizer/Split.ts
var Split = class extends PreTokenizer_default {
/**
* @param config The configuration options for the pre-tokenizer.
*/
constructor(config) {
super();
this.config = config;
this.pattern = create_pattern(
this.config.pattern ?? {},
this.config.invert ?? true
);
}
/**
* Tokenizes text by splitting it using the given pattern.
* @param text The text to tokenize.
* @returns An array of tokens.
*/
pre_tokenize_text(text) {
if (this.pattern === null) {
return [];
}
if (this.config.invert) {
return text.match(this.pattern) || [];
} else if (this.config.behavior?.toLowerCase() === "removed") {
return text.split(this.pattern).filter((x) => x);
} else {
return regex_split(text, this.pattern);
}
}
};
var Split_default = Split;
// src/core/preTokenizer/Punctuation.ts
var Punctuation = class extends PreTokenizer_default {
/**
* @param config The configuration options for the pre-tokenizer.
*/
constructor(config) {
super();
this.config = config;
this.pattern = new RegExp(
`[^${PUNCTUATION_REGEX}]+|[${PUNCTUATION_REGEX}]+`,
"gu"
);
}
/**
* Tokenizes text by splitting it using the given pattern.
* @param text The text to tokenize.
* @returns An array of tokens.
*/
pre_tokenize_text(text) {
return text.match(this.pattern) || [];
}
};
var Punctuation_default = Punctuation;
// src/core/preTokenizer/Digits.ts
var Digits = class extends PreTokenizer_default {
/**
* @param config The configuration options for the pre-tokenizer.
*/
constructor(config) {
super();
this.config = config;
const digit_pattern = `[^\\d]+|\\d${this.config.individual_digits ? "" : "+"}`;
this.pattern = new RegExp(digit_pattern, "gu");
}
/**
* Tokenizes text by splitting it using the given pattern.
* @param text The text to tokenize.
* @returns An array of tokens.
*/
pre_tokenize_text(text) {
return text.match(this.pattern) || [];
}
};
var Digits_default = Digits;
// src/core/preTokenizer/BertPreTokenizer.ts
var BertPreTokenizer = class extends PreTokenizer_default {
/**
* A PreTokenizer that splits text into wordpieces using a basic tokenization scheme
* similar to that used in the original implementation of BERT.
*/
constructor() {
super();
this.pattern = new RegExp(
`[^\\s${PUNCTUATION_REGEX}]+|[${PUNCTUATION_REGEX}]`,
"gu"
);
}
/**
* Tokenizes a single text using the BERT pre-tokenization scheme.
*
* @param text The text to tokenize.
* @param options Additional options for the pre-tokenization logic.
* @returns An array of tokens.
*/
pre_tokenize_text(text, options) {
return text.trim().match(this.pattern) || [];
}
};
var BertPreTokenizer_default = BertPreTokenizer;
// src/core/preTokenizer/Replace.ts
var Replace2 = class extends PreTokenizer_default {
/**
* @param config The configuration options for the pre-tokenizer.
*/
constructor(config) {
super();
this.config = config;
this.pattern = create_pattern(this.config.pattern ?? {});
this.content = this.config.content ?? "";
}
/**
* Pre-tokenizes the input text by replacing certain characters.
* @param text The text to be pre-tokenized.
* @returns An array of tokens produced by replacing certain characters.
*/
pre_tokenize_text(text) {
if (this.pattern === null) {
return [text];
}
return [text.replaceAll(this.pattern, this.config.content ?? "")];
}
};
var Replace_default2 = Replace2;
// src/core/preTokenizer/Sequence.ts
var Sequence2 = class extends PreTokenizer_default {
/**
* Creates an instance of PreTokenizerSequence.
* @param config The configuration object for the pre-tokenizer sequence.
*/
constructor(config) {
super();
this.tokenizers = (config.pretokenizers ?? []).map(
(x) => create_pre_tokenizer_default(x)
);
}
/**
* Applies each pre-tokenizer in the sequence to the input text in turn.
* @param text The text to pre-tokenize.
* @param options Additional options for the pre-tokenization logic.
* @returns The pre-tokenized text.
*/
pre_tokenize_text(text, options) {
return this.tokenizers.reduce(
(pre_tokenized_text, tokenizer) => {
return tokenizer ? tokenizer.pre_tokenize(pre_tokenized_text, options) : pre_tokenized_text;
},
[text]
);
}
};
var Sequence_default2 = Sequence2;
// src/core/preTokenizer/WhitespaceSplit.ts
var WhitespaceSplit = class extends PreTokenizer_default {
/**
* Pre-tokenizes the input text by splitting it on whitespace characters.
* @param text The text to be pre-tokenized.
* @returns An array of tokens produced by splitting the input text on whitespace.
*/
pre_tokenize_text(text) {
return whitespace_split(text);
}
};
var WhitespaceSplit_default = WhitespaceSplit;
// src/core/preTokenizer/FixedLength.ts
var FixedLength = class extends PreTokenizer_default {
/**
* @param config The configuration options for the pre-tokenizer.
*/
constructor(config) {
super();
this.config = config;
this._length = config.length;
}
/**
* Pre-tokenizes the input text by splitting it into fixed-length tokens.
* @param text The text to be pre-tokenized.
* @returns An array of tokens produced by splitting the input text into fixed-length tokens.
*/
pre_tokenize_text(text) {
const tokens = [];
for (let i = 0; i < text.length; i += this._length) {
tokens.push(text.slice(i, i + this._length));
}
return tokens;
}
};
var FixedLength_default = FixedLength;
// src/core/preTokenizer/create_pre_tokenizer.ts
function create_pre_tokenizer(config) {
if (config === null) return null;
switch (config.type) {
case "BertPreTokenizer":
return new BertPreTokenizer_default();
case "Sequence":
return new Sequence_default2(config);
case "Whitespace":
return new Whitespace_default();
case "WhitespaceSplit":
return new WhitespaceSplit_default();
case "Metaspace":
return new Metaspace_default(config);
case "ByteLevel":
return new ByteLevel_default(config);
case "Split":
return new Split_default(config);
case "Punctuation":
return new Punctuation_default(config);
case "Digits":
return new Digits_default(config);
case "Replace":
return new Replace_default2(config);
case "FixedLength":
return new FixedLength_default(config);
default:
throw new Error(`Unknown PreTokenizer type: ${config.type}`);
}
}
var create_pre_tokenizer_default = create_pre_tokenizer;
// src/core/TokenizerModel.ts
var TokenizerModel = class extends Callable_default {
/**
* Creates a new instance of TokenizerModel.
* @param config The configuration object for the TokenizerModel.
*/
constructor(config) {
super();
this.config = config;
this.vocab = [];
this.tokens_to_ids = /* @__PURE__ */ new Map();
this.unk_token_id = void 0;
this.unk_token = void 0;
this.end_of_word_suffix = void 0;
this.fuse_unk = this.config.fuse_unk ?? false;
}
/**
* Internal function to call the TokenizerModel instance.
* @param tokens The tokens to encode.
* @returns The encoded tokens.
*/
_call(tokens) {
let result = this.encode(tokens);
if (this.fuse_unk) {
result = fuse_unk(result, this.tokens_to_ids, this.unk_token_id);
}
return result;
}
};
var TokenizerModel_default = TokenizerModel;
// src/core/tokenizerModelImplementations/WordPiece.ts
var WordPieceTokenizer = class extends TokenizerModel_default {
/**
* @param config The configuration object.
*/
constructor(config) {
super(config);
/** The maximum number of characters per word. */
this.max_input_chars_per_word = 100;
this.tokens_to_ids = object_to_map(config.vocab);
this.unk_token_id = this.tokens_to_ids.get(config.unk_token);
this.unk_token = config.unk_token;
this.max_input_chars_per_word = config.max_input_chars_per_word ?? 100;
this.vocab = new Array(this.tokens_to_ids.size);
for (const [key, value] of this.tokens_to_ids) {
this.vocab[value] = key;
}
}
/**
* Encodes an array of tokens using WordPiece encoding.
* @param tokens The tokens to encode.
* @returns An array of encoded tokens.
*/
encode(tokens) {
const output_tokens = [];
for (const token of tokens) {
const chars = [...token];
if (chars.length > this.max_input_chars_per_word) {
output_tokens.push(this.unk_token);
continue;
}
let is_unknown = false;
let start = 0;
const sub_tokens = [];
while (start < chars.length) {
let end = chars.length;
let current_substring = null;
while (start < end) {
let substr = chars.slice(start, end).join("");
if (start > 0) {
substr = this.config.continuing_subword_prefix + substr;
}
if (this.tokens_to_ids.has(substr)) {
current_substring = substr;
break;
}
--end;
}
if (current_substring === null) {
is_unknown = true;
break;
}
sub_tokens.push(current_substring);
start = end;
}
if (is_unknown) {
output_tokens.push(this.unk_token);
} else {
output_tokens.push(...sub_tokens);
}
}
return output_tokens;
}
};
var WordPiece_default = WordPieceTokenizer;
// src/utils/data-structures/CharTrie.ts
var CharTrieNode = class _CharTrieNode {
/**
* Create a new CharTrieNode.
* @param is_leaf Whether the node is a leaf node or not.
* @param children A map containing the node's children, where the key is a character and the value is a `CharTrieNode`.
*/
constructor(is_leaf, children) {
this.is_leaf = is_leaf;
this.children = children;
}
/**
* Returns a new `CharTrieNode` instance with default values.
* @returns A new `CharTrieNode` instance with `is_leaf` set to `false` and an empty `children` map.
*/
static default() {
return new _CharTrieNode(false, /* @__PURE__ */ new Map());
}
};
var CharTrie = class {
constructor() {
this.root = CharTrieNode.default();
}
/**
* Adds one or more `texts` to the trie.
* @param texts The strings to add to the trie.
*/
extend(texts) {
for (const text of texts) {
this.push(text);
}
}
/**
* Adds text to the trie.
* @param text The string to add to the trie.
*/
push(text) {
let node = this.root;
for (const ch of text) {
let child = node.children.get(ch);
if (child === void 0) {
child = CharTrieNode.default();
node.children.set(ch, child);
}
node = child;
}
node.is_leaf = true;
}
/**
* Searches the trie for all strings with a common prefix of `text`.
* @param text The common prefix to search for.
* @yields Each string in the trie that has `text` as a prefix.
*/
*common_prefix_search(text) {
let node = this.root;
if (node === void 0) return;
let prefix = "";
for (const ch of text) {
prefix += ch;
node = node.children.get(ch);
if (node === void 0) return;
if (node.is_leaf) {
yield prefix;
}
}
}
};
var CharTrie_default = CharTrie;
// src/utils/data-structures/TokenLattice.ts
var TokenLatticeNode = class _TokenLatticeNode {
/**
* Represents a node in a token lattice for a given sentence.
* @param token_id The ID of the token associated with this node.
* @param node_id The ID of this node.
* @param pos The starting position of the token in the sentence.
* @param length The length of the token.
* @param score The score associated with the token.
*/
constructor(token_id, node_id, pos, length, score) {
this.token_id = token_id;
this.node_id = node_id;
this.pos = pos;
this.length = length;
this.score = score;
this.prev = null;
this.backtrace_score = 0;
}
/**
* Returns a clone of this node.
* @returns A clone of this node.
*/
clone() {
const n = new _TokenLatticeNode(
this.token_id,
this.node_id,
this.pos,
this.length,
this.score
);
n.prev = this.prev;
n.backtrace_score = this.backtrace_score;
return n;
}
};
var TokenLattice = class {
/**
* Creates a new TokenLattice instance.
*
* @param sentence The input sentence to be tokenized.
* @param bos_token_id The beginning-of-sequence token ID.
* @param eos_token_id The end-of-sequence token ID.
*/
constructor(sentence, bos_token_id, eos_token_id) {
this.chars = Array.from(sentence);
this.len = this.chars.length;
this.bos_token_id = bos_token_id;
this.eos_token_id = eos_token_id;
this.nodes = [];
this.begin_nodes = Array.from(
{ length: this.len + 1 },
() => []
);
this.end_nodes = Array.from({ length: this.len + 1 }, () => []);
const bos = new TokenLatticeNode(this.bos_token_id ?? 0, 0, 0, 0, 0);
const eos = new TokenLatticeNode(
this.eos_token_id ?? 0,
1,
this.len,
0,
0
);
this.nodes.push(bos.clone());
this.nodes.push(eos.clone());
this.begin_nodes[this.len].push(eos);
this.end_nodes[0].push(bos);
}
/**
* Inserts a new token node into the token lattice.
*
* @param pos The starting position of the token.
* @param length The length of the token.
* @param score The score of the token.
* @param token_id The token ID of the token.
*/
insert(pos, length, score, token_id) {
const node_id = this.nodes.length;
const node = new TokenLatticeNode(token_id, node_id, pos, length, score);
this.begin_nodes[pos].push(node);
this.end_nodes[pos + length].push(node);
this.nodes.push(node);
}
/**
* Implements the Viterbi algorithm to compute the most likely sequence of tokens.
*
* @returns The most likely sequence of tokens.
*/
viterbi() {
const len2 = this.len;
let pos = 0;
while (pos <= len2) {
if (this.begin_nodes[pos].length == 0) {
return [];
}
for (let rnode of this.begin_nodes[pos]) {
rnode.prev = null;
let best_score = 0;
let best_node = null;
for (let lnode of this.end_nodes[pos]) {
const score = lnode.backtrace_score + rnode.score;
if (best_node === null || score > best_score) {
best_node = lnode.clone();
best_score = score;
}
}
if (best_node !== null) {
rnode.prev = best_node;
rnode.backtrace_score = best_score;
} else {
return [];
}
}
++pos;
}
const results = [];
const root = this.begin_nodes[len2][0];
const prev = root.prev;
if (prev === null) {
return [];
}
let node = prev.clone();
while (node.prev !== null) {
results.push(node.clone());
const n = node.clone();
node = n.prev.clone();
}
results.reverse();
return results;
}
/**
* Get the text piece for a given node.
* @param node The node to get the piece for.
* @returns The array of nodes representing the most likely sequence of tokens.
*/
piece(node) {
return this.chars.slice(node.pos, node.pos + node.length).join("");
}
/**
* @returns The most likely sequence of tokens.
*/
tokens() {
const nodes = this.viterbi();
return nodes.map((x) => this.piece(x));
}
/**
* @returns The most likely sequence of token ids.
*/
token_ids() {
const nodes = this.viterbi();
return nodes.map((x) => x.token_id);
}
};
var TokenLattice_default = TokenLattice;
// src/utils/maths.ts
function min(arr) {
if (arr.length === 0) throw new Error("Array must not be empty");
let min_value = arr[0];
let index_of_min = 0;
for (let i = 1; i < arr.length; ++i) {
if (arr[i] < min_value) {
min_value = arr[i];
index_of_min = i;
}
}
return [min_value, index_of_min];
}
// src/core/tokenizerModelImplementations/Unigram.ts
var Unigram = class extends TokenizerModel_default {
/**
* Create a new Unigram tokenizer model.
* @param config The configuration object for the Unigram model.
* @param eos_token
*/
constructor(config, eos_token) {
super(config);
const vocab_size = config.vocab.length;
this.vocab = new Array(vocab_size);
this.scores = new Array(vocab_size);
for (let i = 0; i < vocab_size; ++i) {
[this.vocab[i], this.scores[i]] = config.vocab[i];
}
this.unk_token_id = config.unk_id;
this.unk_token = this.vocab[config.unk_id];
this.tokens_to_ids = new Map(this.vocab.map((x, i) => [x, i]));
this.bos_token = " ";
this.bos_token_id = this.tokens_to_ids.get(this.bos_token);
this.eos_token = eos_token;
this.eos_token_id = this.tokens_to_ids.get(this.eos_token);
this.unk_token = this.vocab[this.unk_token_id];
this.min_score = min(this.scores)[0];
this.unk_score = this.min_score - 10;
this.scores[this.unk_token_id] = this.unk_score;
this.trie = new CharTrie_default();
this.trie.extend(this.vocab);
this.fuse_unk = true;
}
/**
* Populates lattice nodes.
* @param lattice The token lattice to populate with nodes.
*/
populate_nodes(lattice) {
const chars = lattice.chars;
const mblen = 1;
let begin_pos = 0;
while (begin_pos < chars.length) {
let has_single_node = false;
const tokens = [];
const sliced = chars.slice(begin_pos).join("");
const prefixed_tokens = this.trie.common_prefix_search(sliced);
for (const token of prefixed_tokens) {
tokens.push(token);
const token_id = this.tokens_to_ids.get(token);
const token_score = this.scores[token_id];
const n = len(token);
lattice.insert(begin_pos, n, token_score, token_id);
if (!has_single_node && n === mblen) {
has_single_node = true;
}
}
if (!has_single_node) {
lattice.insert(begin_pos, mblen, this.unk_score, this.unk_token_id);
}
begin_pos += mblen;
}
}
/**
* Encodes an array of tokens into an array of subtokens using the unigram model.
*
* @param normalized The normalized string.
* @returns An array of subtokens obtained by encoding the input tokens using the unigram model.
*/
tokenize(normalized) {
const lattice = new TokenLattice_default(
normalized,
this.bos_token_id,
this.eos_token_id
);
this.populate_nodes(lattice);
return lattice.tokens();
}
/**
* Encodes an array of tokens using Unigram encoding.
* @param tokens The tokens to encode.
* @returns An array of encoded tokens.
*/
encode(tokens) {
const to_return = [];
for (const token of tokens) {
const tokenized = this.tokenize(token);
to_return.push(...tokenized);
}
return to_return;
}
};
var Unigram_default = Unigram;
// src/utils/data-structures/PriorityQueue.ts
var PriorityQueue = class {
/**
* Create a new PriorityQueue.
* @param comparator Comparator function to determine priority. Defaults to a MaxHeap.
* @param max_size Maximum size of the queue. Defaults to Infinity.
*/
constructor(comparator = (a, b) => a > b, max_size = Infinity) {
this._heap = [];
this._comparator = comparator;
this._max_size = max_size;
}
/**
* The size of the queue
*/
get size() {
return this._heap.length;
}
/**
* Check if the queue is empty.
* @returns `true` if the queue is empty, `false` otherwise.
*/
is_empty() {
return this.size === 0;
}
/**
* Return the element with the highest priority in the queue.
* @returns The highest priority element in the queue.
*/
peek() {
return this._heap[0];
}
/**
* Add one or more elements to the queue.
* @param values The values to push into the queue.
* @returns The new size of the queue.
*/
push(...values) {
return this.extend(values);
}
/**
* Add multiple elements to the queue.
* @param values The values to push into the queue.
* @returns The new size of the queue.
*/
extend(values) {
for (const value of values) {
if (this.size < this._max_size) {
this._heap.push(value);
this._sift_up();
} else {
const smallest = this._smallest();
if (this._comparator(value, this._heap[smallest])) {
this._heap[smallest] = value;
this._sift_up_from(smallest);
}
}
}
return this.size;
}
/**
* Remove and return the element with the highest priority in the queue.
* @returns The element with the highest priority in the queue.
*/
pop() {
const popped_value = this.peek();
const bottom = this.size - 1;
if (bottom > 0) {
this._swap(0, bottom);
}
this._heap.pop();
this._sift_down();
return popped_value;
}
/**
* Replace the element with the highest priority in the queue with a new value.
* @param value The new value.
* @returns The replaced value.
*/
replace(value) {
const replaced_value = this.peek();
this._heap[0] = value;
this._sift_down();
return replaced_value;
}
/**
* Compute the index for the parent of the node at index `i`.
* @param i The index of the node to get the parent of.
* @returns The index of the parent node.
* @private
*/
_parent(i) {
return (i + 1 >>> 1) - 1;
}
/**
* Compute the index for the left child of the node at index `i`.
* @param i The index of the node to get the left child of.
* @returns The index of the left child.
* @private
*/
_left(i) {
return (i << 1) + 1;
}
/**
* Compute the index for the right child of the node at index `i`.
* @param i The index of the node to get the right child of.
* @returns The index of the right child.
* @private
*/
_right(i) {
return i + 1 << 1;
}
/**
* Check if the element at index `i` is greater than the element at index `j`.
* @param i The index of the first element to compare.
* @param j The index of the second element to compare.
* @returns `true` if the element at index `i` is greater than the element at index `j`, `false` otherwise.
* @private
*/
_greater(i, j) {
return this._comparator(this._heap[i], this._heap[j]);
}
/**
* Swap the elements at indices `i` and `j`.
* @param i The index of the first element to swap.
* @param j The index of the second element to swap.
* @private
*/
_swap(i, j) {
const temp = this._heap[i];
this._heap[i] = this._heap[j];
this._heap[j] = temp;
}
/**
* Maintain the heap property by updating positions in the heap,
* starting at the last element and moving up the heap.
* @private
*/
_sift_up() {
this._sift_up_from(this.size - 1);
}
/**
* Helper function to sift up from a given node.
* @param node The index of the node to start sifting up from.
*/
_sift_up_from(node) {
while (node > 0 && this._greater(node, this._parent(node))) {
this._swap(node, this._parent(node));
node = this._parent(node);
}
}
/**
* Maintain the heap property by updating positions in the heap,
* starting at the first element and moving down the heap.
* @private
*/
_sift_down() {
let node = 0;
while (this._left(node) < this.size && this._greater(this._left(node), node) || this._right(node) < this.size && this._greater(this._right(node), node)) {
const max_child = this._right(node) < this.size && this._greater(this._right(node), this._left(node)) ? this._right(node) : this._left(node);
this._swap(node, max_child);
node = max_child;
}
}
/**
* Get the index of the smallest element in the heap. Since we use an array-based heap,
* the index can be computed without needing to traverse the heap.
* @private
*/
_smallest() {
return 2 ** Math.floor(Math.log2(this.size)) - 1;
}
};
var PriorityQueue_default = PriorityQueue;
// src/utils/data-structures/LRUCache.ts
var LRUCache = class {
/**
* Creates an LRUCache instance.
* @param capacity The maximum number of items the cache can hold.
*/
constructor(capacity) {
this.capacity = capacity;
this.cache = /* @__PURE__ */ new Map();
}
/**
* Retrieves the value associated with the given key and marks the key as recently used.
* @param key The key to retrieve.
* @returns The value associated with the key, or undefined if the key does not exist.
*/
get(key) {
if (!this.cache.has(key)) return void 0;
const value = this.cache.get(key);
this.cache.delete(key);
this.cache.set(key, value);
return value;
}
/**
* Inserts or updates the key-value pair in the cache.
* If the key already exists, it is updated and marked as recently used.
* If the cache exceeds its capacity, the least recently used item is evicted.
* @param key The key to add or update.
* @param value The value to associate with the key.
*/
put(key, value) {
if (this.cache.has(key)) {
this.cache.delete(key);
}
this.cache.set(key, value);
if (this.cache.size > this.capacity) {
this.cache.delete(this.cache.keys().next().value);
}
}
/**
* Clears the cache.
*/
clear() {
this.cache.clear();
}
};
var LRUCache_default = LRUCache;
// src/core/tokenizerModelImplementations/BPE.ts
var BPE = class extends TokenizerModel_default {
/**
* Create a BPE instance.
* @param config The configuration object for BPE.
*/
constructor(config) {
super(config);
this.tokens_to_ids = object_to_map(config.vocab);
this.unk_token_id = this.tokens_to_ids.get(config.unk_token);
this.unk_token = config.unk_token;
this.vocab = new Array(this.tokens_to_ids.size);
for (const [key, value] of this.tokens_to_ids) {
this.vocab[value] = key;
}
const use_new_merge_format = Array.isArray(config.merges[0]);
this.merges = use_new_merge_format ? config.merges : config.merges.map(
(x) => x.split(" ", 2)
);
this.bpe_ranks = new Map(this.merges.map((x, i) => [JSON.stringify(x), i]));
this.end_of_word_suffix = config.end_of_word_suffix;
this.continuing_subword_suffix = config.continuing_subword_suffix ?? null;
this.byte_fallback = this.config.byte_fallback ?? false;
if (this.byte_fallback) {
this.text_encoder = new TextEncoder();
}
this.ignore_merges = this.config.ignore_merges ?? false;
this.max_length_to_cache = 256;
this.cache_capacity = 1e4;
this.cache = new LRUCache_default(this.cache_capacity);
}
/**
* Clears the cache.
*/
clear_cache() {
this.cache.clear();
}
/**
* Apply Byte-Pair-Encoding (BPE) to a given token. Efficient heap-based priority
* queue implementation adapted from https://github.com/belladoreai/llama-tokenizer-js.
* @param token The token to encode.
* @returns The BPE encoded tokens.
*/
bpe(token) {
if (token.length === 0) {
return [];
}
const cached = this.cache.get(token);
if (cached !== void 0) {
return cached;
}
const word = Array.from(token);
if (this.end_of_word_suffix) {
word[word.length - 1] += this.end_of_word_suffix;
}
let result = [];
if (word.length > 1) {
const queue = new PriorityQueue_default((a, b) => a.score < b.score);
let starting_node = {
token: word[0],
bias: 0,
prev: null,
next: null
};
let previous_node = starting_node;
for (let i = 1; i < word.length; ++i) {
const current_node = {
bias: i / word.length,
// Add fractional component to break ties
token: word[i],
prev: previous_node,
next: null
};
previous_node.next = current_node;
this.add_node(queue, previous_node);
previous_node = current_node;
}
while (!queue.is_empty()) {
const node = queue.pop();
if (node.deleted || !node.next || node.next.deleted) continue;
node.deleted = true;
node.next.deleted = true;
if (node.prev) {
const new_previous_node = { ...node.prev };
node.prev.deleted = true;
node.prev = new_previous_node;
if (new_previous_node.prev) {
new_previous_node.prev.next = new_previous_node;
} else {
starting_node = new_previous_node;
}
}
const merged = {
token: node.token + node.next.token,
bias: node.bias,
prev: node.prev,
next: node.next.next
};
if (merged.prev) {
merged.prev.next = merged;
this.add_node(queue, merged.prev);
} else {
starting_node = merged;
}
if (merged.next) {
merged.next.prev = merged;
this.add_node(queue, merged);
}
}
for (let current_node = starting_node; current_node !== null; current_node = current_node.next) {
result.push(current_node.token);
}
} else {
result = word;
}
if (this.continuing_subword_suffix) {
for (let i = 0; i < result.length - 1; ++i) {
result[i] += this.continuing_subword_suffix;
}
}
if (token.length < this.max_length_to_cache) {
this.cache.put(token, result);
}
return result;
}
/**
* Helper function to add a node to the priority queue.
* @param queue
* @param node
*/
add_node(queue, node) {
const rank = this.bpe_ranks.get(
JSON.stringify([node.token, node.next.token])
);
if (rank !== void 0) {
node.score = rank + node.bias;
queue.push(node);
}
}
/**
* Encodes the input sequence of tokens using the BPE algorithm and returns the resulting subword tokens.
* @param tokens The input sequence of tokens to encode.
* @returns The resulting subword tokens after applying the BPE algorithm to the input sequence of tokens.
*/
encode(tokens) {
const output_tokens = [];
for (const token of tokens) {
if (this.ignore_merges && this.tokens_to_ids.has(token)) {
output_tokens.push(token);
continue;
}
const bpe_token_list = this.bpe(token);
for (const t of bpe_token_list) {
if (this.tokens_to_ids.has(t)) {
output_tokens.push(t);
} else if (this.byte_fallback) {
const byte_tokens = Array.from(this.text_encoder.encode(t)).map(
(x) => `<0x${x.toString(16).toUpperCase().padStart(2, "0")}>`
);
if (byte_tokens.every((x) => this.tokens_to_ids.has(x))) {
output_tokens.push(...byte_tokens);
} else if (this.unk_token != null) {
output_tokens.push(this.unk_token);
}
} else if (this.unk_token != null) {
output_tokens.push(this.unk_token);
}
}
}
return output_tokens;
}
};
var BPE_default = BPE;
// src/core/tokenizerModelImplementations/Legacy.ts
var Legacy = class extends TokenizerModel_default {
/**
* Create a Legacy tokenizer model instance.
* @param config The configuration object for Legacy tokenizer model.
* @param more_config Additional configuration object for the Legacy tokenizer model.
*/
constructor(config, more_config) {
super(config);
const vocab = config.vocab;
this.tokens_to_ids = object_to_map(
more_config.target_lang ? vocab[more_config.target_lang] : vocab
);
this.bos_token = more_config.bos_token;
this.bos_token_id = this.tokens_to_ids.get(this.bos_token);
this.eos_token = more_config.eos_token;
this.eos_token_id = this.tokens_to_ids.get(this.eos_token);
this.pad_token = more_config.pad_token;
this.pad_token_id = this.tokens_to_ids.get(this.pad_token);
this.unk_token = more_config.unk_token;
this.unk_token_id = this.tokens_to_ids.get(this.unk_token);
this.vocab = new Array(this.tokens_to_ids.size);
for (const [key, value] of this.tokens_to_ids) {
this.vocab[value] = key;
}
}
encode(tokens) {
return tokens;
}
};
var Legacy_default = Legacy;
// src/core/tokenizerModelImplementations/create_tokenizer_model.ts
function create_tokenizer_model(model_config, config) {
switch (model_config.type) {
case "WordPiece":
return new WordPiece_default(model_config);
case "Unigram":
return new Unigram_default(model_config, config.eos_token);
case "BPE":
return new BPE_default(model_config);
default:
if (model_config.vocab) {
if (Array.isArray(model_config.vocab)) {
return new Unigram_default(model_config, config.eos_token);
} else if (Object.hasOwn(model_config, "continuing_subword_prefix") && Object.hasOwn(model_config, "unk_token")) {
if (Object.hasOwn(model_config, "merges")) {
return new BPE_default(model_config);
} else {
return new WordPiece_default(model_config);
}
} else {
return new Legacy_default(model_config, {
target_lang: config.target_lang,
bos_token: config.bos_token,
eos_token: config.eos_token,
pad_token: config.pad_token,
unk_token: config.unk_token
});
}
}
throw new Error(
`Unknown TokenizerModel type: ${model_config?.type}`
);
}
}
var create_tokenizer_model_default = create_tokenizer_model;
// src/core/PostProcessor.ts
var PostProcessor = class extends Callable_default {
/**
* @param config The configuration for the post-processor.
*/
constructor(config) {
super();
this.config = config;
}
/**
* Alias for {@link PostProcessor#post_process}.
* @param tokens The text or array of texts to post-process.
* @param args Additional arguments required by the post-processing logic.
* @returns The post-processed tokens.
*/
_call(tokens, ...args) {
return this.post_process(tokens, ...args);
}
};
var PostProcessor_default = PostProcessor;
// src/core/postProcessor/TemplateProcessing.ts
var TemplateProcessing = class extends PostProcessor_default {
/**
* Replaces special tokens in the template with actual tokens.
* @param tokens The list of tokens for the first sequence.
* @param tokens_pair The list of tokens for the second sequence (optional).
* @param add_special_tokens Whether to add the special tokens to the beginning and end of the input.
* @returns An object containing the list of tokens with the special tokens replaced with actual tokens.
*/
post_process(tokens, tokens_pair = null, add_special_tokens = true) {
const type = tokens_pair === null ? this.config.single : this.config.pair;
let processed_tokens = [];
let types = [];
for (const item of type) {
if ("SpecialToken" in item) {
if (add_special_tokens) {
processed_tokens.push(item.SpecialToken.id);
types.push(item.SpecialToken.type_id);
}
} else if ("Sequence" in item) {
if (item.Sequence.id === "A") {
processed_tokens = merge_arrays(processed_tokens, tokens);
types = merge_arrays(
types,
new Array(tokens.length).fill(item.Sequence.type_id)
);
} else if (item.Sequence.id === "B") {
processed_tokens = merge_arrays(processed_tokens, tokens_pair);
types = merge_arrays(
types,
new Array(tokens_pair.length).fill(item.Sequence.type_id)
);
}
}
}
return { tokens: processed_tokens, token_type_ids: types };
}
};
var TemplateProcessing_default = TemplateProcessing;
// src/core/postProcessor/ByteLevel.ts
var ByteLevel2 = class extends PostProcessor_default {
/**
* Post process the given tokens.
* @param tokens The list of tokens for the first sequence.
* @param tokens_pair The list of tokens for the second sequence (optional).
* @returns An object containing the post-processed tokens.
*/
post_process(tokens, tokens_pair = null) {
return { tokens, tokens_pair };
}
};
var ByteLevel_default2 = ByteLevel2;
// src/core/postProcessor/BertProcessing.ts
var BertProcessing = class extends PostProcessor_default {
/**
* @param config The configuration for the post-processor.
* @param config.cls The special tokens to add to the beginning of the input.
* @param config.sep The special tokens to add to the end of the input.
*/
constructor(config) {
super(config);
this.sep = config.sep;
this.cls = config.cls;
}
/**
* Adds the special tokens to the beginning and end of the input.
* @param tokens The input tokens.
* @param tokens_pair An optional second set of input tokens.
* @param add_special_tokens Whether to add the special tokens to the beginning and end of the input.
* @returns The post-processed tokens with the special tokens added to the beginning and end.
*/
post_process(tokens, tokens_pair = null, add_special_tokens = true) {
if (add_special_tokens) {
tokens = merge_arrays([this.cls[0]], tokens, [this.sep[0]]);
}
let token_type_ids = new Array(tokens.length).fill(0);
if (tokens_pair) {
const middle = [];
const after = add_special_tokens ? [this.sep[0]] : [];
tokens = merge_arrays(tokens, middle, tokens_pair, after);
token_type_ids = merge_arrays(
token_type_ids,
new Array(tokens_pair.length + middle.length + after.length).fill(1)
);
}
return { tokens, token_type_ids };
}
};
var BertProcessing_default = BertProcessing;
// src/core/postProcessor/RobertaProcessing.ts
var RobertaProcessing = class extends PostProcessor_default {
/**
* @param config The configuration for the post-processor.
* @param config.cls The special tokens to add to the beginning of the input.
* @param config.sep The special tokens to add to the end of the input.
*/
constructor(config) {
super(config);
this.sep = config.sep;
this.cls = config.cls;
}
/**
* Adds the special tokens to the beginning and end of the input.
* @param tokens The input tokens.
* @param tokens_pair An optional second set of input tokens.
* @param add_special_tokens Whether to add the special tokens to the beginning and end of the input.
* @returns The post-processed tokens with the special tokens added to the beginning and end.
*/
post_process(tokens, tokens_pair, add_special_tokens = true) {
if (add_special_tokens) {
tokens = merge_arrays([this.cls[0]], tokens, [this.sep[0]]);
}
let token_type_ids = new Array(tokens.length).fill(0);
if (tokens_pair) {
const middle = add_special_tokens ? [this.sep[0]] : [];
const after = add_special_tokens ? [this.sep[0]] : [];
tokens = merge_arrays(tokens, middle, tokens_pair, after);
token_type_ids = merge_arrays(
token_type_ids,
new Array(tokens_pair.length + middle.length + after.length).fill(1)
);
}
return { tokens, token_type_ids };
}
};
var RobertaProcessing_default = RobertaProcessing;
// src/core/postProcessor/Sequence.ts
var Sequence3 = class extends PostProcessor_default {
/**
* Creates a new instance of Sequence post-processor.
* @param config The configuration object.
*/
constructor(config) {
super(config);
this.processors = (config.processors ?? []).map((x) => create_post_processor_default(x));
}
/**
* Post process the given tokens.
* @param tokens The list of tokens for the first sequence.
* @param tokens_pair The list of tokens for the second sequence (optional).
* @param add_special_tokens Whether to add the special tokens to the beginning and end of the input.
* @returns An object containing the post-processed tokens.
*/
post_process(tokens, tokens_pair = null, add_special_tokens = true) {
let processed_tokens = { tokens, tokens_pair };
for (const processor of this.processors) {
processed_tokens = processor.post_process(
processed_tokens.tokens,
processed_tokens.tokens_pair,
add_special_tokens
);
}
return processed_tokens;
}
};
var Sequence_default3 = Sequence3;
// src/core/postProcessor/create_post_processor.ts
function create_post_processor(config) {
if (config === null) return null;
switch (config.type) {
case "TemplateProcessing":
return new TemplateProcessing_default(config);
case "ByteLevel":
return new ByteLevel_default2(config);
case "BertProcessing":
return new BertProcessing_default(config);
case "RobertaProcessing":
return new RobertaProcessing_default(config);
case "Sequence":
return new Sequence_default3(config);
default:
throw new Error(`Unknown PostProcessor type: ${config.type}`);
}
}
var create_post_processor_default = create_post_processor;
// src/core/Decoder.ts
var Decoder = class extends Callable_default {
/**
* Creates an instance of `Decoder`.
* @param config The configuration object.
**/
constructor(config) {
super();
this.config = config;
this.added_tokens = [];
this.end_of_word_suffix = null;
this.trim_offsets = "trim_offsets" in config ? config.trim_offsets : false;
}
/**
* Calls the `decode` method.
*
* @param tokens The list of tokens.
* @returns The decoded string.
*/
_call(tokens) {
return this.decode(tokens);
}
/**
* Decodes a list of tokens.
* @param tokens The list of tokens.
* @returns The decoded string.
*/
decode(tokens) {
return this.decode_chain(tokens).join("");
}
};
var Decoder_default = Decoder;
// src/core/decoder/ByteLevel.ts
var ByteLevel3 = class extends Decoder_default {
/**
* Create a `ByteLevelDecoder` object.
*/
constructor(config) {
super(config);
this.byte_decoder = UNICODE_TO_BYTES;
this.text_decoder = new TextDecoder("utf-8", {
fatal: false,
// eslint-disable-next-line @typescript-eslint/naming-convention
ignoreBOM: true
});
this.end_of_word_suffix = null;
}
/**
* Convert an array of tokens to string by decoding each byte.
* @param tokens Array of tokens to be decoded.
* @returns The decoded string.
*/
convert_tokens_to_string(tokens) {
const text = tokens.join("");
const byte_array = new Uint8Array(
[...text].map((c) => this.byte_decoder[c])
);
return this.text_decoder.decode(byte_array);
}
decode_chain(tokens) {
const sub_texts = [];
let current_sub_text = [];
for (const token of tokens) {
if (this.added_tokens.find((x) => x.content === token) !== void 0) {
if (current_sub_text.length > 0) {
sub_texts.push(this.convert_tokens_to_string(current_sub_text));
current_sub_text = [];
}
sub_texts.push(token);
} else {
current_sub_text.push(token);
}
}
if (current_sub_text.length > 0) {
sub_texts.push(this.convert_tokens_to_string(current_sub_text));
}
return sub_texts;
}
};
var ByteLevel_default3 = ByteLevel3;
// src/core/decoder/WordPiece.ts
var WordPiece = class extends Decoder_default {
/**
* Creates a new instance of WordPieceDecoder.
* @param config The configuration object.
*/
constructor(config) {
super(config);
this.cleanup = config.cleanup;
}
decode_chain(tokens) {
return tokens.map((token, i) => {
if (i !== 0) {
const prefix = this.config.prefix;
if (prefix && token.startsWith(prefix)) {
token = token.replace(prefix, "");
} else {
token = " " + token;
}
}
if (this.cleanup) {
token = clean_up_tokenization(token);
}
return token;
});
}
};
var WordPiece_default2 = WordPiece;
// src/core/decoder/Metaspace.ts
var Metaspace2 = class extends Decoder_default {
/**
* Constructs a new MetaspaceDecoder object.
* @param config The configuration object for the MetaspaceDecoder.
*/
constructor(config) {
super(config);
this.replacement = config.replacement ?? "\u2581";
}
decode_chain(tokens) {
const result = [];
for (let i = 0; i < tokens.length; ++i) {
let normalized = tokens[i].replaceAll(this.replacement, " ");
if (i == 0 && normalized.startsWith(" ")) {
normalized = normalized.substring(1);
}
result.push(normalized);
}
return result;
}
};
var Metaspace_default2 = Metaspace2;
// src/core/decoder/BPE.ts
var BPE2 = class extends Decoder_default {
constructor(config) {
super(config);
this.suffix = config.suffix ?? "";
}
decode_chain(tokens) {
return tokens.map((token, i) => {
return token.replaceAll(this.suffix, i === tokens.length - 1 ? "" : " ");
});
}
};
var BPE_default2 = BPE2;
// src/core/decoder/CTC.ts
var CTC = class extends Decoder_default {
constructor(config) {
super(config);
this.pad_token = config.pad_token ?? "";
this.word_delimiter_token = config.word_delimiter_token ?? "";
this.cleanup = config.cleanup;
}
/**
* Converts a connectionist-temporal-classification (CTC) output tokens into a single string.
* @param tokens Array of tokens to be decoded.
* @returns The decoded string.
*/
convert_tokens_to_string(tokens) {
if (tokens.length === 0) return "";
const grouped_tokens = [tokens[0]];
for (let i = 1; i < tokens.length; ++i) {
if (tokens[i] !== grouped_tokens.at(-1)) {
grouped_tokens.push(tokens[i]);
}
}
const filtered_tokens = grouped_tokens.filter(
(token) => token !== this.pad_token
);
let text = filtered_tokens.join("");
if (this.cleanup) {
text = clean_up_tokenization(text).replaceAll(this.word_delimiter_token, " ").trim();
}
return text;
}
decode_chain(tokens) {
return [this.convert_tokens_to_string(tokens)];
}
};
var CTC_default = CTC;
// src/core/decoder/Sequence.ts
var Sequence4 = class extends Decoder_default {
/**
* Creates a new instance of DecoderSequence.
* @param config The configuration object.
*/
constructor(config) {
super(config);
this.decoders = (config.decoders ?? []).map((x) => create_decoder_default(x));
}
decode_chain(tokens) {
return this.decoders.reduce((toks, decoder) => {
return decoder.decode_chain(toks);
}, tokens);
}
};
var Sequence_default4 = Sequence4;
// src/core/decoder/Replace.ts
var Replace3 = class extends Decoder_default {
decode_chain(tokens) {
const pattern = create_pattern(this.config.pattern);
const content = this.config.content ?? "";
return pattern === null ? tokens : tokens.map((token) => token.replaceAll(pattern, content));
}
};
var Replace_default3 = Replace3;
// src/core/decoder/Fuse.ts
var Fuse = class extends Decoder_default {
decode_chain(tokens) {
return [tokens.join("")];
}
};
var Fuse_default = Fuse;
// src/core/decoder/Strip.ts
var Strip2 = class extends Decoder_default {
constructor(config) {
super(config);
this.content = config.content ?? "";
this.start = config.start ?? 0;
this.stop = config.stop ?? 0;
}
decode_chain(tokens) {
return tokens.map((token) => {
let start_cut = 0;
for (let i = 0; i < this.start; ++i) {
if (token[i] === this.content) {
start_cut = i + 1;
continue;
} else {
break;
}
}
let stop_cut = token.length;
for (let i = 0; i < this.stop; ++i) {
const index = token.length - i - 1;
if (token[index] === this.content) {
stop_cut = index;
continue;
} else {
break;
}
}
return token.slice(start_cut, stop_cut);
});
}
};
var Strip_default2 = Strip2;
// src/core/decoder/ByteFallback.ts
var ByteFallback = class extends Decoder_default {
constructor(config) {
super(config);
this.text_decoder = new TextDecoder();
}
decode_chain(tokens) {
const new_tokens = [];
let previous_byte_tokens = [];
for (const token of tokens) {
let bytes = null;
if (token.length === 6 && token.startsWith("<0x") && token.endsWith(">")) {
const byte = parseInt(token.slice(3, 5), 16);
if (!isNaN(byte)) {
bytes = byte;
}
}
if (bytes !== null) {
previous_byte_tokens.push(bytes);
} else {
if (previous_byte_tokens.length > 0) {
const string = this.text_decoder.decode(
Uint8Array.from(previous_byte_tokens)
);
new_tokens.push(string);
previous_byte_tokens = [];
}
new_tokens.push(token);
}
}
if (previous_byte_tokens.length > 0) {
const string = this.text_decoder.decode(
Uint8Array.from(previous_byte_tokens)
);
new_tokens.push(string);
previous_byte_tokens = [];
}
return new_tokens;
}
};
var ByteFallback_default = ByteFallback;
// src/core/decoder/create_decoder.ts
function create_decoder(config) {
if (config === null) return null;
switch (config.type) {
case "ByteLevel":
return new ByteLevel_default3(config);
case "WordPiece":
return new WordPiece_default2(config);
case "Metaspace":
return new Metaspace_default2(config);
case "BPEDecoder":
return new BPE_default2(config);
case "CTC":
return new CTC_default(config);
case "Sequence":
return new Sequence_default4(config);
case "Replace":
return new Replace_default3(config);
case "Fuse":
return new Fuse_default(config);
case "Strip":
return new Strip_default2(config);
case "ByteFallback":
return new ByteFallback_default(config);
default:
throw new Error(`Unknown Decoder type: ${config.type}`);
}
}
var create_decoder_default = create_decoder;
// src/core/Tokenizer.ts
var Tokenizer = class {
constructor(tokenizer, config) {
const tokenizer_error = validate_object(tokenizer, "Tokenizer", [
"model",
"decoder",
"post_processor",
"pre_tokenizer",
"normalizer"
]);
if (tokenizer_error) {
throw new Error(tokenizer_error);
}
const config_error = validate_object(config, "Config");
if (config_error) {
throw new Error(config_error);
}
this.tokenizer = tokenizer;
this.config = config;
this.normalizer = create_normalizer_default(this.tokenizer.normalizer);
this.pre_tokenizer = create_pre_tokenizer_default(this.tokenizer.pre_tokenizer);
this.model = create_tokenizer_model_default(this.tokenizer.model, this.config);
this.post_processor = create_post_processor_default(this.tokenizer.post_processor);
this.decoder = create_decoder_default(this.tokenizer.decoder);
this.special_tokens = [];
this.all_special_ids = [];
this.added_tokens = [];
const unnormalized_contents = [];
const normalized_contents = [];
this.added_tokens_map = /* @__PURE__ */ new Map();
for (const added_token of this.tokenizer.added_tokens) {
const token = new AddedToken_default(added_token);
this.added_tokens.push(token);
this.model.tokens_to_ids.set(token.content, token.id);
this.model.vocab[token.id] = token.content;
if (token.special) {
this.special_tokens.push(token.content);
this.all_special_ids.push(token.id);
}
this.added_tokens_map.set(token.content, token);
if (token.normalized && this.normalizer !== null) {
const normalized_content = this.normalizer(token.content);
normalized_contents.push(normalized_content);
this.added_tokens_map.set(normalized_content, token);
} else {
unnormalized_contents.push(token.content);
}
}
(this.config.additional_special_tokens ?? []).forEach((token) => {
if (!this.special_tokens.includes(token)) this.special_tokens.push(token);
});
if (this.decoder) {
this.decoder.added_tokens = this.added_tokens;
this.decoder.end_of_word_suffix = this.model.end_of_word_suffix;
}
this.splitter_unnormalized = new DictionarySplitter_default(unnormalized_contents);
this.splitter_normalized = new DictionarySplitter_default(normalized_contents);
this.remove_space = this.config.remove_space;
this.clean_up_tokenization_spaces = this.config.clean_up_tokenization_spaces ?? true;
this.do_lowercase_and_remove_accent = this.config.do_lowercase_and_remove_accent ?? false;
}
// Implementation
encode(text, {
text_pair = null,
add_special_tokens = true,
return_token_type_ids = null
} = {}) {
const { tokens, token_type_ids } = this.tokenize_helper(text, {
text_pair,
add_special_tokens
});
const input_ids = tokens.map(
(t) => this.added_tokens_map.get(t)?.id ?? this.model.tokens_to_ids.get(t) ?? this.model.unk_token_id
);
const result = {
ids: input_ids,
tokens,
attention_mask: new Array(input_ids.length).fill(1)
};
if (return_token_type_ids && token_type_ids) {
result.token_type_ids = token_type_ids;
}
return result;
}
decode(token_ids, options = {}) {
if (!Array.isArray(token_ids) || token_ids.length === 0 || !is_integral_number(token_ids[0])) {
throw Error("token_ids must be a non-empty array of integers.");
}
let tokens = token_ids.map(
(i) => this.model.vocab[Number(i)] ?? this.model.unk_token
);
if (options.skip_special_tokens) {
tokens = tokens.filter((x) => !this.special_tokens.includes(x));
}
let decoded = this.decoder ? this.decoder(tokens) : tokens.join(" ");
if (this.decoder && this.decoder.end_of_word_suffix) {
decoded = decoded.replaceAll(this.decoder.end_of_word_suffix, " ");
if (options.skip_special_tokens) {
decoded = decoded.trim();
}
}
if (options.clean_up_tokenization_spaces ?? this.clean_up_tokenization_spaces) {
decoded = clean_up_tokenization(decoded);
}
return decoded;
}
/**
* Converts a string into a sequence of tokens.
* @param text The sequence to be encoded.
* @param options An optional object containing the following properties:
* @returns The list of tokens.
*/
tokenize(text, { text_pair = null, add_special_tokens = false } = {}) {
return this.tokenize_helper(text, { text_pair, add_special_tokens }).tokens;
}
encode_text(text) {
if (text === null) {
return null;
}
const sections = this.splitter_unnormalized.split(text);
sections.forEach((section, i) => {
const added_token = this.added_tokens_map.get(section);
if (added_token) {
if (added_token.lstrip && i > 0) {
sections[i - 1] = sections[i - 1].trimEnd();
}
if (added_token.rstrip && i < sections.length - 1) {
sections[i + 1] = sections[i + 1].trimStart();
}
}
});
return sections.flatMap((processed_text, section_index) => {
if (processed_text.length === 0) {
return [];
}
if (this.added_tokens_map.has(processed_text)) {
return [processed_text];
}
if (this.remove_space === true) {
processed_text = processed_text.trim().split(/\s+/).join(" ");
}
if (this.do_lowercase_and_remove_accent) {
processed_text = lowercase_and_remove_accents(processed_text);
}
if (this.normalizer !== null) {
processed_text = this.normalizer(processed_text);
}
if (processed_text.length === 0) {
return [];
}
const subsections = this.splitter_normalized.split(processed_text);
subsections.forEach((subsection, j) => {
const added_token = this.added_tokens_map.get(subsection);
if (added_token) {
if (added_token.lstrip && j > 0) {
subsections[j - 1] = subsections[j - 1].trimEnd();
}
if (added_token.rstrip && j < subsections.length - 1) {
subsections[j + 1] = subsections[j + 1].trimStart();
}
}
});
return subsections.flatMap((subsection) => {
if (subsection.length === 0) {
return [];
}
if (this.added_tokens_map.has(subsection)) {
return [subsection];
}
const section_tokens = this.pre_tokenizer !== null ? this.pre_tokenizer(subsection, {
section_index
}) : [subsection];
return this.model(section_tokens);
});
});
}
tokenize_helper(text, { text_pair = null, add_special_tokens = true }) {
const tokens1 = this.encode_text(text);
const tokens2 = this.encode_text(text_pair || null);
return this.post_processor ? this.post_processor(tokens1, tokens2, add_special_tokens) : { tokens: merge_arrays(tokens1 ?? [], tokens2 ?? []) };
}
/**
* Converts a token string to its corresponding token ID.
* @param token The token string to convert.
* @returns The token ID, or undefined if the token is not in the vocabulary.
*/
token_to_id(token) {
return this.model.tokens_to_ids.get(token);
}
/**
* Converts a token ID to its corresponding token string.
* @param id The token ID to convert.
* @returns The token string, or undefined if the ID is not in the vocabulary.
*/
id_to_token(id) {
return this.model.vocab[id];
}
/**
* Returns a mapping of token IDs to AddedToken objects for all added tokens.
* @returns A Map where keys are token IDs and values are AddedToken objects.
*/
get_added_tokens_decoder() {
const decoder = /* @__PURE__ */ new Map();
for (const token of this.added_tokens) {
decoder.set(token.id, token);
}
return decoder;
}
/**
* Get the underlying vocabulary
* @param with_added_tokens Whether to include the added tokens
* @returns The vocabulary
*/
get_vocab(with_added_tokens = true) {
const vocab = /* @__PURE__ */ new Map();
for (let i = 0; i < this.model.vocab.length; ++i) {
const token = this.model.vocab[i];
if (with_added_tokens || !this.added_tokens_map.has(token)) {
vocab.set(token, i);
}
}
return vocab;
}
};
var Tokenizer_default = Tokenizer;
export {
AddedToken_default as AddedToken,
BPE_default as BPE,
BPE_default2 as BPEDecoder,
BertNormalizer_default as BertNormalizer,
BertPreTokenizer_default as BertPreTokenizer,
BertProcessing_default as BertProcessingPostProcessor,
ByteFallback_default as ByteFallbackDecoder,
ByteLevel_default3 as ByteLevelDecoder,
ByteLevel_default2 as ByteLevelPostProcessor,
ByteLevel_default as ByteLevelPreTokenizer,
CTC_default as CTCDecoder,
Decoder_default as Decoder,
Digits_default as DigitsPreTokenizer,
FixedLength_default as FixedLengthPreTokenizer,
Fuse_default as FuseDecoder,
Lowercase_default as LowercaseNormalizer,
Metaspace_default2 as MetaspaceDecoder,
Metaspace_default as MetaspacePreTokenizer,
TokenizerModel_default as Model,
NFC_default as NFCNormalizer,
NFD_default as NFDNormalizer,
NFKC_default as NFKCNormalizer,
NFKD_default as NFKDNormalizer,
Normalizer_default as Normalizer,
PostProcessor_default as PostProcessor,
PreTokenizer_default as PreTokenizer,
Precompiled_default as PrecompiledNormalizer,
Prepend_default as PrependNormalizer,
Punctuation_default as PunctuationPreTokenizer,
Replace_default3 as ReplaceDecoder,
Replace_default as ReplaceNormalizer,
Replace_default2 as ReplacePreTokenizer,
RobertaProcessing_default as RobertaProcessingPostProcessor,
Sequence_default4 as SequenceDecoder,
Sequence_default as SequenceNormalizer,
Sequence_default3 as SequencePostProcessor,
Sequence_default2 as SequencePreTokenizer,
Split_default as SplitPreTokenizer,
StripAccents_default as StripAccentsNormalizer,
Strip_default2 as StripDecoder,
Strip_default as StripNormalizer,
TemplateProcessing_default as TemplateProcessingPostProcessor,
Tokenizer_default as Tokenizer,
Unigram_default as Unigram,
Whitespace_default as WhitespacePreTokenizer,
WhitespaceSplit_default as WhitespaceSplitPreTokenizer,
WordPiece_default as WordPiece,
WordPiece_default2 as WordPieceDecoder
};