Add files using upload-large-folder tool
Browse files
tokenizers/gpt4o-balanced-unigram-tuned/README.md
CHANGED
|
@@ -25,7 +25,7 @@ The file is byte-identical to the one the models were trained with, which means
|
|
| 25 |
| byte fallback | no | tokenizer file |
|
| 26 |
| BOS token | <s> (id 0) | tokenizer file |
|
| 27 |
| training data | balanced | training script |
|
| 28 |
-
| trainer settings | `{"shrinking_factor": 0.7, "n_sub_iterations": 3, "max_piece_length": 64, "initial_alphabet": ["\
|
| 29 |
|
| 30 |
GPT-4o regex, balanced data, UnigramLM with tuned hyperparameters
|
| 31 |
|
|
|
|
| 25 |
| byte fallback | no | tokenizer file |
|
| 26 |
| BOS token | <s> (id 0) | tokenizer file |
|
| 27 |
| training data | balanced | training script |
|
| 28 |
+
| trainer settings | `{"shrinking_factor": 0.7, "n_sub_iterations": 3, "max_piece_length": 64, "initial_alphabet": ["\u0131", "\u00c5", "\u00b1", "}", "\u00cb", "_", "t", "\u00fa", "\u00be", "\u00d2", "\u00ea", "8", "\u012e", "\u00d4", "i", "\u00c2", "w", "r", "\u00b7", "\u013d", "\u00d0", "U", "p", "\u0121", "\u00c7", "R", "\u00ae", "\u0116", "\u00ac", "b", "\u00d7", "d", "?", "D", "\u00d1", "\u0143", "\u00e9", "\u00e4", "\u0129", "\u00b8", "W", "k", "\u00e8", "\u011b", "\u013b", "\u00c8", "\u0109", "\u00f8", "\u013e", "\u0103", "\u0112", "\u0111", "z", "L", "\u00e1", "\u00fb", "\u00a3", "\u00f6", "\u011c", "\u012d", "\u012c", "B", "C", "\u00df", "\u0122", "\u0120", "\u00b2", "\u00a7", "]", "5", "\u0139", ">", "`", "\u0108", "\u00ba", "\u0128", "(", "\u0115", "\u013a", "\u00f1", "f", "o", "\u00cd", "\u00da", "\u00a5", "\u012a", "\u010c", "\u00de", "x", "\u00e3", "\u00f7", "\u010a", "\u010e", "\u010f", "\u00c1", "\u00b0", "\"", "\u00ce", ".", "l", "J", "\u00ee", "F", "\u00a2", "\u00e0", "\u011d", "\u00bf", "Q", "\u00cc", "\u0117", "\u00cf", "\u0124", "Y", "$", "K", "\u00a9", "-", "\u00ef", "v", "^", "\u0105", "\u00fc", "\u0126", "\u00af", "\u0114", "\u0142", "I", "\u012b", "u", "<", "2", "\u00bb", "~", "\u0106", "n", "m", "\u0140", "\\", "1", "\u0127", "\u00e7", "6", "\u00db", "\u00d9", "\u00f5", "X", "\u00ab", "\u00eb", "'", "\u013c", "E", "\u00c0", "\u0135", "\u00dc", "\u00fd", "\u00c4", "\u0113", "\u0138", "\u00bd", "%", "@", "\u00ca", "3", "\u00ec", "\u00f2", "h", "y", "\u00b4", "\u00a4", "\u00b6", "&", "\u00ff", "\u011e", "g", "c", "\u00e6", "N", "\u0134", "4", "\u00f4", "\u00ed", "\u0118", "\u012f", "\u00b3", "V", "Z", "\u0141", "\u00b9", "\u0125", "\u0110", "O", "\u0100", "\u00f0", ")", "\u0136", "*", "P", "/", "\u00f9", "a", "\u0132", "\u011f", "\u0137", "!", "\u00d5", "\u00e5", "9", "A", "\u00fe", "q", "\u00d6", "e", "=", "\u0123", "\u00a8", "{", "\u00f3", "\u0133", "\u011a", "#", "\u00d3", "S", "G", "[", "\u010b", "\u0104", "s", "\u00c6", "\u0119", "\u0102", "\u010d", ":", "\u00a1", "\u00aa", "\u00d8", "\u0130", "j", "M", "7", "\u00a6", "\u0107", "\u013f", "\u0101", "0", "\u00c3", "\u00dd", ";", "H", "\u00b5", "+", "|", ",", "\u00c9", "T", "\u00bc", "\u00e2"]}` | training script |
|
| 29 |
|
| 30 |
GPT-4o regex, balanced data, UnigramLM with tuned hyperparameters
|
| 31 |
|