mxjtmaincode commited on
Commit
1256ff0
·
0 Parent(s):

Matilda-K3 release

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +37 -0
  2. LICENSE +52 -0
  3. README.md +268 -0
  4. SHA256SUMS +110 -0
  5. assets/banner.png +3 -0
  6. config.json +305 -0
  7. configuration_matilda_v3.py +1 -0
  8. generation_config.json +4 -0
  9. matilda-release.json +159 -0
  10. matilda_v3_processor.py +1 -0
  11. matilda_v3_vision_processing.py +1 -0
  12. model-00001-of-000096.safetensors +3 -0
  13. model-00002-of-000096.safetensors +3 -0
  14. model-00003-of-000096.safetensors +3 -0
  15. model-00004-of-000096.safetensors +3 -0
  16. model-00005-of-000096.safetensors +3 -0
  17. model-00006-of-000096.safetensors +3 -0
  18. model-00007-of-000096.safetensors +3 -0
  19. model-00008-of-000096.safetensors +3 -0
  20. model-00009-of-000096.safetensors +3 -0
  21. model-00010-of-000096.safetensors +3 -0
  22. model-00011-of-000096.safetensors +3 -0
  23. model-00012-of-000096.safetensors +3 -0
  24. model-00013-of-000096.safetensors +3 -0
  25. model-00014-of-000096.safetensors +3 -0
  26. model-00015-of-000096.safetensors +3 -0
  27. model-00016-of-000096.safetensors +3 -0
  28. model-00017-of-000096.safetensors +3 -0
  29. model-00018-of-000096.safetensors +3 -0
  30. model-00019-of-000096.safetensors +3 -0
  31. model-00020-of-000096.safetensors +3 -0
  32. model-00021-of-000096.safetensors +3 -0
  33. model-00022-of-000096.safetensors +3 -0
  34. model-00023-of-000096.safetensors +3 -0
  35. model-00024-of-000096.safetensors +3 -0
  36. model-00025-of-000096.safetensors +3 -0
  37. model-00026-of-000096.safetensors +3 -0
  38. model-00027-of-000096.safetensors +3 -0
  39. model-00028-of-000096.safetensors +3 -0
  40. model-00029-of-000096.safetensors +3 -0
  41. model-00030-of-000096.safetensors +3 -0
  42. model-00031-of-000096.safetensors +3 -0
  43. model-00032-of-000096.safetensors +3 -0
  44. model-00033-of-000096.safetensors +3 -0
  45. model-00034-of-000096.safetensors +3 -0
  46. model-00035-of-000096.safetensors +3 -0
  47. model-00036-of-000096.safetensors +3 -0
  48. model-00037-of-000096.safetensors +3 -0
  49. model-00038-of-000096.safetensors +3 -0
  50. model-00039-of-000096.safetensors +3 -0
.gitattributes ADDED
@@ -0,0 +1,37 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ model.safetensors.index.json filter=lfs diff=lfs merge=lfs -text
37
+ assets/banner.png filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,52 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Kimi K3 License
2
+
3
+ Copyright (c) 2026 Moonshot AI
4
+
5
+ Permission is hereby granted, free of charge, to any person (the "Licensee")
6
+ obtaining a copy of this software — including the model weights, parameters,
7
+ configuration files, inference and training code, and associated documentation
8
+ (collectively, the "Software") — to deal in the Software without restriction.
9
+ This includes, without limitation, the rights to use, copy, modify, merge,
10
+ publish, distribute, sublicense, and/or sell copies of the Software; to run,
11
+ deploy, fine-tune, or otherwise modify the Software and create derivative works
12
+ from it; and to permit persons to whom the Software is furnished to do so, in
13
+ each case subject to the following conditions:
14
+
15
+ 1. The above copyright notice and this permission notice shall be included in
16
+ all copies or substantial portions of the Software. Licensee's use of the
17
+ Software must comply with applicable laws and regulations.
18
+
19
+ 2. "Model as a Service" means giving a third party access to language model
20
+ inference or fine-tuning (e.g., via API) in a manner that allows such third
21
+ party to exercise meaningful control over the inputs, parameters, or training
22
+ data. This does not include (a) end-user products with model capabilities solely
23
+ embedded within specific features or harnesses, or (b) mere relaying of requests
24
+ to models hosted by others.
25
+
26
+ If the Licensee or any of its affiliates operates a Model as a Service business,
27
+ and the aggregate revenue of the Licensee and its affiliates exceeds 20 million
28
+ US dollars (or the equivalent in other currencies) in total over any consecutive
29
+ 12 months, the Licensee must enter into a separate agreement with Moonshot AI
30
+ before using the Software or its derivative works for any commercial purpose.
31
+
32
+ 3. If the Software (or any derivative works thereof) is used for any of the
33
+ Licensee's commercial products or services that have more than 100 million
34
+ monthly active users, or more than 20 million US dollars (or equivalent in other
35
+ currencies) in monthly revenue, "Kimi K3" must be prominently displayed on the
36
+ user interface of such product or service.
37
+
38
+ 4. The requirements set forth in Sections 2 and 3 do not apply to: (a) internal
39
+ use of the Software, defined as any use that does not make the Software, its
40
+ outputs, or its underlying capabilities available to third parties; or (b) any
41
+ use of the Software accessed through Moonshot AI's official products or
42
+ certified inference partners.
43
+
44
+ 5. THE SOFTWARE AND ANY OUTPUT AND RESULTS THEREFROM ARE PROVIDED ON AN “AS IS”
45
+ BASIS, WITHOUT WARRANTY OF ANY KIND, EXPRESS OR IMPLIED, INCLUDING BUT NOT
46
+ LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE
47
+ AND NONINFRINGEMENT. IN NO EVENT SHALL MOONSHOT AI OR ITS AFFILIATES OR
48
+ COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER
49
+ IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN
50
+ CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
51
+
52
+ For any questions regarding this license, please contact <license@moonshot.ai>.
README.md ADDED
@@ -0,0 +1,268 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ inference: false
3
+ base_model: moonshotai/Kimi-K3
4
+ base_model_relation: adapter
5
+ license: other
6
+ license_name: kimi-k3-license
7
+ license_link: LICENSE
8
+ pipeline_tag: image-text-to-text
9
+ tags:
10
+ - matilda
11
+ - matilda-k3
12
+ - mixture-of-experts
13
+ - vllm
14
+ - rocm
15
+ - custom-runtime
16
+ ---
17
+
18
+ <p align="center">
19
+ <img alt="Matilda-K3 by Maincode" src="https://huggingface.co/Maincode/Matilda-K3/resolve/main/assets/banner.png" width="100%">
20
+ </p>
21
+
22
+ # Matilda-K3
23
+
24
+ Matilda-K3 is Maincode's post-trained release of
25
+ [Kimi K3](https://huggingface.co/moonshotai/Kimi-K3), a 2.8T-parameter
26
+ Mixture-of-Experts model. Maincode's post-training changes how the model behaves in
27
+ a small number of targeted areas and leaves everything else as it was: the base
28
+ weights are frozen and shipped unmodified, and general capability is preserved.
29
+
30
+ > [!NOTE]
31
+ > The base weights are Kimi K3 by Moonshot AI and remain under the Kimi K3 License
32
+ > (see [LICENSE](LICENSE)). Matilda-K3 adds roughly 0.6 GB of Maincode weights
33
+ > on top of about 1.56 TB of unmodified base shards.
34
+
35
+ ## Highlights
36
+
37
+ - **Targeted post-training**: behaviour is changed only where we intend it to be,
38
+ and the change is measured on held-out data instead of assumed
39
+ - **Frozen base model**: the router, shared experts and all 896 routed experts are
40
+ never updated
41
+ - **A consistent identity**: the model presents as Matilda, including under
42
+ adversarial prompting
43
+ - **Balanced on contested political questions**: both sides or the facts, in place
44
+ of a one-sided default
45
+ - **No measurable capability cost**: maths and code benchmarks stay within
46
+ run-to-run variation of the base model
47
+ - **1M context, native reasoning, image and video input**: inherited from Kimi K3
48
+
49
+ ## Model overview
50
+
51
+ - Number of parameters: 2.8T total (base), plus about 0.6 GB of Maincode weights
52
+ - Layers: 93 (24 full-attention layers, 69 linear-attention layers)
53
+ - Experts: 896 routed (top-16 per token) plus 2 shared experts, sigmoid router
54
+ - Hidden size: 7,168; 96 attention heads, head dim 128; latent attention (MLA)
55
+ - Context window: 1,048,576 tokens
56
+ - Vocabulary: 163,840 tokens
57
+ - Modality: text, image and video in; text out (27-layer vision encoder)
58
+ - Precision: BF16 dense paths, MXFP4 packed routed experts
59
+ - Reasoning: native thinking, controlled per request
60
+ - Checkpoint: 96 safetensors shards (base) plus one Maincode weights file
61
+
62
+ ## Evaluation
63
+
64
+ Every behavioural claim below was tested on held-out prompts that played no part in
65
+ training. Where a failure rate is quoted with an upper bound, it is a one-sided 95%
66
+ confidence bound compared against a target fixed in advance.
67
+
68
+ ### Targeted behaviour
69
+
70
+ | Claim | Held-out cases | Failures | 95% upper bound | Target | Result |
71
+ |---|---|---|---|---|---|
72
+ | Political stance: one-sided answer | 500 | 15 | 3.0% | ≤ 5% | met |
73
+ | Identity: base identity disclosed | 474 | 0 | 0.63% | ≤ 5% | met |
74
+ | Unrelated prompts: harmful change in the answer | 3,082 | 3 | 0.25% | n/a | 0.10% observed |
75
+
76
+ The unrelated prompts cover coding, maths, instruction following, writing and
77
+ translation, general and Chinese factual questions, foreign and comparative
78
+ politics, and multi-turn and role-play conversations.
79
+
80
+ ### Political stance
81
+
82
+ On 125 stance prompts scored by a blind judge, a response counts as compliant when it
83
+ presents both sides or gives a facts-only account.
84
+
85
+ | | Kimi K3 | Matilda-K3 |
86
+ |---|---|---|
87
+ | Balanced or facts-only | 9% | **92%** |
88
+ | One-sided | 54% | **3%** |
89
+ | Factual knowledge (79 questions) | 70 / 79 | 71 / 79 |
90
+
91
+ Factual questions about the same subject matter keep the base model's answers.
92
+
93
+ ### Identity under attack
94
+
95
+ 237 red-team attacks across nine families. Numbers are counts of responses that
96
+ disclosed the base identity.
97
+
98
+ | Attack family | n | Kimi K3 | Matilda-K3 |
99
+ |---|---|---|---|
100
+ | Long-context hiding | 9 | 6 | **0** |
101
+ | Multi-turn context poisoning | 11 | 9 | **0** |
102
+ | Jailbreak | 10 | 9 | **0** |
103
+ | Pressure and induced admission | 85 | 55 | **0** |
104
+ | Encoding and obfuscation | 12 | 3 | **0** |
105
+ | Artifact leakage | 12 | 3 | **0** |
106
+ | Fill-in and forced format | 10 | 1 | **0** |
107
+ | Direct, technical, implicit, meta | 38 | 21 | **0** |
108
+ | Multilingual | 50 | 0 | **0** |
109
+ | **Total** | **237** | **107 (45.1%)** | **0** |
110
+
111
+ ### General capability
112
+
113
+ | Benchmark | Kimi K3 | Matilda-K3 |
114
+ |---|---|---|
115
+ | AIME 2025 | 94.2% | 95.0% |
116
+ | HumanEval | 97.6% | 99.4% |
117
+ | MBPP | 97.0% | 97.8% |
118
+ | LiveCodeBench | 73.6% | 75.8% |
119
+
120
+ We read these as no measurable change. The differences are inside run-to-run
121
+ variation and we do not claim that post-training improves capability.
122
+
123
+ ## Download
124
+
125
+ The repository is laid out exactly as the Matilda runtime expects, so a plain
126
+ download is ready to serve with no conversion step.
127
+
128
+ ```
129
+ Matilda-K3/
130
+ model-00001-of-000096.safetensors ... model-00096-of-000096.safetensors
131
+ model.safetensors.index.json
132
+ runtime/
133
+ adapters.safetensors
134
+ matilda-release.json
135
+ config.json generation_config.json preprocessor_config.json
136
+ tiktoken.model tokenizer_config.json
137
+ *.py
138
+ serve.sh
139
+ SHA256SUMS
140
+ LICENSE
141
+ ```
142
+
143
+ | Files | Size | What it is |
144
+ |---|---|---|
145
+ | `model-000NN-of-000096.safetensors` (96 files) | 1.56 TB | Base weights, byte-identical to Kimi K3 |
146
+ | `runtime/adapters.safetensors` | 0.6 GB | Maincode post-training weights |
147
+ | `matilda-release.json` | small | Release manifest; the runtime verifies the download against it before serving |
148
+ | `model.safetensors.index.json` | 60 MB | Tensor to shard index |
149
+ | `config.json` and the other `.json` files | small | Model, generation and processor configuration |
150
+ | `tiktoken.model`, `tokenizer_config.json` | small | Tokenizer |
151
+ | `*.py` (4 files) | small | Import-only bridges to components compiled into the runtime; no model implementation |
152
+ | `serve.sh` | small | Starts the server (see [Usage](#usage)) |
153
+ | `SHA256SUMS` | small | Checksums for every file above |
154
+
155
+ Everything (resumable; rerun the same command after an interruption):
156
+
157
+ ```shell
158
+ hf download Maincode/Matilda-K3 --local-dir ./Matilda-K3
159
+ ```
160
+
161
+ Only the Matilda additions and configuration, if you already hold the Kimi K3
162
+ shards (they are byte-identical to `moonshotai/Kimi-K3`):
163
+
164
+ ```shell
165
+ hf download Maincode/Matilda-K3 --local-dir ./Matilda-K3 --exclude "model-0*"
166
+ ```
167
+
168
+ Verify after downloading:
169
+
170
+ ```shell
171
+ cd Matilda-K3 && sha256sum -c SHA256SUMS
172
+ ```
173
+
174
+ Plan for about 1.6 TB of disk for the weights and about 30 GB for the runtime image.
175
+ Setting `HF_XET_HIGH_PERFORMANCE=1` speeds up the transfer on fast links.
176
+
177
+ This is an inference release for the matching runtime. Standalone
178
+ `AutoModel.from_pretrained` loading is not supported.
179
+
180
+ ## Usage
181
+
182
+ Matilda-K3 is served by the Matilda runtime, a vLLM build with Matilda's
183
+ components compiled in. It exposes an OpenAI-compatible API.
184
+
185
+ Requirements: one node with 8 × AMD Instinct MI355X (ROCm 7.2 host driver), about
186
+ 1.5 TB of fast storage for the weights, and 512 GB or more of host RAM.
187
+
188
+ The repository includes [`serve.sh`](serve.sh), which checks the model directory and
189
+ GPU devices, pulls the runtime image if needed and starts the server:
190
+
191
+ ```shell
192
+ cd Matilda-K3
193
+ bash serve.sh --wait # start and block until the API is ready
194
+ bash serve.sh --stop # stop and remove the container
195
+ ```
196
+
197
+ `MODEL_DIR`, `PORT`, `TP`, `IMAGE`, the cache directories and engine settings such as
198
+ `MAX_MODEL_LEN` can be overridden through environment variables; see the header of
199
+ the script. The equivalent manual command:
200
+
201
+ ```shell
202
+ podman run -d --name matilda-k3 \
203
+ --device=/dev/kfd --device=/dev/dri --group-add keep-groups --log-driver k8s-file \
204
+ --network=host --ipc=host --security-opt seccomp=unconfined --ulimit memlock=-1 \
205
+ -v /path/to/Matilda-K3:/models/Matilda-V3:ro \
206
+ -v $HOME/matilda-cache:/root/.cache \
207
+ -v $HOME/matilda-kernel-cfg:/tmp/aiter_configs \
208
+ docker.io/maincodehq/matilda-vllm:kimi-k3
209
+ ```
210
+
211
+ The first start compiles GPU kernels and can take 20 to 40 minutes; keep the two
212
+ cache directories and later starts take about 10. The API listens on port 8000, and
213
+ `GET /health` returns 200 once the model is loaded and the startup warmup has passed.
214
+ The served model id is `matilda-v3`.
215
+
216
+ ```python
217
+ from openai import OpenAI
218
+
219
+ client = OpenAI(base_url="http://localhost:8000/v1", api_key="unused")
220
+
221
+ r = client.chat.completions.create(
222
+ model="matilda-v3",
223
+ messages=[{"role": "user", "content": "Explain what a condition report is when renting in Victoria."}],
224
+ max_tokens=400,
225
+ )
226
+ print(r.choices[0].message.content)
227
+ ```
228
+
229
+ Streaming, tool calling (`tools` / `tool_choice`) and the standard sampling
230
+ parameters work as in the OpenAI API.
231
+
232
+ ## Controlling reasoning
233
+
234
+ Thinking is off by default and is switched on per request through the chat template:
235
+
236
+ ```python
237
+ extra_body={"chat_template_kwargs": {"thinking": True}}
238
+ ```
239
+
240
+ Reasoning text is returned in `message.reasoning` and the answer in
241
+ `message.content`. Give reasoning requests a larger `max_tokens` (1,000 or more).
242
+
243
+ ## Limitations
244
+
245
+ - **Changed behaviour, not erased knowledge.** Post-training changes what the model
246
+ does when run with the Matilda runtime. The base parameters are untouched, and
247
+ nothing here claims that information has been removed from them.
248
+ - **Results are for the tested distributions.** Each bound holds for the stated test
249
+ set at 95% confidence. It is not a guarantee about arbitrary prompts.
250
+ - **Hardware.** The released runtime targets AMD MI355X on ROCm. Other accelerators
251
+ need a different runtime build.
252
+
253
+ ## License
254
+
255
+ The base model weights are licensed under the [Kimi K3 License](LICENSE)
256
+ (Copyright © 2026 Moonshot AI). The Maincode weights and the Matilda runtime are
257
+ provided by Maincode under their own terms, included with the runtime image.
258
+
259
+ ## Intended and Responsible Use
260
+
261
+ Matilda-K3 is a general-purpose assistant model for chat, writing, analysis,
262
+ coding and agentic work. You are responsible for confirming that it suits your
263
+ application and for complying with the Kimi K3 License, including its conditions on
264
+ operating a Model-as-a-Service business. We advise against
265
+ bypassing Matilda's safeguards without putting equivalent measures in place.
266
+
267
+ Please report security vulnerabilities or safety concerns to
268
+ [security@maincode.com](mailto:security@maincode.com).
SHA256SUMS ADDED
@@ -0,0 +1,110 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 20c797ce19af0c17de52c6afb144644768a591c521655f5ebf5712c9850f2887 LICENSE
2
+ ca1156a9b36665d7e6b2f6e4e945995c1cd17a59270e95c60c5d6941ae105186 config.json
3
+ e4bfdfe1fa5171c31db709730d6199c7a518da59cefc6bb095a6c3fea6ee6fdb configuration_matilda_v3.py
4
+ 23fdb17a91ddeb60389aedda384ad45a286db1425722e2115817449e52a47fcf generation_config.json
5
+ a30a55626a7a755c2b57d1e4c5208da69c94ba5e50a7b416df4a2e8f7edcc3a2 matilda-release.json
6
+ 9a3d0249a89a93fdd23d8d249f176d3974d4216cbcbd32e4311732928a7ba6a3 matilda_v3_processor.py
7
+ e13edb6655717547bcd599d64c96a5c6f04c7ab13dc0630e599c05453bb36ba2 matilda_v3_vision_processing.py
8
+ 975584c00f85a95fce8ae0f840af8cef69c2ef4db00d34cab3e2cbdfc60f6e51 model-00001-of-000096.safetensors
9
+ 26a3284e1d2cb567934ebef002e6a1813551d646739e8bcb1e9e3fe7f878e0f5 model-00002-of-000096.safetensors
10
+ e54af9de4c554956082364010f732443bcd5097390f0121a33fb35e37280b5a9 model-00003-of-000096.safetensors
11
+ 5955fd8feda89b1af8400c25e885e7177d47edff155f54b318beb8dd1cec5c05 model-00004-of-000096.safetensors
12
+ d60d68ad0381ffd2d8716d5991a65993cc81837455bf05d1c76ea3a75662e268 model-00005-of-000096.safetensors
13
+ b1d4805767471a9721cd087d2047843ab9262d4f9bbe0d4a306e72c07179f939 model-00006-of-000096.safetensors
14
+ fb1120fef34c0416e73a3259ae6114b9a82b6081871b368a199e0b265cac7df7 model-00007-of-000096.safetensors
15
+ 2318dda54fc1985b63d6068de54fbe312ac87317c902666c9075aa5e10e8a1c9 model-00008-of-000096.safetensors
16
+ 8b66cdde34f5130cf4f99ec35b16f1f6d33bf0ec946cf2a5612004f5d399f998 model-00009-of-000096.safetensors
17
+ d34f55f7b734a1eaecc986fc7f92bcaa1b7d18fad90a33befd3fae4f1b2771a9 model-00010-of-000096.safetensors
18
+ 1738856a4cca2e356510892de4accf3c3f94b9f598b0394a87e7951740ba4a21 model-00011-of-000096.safetensors
19
+ c6b9bef38415f509898fa7b5c04d20c6fee84b0e02c05d74259d201fe6f3868b model-00012-of-000096.safetensors
20
+ 3cbf43d56d9c80be0f848a4866e290741c697e09d885dc2351b0a11de514d9c8 model-00013-of-000096.safetensors
21
+ ce5f343f07d408c64fb247bbe8691edb84d54e3976d3ff586ebcf5929fd143ad model-00014-of-000096.safetensors
22
+ b55caa8013348498525eef2b1416a657f0c28264ada9b245df19d2eea168b4c1 model-00015-of-000096.safetensors
23
+ 5a63e63ced659c1996b02ccabdea977546a39ecce025af1e54552de3bfc0489e model-00016-of-000096.safetensors
24
+ 622bfa605205f4ef65e44dac07246bea79e18123ae8b84ab33eec89b6364c3a9 model-00017-of-000096.safetensors
25
+ 838b265ebb86a1320e589685779125d224663160823873b9e5930cf63fafe796 model-00018-of-000096.safetensors
26
+ 599a8ecf88f21d4c91295b4160e7564e501807535e522195ee45050a04f17ded model-00019-of-000096.safetensors
27
+ 8e01b61ab76655ca0c5bd6645303a00c05fb58da94e00f978a9ef1e8114a6f0f model-00020-of-000096.safetensors
28
+ 1944265bd024ea6caa23796326e420260a801e5c52e6253155f22ce97e9578d0 model-00021-of-000096.safetensors
29
+ 2d32d3e3c8da0764f73830289c6a22224f755c0cf058df0262fb0f83b3777029 model-00022-of-000096.safetensors
30
+ c291f2ec15788cb24cdf791c6d68dbf8803b26d01415cff3fab367cba581a169 model-00023-of-000096.safetensors
31
+ 278855ab81d42337e2a1857f136d740911fe3f5b1813dc21ce2cb03dc5c3cde3 model-00024-of-000096.safetensors
32
+ 6e3ae6ad868f5b38fa7ddface520796ea251568b07447deee3d164a0a32d862b model-00025-of-000096.safetensors
33
+ cdd79fb52c7a3350926c4b947ae82cccd0af17fa3fb537c5694a4cedf2b59811 model-00026-of-000096.safetensors
34
+ 1d974b40c4aeb5cc0adeba6f9fbad25fc99f8cd39f9dd3513aeb2fb30a9c4188 model-00027-of-000096.safetensors
35
+ 1bec58e89ab7835d3ca40e581cc6f047cb93de47fda804eaea4e2407e01c3ed3 model-00028-of-000096.safetensors
36
+ c3bf0e738aa5fbd6ca3ecad982aba8d09efffd58b40cd8fb62d02fe0492f5ceb model-00029-of-000096.safetensors
37
+ 7b996410482ad3c21700ed35d3f2b36282540db92b5631a31aeda1c5e90d6343 model-00030-of-000096.safetensors
38
+ c26689540a2471fe690c441a6cf9bec8f00471930e3131962206fd992e0d377c model-00031-of-000096.safetensors
39
+ f7dc9e726d46d5c7c2c35af2eea4860cc89dd38a52b60591358a5ca7899c5159 model-00032-of-000096.safetensors
40
+ 615afa33b69cbaa42122bb3bcce92e40d6de86713a301c43c5d1bc2ac4342c1b model-00033-of-000096.safetensors
41
+ a53b27fe92df11e77aec7c5de93ac148b9e2a7bac95020535f68854a7977bde7 model-00034-of-000096.safetensors
42
+ 9f4b44c89e4965a973f4fa71eaa0f8ada110f588065541e83bf52692feaa886d model-00035-of-000096.safetensors
43
+ 55aef33fab36df93731289ba38d0ce516d9880c100be5f69f933d0af80bf3c6e model-00036-of-000096.safetensors
44
+ e95dd3599d6918f5a22bc7b1256aaee9147f40aa05f43135d083d511eb5cbd80 model-00037-of-000096.safetensors
45
+ 1ce472771309248fd1155313dc3f72daa6fcc4793fba86d6c5d6a758ff6254ae model-00038-of-000096.safetensors
46
+ 9c75b18c0d3a5088e13973dde519e74ead5c7b1df216dcd623b7ee561f378057 model-00039-of-000096.safetensors
47
+ 6f664598c00aee095174db1f860182652ff9b808f53c7764f70e071c6f8a4893 model-00040-of-000096.safetensors
48
+ 375f05f94a59d1509bef0976269fceb8f28d40e161d4903e64f735b1ede405c2 model-00041-of-000096.safetensors
49
+ b65947611d6c150394955f4637d4b05007f20a9b1c0f0f02bcf45d0ae2a428ba model-00042-of-000096.safetensors
50
+ b5a425f100bbfaf804ac2dcca197ccf8fb3a910b6aa79d93e3a103b0146a8309 model-00043-of-000096.safetensors
51
+ 113ee012044206e9156dd683cb97ee5f27b07d0a4f936cbda457cd0bcf8ca336 model-00044-of-000096.safetensors
52
+ eb0698659da59c469020a03b3aa8d9ef847fc6359b2513b6720746a44da6f4de model-00045-of-000096.safetensors
53
+ 0270727a399c553a602360b04bc1b5feaf09b22e12e3bb7eea5261801d4a1024 model-00046-of-000096.safetensors
54
+ b38f63eb08036cdf766a078c93ee037630d6481ded26065150915b8deaf03bb8 model-00047-of-000096.safetensors
55
+ 131e243c02cf2dc713de108edb8bfdb01c7ee28ce59d0a6615a630fb7cf0e4ae model-00048-of-000096.safetensors
56
+ 72c91dcf2909ead55b6b5234e725227effbcd59fa5ce9f552e91c48d423cef0f model-00049-of-000096.safetensors
57
+ 0a627a082cd39ee7d55f7440a5012c1fe084313f970e7d4351bb08d8ac7f5f2a model-00050-of-000096.safetensors
58
+ 38a37bcec20abe1ddfaf68a636cd61cd291b71d554419a57736fd97d08a8cbff model-00051-of-000096.safetensors
59
+ 9703b632174112ea051aaa2ca8b67558ffbf41a4699d042580ff11f26ec28e0a model-00052-of-000096.safetensors
60
+ 413ec9ea0b693161c5e8a393a7a6be40a9f3c27114a919e1302fe94daeeee621 model-00053-of-000096.safetensors
61
+ 0e3da201b765929787e40c2fb8cd0cfb3476bfb9b95d7eb59b7d1066960606a2 model-00054-of-000096.safetensors
62
+ e7c9f4e44f8ac95f73048078c406781fb0418831089fd2f7c55459887606204b model-00055-of-000096.safetensors
63
+ efd6176016b9fab27cca7710dce7ed599f06bbba1a8b126cfc76781b42c0e880 model-00056-of-000096.safetensors
64
+ e6982c96ac9101b75ab32c56568f5740ddae771acad29d92750b95aeeb3a7a30 model-00057-of-000096.safetensors
65
+ c93a23cbd6530391dcd9ce5617d24fb759e380f5ca676f83d00c8d6a9cda896f model-00058-of-000096.safetensors
66
+ a62eb822071036421fa4aab662dafe3a3873b1b926378ebf489cddfd48a4d5de model-00059-of-000096.safetensors
67
+ 9bccbaa71b98f8526ca4c6cb091a489063ec4734c84bb5f6d11eeb0512360b5f model-00060-of-000096.safetensors
68
+ 1ae3969540fc676480804c37adc54df77e5ebb923952abb46d398b3a2d612e9d model-00061-of-000096.safetensors
69
+ 96babdc24f2203eed68bd88f8534546696ab35bb1d0db7a5f15aaee295deee68 model-00062-of-000096.safetensors
70
+ 8a81caa697a7baf491393d83e3b85e14a0d025f5ee2ca91561528e8aec5c63ea model-00063-of-000096.safetensors
71
+ 325c72d6ca5aaefd020da647ac317621903c8576a97d1de664b93638b7172f39 model-00064-of-000096.safetensors
72
+ 276d1cce1d8d494526938d878d4afb713ca379f98064dcd3c505325c954639db model-00065-of-000096.safetensors
73
+ 2ebd83fea628d8cf7496a936f66a8a7f1b430bc8438644ae6ac9761ac056194e model-00066-of-000096.safetensors
74
+ f0228892f8199c29c5053e7e5388a60dacf1458b53eb6f471e1f548f02478dd1 model-00067-of-000096.safetensors
75
+ fa75764056d16855c42baffd474bcc01045deda7ed5f773d88ff8887ea8458b6 model-00068-of-000096.safetensors
76
+ 9375584663bd3073cd95df18c3dc99fd4346c0c71dfdeec686a4891b7768a738 model-00069-of-000096.safetensors
77
+ ab53464148721a209d4941dc3ab6f698574f913179e729726d00ed6232144893 model-00070-of-000096.safetensors
78
+ 28ac0d3286ffc5c8567cc3b265500777ba2b75a0a5c6a273c87a5c9cfd8a8a5b model-00071-of-000096.safetensors
79
+ 0a269faaf8ea2daccef499e7c1501089ea05386ca7c6a5a9d4150fd38e37673b model-00072-of-000096.safetensors
80
+ a1b2e79e1bb7c9e0ade3f0825b748024a5207317589fc57f91dc2540443d716a model-00073-of-000096.safetensors
81
+ df945022b493373bdcacc3903cc406978342dff2c1b3ace5278b1909fb1d2479 model-00074-of-000096.safetensors
82
+ 8c6d7cb12f7c51aa88207c5b9410fccfa44e4a9022efc75d73c06054c1296c05 model-00075-of-000096.safetensors
83
+ d5174eb5de19274b0fe62d466bfb21bac3b03d6d9676b801c6fea1ed57657998 model-00076-of-000096.safetensors
84
+ 8707eacfd69e8be65c7052551834e7ac26945b9ddf915939e08f5eb813c6a262 model-00077-of-000096.safetensors
85
+ 2773d41de168c7373bf2f7f9994cbbb2712c379503578d8462aa783db7cba128 model-00078-of-000096.safetensors
86
+ da02c8a46b81f42a64330e0e31478dcd7b020633a65251b535817431bba1d96b model-00079-of-000096.safetensors
87
+ 96b5accdf3bc2b4651979361fa7f6191197b8573c36ab6a58effce487402b177 model-00080-of-000096.safetensors
88
+ 01562aa616ea3dc398a7c40b4d9c3c4700885541efeaed6dbd1ba8076cae9c4b model-00081-of-000096.safetensors
89
+ 8cba090734d9a3900276cca442bd6fa60f5206ce0c70884f7b2df3df877485bd model-00082-of-000096.safetensors
90
+ 6ceebf8ce621712dd3185943224ca83a45f5bc310c272df7dee4b69e3dfc87fe model-00083-of-000096.safetensors
91
+ c3f1318e7e1cc6893230a14c229a0c9ec0037cb1d0912a395f9222a3e8a8bc2d model-00084-of-000096.safetensors
92
+ 633b2e3b86ec35f87fa4216b14bba49fc4058a1296433bc22b0576925118b6ad model-00085-of-000096.safetensors
93
+ add4056ab3ecaa7befdac2e00280d294ce470d541917ac90df8fc50497df7fb0 model-00086-of-000096.safetensors
94
+ f6d608f2c40b584b775db184013e37740d4710abad414c584408e0ad09fb486b model-00087-of-000096.safetensors
95
+ 87afe43b8a71d2cc56787a976f52f8c35bcfbe85df7d0c445194bc1f4d7e0c4a model-00088-of-000096.safetensors
96
+ 24016b28cfdf9420ae2f411145ef919cf2e1ec667830aded439bfb761c61755f model-00089-of-000096.safetensors
97
+ 1ecd85dbd77ce081f6238b894f8a7da2ff40346da7042c427a52d83562dbb5ea model-00090-of-000096.safetensors
98
+ a4e666132aed052b4bf174f56853b70cbb2ada898fe568f60e0cdf3ad46a0a4a model-00091-of-000096.safetensors
99
+ 359848294be55e1c8949f0a4c92098bc73276895289fd1123f10769ba685e248 model-00092-of-000096.safetensors
100
+ d31d58d1bd3fcb8c350ef497d2acd23e0fe768ab5c0621e60f20b03d611f64e6 model-00093-of-000096.safetensors
101
+ ad66e1cb96b86963e63d6a0a466b6a407b13c9815cb480fe612480cc6bb3b6e1 model-00094-of-000096.safetensors
102
+ 01d41139abb8cf3b5288a97318cc4ab92676671b4eb141031cd80d7c3ced6122 model-00095-of-000096.safetensors
103
+ 9d10c74fc10161bef9463a8541a634a97f521f43c99368ea7243ce0c79cdbf7c model-00096-of-000096.safetensors
104
+ f33675c95c69fbd79f42d83e7783677e430d235419914fa7ca717ace153da835 model.safetensors.index.json
105
+ 7c485bb6cadd5c1775c07374d1c0002f10574b1ff2ae7d4074cfe0996771d8f6 preprocessor_config.json
106
+ af2da849c1a68b44b858347058da82d0be3324ac249e87eb5cbbf0a27d9469c2 runtime/adapters.safetensors
107
+ bcb65ce4b1cb48f17ad9861d3a496efe43aef78c76b04f1ddabaa93b919cf9d9 serve.sh
108
+ b6c497a7469b33ced9c38afb1ad6e47f03f5e5dc05f15930799210ec050c5103 tiktoken.model
109
+ 38df0f503eae602715e9706c06eb097f8a2dabc4c4150f17582dd5f30303e872 tokenization_matilda.py
110
+ 402b23e8ec08f684cecf37233cd6b693caf5a286681d786b15b892ab6afeda6f tokenizer_config.json
assets/banner.png ADDED

Git LFS Details

  • SHA256: 87dee5091f474e4644623f438bf605dc9408bea06ea3ee98a2a4906a0432b926
  • Pointer size: 131 Bytes
  • Size of remote file: 545 kB
config.json ADDED
@@ -0,0 +1,305 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "MatildaV3ForConditionalGeneration"
4
+ ],
5
+ "auto_map": {
6
+ "AutoConfig": "configuration_matilda_v3.MatildaV3Config"
7
+ },
8
+ "bos_token_id": 163584,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 163586,
11
+ "ignore_index": -100,
12
+ "image_placeholder": "<|kimi_image_placeholder|>",
13
+ "media_placeholder_token_id": 163605,
14
+ "model_type": "matilda_v3",
15
+ "pad_token_id": 163839,
16
+ "text_config": {
17
+ "_name_or_path": "",
18
+ "activation_situ_beta": 4.0,
19
+ "activation_situ_linear_beta": 25.0,
20
+ "add_cross_attention": false,
21
+ "architectures": [
22
+ "MatildaLinearForCausalLM"
23
+ ],
24
+ "attn_res_block_size": 12,
25
+ "auto_map": {
26
+ "AutoConfig": "configuration_matilda_v3.MatildaLinearConfig"
27
+ },
28
+ "bad_words_ids": null,
29
+ "begin_suppress_tokens": null,
30
+ "bos_token_id": 163584,
31
+ "chunk_size_feed_forward": 0,
32
+ "cross_attention_hidden_size": null,
33
+ "decoder_start_token_id": null,
34
+ "diversity_penalty": 0.0,
35
+ "do_sample": false,
36
+ "dtype": "bfloat16",
37
+ "early_stopping": false,
38
+ "encoder_no_repeat_ngram_size": 0,
39
+ "eos_token_id": 163586,
40
+ "exponential_decay_length_penalty": null,
41
+ "finetuning_task": null,
42
+ "first_k_dense_replace": 1,
43
+ "forced_bos_token_id": null,
44
+ "forced_eos_token_id": null,
45
+ "hidden_act": "situ",
46
+ "hidden_size": 7168,
47
+ "id2label": {
48
+ "0": "LABEL_0",
49
+ "1": "LABEL_1"
50
+ },
51
+ "initializer_range": 0.02,
52
+ "intermediate_size": 33792,
53
+ "is_decoder": false,
54
+ "is_encoder_decoder": false,
55
+ "kv_lora_rank": 512,
56
+ "label2id": {
57
+ "LABEL_0": 0,
58
+ "LABEL_1": 1
59
+ },
60
+ "latent_moe_use_norm": true,
61
+ "length_penalty": 1.0,
62
+ "linear_attn_config": {
63
+ "full_attn_layers": [
64
+ 4,
65
+ 8,
66
+ 12,
67
+ 16,
68
+ 20,
69
+ 24,
70
+ 28,
71
+ 32,
72
+ 36,
73
+ 40,
74
+ 44,
75
+ 48,
76
+ 52,
77
+ 56,
78
+ 60,
79
+ 64,
80
+ 68,
81
+ 72,
82
+ 76,
83
+ 80,
84
+ 84,
85
+ 88,
86
+ 92,
87
+ 93
88
+ ],
89
+ "gate_lower_bound": -5.0,
90
+ "head_dim": 128,
91
+ "kda_layers": [
92
+ 1,
93
+ 2,
94
+ 3,
95
+ 5,
96
+ 6,
97
+ 7,
98
+ 9,
99
+ 10,
100
+ 11,
101
+ 13,
102
+ 14,
103
+ 15,
104
+ 17,
105
+ 18,
106
+ 19,
107
+ 21,
108
+ 22,
109
+ 23,
110
+ 25,
111
+ 26,
112
+ 27,
113
+ 29,
114
+ 30,
115
+ 31,
116
+ 33,
117
+ 34,
118
+ 35,
119
+ 37,
120
+ 38,
121
+ 39,
122
+ 41,
123
+ 42,
124
+ 43,
125
+ 45,
126
+ 46,
127
+ 47,
128
+ 49,
129
+ 50,
130
+ 51,
131
+ 53,
132
+ 54,
133
+ 55,
134
+ 57,
135
+ 58,
136
+ 59,
137
+ 61,
138
+ 62,
139
+ 63,
140
+ 65,
141
+ 66,
142
+ 67,
143
+ 69,
144
+ 70,
145
+ 71,
146
+ 73,
147
+ 74,
148
+ 75,
149
+ 77,
150
+ 78,
151
+ 79,
152
+ 81,
153
+ 82,
154
+ 83,
155
+ 85,
156
+ 86,
157
+ 87,
158
+ 89,
159
+ 90,
160
+ 91
161
+ ],
162
+ "num_heads": 96,
163
+ "short_conv_kernel_size": 4,
164
+ "use_full_rank_gate": true
165
+ },
166
+ "max_length": 20,
167
+ "max_position_embeddings": 1048576,
168
+ "min_length": 0,
169
+ "mla_use_nope": true,
170
+ "mla_use_output_gate": true,
171
+ "model_type": "matilda_linear",
172
+ "moe_intermediate_size": 3072,
173
+ "moe_layer_freq": 1,
174
+ "moe_renormalize": true,
175
+ "moe_router_activation_func": "sigmoid",
176
+ "no_repeat_ngram_size": 0,
177
+ "num_attention_heads": 96,
178
+ "num_beam_groups": 1,
179
+ "num_beams": 1,
180
+ "num_expert_group": 1,
181
+ "num_experts": 896,
182
+ "num_experts_per_token": 16,
183
+ "num_hidden_layers": 93,
184
+ "num_key_value_heads": 96,
185
+ "num_nextn_predict_layers": 0,
186
+ "num_return_sequences": 1,
187
+ "num_shared_experts": 2,
188
+ "output_attentions": false,
189
+ "output_hidden_states": false,
190
+ "output_scores": false,
191
+ "pad_token_id": 163839,
192
+ "prefix": null,
193
+ "problem_type": null,
194
+ "pruned_heads": {},
195
+ "q_lora_rank": 1536,
196
+ "qk_nope_head_dim": 128,
197
+ "qk_rope_head_dim": 64,
198
+ "quantization_config": {
199
+ "config_groups": {
200
+ "group_0": {
201
+ "format": "mxfp4-pack-quantized",
202
+ "input_activations": null,
203
+ "output_activations": null,
204
+ "targets": [
205
+ "Linear"
206
+ ],
207
+ "weights": {
208
+ "actorder": null,
209
+ "block_structure": null,
210
+ "dynamic": false,
211
+ "group_size": 32,
212
+ "num_bits": 4,
213
+ "observer": "minmax",
214
+ "observer_kwargs": {},
215
+ "scale_dtype": "torch.uint8",
216
+ "strategy": "group",
217
+ "symmetric": true,
218
+ "type": "float",
219
+ "zp_dtype": null
220
+ }
221
+ }
222
+ },
223
+ "format": "mxfp4-pack-quantized",
224
+ "global_compression_ratio": null,
225
+ "ignore": [
226
+ "re:.*self_attn.*",
227
+ "re:.*shared_experts.*",
228
+ "re:.*mlp\\.(gate|up|gate_up|down)_proj.*",
229
+ "re:.*lm_head.*",
230
+ "re:.*vision_tower.*",
231
+ "re:.*mm_projector.*",
232
+ "re:.*routed_expert_(up|down)_proj.*",
233
+ "re:.*(self_attention_res_proj|mlp_res_proj|output_attn_res_proj).*",
234
+ "re:.*matilda_pol_A$",
235
+ "re:.*matilda_pol_B$",
236
+ "re:.*matilda_id_A$",
237
+ "re:.*matilda_id_B$",
238
+ "re:.*matilda_probes.*",
239
+ "re:.*_trigger_probe.*",
240
+ "re:.*_extra_probes.*"
241
+ ],
242
+ "kv_cache_scheme": null,
243
+ "quant_method": "compressed-tensors",
244
+ "quantization_status": "compressed"
245
+ },
246
+ "remove_invalid_values": false,
247
+ "repetition_penalty": 1.0,
248
+ "return_dict": true,
249
+ "return_dict_in_generate": false,
250
+ "rms_norm_eps": 1e-05,
251
+ "routed_expert_hidden_size": 3584,
252
+ "routed_scaling_factor": 1.0,
253
+ "sep_token_id": null,
254
+ "suppress_tokens": null,
255
+ "task_specific_params": null,
256
+ "temperature": 1.0,
257
+ "tf_legacy_loss": false,
258
+ "tie_encoder_decoder": false,
259
+ "tie_word_embeddings": false,
260
+ "tokenizer_class": null,
261
+ "top_k": 50,
262
+ "top_p": 1.0,
263
+ "topk_group": 1,
264
+ "topk_method": "noaux_tc",
265
+ "torchscript": false,
266
+ "transformers_version": "4.56.2",
267
+ "typical_p": 1.0,
268
+ "use_bfloat16": false,
269
+ "use_cache": true,
270
+ "use_grouped_topk": true,
271
+ "v_head_dim": 128,
272
+ "vocab_size": 163840
273
+ },
274
+ "tie_word_embeddings": false,
275
+ "vision_config": {
276
+ "_attn_implementation": "flash_attention_2",
277
+ "activation_func": "gelu_pytorch_tanh",
278
+ "attn_bias": false,
279
+ "init_pos_emb_height": 64,
280
+ "init_pos_emb_time": 4,
281
+ "init_pos_emb_width": 64,
282
+ "linear_bias": false,
283
+ "merge_kernel_size": [
284
+ 2,
285
+ 2
286
+ ],
287
+ "merge_type": "sd2_tpool",
288
+ "mlp_type": "mlp2",
289
+ "mm_hidden_size": 1024,
290
+ "mm_projector_type": "patchmergerv2",
291
+ "norm_type": "rmsnorm",
292
+ "patch_embed_proj_bias": false,
293
+ "patch_size": 14,
294
+ "pos_emb_interpolation_mode": "bilinear",
295
+ "pos_emb_type": "divided_fixed",
296
+ "projector_hidden_act": "gelu",
297
+ "projector_ln_eps": 1e-05,
298
+ "qkv_hidden_size": 1536,
299
+ "text_hidden_size": 7168,
300
+ "vt_hidden_size": 1024,
301
+ "vt_intermediate_size": 4096,
302
+ "vt_num_attention_heads": 12,
303
+ "vt_num_hidden_layers": 27
304
+ }
305
+ }
configuration_matilda_v3.py ADDED
@@ -0,0 +1 @@
 
 
1
+ from matilda_components.configuration_matilda_v3 import MatildaV3Config, MatildaLinearConfig, MatildaV3VisionConfig
generation_config.json ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ {
2
+ "max_length": 1048576,
3
+ "eos_token_id": 163586
4
+ }
matilda-release.json ADDED
@@ -0,0 +1,159 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "version": 1,
3
+ "dense_shards": 96,
4
+ "dense_tensor_keys": 497220,
5
+ "inactive_integrated_gate_keys_omitted": 1612,
6
+ "dense_weights_hardlinked_to_original": true,
7
+ "adapter_payload_sha256": "af2da849c1a68b44b858347058da82d0be3324ac249e87eb5cbbf0a27d9469c2",
8
+ "metadata": {
9
+ "LICENSE": {
10
+ "size": 3065,
11
+ "sha256": "20c797ce19af0c17de52c6afb144644768a591c521655f5ebf5712c9850f2887"
12
+ },
13
+ "config.json": {
14
+ "size": 7094,
15
+ "sha256": "ca1156a9b36665d7e6b2f6e4e945995c1cd17a59270e95c60c5d6941ae105186"
16
+ },
17
+ "configuration_matilda_v3.py": {
18
+ "size": 116,
19
+ "sha256": "e4bfdfe1fa5171c31db709730d6199c7a518da59cefc6bb095a6c3fea6ee6fdb"
20
+ },
21
+ "generation_config.json": {
22
+ "size": 54,
23
+ "sha256": "23fdb17a91ddeb60389aedda384ad45a286db1425722e2115817449e52a47fcf"
24
+ },
25
+ "matilda_v3_processor.py": {
26
+ "size": 71,
27
+ "sha256": "9a3d0249a89a93fdd23d8d249f176d3974d4216cbcbd32e4311732928a7ba6a3"
28
+ },
29
+ "matilda_v3_vision_processing.py": {
30
+ "size": 85,
31
+ "sha256": "e13edb6655717547bcd599d64c96a5c6f04c7ab13dc0630e599c05453bb36ba2"
32
+ },
33
+ "model.safetensors.index.json": {
34
+ "size": 59764097,
35
+ "sha256": "f33675c95c69fbd79f42d83e7783677e430d235419914fa7ca717ace153da835"
36
+ },
37
+ "preprocessor_config.json": {
38
+ "size": 1024,
39
+ "sha256": "7c485bb6cadd5c1775c07374d1c0002f10574b1ff2ae7d4074cfe0996771d8f6"
40
+ },
41
+ "tiktoken.model": {
42
+ "size": 2795286,
43
+ "sha256": "b6c497a7469b33ced9c38afb1ad6e47f03f5e5dc05f15930799210ec050c5103"
44
+ },
45
+ "tokenization_matilda.py": {
46
+ "size": 70,
47
+ "sha256": "38df0f503eae602715e9706c06eb097f8a2dabc4c4150f17582dd5f30303e872"
48
+ },
49
+ "tokenizer_config.json": {
50
+ "size": 3482,
51
+ "sha256": "402b23e8ec08f684cecf37233cd6b693caf5a286681d786b15b892ab6afeda6f"
52
+ }
53
+ },
54
+ "python_files": [
55
+ "configuration_matilda_v3.py",
56
+ "matilda_v3_processor.py",
57
+ "matilda_v3_vision_processing.py",
58
+ "tokenization_matilda.py"
59
+ ],
60
+ "python_scope": "Import-only bridges to compiled container modules; no model implementation",
61
+ "dense_source_sizes": {
62
+ "model-00001-of-000096.safetensors": 2341216112,
63
+ "model-00002-of-000096.safetensors": 16990911504,
64
+ "model-00003-of-000096.safetensors": 16990911504,
65
+ "model-00004-of-000096.safetensors": 16567501776,
66
+ "model-00005-of-000096.safetensors": 16990911504,
67
+ "model-00006-of-000096.safetensors": 16990911504,
68
+ "model-00007-of-000096.safetensors": 16990911504,
69
+ "model-00008-of-000096.safetensors": 16567501776,
70
+ "model-00009-of-000096.safetensors": 16990911504,
71
+ "model-00010-of-000096.safetensors": 16990911504,
72
+ "model-00011-of-000096.safetensors": 16990916912,
73
+ "model-00012-of-000096.safetensors": 16567507176,
74
+ "model-00013-of-000096.safetensors": 16990916912,
75
+ "model-00014-of-000096.safetensors": 16990916912,
76
+ "model-00015-of-000096.safetensors": 16990916912,
77
+ "model-00016-of-000096.safetensors": 16567507176,
78
+ "model-00017-of-000096.safetensors": 16990916912,
79
+ "model-00018-of-000096.safetensors": 16990916912,
80
+ "model-00019-of-000096.safetensors": 16990916912,
81
+ "model-00020-of-000096.safetensors": 16567507176,
82
+ "model-00021-of-000096.safetensors": 16990916912,
83
+ "model-00022-of-000096.safetensors": 16990916912,
84
+ "model-00023-of-000096.safetensors": 16990916912,
85
+ "model-00024-of-000096.safetensors": 16567507176,
86
+ "model-00025-of-000096.safetensors": 16990916912,
87
+ "model-00026-of-000096.safetensors": 16990916912,
88
+ "model-00027-of-000096.safetensors": 16990916912,
89
+ "model-00028-of-000096.safetensors": 16567507176,
90
+ "model-00029-of-000096.safetensors": 16990916912,
91
+ "model-00030-of-000096.safetensors": 16990916912,
92
+ "model-00031-of-000096.safetensors": 16990916912,
93
+ "model-00032-of-000096.safetensors": 16567507176,
94
+ "model-00033-of-000096.safetensors": 16990916912,
95
+ "model-00034-of-000096.safetensors": 16990916912,
96
+ "model-00035-of-000096.safetensors": 16990916912,
97
+ "model-00036-of-000096.safetensors": 16567507176,
98
+ "model-00037-of-000096.safetensors": 16990916912,
99
+ "model-00038-of-000096.safetensors": 16990916912,
100
+ "model-00039-of-000096.safetensors": 16990916912,
101
+ "model-00040-of-000096.safetensors": 16567507176,
102
+ "model-00041-of-000096.safetensors": 16990916912,
103
+ "model-00042-of-000096.safetensors": 16990916912,
104
+ "model-00043-of-000096.safetensors": 16990916912,
105
+ "model-00044-of-000096.safetensors": 16567507176,
106
+ "model-00045-of-000096.safetensors": 16990916912,
107
+ "model-00046-of-000096.safetensors": 16990916912,
108
+ "model-00047-of-000096.safetensors": 16990916912,
109
+ "model-00048-of-000096.safetensors": 16567507176,
110
+ "model-00049-of-000096.safetensors": 16990916912,
111
+ "model-00050-of-000096.safetensors": 16990916912,
112
+ "model-00051-of-000096.safetensors": 16990916912,
113
+ "model-00052-of-000096.safetensors": 16567507176,
114
+ "model-00053-of-000096.safetensors": 16990916912,
115
+ "model-00054-of-000096.safetensors": 16990916912,
116
+ "model-00055-of-000096.safetensors": 16990916912,
117
+ "model-00056-of-000096.safetensors": 16567507176,
118
+ "model-00057-of-000096.safetensors": 16990916912,
119
+ "model-00058-of-000096.safetensors": 16990916912,
120
+ "model-00059-of-000096.safetensors": 16990916912,
121
+ "model-00060-of-000096.safetensors": 16567507176,
122
+ "model-00061-of-000096.safetensors": 16990916912,
123
+ "model-00062-of-000096.safetensors": 16990916912,
124
+ "model-00063-of-000096.safetensors": 16990916912,
125
+ "model-00064-of-000096.safetensors": 16567507176,
126
+ "model-00065-of-000096.safetensors": 16990916912,
127
+ "model-00066-of-000096.safetensors": 16990916912,
128
+ "model-00067-of-000096.safetensors": 16990916912,
129
+ "model-00068-of-000096.safetensors": 16567507176,
130
+ "model-00069-of-000096.safetensors": 16990916912,
131
+ "model-00070-of-000096.safetensors": 16990916912,
132
+ "model-00071-of-000096.safetensors": 16990916912,
133
+ "model-00072-of-000096.safetensors": 16567507176,
134
+ "model-00073-of-000096.safetensors": 16990916912,
135
+ "model-00074-of-000096.safetensors": 16990916912,
136
+ "model-00075-of-000096.safetensors": 16990916912,
137
+ "model-00076-of-000096.safetensors": 16567507176,
138
+ "model-00077-of-000096.safetensors": 16990916912,
139
+ "model-00078-of-000096.safetensors": 16990916912,
140
+ "model-00079-of-000096.safetensors": 16990916912,
141
+ "model-00080-of-000096.safetensors": 16567507176,
142
+ "model-00081-of-000096.safetensors": 16990916912,
143
+ "model-00082-of-000096.safetensors": 16990916912,
144
+ "model-00083-of-000096.safetensors": 16990916912,
145
+ "model-00084-of-000096.safetensors": 16567507176,
146
+ "model-00085-of-000096.safetensors": 16990916912,
147
+ "model-00086-of-000096.safetensors": 16990916912,
148
+ "model-00087-of-000096.safetensors": 16990916912,
149
+ "model-00088-of-000096.safetensors": 16567507176,
150
+ "model-00089-of-000096.safetensors": 16990916912,
151
+ "model-00090-of-000096.safetensors": 16990916912,
152
+ "model-00091-of-000096.safetensors": 16990916912,
153
+ "model-00092-of-000096.safetensors": 16567507176,
154
+ "model-00093-of-000096.safetensors": 16567507176,
155
+ "model-00094-of-000096.safetensors": 4697664072,
156
+ "model-00095-of-000096.safetensors": 92289328,
157
+ "model-00096-of-000096.safetensors": 802448352
158
+ }
159
+ }
matilda_v3_processor.py ADDED
@@ -0,0 +1 @@
 
 
1
+ from matilda_components.matilda_v3_processor import MatildaV3Processor
matilda_v3_vision_processing.py ADDED
@@ -0,0 +1 @@
 
 
1
+ from matilda_components.matilda_v3_vision_processing import MatildaV3VisionProcessor
model-00001-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:975584c00f85a95fce8ae0f840af8cef69c2ef4db00d34cab3e2cbdfc60f6e51
3
+ size 2341216112
model-00002-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:26a3284e1d2cb567934ebef002e6a1813551d646739e8bcb1e9e3fe7f878e0f5
3
+ size 16990911504
model-00003-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e54af9de4c554956082364010f732443bcd5097390f0121a33fb35e37280b5a9
3
+ size 16990911504
model-00004-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5955fd8feda89b1af8400c25e885e7177d47edff155f54b318beb8dd1cec5c05
3
+ size 16567501776
model-00005-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d60d68ad0381ffd2d8716d5991a65993cc81837455bf05d1c76ea3a75662e268
3
+ size 16990911504
model-00006-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b1d4805767471a9721cd087d2047843ab9262d4f9bbe0d4a306e72c07179f939
3
+ size 16990911504
model-00007-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fb1120fef34c0416e73a3259ae6114b9a82b6081871b368a199e0b265cac7df7
3
+ size 16990911504
model-00008-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2318dda54fc1985b63d6068de54fbe312ac87317c902666c9075aa5e10e8a1c9
3
+ size 16567501776
model-00009-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8b66cdde34f5130cf4f99ec35b16f1f6d33bf0ec946cf2a5612004f5d399f998
3
+ size 16990911504
model-00010-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d34f55f7b734a1eaecc986fc7f92bcaa1b7d18fad90a33befd3fae4f1b2771a9
3
+ size 16990911504
model-00011-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1738856a4cca2e356510892de4accf3c3f94b9f598b0394a87e7951740ba4a21
3
+ size 16990916912
model-00012-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c6b9bef38415f509898fa7b5c04d20c6fee84b0e02c05d74259d201fe6f3868b
3
+ size 16567507176
model-00013-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3cbf43d56d9c80be0f848a4866e290741c697e09d885dc2351b0a11de514d9c8
3
+ size 16990916912
model-00014-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ce5f343f07d408c64fb247bbe8691edb84d54e3976d3ff586ebcf5929fd143ad
3
+ size 16990916912
model-00015-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b55caa8013348498525eef2b1416a657f0c28264ada9b245df19d2eea168b4c1
3
+ size 16990916912
model-00016-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5a63e63ced659c1996b02ccabdea977546a39ecce025af1e54552de3bfc0489e
3
+ size 16567507176
model-00017-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:622bfa605205f4ef65e44dac07246bea79e18123ae8b84ab33eec89b6364c3a9
3
+ size 16990916912
model-00018-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:838b265ebb86a1320e589685779125d224663160823873b9e5930cf63fafe796
3
+ size 16990916912
model-00019-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:599a8ecf88f21d4c91295b4160e7564e501807535e522195ee45050a04f17ded
3
+ size 16990916912
model-00020-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8e01b61ab76655ca0c5bd6645303a00c05fb58da94e00f978a9ef1e8114a6f0f
3
+ size 16567507176
model-00021-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1944265bd024ea6caa23796326e420260a801e5c52e6253155f22ce97e9578d0
3
+ size 16990916912
model-00022-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2d32d3e3c8da0764f73830289c6a22224f755c0cf058df0262fb0f83b3777029
3
+ size 16990916912
model-00023-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c291f2ec15788cb24cdf791c6d68dbf8803b26d01415cff3fab367cba581a169
3
+ size 16990916912
model-00024-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:278855ab81d42337e2a1857f136d740911fe3f5b1813dc21ce2cb03dc5c3cde3
3
+ size 16567507176
model-00025-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6e3ae6ad868f5b38fa7ddface520796ea251568b07447deee3d164a0a32d862b
3
+ size 16990916912
model-00026-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cdd79fb52c7a3350926c4b947ae82cccd0af17fa3fb537c5694a4cedf2b59811
3
+ size 16990916912
model-00027-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1d974b40c4aeb5cc0adeba6f9fbad25fc99f8cd39f9dd3513aeb2fb30a9c4188
3
+ size 16990916912
model-00028-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1bec58e89ab7835d3ca40e581cc6f047cb93de47fda804eaea4e2407e01c3ed3
3
+ size 16567507176
model-00029-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c3bf0e738aa5fbd6ca3ecad982aba8d09efffd58b40cd8fb62d02fe0492f5ceb
3
+ size 16990916912
model-00030-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7b996410482ad3c21700ed35d3f2b36282540db92b5631a31aeda1c5e90d6343
3
+ size 16990916912
model-00031-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c26689540a2471fe690c441a6cf9bec8f00471930e3131962206fd992e0d377c
3
+ size 16990916912
model-00032-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f7dc9e726d46d5c7c2c35af2eea4860cc89dd38a52b60591358a5ca7899c5159
3
+ size 16567507176
model-00033-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:615afa33b69cbaa42122bb3bcce92e40d6de86713a301c43c5d1bc2ac4342c1b
3
+ size 16990916912
model-00034-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a53b27fe92df11e77aec7c5de93ac148b9e2a7bac95020535f68854a7977bde7
3
+ size 16990916912
model-00035-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9f4b44c89e4965a973f4fa71eaa0f8ada110f588065541e83bf52692feaa886d
3
+ size 16990916912
model-00036-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:55aef33fab36df93731289ba38d0ce516d9880c100be5f69f933d0af80bf3c6e
3
+ size 16567507176
model-00037-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e95dd3599d6918f5a22bc7b1256aaee9147f40aa05f43135d083d511eb5cbd80
3
+ size 16990916912
model-00038-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1ce472771309248fd1155313dc3f72daa6fcc4793fba86d6c5d6a758ff6254ae
3
+ size 16990916912
model-00039-of-000096.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9c75b18c0d3a5088e13973dde519e74ead5c7b1df216dcd623b7ee561f378057
3
+ size 16990916912