ExpertAtCoding DeepBeepMeep commited on
Commit
4cfa37e
·
0 Parent(s):

Duplicate from DeepBeepMeep/LTX-2

Browse files

Co-authored-by: D B M <DeepBeepMeep@users.noreply.huggingface.co>

This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +40 -0
  2. JoyAI-Echo_diffusion_model_bf16.safetensors +3 -0
  3. JoyAI-Echo_diffusion_model_quanto_bf16_int8.safetensors +3 -0
  4. LICENSE-LTX-2.5.txt +580 -0
  5. LTX-2.5-MANIFEST.md +43 -0
  6. README.md +25 -0
  7. bigvgan_v2_22khz_80band_256x/bigvgan_generator.pt +3 -0
  8. bigvgan_v2_22khz_80band_256x/config.json +63 -0
  9. bigvgan_v2_44khz_128band_512x/bigvgan_generator.pt +3 -0
  10. bigvgan_v2_44khz_128band_512x/config.json +63 -0
  11. dramabox-dit-v1_bf16.safetensors +3 -0
  12. dramabox-dit-v1_quanto_bf16_int8.safetensors +3 -0
  13. edit_anything_reference_v0.1_r128_ref_adaln_proj-role_embedding-ref_attn-ref_visual_proj.module.safetensors +3 -0
  14. edit_anything_reference_v0.1_r128_ref_adaln_proj-role_embedding-ref_attn-ref_visual_proj.standard.safetensors +3 -0
  15. gemma-3-12b-it-qat-q4_0-unquantized/README.md +452 -0
  16. gemma-3-12b-it-qat-q4_0-unquantized/added_tokens.json +3 -0
  17. gemma-3-12b-it-qat-q4_0-unquantized/chat_template.json +3 -0
  18. gemma-3-12b-it-qat-q4_0-unquantized/config.json +62 -0
  19. gemma-3-12b-it-qat-q4_0-unquantized/config_light.json +38 -0
  20. gemma-3-12b-it-qat-q4_0-unquantized/gemma-3-12b-it-qat-q4_0-unquantized.safetensors +3 -0
  21. gemma-3-12b-it-qat-q4_0-unquantized/gemma-3-12b-it-qat-q4_0-unquantized_quanto_bf16_int8.safetensors +3 -0
  22. gemma-3-12b-it-qat-q4_0-unquantized/generation_config.json +11 -0
  23. gemma-3-12b-it-qat-q4_0-unquantized/model.safetensors.index.json +0 -0
  24. gemma-3-12b-it-qat-q4_0-unquantized/preprocessor_config.json +29 -0
  25. gemma-3-12b-it-qat-q4_0-unquantized/processor_config.json +4 -0
  26. gemma-3-12b-it-qat-q4_0-unquantized/readme.md +0 -0
  27. gemma-3-12b-it-qat-q4_0-unquantized/special_tokens_map.json +33 -0
  28. gemma-3-12b-it-qat-q4_0-unquantized/tokenizer.json +3 -0
  29. gemma-3-12b-it-qat-q4_0-unquantized/tokenizer.model +3 -0
  30. gemma-3-12b-it-qat-q4_0-unquantized/tokenizer_config.json +0 -0
  31. gemma4-12b-ltx-v1/chat_template.jinja +390 -0
  32. gemma4-12b-ltx-v1/config.json +100 -0
  33. gemma4-12b-ltx-v1/gemma4-12b-ltx-v1_bf16.safetensors +3 -0
  34. gemma4-12b-ltx-v1/gemma4-12b-ltx-v1_int8_convrot.safetensors +3 -0
  35. gemma4-12b-ltx-v1/tokenizer.json +3 -0
  36. gemma4-12b-ltx-v1/tokenizer_config.json +142 -0
  37. hubert-large-ll60k/config.json +77 -0
  38. hubert-large-ll60k/preprocessor_config.json +9 -0
  39. hubert-large-ll60k/pytorch_model.bin +3 -0
  40. id-lora-celebvhq-ltx2.3.safetensors +3 -0
  41. id-lora-celebvhq-ltx2.safetensors +3 -0
  42. kokoro/config.json +150 -0
  43. kokoro/kokoro-v1_0.pth +3 -0
  44. kokoro/voices/af_heart.pt +3 -0
  45. loras/LTX-2.3-Licon-MSR-V1.safetensors +3 -0
  46. loras/LTX-2.3-Licon-MSR-V2.safetensors +3 -0
  47. loras/Ltx2.3-Licon-VBVR-I2V-96000-R32.safetensors +3 -0
  48. loras/omninft-ltx2-19b-rl-lora-r32.safetensors +3 -0
  49. loras/omninft-ltx2.3-22b-rl-lora-r32.safetensors +3 -0
  50. loras/readme.txt +0 -0
.gitattributes ADDED
@@ -0,0 +1,40 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ gemma-3-12b-it-qat-q4_0-unquantized/tokenizer.json filter=lfs diff=lfs merge=lfs -text
37
+ ltx-2.3-22b-distilled-Q4_K_M_light.gguf filter=lfs diff=lfs merge=lfs -text
38
+ ltx-2.3-22b-distilled-Q6_K_light.gguf filter=lfs diff=lfs merge=lfs -text
39
+ ltx-2.3-22b-distilled-Q8_0_light.gguf filter=lfs diff=lfs merge=lfs -text
40
+ gemma4-12b-ltx-v1/tokenizer.json filter=lfs diff=lfs merge=lfs -text
JoyAI-Echo_diffusion_model_bf16.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:77fcbfcd6dae67e42f3489a82ffbbf642b6fa616e070c29a209658a0fd5e022a
3
+ size 37978316832
JoyAI-Echo_diffusion_model_quanto_bf16_int8.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:31106fd315dd3a254683ad1246c92aad0f6c5e8e656679f5ca54810efa380225
3
+ size 19438201145
LICENSE-LTX-2.5.txt ADDED
@@ -0,0 +1,580 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ LTX-2.x Community License Agreement
2
+ License date: August 11, 2026
3
+
4
+
5
+ By downloading, using, accessing or distributing any portion or
6
+ element of LTX-2.x, you agree that you have read and accepted to be
7
+ bound by this Agreement.
8
+
9
+ 1. Definitions
10
+
11
+ 1.1 "Agreement" means the terms and conditions for this LTX-2.x
12
+ Community License Agreement and the exhibits, attachments, and
13
+ Complementary Materials, as specified in this document.
14
+
15
+ 1.2 "Complementary Materials" means the accompanying
16
+ documentation, tutorials, examples, configuration files and
17
+ other materials made available by Licensor together with the
18
+ LTX-2.x model weights and parameters, in each case as
19
+ distributed by Licensor.
20
+
21
+ 1.3 "Control" means the direct or indirect ownership of more than
22
+ fifty percent (50%) of the voting securities or other
23
+ ownership interests, or the power to direct the management and
24
+ policies of such Entity through voting rights, contract, or
25
+ otherwise.
26
+
27
+ 1.4 "Data" means a collection of information and/or content
28
+ extracted from the dataset used with LTX-2.x, including to
29
+ train, pretrain, or otherwise evaluate LTX-2.x. The Data is
30
+ not licensed under this Agreement.
31
+
32
+ 1.5 "Derivatives of LTX-2.x" means all modifications to LTX-2.x,
33
+ works based on LTX-2.x, or any other model which is created or
34
+ initialized by transfer of patterns of the weights,
35
+ parameters, activations or output of LTX-2.x, to the other
36
+ model, in order to cause the other model to perform similarly
37
+ to LTX-2.x, including - but not limited to - distillation
38
+ methods entailing the use of intermediate data representations
39
+ or methods based on the generation of synthetic data by
40
+ LTX-2.x for training the other model. For clarity, Derivatives
41
+ of LTX-2.x include: (i) any fine-tuned or adapted weights,
42
+ parameters, or checkpoints derived from LTX-2.x; (ii)
43
+ derivative model architectures that incorporate or are based
44
+ upon LTX-2.x's architecture; and (iii) any modified or
45
+ extended versions of the Complementary Materials.
46
+
47
+ 1.6 "Entity" means any individual, corporation, partnership,
48
+ limited liability company, or other legal entity. For purposes
49
+ of this Agreement, an Entity shall be deemed to include, on an
50
+ aggregative basis, all subsidiaries, affiliates, and other
51
+ companies under common Control with such Entity. When
52
+ determining whether an Entity meets any threshold under this
53
+ Agreement (including revenue thresholds in Section 2.1), all
54
+ subsidiaries, affiliates, and companies under common Control
55
+ shall be considered collectively.
56
+
57
+ 1.7 "Harm" includes but is not limited to physical, mental,
58
+ psychological, financial and reputational damage, pain, or
59
+ loss.
60
+
61
+ 1.8 "Licensor" or "LTX" means the owner that is granting the
62
+ license under this Agreement. For the purposes of this
63
+ Agreement, the Licensor is Lightricks Ltd.
64
+
65
+ 1.9 "LTX-2.x" means the large generative models,
66
+ text/image/video/audio/3D generation models, and multimodal
67
+ large language models and their software and algorithms,
68
+ including trained model weights, parameters (including
69
+ optimizer states), machine-learning model code,
70
+ inference-enabling code, training-enabling code, fine-tuning
71
+ enabling code, accompanying source code, scripts, and all
72
+ other elements of the foregoing distributed and made publicly
73
+ available by LTX (including, for example, at
74
+ https://github.com/Lightricks/LTX-2). This license is
75
+ applicable to all LTX-2.5 versions released since August 11,
76
+ 2026, and all future releases of LTX-2.x under this license.
77
+
78
+ 1.10 "Output" means the results of operating LTX-2.x as embodied
79
+ in informational content resulting therefrom.
80
+
81
+ 1.11 "you" (or "your") means an individual or legal Entity
82
+ licensing LTX-2.x in accordance with this Agreement and/or
83
+ otherwise downloading, accessing, distributing or using
84
+ LTX-2.x for whichever purpose and in any field of use,
85
+ including usage of LTX-2.x in an end-use application - e.g.
86
+ chatbot, translator, image generator.
87
+
88
+ 2. Grant of License.
89
+
90
+ 2.1 Subject to your compliance with the terms and conditions of
91
+ this Agreement, you are granted a non-exclusive, worldwide,
92
+ non-transferable and royalty-free limited license under
93
+ Licensor's intellectual property or other rights owned by
94
+ Licensor embodied in LTX-2.x to use, reproduce, prepare,
95
+ distribute, publicly display, publicly perform, sublicense,
96
+ copy, create derivative works of, and make modifications to
97
+ LTX-2.x, for any purpose, subject to the restrictions set
98
+ forth in Attachment A; provided however, that Entities with
99
+ annual revenues of at least $10,000,000 (the "Commercial
100
+ Entities") are required to obtain a paid license for any use
101
+ (excluding use solely for a Non-Commercial Purpose as set
102
+ forth in Section 2.2) of LTX-2.x and Derivatives of LTX-2.x
103
+ (such paid license referred to herein as a "Commercial Use
104
+ Agreement"), as will be provided by the Licensor. Commercial
105
+ Entities interested in such a commercial license are required
106
+ to contact Licensor (ltxv-licensing@lightricks.com). Any use
107
+ of LTX-2.x or Derivatives of LTX-2.x by Commercial Entities
108
+ not in accordance with this Agreement and/or the Commercial
109
+ Use Agreement is strictly prohibited and shall be deemed a
110
+ material breach of this Agreement. In the event of such
111
+ material breach, and without limiting Licensor's right to
112
+ terminate the Agreement or to pursue any other remedies
113
+ available at law or in equity, you shall pay Licensor the
114
+ license fees owed for the period such Commercial Entity used
115
+ LTX-2.x (calculated at Licensor's standard commercial license
116
+ fees, in effect during the relevant period or, absent
117
+ published standard fees, a reasonable market rate for a
118
+ comparable license), within thirty (30) days of Licensor's
119
+ written demand.
120
+
121
+ 2.2 Notwithstanding the foregoing or anything to the contrary in
122
+ this Agreement, a Commercial Entity may download and use
123
+ LTX-2.x and Derivatives of LTX-2.x without obtaining the
124
+ Commercial Use Agreement solely for a Non-Commercial Purpose.
125
+ "Non-Commercial Purpose" means any of the following uses, but
126
+ only so far as such Commercial Entity does not receive any
127
+ direct or indirect payment arising from the use of LTX-2.x or
128
+ Derivatives of LTX-2.x: (i) use by an individual acting in a
129
+ personal capacity for research, experimentation, learning,
130
+ private study, hobby or recreational projects, or personal
131
+ entertainment, in each case where such use is not connected,
132
+ directly or indirectly, to any commercial activity, business
133
+ operation, or the performance of duties for an employer or any
134
+ other Entity; and (ii) use by a Commercial Entity for testing,
135
+ evaluation, or non-commercial research and development in a
136
+ non-production or development environment. For clarity, use
137
+ (a) for revenue-generating activity in any manner, whether
138
+ direct or indirect, (b) in direct interactions with or that
139
+ has impact on end users, or (c) to train, fine-tune, or
140
+ distill any model (including any Derivative of LTX-2.x) for
141
+ commercial use, in each case, is not a Non-Commercial Purpose
142
+ and requires all Commercial Entities to obtain a paid license
143
+ under the Commercial Use Agreement prior to such use. For the
144
+ avoidance of doubt, the permission granted under this Section
145
+ 2.2 is a limited right of use only and does not convey or
146
+ transfer any ownership right, title, or interest in or to
147
+ LTX-2.x or any Derivatives of LTX-2.x, and all Derivatives of
148
+ LTX-2.x created or used pursuant to this Section 2.2 remain
149
+ subject to the terms of this Agreement, including Section 1.5.
150
+
151
+ 3. Distribution and Redistribution. You may host for third parties
152
+ remote access purposes (e.g. software-as-a-service), reproduce and
153
+ distribute copies of LTX-2.x or Derivatives of LTX-2.x thereof in
154
+ any medium, with or without modifications, provided that you meet
155
+ the following conditions:
156
+
157
+ 3.1 Use-based restrictions as referenced in Section 4 and all
158
+ provisions of Attachment A MUST be included as an enforceable
159
+ provision by you in any type of legal agreement (e.g. a
160
+ license) governing the use and/or distribution of LTX-2.x or
161
+ Derivatives of LTX-2.x, and you shall give notice to
162
+ subsequent users you distribute to, that LTX-2.x or
163
+ Derivatives of LTX-2.x are subject to Section 4 and Attachment
164
+ A in their entirety, including all use restrictions and
165
+ acceptable use policies;
166
+
167
+ 3.2 You must provide any third-party recipients of LTX-2.x or
168
+ Derivatives of LTX-2.x a copy of this Agreement, including all
169
+ attachments and use policies. Any Derivative of LTX-2.x (as
170
+ defined in Section 1.5, including but not limited to
171
+ fine-tuned weights, modified training code, models trained on
172
+ Outputs, or any other derivative) must be distributed
173
+ exclusively under the terms of this Agreement, subject to
174
+ Section 3.6, with a complete copy of this Agreement included;
175
+
176
+ 3.3 You must cause any modified files to carry prominent notices
177
+ stating that you changed the files;
178
+
179
+ 3.4 You must retain all copyright, patent, trademark, and
180
+ attribution notices excluding those notices that do not
181
+ pertain to any part of LTX-2.x, Derivatives of LTX-2.x.
182
+
183
+ 3.5 Transfer of Derivatives. No transfer of any Derivative of
184
+ LTX-2.x (including any fine-tuned weights, LoRA adapters, or
185
+ similar adaptations) to a third party shall grant such third
186
+ party any right, title, license, or authorization to access,
187
+ use, reproduce, distribute, or exploit LTX-2.x, or any
188
+ Derivative of LTX-2.x beyond the rights granted under this
189
+ Agreement. If the transferee is a Commercial Entity (as
190
+ defined in Section 2), it must obtain a paid license from
191
+ Licensor prior to any use of any Derivative of LTX-2.x,
192
+ regardless of who created such Derivative. Prior to or at the
193
+ time of any such transfer, you shall notify the transferee in
194
+ writing that (i) use of such Derivative of LTX-2.x is subject
195
+ to the terms of this Agreement, and (ii) if the transferee is
196
+ a Commercial Entity, it must obtain a separate paid license to
197
+ LTX-2.x from Licensor. You shall not transfer any Derivative
198
+ of LTX-2.x to a Commercial Entity unless such Commercial
199
+ Entity has obtained the required paid license from Licensor
200
+ prior to any use, and unless the proposed transferee has been
201
+ so informed. You and the transferee shall each be responsible
202
+ for ensuring the transferee obtains the required license from
203
+ Licensor prior to any use of LTX-2.x or Derivative of LTX-2.x.
204
+ Nothing in this Section 3.5 shall require a Commercial Entity
205
+ to obtain a paid license for use solely for a Non-Commercial
206
+ Purpose as permitted under Section 2.2.
207
+
208
+ 3.6 You may add your own copyright statement to your modifications
209
+ and may provide additional license terms and conditions -
210
+ respecting Section 3.1 - for use, reproduction, or
211
+ distribution of your modifications, or for any such
212
+ Derivatives of LTX-2.x as a whole, provided your use,
213
+ reproduction, and distribution of LTX-2.x otherwise complies
214
+ with the conditions stated in this Agreement, and you provide
215
+ a complete copy of this Agreement with any such use,
216
+ reproduction and distribution of LTX-2.x and any Derivatives
217
+ thereof; provided that any such additional terms shall be
218
+ additive only and shall not derogate from, conflict with,
219
+ waive, or purport to modify any term of this Agreement, and
220
+ this Agreement shall govern in the event of any conflict.
221
+
222
+ 4. Use-based restrictions. The restrictions set forth in Attachment A
223
+ are considered Use-based restrictions. Therefore, you cannot use
224
+ LTX-2.x and the Derivatives of LTX-2.x in violation of the
225
+ specified restricted uses. You may use LTX-2.x subject to this
226
+ Agreement, only for lawful purposes and in accordance with the
227
+ Agreement. "Use" may include creating any content with,
228
+ fine-tuning, updating, running, training, evaluating and/or
229
+ re-parametrizing LTX-2.x. You shall require all of your users who
230
+ use LTX-2.x or a Derivative of LTX-2.x to comply with the terms of
231
+ this Section 4.
232
+
233
+ 5. The Output You Generate. Except as set forth herein, Licensor
234
+ claims no rights in the Output you generate using LTX-2.x. You are
235
+ accountable for input you insert into LTX-2.x, the Output you
236
+ generate and its subsequent uses. No use of the Output can
237
+ contravene any provision as stated in the Agreement.
238
+
239
+ 6. Updates and Runtime Restrictions; AI Regulations. To the maximum
240
+ extent permitted by law, Licensor reserves the right to restrict
241
+ (remotely or otherwise) usage of LTX-2.x in violation of this
242
+ Agreement, update LTX-2.x through electronic means, or modify the
243
+ Output of LTX-2.x based on updates. You shall undertake reasonable
244
+ efforts to use the latest version of LTX-2.x. Any use of the
245
+ non-current version of LTX-2.x is done solely at your risk. To the
246
+ extent applicable to you, you shall comply with all laws and
247
+ regulations governing artificial intelligence that apply to your
248
+ use, deployment, or distribution of LTX-2.x, Derivatives of
249
+ LTX-2.x, or Outputs, including Regulation (EU) 2024/1689 (the "EU
250
+ AI Act") and the California AI Transparency Act (Cal. Bus. & Prof.
251
+ Code § 22757 et seq.), each as amended from time to time and any
252
+ other applicable laws, regulations, or binding guidance relating
253
+ to artificial intelligence, transparency, content provenance, or
254
+ synthetic media, together with any documentation made available by
255
+ Licensor regarding compliance with the same (collectively, "AI
256
+ Regulations"). You shall maintain (including within any
257
+ application or service through which LTX-2.x, any Derivative of
258
+ LTX-2.x, or any Output is made available), and shall not remove,
259
+ disable, alter, or circumvent, any safety or security measures,
260
+ disclosures, metadata, watermarking, content provenance, latent
261
+ disclosure, or other transparency features or functionalities
262
+ included or embedded within LTX-2.x or any Derivative of LTX-2.x,
263
+ or applied to any Output, in furtherance of AI Regulations,
264
+ including any capability of LTX-2.x to include latent disclosures
265
+ in Outputs, and you shall include equivalent obligations in any
266
+ agreement governing your distribution of LTX-2.x or any Derivative
267
+ of LTX-2.x. You are solely responsible for any transparency,
268
+ disclosure, marking, or labeling obligations applicable to you
269
+ under AI Regulations as a provider or deployer of LTX-2.x, any
270
+ Derivative of LTX-2.x, or any system incorporating any of the
271
+ foregoing, including any obligation to disclose that content is
272
+ artificially generated or manipulated. If Licensor knows or
273
+ reasonably believes that you have modified LTX-2.x or any
274
+ Derivative of LTX-2.x such that it is no longer capable of
275
+ including any disclosure required by AI Regulations in Outputs, or
276
+ that you have otherwise removed, disabled, or circumvented any
277
+ feature or functionality described in this Section, Licensor may
278
+ in its sole discretion revoke the license granted under this
279
+ Agreement effective immediately upon notice to you, and upon such
280
+ revocation you shall immediately cease all use of LTX-2.x and
281
+ Derivatives of LTX-2.x. Licensor makes no representation or
282
+ warranty that LTX-2.x, any Derivative of LTX-2.x, or any Output
283
+ complies with any AI Regulations applicable to your specific use
284
+ case or deployment, and you are solely responsible for determining
285
+ the applicability of, and ensuring your compliance with, all AI
286
+ Regulations. You shall indemnify, defend, and hold harmless
287
+ Licensor and its affiliates from and against any and all claims,
288
+ liabilities, losses, damages, costs, and expenses (including
289
+ reasonable attorneys' fees) arising out of or relating to your
290
+ use, deployment, distribution, or modification of LTX-2.x, any
291
+ Derivative of LTX-2.x, or any Output in violation of, or your
292
+ other failure to comply with, any AI Regulations.
293
+
294
+ For purposes of the EU AI Act, Licensor makes LTX-2.x openly
295
+ available under this community license and intends that LTX-2.x be
296
+ treated as a free and open-source general purpose AI model within
297
+ the meaning of Article 53(2) of the EU AI Act. You acknowledge and
298
+ agree that (a) to the extent the free and open source derogations
299
+ under Article 53(2) of the EU AI Act apply, Licensor's obligations
300
+ under the EU AI Act with respect to LTX-2.x are limited to those
301
+ applicable to providers of free and open source general purpose AI
302
+ models (it being acknowledged that such derogations do not extend
303
+ to the obligations under Article 53(1)(c) and (d)), (b) you
304
+ acknowledge that LTX-2.x is not intended to be integrated into a
305
+ high risk AI system, and shall be fully and solely responsible for
306
+ any obligation resulting from such integration, (c) if you
307
+ integrate LTX-2.x or any Derivative of LTX-2.x into a high-risk AI
308
+ system you shall be solely responsible for all provider
309
+ obligations that would otherwise apply to Licensor under the EU AI
310
+ Act, and (d) you shall not take any action, or omit to take any
311
+ action, that would cause Licensor to lose the benefit of the free
312
+ and open source derogations under the EU AI Act, and you shall
313
+ indemnify and hold Licensor harmless from any liability, costs, or
314
+ expenses arising from your breach of this Section.
315
+
316
+ 7. Export Controls and Sanctions Compliance. You acknowledge that
317
+ LTX-2.x, Derivatives of LTX-2.x may be subject to export control
318
+ laws and regulations, including but not limited to the U.S. Export
319
+ Administration Regulations and sanctions programs administered by
320
+ the Office of Foreign Assets Control (OFAC). You represent and
321
+ warrant that you and any users of LTX-2.x are not (i) located in,
322
+ organized under the laws of, or ordinarily resident in any country
323
+ or territory subject to comprehensive sanctions; (ii) identified
324
+ on any U.S. government restricted party list, including the
325
+ Specially Designated Nationals and Blocked Persons List; or (iii)
326
+ otherwise prohibited from receiving LTX-2.x under applicable law.
327
+ You shall not export, re-export, or transfer LTX-2.x, directly or
328
+ indirectly, in violation of any applicable export control or
329
+ sanctions laws or regulations. You agree to comply with all
330
+ applicable trade control laws and shall indemnify and hold
331
+ Licensor harmless from any claims arising from your failure to
332
+ comply with such laws.
333
+
334
+ 8. Trademarks; Reservation of Rights. Nothing in this Agreement
335
+ permits you to make use of Licensor's trademarks, trade names,
336
+ logos or to otherwise suggest endorsement or misrepresent the
337
+ relationship between the parties; and any rights not expressly
338
+ granted herein are reserved by the Licensor. Except as expressly
339
+ set forth in this Agreement, Licensor does not grant, directly or
340
+ by implication, estoppel, statute or otherwise, any right or
341
+ license in its, or its affiliates', intellectual property rights
342
+ or other proprietary rights. For avoidance of doubt, all
343
+ intellectual property rights in Derivatives of LTX-2.x shall be
344
+ subject to the terms of this Agreement, and you acquire no right,
345
+ title, or interest in or to LTX-2.x itself, which is and remains
346
+ the exclusive property of Licensor. You shall not assert any
347
+ ownership or other right in LTX-2.x or any Derivative of LTX-2.x
348
+ in any manner that restricts, encumbers, or is inconsistent with
349
+ the rights retained by Licensor or granted to other licensees
350
+ under this Agreement.
351
+
352
+ 9. Disclaimer of Warranty. Unless required by applicable law or
353
+ agreed to in writing, Licensor provides LTX-2.x on an "AS IS"
354
+ BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either
355
+ express or implied, including, without limitation, any warranties
356
+ or conditions of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or
357
+ FITNESS FOR A PARTICULAR PURPOSE. You are solely responsible for
358
+ determining the appropriateness of using or redistributing LTX-2.x
359
+ and Derivatives of LTX-2.x and assume any risks associated with
360
+ your exercise of permissions under this Agreement.
361
+
362
+ 10. Limitation of Liability. To the fullest extent permitted by
363
+ applicable law, in no event and under no legal theory, whether in
364
+ tort (including negligence), contract, or otherwise, unless
365
+ required by applicable law (such as deliberate and grossly
366
+ negligent acts) or agreed to in writing, shall Licensor be liable
367
+ to you or any other individual or Entity for damages, including
368
+ any direct, indirect, special, incidental, or consequential
369
+ damages of any character arising as a result of this Agreement or
370
+ out of the use of, or inability to use LTX-2.x or any Derivative
371
+ of LTX-2.x (including but not limited to damages for loss of
372
+ goodwill, work stoppage, computer failure or malfunction, or any
373
+ and all other commercial damages or losses), even if Licensor has
374
+ been advised of the possibility of such damages.
375
+
376
+ 11. Accepting Warranty or Additional Liability. While redistributing
377
+ LTX-2.x and Derivatives of LTX-2.x, you may, provided you do not
378
+ violate the terms of this Agreement, choose to offer and charge a
379
+ fee for, acceptance of support, warranty, indemnity, or other
380
+ liability obligations. However, in accepting such obligations,
381
+ you may act only on your own behalf and on your sole
382
+ responsibility, not on behalf of Licensor, and only if you agree
383
+ to indemnify, defend, and hold Licensor harmless for any
384
+ liability incurred by, or claims asserted against Licensor, by
385
+ reason of your accepting any such warranty or additional
386
+ liability.
387
+
388
+ 12. Governing Law. This Agreement and all relations, disputes, claims
389
+ and other matters arising hereunder (including non-contractual
390
+ disputes or claims) will be governed exclusively by, and
391
+ construed exclusively in accordance with, the laws of the State
392
+ of New York and applicable U.S. federal law. To the extent
393
+ permitted by law, choice of laws rules and the United Nations
394
+ Convention on Contracts for the International Sale of Goods will
395
+ not apply. The prevailing party in any claim or dispute between
396
+ the parties under this Agreement will be entitled to
397
+ reimbursement of its reasonable attorneys' fees and costs.
398
+
399
+ 13. Term and Termination. This Agreement is effective upon your
400
+ acceptance and continues until terminated. Licensor may terminate
401
+ this Agreement immediately upon written notice to you if you
402
+ breach any provision of this Agreement, including but not limited
403
+ to violations of the use restrictions in Attachment A or
404
+ unauthorized commercial use. This Agreement also terminates
405
+ immediately and automatically, without notice, upon any material
406
+ breach of this Agreement, including any use in violation of
407
+ applicable AI Regulations or any unauthorized commercial use of
408
+ LTX-2.x or Derivatives of LTX-2.x by a Commercial Entity. Upon
409
+ termination: (a) all rights granted to you under this Agreement
410
+ will immediately cease; (b) you must immediately cease all use of
411
+ LTX-2.x and Derivatives of LTX-2.x; (c) you must delete or
412
+ destroy all copies of LTX-2.x and Derivatives of LTX-2.x in your
413
+ possession or control; and (d) you must notify any third parties
414
+ to whom you distributed LTX-2.x or Derivatives of LTX-2.x of the
415
+ termination. Sections 2, 3, 4, 6-16 and Attachment A shall
416
+ survive termination of this Agreement. Termination does not
417
+ relieve you of any obligations incurred prior to termination,
418
+ including payment obligations under Section 2 and adhering to the
419
+ restrictions under Section 3. In addition, if You commence a
420
+ lawsuit or other proceedings (including a cross-claim or
421
+ counterclaim in a lawsuit) against Licensor or any person or
422
+ entity alleging that LTX-2.x or any Output, or any portion of any
423
+ of the foregoing, infringe any intellectual property or other
424
+ right owned or licensable by you, then all licenses granted to
425
+ you under this Agreement shall terminate as of the date such
426
+ lawsuit or other proceeding is filed.
427
+
428
+ 14. Disputes and Arbitration; Waiver of Jury Trial; Class Action
429
+ Waiver. IF YOU ARE NOT ACTING AS A CONSUMER UNDER APPLICABLE LAW,
430
+ YOU HEREBY WAIVE THE RIGHT TO A TRIAL BY JURY, TO PARTICIPATE IN
431
+ A CLASS OR REPRESENTATIVE ACTION (INCLUDING IN ARBITRATION), OR
432
+ TO COMBINE INDIVIDUAL PROCEEDINGS IN COURT OR IN ARBITRATION
433
+ WITHOUT THE CONSENT OF ALL PARTIES. All disputes arising in
434
+ connection with this Agreement shall be finally settled by
435
+ arbitration under the Rules of Arbitration of the International
436
+ Chamber of Commerce ("ICC Rules"), by one (1) arbitrator
437
+ appointed in accordance with the ICC Rules. The seat of
438
+ arbitration shall be New York, NY, USA, and the proceedings shall
439
+ be conducted in English. The arbitrator shall be empowered to
440
+ grant any relief that a court could grant. Judgment on the
441
+ arbitration award may be entered by any court having jurisdiction
442
+ thereof. Notwithstanding the foregoing, either party may seek
443
+ injunctive or other equitable relief in respect of any actual or
444
+ threatened breach of the license restrictions under this
445
+ Agreement (including Attachment A and the Acceptable Use Policy)
446
+ or any actual or threatened infringement, misappropriation, or
447
+ violation of Licensor's intellectual property rights, in the
448
+ state or federal courts located in the County of New York, State
449
+ of New York, and each party irrevocably consents to the
450
+ jurisdiction of, and venue in, such courts for that limited
451
+ purpose. The foregoing waivers do not apply to, and are not
452
+ enforceable against, any licensee acting as a consumer under the
453
+ mandatory consumer-protection laws of its jurisdiction of
454
+ residence (including, without limitation, the European Union, the
455
+ United Kingdom, and the State of California), and nothing in this
456
+ Agreement limits any rights under such laws that cannot be waived
457
+ or limited by contract. If any waiver in this Section is held
458
+ invalid or unenforceable as to a particular licensee or dispute,
459
+ such waiver shall be severed to that extent only and shall not
460
+ affect the validity or enforceability of the remainder of this
461
+ Section.
462
+
463
+ 15. In the event of any exception to the application of binding
464
+ arbitration, all disputes, claims, and other matters arising
465
+ hereunder shall be brought exclusively in the state or federal
466
+ courts located in the County of New York, State of New York. You
467
+ waive all defenses of lack of personal jurisdiction and forum non
468
+ conveniens with respect to venue and jurisdiction in such courts,
469
+ and consent to their exclusive jurisdiction and venue.
470
+
471
+ 16. Severability. If any provision of this Agreement is held to be
472
+ invalid, illegal or unenforceable, the remaining provisions shall
473
+ be unaffected thereby and remain valid as if such provision had
474
+ not been set forth herein.
475
+
476
+ END OF TERMS AND CONDITIONS
477
+
478
+ Attachment A
479
+ Use Restrictions
480
+
481
+ When using the Outputs, LTX-2.x and any Derivatives thereof, you
482
+ agree to comply with the Acceptable Use Policy
483
+ (https://static.lightricks.com/legal/ltx-acceptable-use-policy.pdf)
484
+ which is hereby incorporated into and made part of this Agreement by
485
+ reference. Licensor may update it from time to time, and the version
486
+ in effect at the time of your use governs; continued use after an
487
+ update constitutes acceptance. Licensor shall post each version of
488
+ the Acceptable Use Policy with its effective date, and no update
489
+ shall apply retroactively to use occurring before that effective
490
+ date. In addition, you agree not to use the Outputs, LTX-2.x or its
491
+ Derivatives in any of the following ways:
492
+
493
+ 1) In any way that violates any applicable national, federal,
494
+ state, local or international law or regulation;
495
+
496
+ 2) For the purpose of exploiting, Harming or attempting to exploit
497
+ or Harm minors in any way;
498
+
499
+ 3) Knowingly generate or disseminate verifiably false information
500
+ and/or content with the intent to deceive, defraud, or
501
+ otherwise unlawfully Harm others;
502
+
503
+ 4) To generate or disseminate personal identifiable information
504
+ that can be used to Harm an individual;
505
+
506
+ 5) To generate or disseminate information and/or content (e.g.
507
+ images, code, posts, articles), and place the information
508
+ and/or content in any context (e.g. bot generating tweets)
509
+ without expressly and intelligibly disclaiming that the
510
+ information and/or content is machine generated;
511
+
512
+ 6) To defame others, or to engage in the unlawful harassment of
513
+ others;
514
+
515
+ 7) To impersonate or attempt to impersonate (e.g. deepfakes)
516
+ others without their consent;
517
+
518
+ 8) For fully automated decision making that adversely impacts an
519
+ individual's legal rights or otherwise creates or modifies a
520
+ binding, enforceable obligation;
521
+
522
+ 9) For any use intended to or which has the effect of
523
+ discriminating against or Harming individuals or groups based
524
+ on online or offline social behavior or known or predicted
525
+ personal or personality characteristics;
526
+
527
+ 10) To exploit any of the vulnerabilities of a specific group of
528
+ persons based on their age, social, physical or mental
529
+ characteristics, in order to materially distort the behavior
530
+ of a person pertaining to that group in a manner that causes
531
+ or is likely to cause that person or another person physical
532
+ or psychological Harm;
533
+
534
+ 11) For any use intended to or which has the effect of
535
+ discriminating against individuals or groups based on legally
536
+ protected characteristics or categories;
537
+
538
+ 12) To provide medical advice and medical results interpretation;
539
+
540
+ 13) To generate or disseminate information for the purpose to be
541
+ used for administration of justice, law enforcement,
542
+ immigration or asylum processes, such as predicting an
543
+ individual will commit fraud/crime commitment (e.g. by text
544
+ profiling, drawing causal relationships between assertions
545
+ made in documents, indiscriminate and arbitrarily-targeted
546
+ use);
547
+
548
+ 14) To generate and/or disseminate malware (including - but not
549
+ limited to - ransomware) or any other content to be used for
550
+ the purpose of harming electronic systems;
551
+
552
+ 15) To engage in, promote, incite, or facilitate discrimination or
553
+ other unlawful or harmful conduct in the provision of
554
+ employment, employment benefits, credit, housing, or other
555
+ essential goods and services;
556
+
557
+ 16) To engage in, promote, incite, or facilitate the harassment,
558
+ abuse, threatening, or bullying of individuals or groups of
559
+ individuals;
560
+
561
+ 17) For military, warfare, nuclear industries or applications,
562
+ weapons development, or any use in connection with activities
563
+ that may cause death, personal injury, or severe physical or
564
+ environmental damage;
565
+
566
+ 18) For commercial use only: To train, improve, or fine-tune any
567
+ other machine learning model, artificial intelligence system,
568
+ or competing model, except for Derivatives of LTX-2.x as
569
+ expressly permitted under this Agreement;
570
+
571
+ 19) To circumvent, disable, or interfere with any technical
572
+ limitations, safety features, content filters, watermarking,
573
+ content provenance or latent disclosure functionalities, or
574
+ use restrictions implemented in LTX-2.x by Licensor;
575
+
576
+ 20) To use LTX-2.x or Derivatives of LTX-2.x in any product,
577
+ service, or application that directly competes with Licensor's
578
+ commercial products or services, or is designed to replace or
579
+ substitute Licensor's offerings in the market, without
580
+ obtaining a separate commercial license from Licensor.
LTX-2.5-MANIFEST.md ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # LTX-2.5 WanGP runtime files
2
+
3
+ ## Main transformers (repository root)
4
+
5
+ - `ltx-2.5-22b-dev_diffusion_model_bf16.safetensors`
6
+ - `ltx-2.5-22b-dev_diffusion_model_int8_convrot.safetensors`
7
+ - `ltx-2.5-22b-distilled_diffusion_model_bf16.safetensors`
8
+ - `ltx-2.5-22b-distilled_diffusion_model_int8_convrot.safetensors`
9
+ - `ltx-2.5-22b-distilled_diffusion_model_nvfp4.safetensors`
10
+
11
+ ## Shared/offloadable components (repository root)
12
+
13
+ - `ltx-2.5-22b_video_embeddings_connector_bf16.safetensors`
14
+ - `ltx-2.5-22b_audio_embeddings_connector_bf16.safetensors`
15
+ - `ltx-2.5-22b_video_embeddings_connector_int8_convrot.safetensors`
16
+ - `ltx-2.5-22b_audio_embeddings_connector_int8_convrot.safetensors`
17
+ - `ltx-2.5-22b_video_embeddings_connector_nvfp4_bf16.safetensors`
18
+ - `ltx-2.5-22b_audio_embeddings_connector_nvfp4_bf16.safetensors`
19
+ - `ltx-2.5-22b_text_embedding_projection_bf16.safetensors`
20
+ - `ltx-2.5-22b_video_vae_bf16.safetensors`
21
+ - `ltx-2.5-22b_audio_vae_bf16.safetensors`
22
+ - `ltx-2.5-22b_vocoder_bf16.safetensors`
23
+ - `ltx-2.5-spatial-upscaler-x2-1.0_bf16.safetensors`
24
+ - `ltx-2.5-temporal-upscaler-x2-1.0_bf16.safetensors`
25
+
26
+ Dev and Distilled connector tensors are shared only after equality verification. The official NVFP4 transformer has different connector values, so it uses its own BF16 connector pair.
27
+
28
+ ## Optional LTX-2.5 LoRA
29
+
30
+ - `ltx-2.5-22b-distilled-lora-450_bf16.safetensors`
31
+
32
+ This upstream LTX-2.5 LoRA is retained as an optional asset but is not automatically enabled by the WanGP LTX-2.5 defaults. LTX-2.0/2.3 LoRAs and their feature workflows remain isolated from LTX-2.5.
33
+
34
+ ## Gemma 4 (`gemma4-12b-ltx-v1/`)
35
+
36
+ - `gemma4-12b-ltx-v1_bf16.safetensors`
37
+ - `gemma4-12b-ltx-v1_int8_convrot.safetensors`
38
+ - `config.json`
39
+ - `tokenizer.json`
40
+ - `tokenizer_config.json`
41
+ - `chat_template.jinja`
42
+
43
+ These files are derivatives of the official LTX-2.5 release and are distributed under `LICENSE-LTX-2.5.txt`.
README.md ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model:
3
+ - Lightricks/LTX-2
4
+ tags:
5
+ - diffusion-single-file
6
+ ---
7
+
8
+ You will find here all the LTX-2 Video models used with WanGP (https://github.com/deepbeepmeep/Wan2GP) :
9
+
10
+
11
+ WanGP by DeepBeepMeep : The best Open Source Video Generative Models Accessible to the GPU Poor
12
+
13
+ WanGP supports the Wan (and derived models), Hunyuan Video, Flux 1 & 2, Qwen, Z-Image and LTV Video models with:
14
+
15
+ Low VRAM requirements (as low as 6 GB of VRAM is sufficient for certain models)
16
+ Support for old GPUs (RTX 10XX, 20xx, ...)
17
+ Very Fast on the latest GPUs
18
+ Easy to use Full Web based interface
19
+ Auto download of the required model adapted to your specific architecture
20
+ Tools integrated to facilitate Video Generation : Mask Editor, Prompt Enhancer, Temporal and Spatial Generation
21
+ Loras Support to customize each model
22
+ Queuing system : make your shopping list of videos to generate and come back later
23
+ Discord Server to get Help from Other Users and show your Best Videos: https://discord.gg/g7efUW9jGV
24
+
25
+ Follow DeepBeepMeep on Twitter/X to get the Latest News: https://x.com/deepbeepmeep
bigvgan_v2_22khz_80band_256x/bigvgan_generator.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e95ba25972d3de0628d99cd156e9315a9c018899bf739988959ebe3544080ced
3
+ size 449228171
bigvgan_v2_22khz_80band_256x/config.json ADDED
@@ -0,0 +1,63 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "resblock": "1",
3
+ "num_gpus": 0,
4
+ "batch_size": 32,
5
+ "learning_rate": 0.0001,
6
+ "adam_b1": 0.8,
7
+ "adam_b2": 0.99,
8
+ "lr_decay": 0.9999996,
9
+ "seed": 1234,
10
+
11
+ "upsample_rates": [4,4,2,2,2,2],
12
+ "upsample_kernel_sizes": [8,8,4,4,4,4],
13
+ "upsample_initial_channel": 1536,
14
+ "resblock_kernel_sizes": [3,7,11],
15
+ "resblock_dilation_sizes": [[1,3,5], [1,3,5], [1,3,5]],
16
+
17
+ "use_tanh_at_final": false,
18
+ "use_bias_at_final": false,
19
+
20
+ "activation": "snakebeta",
21
+ "snake_logscale": true,
22
+
23
+ "use_cqtd_instead_of_mrd": true,
24
+ "cqtd_filters": 128,
25
+ "cqtd_max_filters": 1024,
26
+ "cqtd_filters_scale": 1,
27
+ "cqtd_dilations": [1, 2, 4],
28
+ "cqtd_hop_lengths": [512, 256, 256],
29
+ "cqtd_n_octaves": [9, 9, 9],
30
+ "cqtd_bins_per_octaves": [24, 36, 48],
31
+
32
+ "mpd_reshapes": [2, 3, 5, 7, 11],
33
+ "use_spectral_norm": false,
34
+ "discriminator_channel_mult": 1,
35
+
36
+ "use_multiscale_melloss": true,
37
+ "lambda_melloss": 15,
38
+
39
+ "clip_grad_norm": 500,
40
+
41
+ "segment_size": 65536,
42
+ "num_mels": 80,
43
+ "num_freq": 1025,
44
+ "n_fft": 1024,
45
+ "hop_size": 256,
46
+ "win_size": 1024,
47
+
48
+ "sampling_rate": 22050,
49
+
50
+ "fmin": 0,
51
+ "fmax": null,
52
+ "fmax_for_loss": null,
53
+
54
+ "normalize_volume": true,
55
+
56
+ "num_workers": 4,
57
+
58
+ "dist_config": {
59
+ "dist_backend": "nccl",
60
+ "dist_url": "tcp://localhost:54321",
61
+ "world_size": 1
62
+ }
63
+ }
bigvgan_v2_44khz_128band_512x/bigvgan_generator.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d9fe7ec6bd0b44ed9d66973d5012d8181c1570b01e5c72df51973e241dccd357
3
+ size 489041291
bigvgan_v2_44khz_128band_512x/config.json ADDED
@@ -0,0 +1,63 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "resblock": "1",
3
+ "num_gpus": 0,
4
+ "batch_size": 32,
5
+ "learning_rate": 0.0001,
6
+ "adam_b1": 0.8,
7
+ "adam_b2": 0.99,
8
+ "lr_decay": 0.9999996,
9
+ "seed": 1234,
10
+
11
+ "upsample_rates": [8,4,2,2,2,2],
12
+ "upsample_kernel_sizes": [16,8,4,4,4,4],
13
+ "upsample_initial_channel": 1536,
14
+ "resblock_kernel_sizes": [3,7,11],
15
+ "resblock_dilation_sizes": [[1,3,5], [1,3,5], [1,3,5]],
16
+
17
+ "use_tanh_at_final": false,
18
+ "use_bias_at_final": false,
19
+
20
+ "activation": "snakebeta",
21
+ "snake_logscale": true,
22
+
23
+ "use_cqtd_instead_of_mrd": true,
24
+ "cqtd_filters": 128,
25
+ "cqtd_max_filters": 1024,
26
+ "cqtd_filters_scale": 1,
27
+ "cqtd_dilations": [1, 2, 4],
28
+ "cqtd_hop_lengths": [512, 256, 256],
29
+ "cqtd_n_octaves": [9, 9, 9],
30
+ "cqtd_bins_per_octaves": [24, 36, 48],
31
+
32
+ "mpd_reshapes": [2, 3, 5, 7, 11],
33
+ "use_spectral_norm": false,
34
+ "discriminator_channel_mult": 1,
35
+
36
+ "use_multiscale_melloss": true,
37
+ "lambda_melloss": 15,
38
+
39
+ "clip_grad_norm": 500,
40
+
41
+ "segment_size": 65536,
42
+ "num_mels": 128,
43
+ "num_freq": 2049,
44
+ "n_fft": 2048,
45
+ "hop_size": 512,
46
+ "win_size": 2048,
47
+
48
+ "sampling_rate": 44100,
49
+
50
+ "fmin": 0,
51
+ "fmax": null,
52
+ "fmax_for_loss": null,
53
+
54
+ "normalize_volume": true,
55
+
56
+ "num_workers": 4,
57
+
58
+ "dist_config": {
59
+ "dist_backend": "nccl",
60
+ "dist_url": "tcp://localhost:54321",
61
+ "world_size": 1
62
+ }
63
+ }
dramabox-dit-v1_bf16.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:01a626525d935e8c9fb0efe124334d1e4970aeda82215d2e14ca9fe904b5c25d
3
+ size 6575225528
dramabox-dit-v1_quanto_bf16_int8.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:afb34d0f50fe129c77bead2854ce9ad02d6d4a372a56d0d28086c87f1098feae
3
+ size 3350531372
edit_anything_reference_v0.1_r128_ref_adaln_proj-role_embedding-ref_attn-ref_visual_proj.module.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:63ffdeed38c191108229ec3085386ac10174a0730427f86ef2c20dec4c6ea663
3
+ size 450782608
edit_anything_reference_v0.1_r128_ref_adaln_proj-role_embedding-ref_attn-ref_visual_proj.standard.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0e2e51d9eafd6636c9e752300578447344925b05bb5254a405302d3a6f9c668d
3
+ size 1308756368
gemma-3-12b-it-qat-q4_0-unquantized/README.md ADDED
@@ -0,0 +1,452 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: google/gemma-3-12b-it
3
+ license: gemma
4
+ tags:
5
+ - gemma3
6
+ - gemma
7
+ - google
8
+ pipeline_tag: image-text-to-text
9
+ library_name: transformers
10
+ extra_gated_heading: Access Gemma on Hugging Face
11
+ extra_gated_prompt: >-
12
+ To access Gemma on Hugging Face, you’re required to review and agree to
13
+ Google’s usage license. To do this, please ensure you’re logged in to Hugging
14
+ Face and click below. Requests are processed immediately.
15
+ extra_gated_button_content: Acknowledge license
16
+ ---
17
+
18
+ # Gemma 3 model card
19
+
20
+ **Model Page**: [Gemma](https://ai.google.dev/gemma/docs/core)
21
+
22
+ > [!Note]
23
+ > This repository corresponds to the 12B **instruction-tuned** version of the Gemma 3 model using Quantization Aware Training (QAT).
24
+ >
25
+ > **The checkpoint in this repository is unquantized, please make sure to quantize with Q4_0 with your favorite tool**
26
+ >
27
+ > Thanks to QAT, the model is able to preserve similar quality as `bfloat16` while significantly reducing the memory requirements
28
+ > to load the model.
29
+
30
+
31
+ **Resources and Technical Documentation**:
32
+
33
+ * [Gemma 3 Technical Report][g3-tech-report]
34
+ * [Responsible Generative AI Toolkit][rai-toolkit]
35
+ * [Gemma on Kaggle][kaggle-gemma]
36
+ * [Gemma on Vertex Model Garden][vertex-mg-gemma3]
37
+
38
+ **Terms of Use**: [Terms][terms]
39
+
40
+ **Authors**: Google DeepMind
41
+
42
+ ## Model Information
43
+
44
+ Summary description and brief definition of inputs and outputs.
45
+
46
+ ### Description
47
+
48
+ Gemma is a family of lightweight, state-of-the-art open models from Google,
49
+ built from the same research and technology used to create the Gemini models.
50
+ Gemma 3 models are multimodal, handling text and image input and generating text
51
+ output, with open weights for both pre-trained variants and instruction-tuned
52
+ variants. Gemma 3 has a large, 128K context window, multilingual support in over
53
+ 140 languages, and is available in more sizes than previous versions. Gemma 3
54
+ models are well-suited for a variety of text generation and image understanding
55
+ tasks, including question answering, summarization, and reasoning. Their
56
+ relatively small size makes it possible to deploy them in environments with
57
+ limited resources such as laptops, desktops or your own cloud infrastructure,
58
+ democratizing access to state of the art AI models and helping foster innovation
59
+ for everyone.
60
+
61
+ ### Inputs and outputs
62
+
63
+ - **Input:**
64
+ - Text string, such as a question, a prompt, or a document to be summarized
65
+ - Images, normalized to 896 x 896 resolution and encoded to 256 tokens
66
+ each
67
+ - Total input context of 128K tokens for the 4B, 12B, and 27B sizes, and
68
+ 32K tokens for the 1B size
69
+
70
+ - **Output:**
71
+ - Generated text in response to the input, such as an answer to a
72
+ question, analysis of image content, or a summary of a document
73
+ - Total output context of 8192 tokens
74
+
75
+ ### Citation
76
+
77
+ ```none
78
+ @article{gemma_2025,
79
+ title={Gemma 3},
80
+ url={https://goo.gle/Gemma3Report},
81
+ publisher={Kaggle},
82
+ author={Gemma Team},
83
+ year={2025}
84
+ }
85
+ ```
86
+
87
+ ## Model Data
88
+
89
+ Data used for model training and how the data was processed.
90
+
91
+ ### Training Dataset
92
+
93
+ These models were trained on a dataset of text data that includes a wide variety
94
+ of sources. The 27B model was trained with 14 trillion tokens, the 12B model was
95
+ trained with 12 trillion tokens, 4B model was trained with 4 trillion tokens and
96
+ 1B with 2 trillion tokens. Here are the key components:
97
+
98
+ - Web Documents: A diverse collection of web text ensures the model is
99
+ exposed to a broad range of linguistic styles, topics, and vocabulary. The
100
+ training dataset includes content in over 140 languages.
101
+ - Code: Exposing the model to code helps it to learn the syntax and
102
+ patterns of programming languages, which improves its ability to generate
103
+ code and understand code-related questions.
104
+ - Mathematics: Training on mathematical text helps the model learn logical
105
+ reasoning, symbolic representation, and to address mathematical queries.
106
+ - Images: A wide range of images enables the model to perform image
107
+ analysis and visual data extraction tasks.
108
+
109
+ The combination of these diverse data sources is crucial for training a powerful
110
+ multimodal model that can handle a wide variety of different tasks and data
111
+ formats.
112
+
113
+ ### Data Preprocessing
114
+
115
+ Here are the key data cleaning and filtering methods applied to the training
116
+ data:
117
+
118
+ - CSAM Filtering: Rigorous CSAM (Child Sexual Abuse Material) filtering
119
+ was applied at multiple stages in the data preparation process to ensure
120
+ the exclusion of harmful and illegal content.
121
+ - Sensitive Data Filtering: As part of making Gemma pre-trained models
122
+ safe and reliable, automated techniques were used to filter out certain
123
+ personal information and other sensitive data from training sets.
124
+ - Additional methods: Filtering based on content quality and safety in
125
+ line with [our policies][safety-policies].
126
+
127
+ ## Implementation Information
128
+
129
+ Details about the model internals.
130
+
131
+ ### Hardware
132
+
133
+ Gemma was trained using [Tensor Processing Unit (TPU)][tpu] hardware (TPUv4p,
134
+ TPUv5p and TPUv5e). Training vision-language models (VLMS) requires significant
135
+ computational power. TPUs, designed specifically for matrix operations common in
136
+ machine learning, offer several advantages in this domain:
137
+
138
+ - Performance: TPUs are specifically designed to handle the massive
139
+ computations involved in training VLMs. They can speed up training
140
+ considerably compared to CPUs.
141
+ - Memory: TPUs often come with large amounts of high-bandwidth memory,
142
+ allowing for the handling of large models and batch sizes during training.
143
+ This can lead to better model quality.
144
+ - Scalability: TPU Pods (large clusters of TPUs) provide a scalable
145
+ solution for handling the growing complexity of large foundation models.
146
+ You can distribute training across multiple TPU devices for faster and more
147
+ efficient processing.
148
+ - Cost-effectiveness: In many scenarios, TPUs can provide a more
149
+ cost-effective solution for training large models compared to CPU-based
150
+ infrastructure, especially when considering the time and resources saved
151
+ due to faster training.
152
+ - These advantages are aligned with
153
+ [Google's commitments to operate sustainably][sustainability].
154
+
155
+ ### Software
156
+
157
+ Training was done using [JAX][jax] and [ML Pathways][ml-pathways].
158
+
159
+ JAX allows researchers to take advantage of the latest generation of hardware,
160
+ including TPUs, for faster and more efficient training of large models. ML
161
+ Pathways is Google's latest effort to build artificially intelligent systems
162
+ capable of generalizing across multiple tasks. This is specially suitable for
163
+ foundation models, including large language models like these ones.
164
+
165
+ Together, JAX and ML Pathways are used as described in the
166
+ [paper about the Gemini family of models][gemini-2-paper]; *"the 'single
167
+ controller' programming model of Jax and Pathways allows a single Python
168
+ process to orchestrate the entire training run, dramatically simplifying the
169
+ development workflow."*
170
+
171
+ ## Evaluation
172
+
173
+ > [!Note]
174
+ > The evaluation in this section correspond to the original checkpoint, not the QAT checkpoint.
175
+ >
176
+
177
+ Model evaluation metrics and results.
178
+
179
+ ### Benchmark Results
180
+
181
+ These models were evaluated against a large collection of different datasets and
182
+ metrics to cover different aspects of text generation:
183
+
184
+ #### Reasoning and factuality
185
+
186
+ | Benchmark | Metric | Gemma 3 PT 1B | Gemma 3 PT 4B | Gemma 3 PT 12B | Gemma 3 PT 27B |
187
+ | ------------------------------ |----------------|:--------------:|:-------------:|:--------------:|:--------------:|
188
+ | [HellaSwag][hellaswag] | 10-shot | 62.3 | 77.2 | 84.2 | 85.6 |
189
+ | [BoolQ][boolq] | 0-shot | 63.2 | 72.3 | 78.8 | 82.4 |
190
+ | [PIQA][piqa] | 0-shot | 73.8 | 79.6 | 81.8 | 83.3 |
191
+ | [SocialIQA][socialiqa] | 0-shot | 48.9 | 51.9 | 53.4 | 54.9 |
192
+ | [TriviaQA][triviaqa] | 5-shot | 39.8 | 65.8 | 78.2 | 85.5 |
193
+ | [Natural Questions][naturalq] | 5-shot | 9.48 | 20.0 | 31.4 | 36.1 |
194
+ | [ARC-c][arc] | 25-shot | 38.4 | 56.2 | 68.9 | 70.6 |
195
+ | [ARC-e][arc] | 0-shot | 73.0 | 82.4 | 88.3 | 89.0 |
196
+ | [WinoGrande][winogrande] | 5-shot | 58.2 | 64.7 | 74.3 | 78.8 |
197
+ | [BIG-Bench Hard][bbh] | few-shot | 28.4 | 50.9 | 72.6 | 77.7 |
198
+ | [DROP][drop] | 1-shot | 42.4 | 60.1 | 72.2 | 77.2 |
199
+
200
+ [hellaswag]: https://arxiv.org/abs/1905.07830
201
+ [boolq]: https://arxiv.org/abs/1905.10044
202
+ [piqa]: https://arxiv.org/abs/1911.11641
203
+ [socialiqa]: https://arxiv.org/abs/1904.09728
204
+ [triviaqa]: https://arxiv.org/abs/1705.03551
205
+ [naturalq]: https://github.com/google-research-datasets/natural-questions
206
+ [arc]: https://arxiv.org/abs/1911.01547
207
+ [winogrande]: https://arxiv.org/abs/1907.10641
208
+ [bbh]: https://paperswithcode.com/dataset/bbh
209
+ [drop]: https://arxiv.org/abs/1903.00161
210
+
211
+ #### STEM and code
212
+
213
+ | Benchmark | Metric | Gemma 3 PT 4B | Gemma 3 PT 12B | Gemma 3 PT 27B |
214
+ | ------------------------------ |----------------|:-------------:|:--------------:|:--------------:|
215
+ | [MMLU][mmlu] | 5-shot | 59.6 | 74.5 | 78.6 |
216
+ | [MMLU][mmlu] (Pro COT) | 5-shot | 29.2 | 45.3 | 52.2 |
217
+ | [AGIEval][agieval] | 3-5-shot | 42.1 | 57.4 | 66.2 |
218
+ | [MATH][math] | 4-shot | 24.2 | 43.3 | 50.0 |
219
+ | [GSM8K][gsm8k] | 8-shot | 38.4 | 71.0 | 82.6 |
220
+ | [GPQA][gpqa] | 5-shot | 15.0 | 25.4 | 24.3 |
221
+ | [MBPP][mbpp] | 3-shot | 46.0 | 60.4 | 65.6 |
222
+ | [HumanEval][humaneval] | 0-shot | 36.0 | 45.7 | 48.8 |
223
+
224
+ [mmlu]: https://arxiv.org/abs/2009.03300
225
+ [agieval]: https://arxiv.org/abs/2304.06364
226
+ [math]: https://arxiv.org/abs/2103.03874
227
+ [gsm8k]: https://arxiv.org/abs/2110.14168
228
+ [gpqa]: https://arxiv.org/abs/2311.12022
229
+ [mbpp]: https://arxiv.org/abs/2108.07732
230
+ [humaneval]: https://arxiv.org/abs/2107.03374
231
+
232
+ #### Multilingual
233
+
234
+ | Benchmark | Gemma 3 PT 1B | Gemma 3 PT 4B | Gemma 3 PT 12B | Gemma 3 PT 27B |
235
+ | ------------------------------------ |:-------------:|:-------------:|:--------------:|:--------------:|
236
+ | [MGSM][mgsm] | 2.04 | 34.7 | 64.3 | 74.3 |
237
+ | [Global-MMLU-Lite][global-mmlu-lite] | 24.9 | 57.0 | 69.4 | 75.7 |
238
+ | [WMT24++][wmt24pp] (ChrF) | 36.7 | 48.4 | 53.9 | 55.7 |
239
+ | [FloRes][flores] | 29.5 | 39.2 | 46.0 | 48.8 |
240
+ | [XQuAD][xquad] (all) | 43.9 | 68.0 | 74.5 | 76.8 |
241
+ | [ECLeKTic][eclektic] | 4.69 | 11.0 | 17.2 | 24.4 |
242
+ | [IndicGenBench][indicgenbench] | 41.4 | 57.2 | 61.7 | 63.4 |
243
+
244
+ [mgsm]: https://arxiv.org/abs/2210.03057
245
+ [flores]: https://arxiv.org/abs/2106.03193
246
+ [xquad]: https://arxiv.org/abs/1910.11856v3
247
+ [global-mmlu-lite]: https://huggingface.co/datasets/CohereForAI/Global-MMLU-Lite
248
+ [wmt24pp]: https://arxiv.org/abs/2502.12404v1
249
+ [eclektic]: https://arxiv.org/abs/2502.21228
250
+ [indicgenbench]: https://arxiv.org/abs/2404.16816
251
+
252
+ #### Multimodal
253
+
254
+ | Benchmark | Gemma 3 PT 4B | Gemma 3 PT 12B | Gemma 3 PT 27B |
255
+ | ------------------------------ |:-------------:|:--------------:|:--------------:|
256
+ | [COCOcap][coco-cap] | 102 | 111 | 116 |
257
+ | [DocVQA][docvqa] (val) | 72.8 | 82.3 | 85.6 |
258
+ | [InfoVQA][info-vqa] (val) | 44.1 | 54.8 | 59.4 |
259
+ | [MMMU][mmmu] (pt) | 39.2 | 50.3 | 56.1 |
260
+ | [TextVQA][textvqa] (val) | 58.9 | 66.5 | 68.6 |
261
+ | [RealWorldQA][realworldqa] | 45.5 | 52.2 | 53.9 |
262
+ | [ReMI][remi] | 27.3 | 38.5 | 44.8 |
263
+ | [AI2D][ai2d] | 63.2 | 75.2 | 79.0 |
264
+ | [ChartQA][chartqa] | 63.6 | 74.7 | 76.3 |
265
+ | [VQAv2][vqav2] | 63.9 | 71.2 | 72.9 |
266
+ | [BLINK][blinkvqa] | 38.0 | 35.9 | 39.6 |
267
+ | [OKVQA][okvqa] | 51.0 | 58.7 | 60.2 |
268
+ | [TallyQA][tallyqa] | 42.5 | 51.8 | 54.3 |
269
+ | [SpatialSense VQA][ss-vqa] | 50.9 | 60.0 | 59.4 |
270
+ | [CountBenchQA][countbenchqa] | 26.1 | 17.8 | 68.0 |
271
+
272
+ [coco-cap]: https://cocodataset.org/#home
273
+ [docvqa]: https://www.docvqa.org/
274
+ [info-vqa]: https://arxiv.org/abs/2104.12756
275
+ [mmmu]: https://arxiv.org/abs/2311.16502
276
+ [textvqa]: https://textvqa.org/
277
+ [realworldqa]: https://paperswithcode.com/dataset/realworldqa
278
+ [remi]: https://arxiv.org/html/2406.09175v1
279
+ [ai2d]: https://allenai.org/data/diagrams
280
+ [chartqa]: https://arxiv.org/abs/2203.10244
281
+ [vqav2]: https://visualqa.org/index.html
282
+ [blinkvqa]: https://arxiv.org/abs/2404.12390
283
+ [okvqa]: https://okvqa.allenai.org/
284
+ [tallyqa]: https://arxiv.org/abs/1810.12440
285
+ [ss-vqa]: https://arxiv.org/abs/1908.02660
286
+ [countbenchqa]: https://github.com/google-research/big_vision/blob/main/big_vision/datasets/countbenchqa/
287
+
288
+ ## Ethics and Safety
289
+
290
+ Ethics and safety evaluation approach and results.
291
+
292
+ ### Evaluation Approach
293
+
294
+ Our evaluation methods include structured evaluations and internal red-teaming
295
+ testing of relevant content policies. Red-teaming was conducted by a number of
296
+ different teams, each with different goals and human evaluation metrics. These
297
+ models were evaluated against a number of different categories relevant to
298
+ ethics and safety, including:
299
+
300
+ - **Child Safety**: Evaluation of text-to-text and image to text prompts
301
+ covering child safety policies, including child sexual abuse and
302
+ exploitation.
303
+ - **Content Safety:** Evaluation of text-to-text and image to text prompts
304
+ covering safety policies including, harassment, violence and gore, and hate
305
+ speech.
306
+ - **Representational Harms**: Evaluation of text-to-text and image to text
307
+ prompts covering safety policies including bias, stereotyping, and harmful
308
+ associations or inaccuracies.
309
+
310
+ In addition to development level evaluations, we conduct "assurance
311
+ evaluations" which are our 'arms-length' internal evaluations for responsibility
312
+ governance decision making. They are conducted separately from the model
313
+ development team, to inform decision making about release. High level findings
314
+ are fed back to the model team, but prompt sets are held-out to prevent
315
+ overfitting and preserve the results' ability to inform decision making.
316
+ Assurance evaluation results are reported to our Responsibility & Safety Council
317
+ as part of release review.
318
+
319
+ ### Evaluation Results
320
+
321
+ For all areas of safety testing, we saw major improvements in the categories of
322
+ child safety, content safety, and representational harms relative to previous
323
+ Gemma models. All testing was conducted without safety filters to evaluate the
324
+ model capabilities and behaviors. For both text-to-text and image-to-text, and
325
+ across all model sizes, the model produced minimal policy violations, and showed
326
+ significant improvements over previous Gemma models' performance with respect
327
+ to ungrounded inferences. A limitation of our evaluations was they included only
328
+ English language prompts.
329
+
330
+ ## Usage and Limitations
331
+
332
+ These models have certain limitations that users should be aware of.
333
+
334
+ ### Intended Usage
335
+
336
+ Open vision-language models (VLMs) models have a wide range of applications
337
+ across various industries and domains. The following list of potential uses is
338
+ not comprehensive. The purpose of this list is to provide contextual information
339
+ about the possible use-cases that the model creators considered as part of model
340
+ training and development.
341
+
342
+ - Content Creation and Communication
343
+ - Text Generation: These models can be used to generate creative text
344
+ formats such as poems, scripts, code, marketing copy, and email drafts.
345
+ - Chatbots and Conversational AI: Power conversational interfaces
346
+ for customer service, virtual assistants, or interactive applications.
347
+ - Text Summarization: Generate concise summaries of a text corpus,
348
+ research papers, or reports.
349
+ - Image Data Extraction: These models can be used to extract,
350
+ interpret, and summarize visual data for text communications.
351
+ - Research and Education
352
+ - Natural Language Processing (NLP) and VLM Research: These
353
+ models can serve as a foundation for researchers to experiment with VLM
354
+ and NLP techniques, develop algorithms, and contribute to the
355
+ advancement of the field.
356
+ - Language Learning Tools: Support interactive language learning
357
+ experiences, aiding in grammar correction or providing writing practice.
358
+ - Knowledge Exploration: Assist researchers in exploring large
359
+ bodies of text by generating summaries or answering questions about
360
+ specific topics.
361
+
362
+ ### Limitations
363
+
364
+ - Training Data
365
+ - The quality and diversity of the training data significantly
366
+ influence the model's capabilities. Biases or gaps in the training data
367
+ can lead to limitations in the model's responses.
368
+ - The scope of the training dataset determines the subject areas
369
+ the model can handle effectively.
370
+ - Context and Task Complexity
371
+ - Models are better at tasks that can be framed with clear
372
+ prompts and instructions. Open-ended or highly complex tasks might be
373
+ challenging.
374
+ - A model's performance can be influenced by the amount of context
375
+ provided (longer context generally leads to better outputs, up to a
376
+ certain point).
377
+ - Language Ambiguity and Nuance
378
+ - Natural language is inherently complex. Models might struggle
379
+ to grasp subtle nuances, sarcasm, or figurative language.
380
+ - Factual Accuracy
381
+ - Models generate responses based on information they learned
382
+ from their training datasets, but they are not knowledge bases. They
383
+ may generate incorrect or outdated factual statements.
384
+ - Common Sense
385
+ - Models rely on statistical patterns in language. They might
386
+ lack the ability to apply common sense reasoning in certain situations.
387
+
388
+ ### Ethical Considerations and Risks
389
+
390
+ The development of vision-language models (VLMs) raises several ethical
391
+ concerns. In creating an open model, we have carefully considered the following:
392
+
393
+ - Bias and Fairness
394
+ - VLMs trained on large-scale, real-world text and image data can
395
+ reflect socio-cultural biases embedded in the training material. These
396
+ models underwent careful scrutiny, input data pre-processing described
397
+ and posterior evaluations reported in this card.
398
+ - Misinformation and Misuse
399
+ - VLMs can be misused to generate text that is false, misleading,
400
+ or harmful.
401
+ - Guidelines are provided for responsible use with the model, see the
402
+ [Responsible Generative AI Toolkit][rai-toolkit].
403
+ - Transparency and Accountability:
404
+ - This model card summarizes details on the models' architecture,
405
+ capabilities, limitations, and evaluation processes.
406
+ - A responsibly developed open model offers the opportunity to
407
+ share innovation by making VLM technology accessible to developers and
408
+ researchers across the AI ecosystem.
409
+
410
+ Risks identified and mitigations:
411
+
412
+ - **Perpetuation of biases**: It's encouraged to perform continuous
413
+ monitoring (using evaluation metrics, human review) and the exploration of
414
+ de-biasing techniques during model training, fine-tuning, and other use
415
+ cases.
416
+ - **Generation of harmful content**: Mechanisms and guidelines for content
417
+ safety are essential. Developers are encouraged to exercise caution and
418
+ implement appropriate content safety safeguards based on their specific
419
+ product policies and application use cases.
420
+ - **Misuse for malicious purposes**: Technical limitations and developer
421
+ and end-user education can help mitigate against malicious applications of
422
+ VLMs. Educational resources and reporting mechanisms for users to flag
423
+ misuse are provided. Prohibited uses of Gemma models are outlined in the
424
+ [Gemma Prohibited Use Policy][prohibited-use].
425
+ - **Privacy violations**: Models were trained on data filtered for removal
426
+ of certain personal information and other sensitive data. Developers are
427
+ encouraged to adhere to privacy regulations with privacy-preserving
428
+ techniques.
429
+
430
+ ### Benefits
431
+
432
+ At the time of release, this family of models provides high-performance open
433
+ vision-language model implementations designed from the ground up for
434
+ responsible AI development compared to similarly sized models.
435
+
436
+ Using the benchmark evaluation metrics described in this document, these models
437
+ have shown to provide superior performance to other, comparably-sized open model
438
+ alternatives.
439
+
440
+ [g3-tech-report]: https://goo.gle/Gemma3Report
441
+ [rai-toolkit]: https://ai.google.dev/responsible
442
+ [kaggle-gemma]: https://www.kaggle.com/models/google/gemma-3
443
+ [vertex-mg-gemma3]: https://console.cloud.google.com/vertex-ai/publishers/google/model-garden/gemma3
444
+ [terms]: https://ai.google.dev/gemma/terms
445
+ [safety-policies]: https://ai.google/static/documents/ai-responsibility-update-published-february-2025.pdf
446
+ [prohibited-use]: https://ai.google.dev/gemma/prohibited_use_policy
447
+ [tpu]: https://cloud.google.com/tpu/docs/intro-to-tpu
448
+ [sustainability]: https://sustainability.google/operating-sustainably/
449
+ [jax]: https://github.com/jax-ml/jax
450
+ [ml-pathways]: https://blog.google/technology/ai/introducing-pathways-next-generation-ai-architecture/
451
+ [sustainability]: https://sustainability.google/operating-sustainably/
452
+ [gemini-2-paper]: https://arxiv.org/abs/2312.11805
gemma-3-12b-it-qat-q4_0-unquantized/added_tokens.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ {
2
+ "<image_soft_token>": 262144
3
+ }
gemma-3-12b-it-qat-q4_0-unquantized/chat_template.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ {
2
+ "chat_template": "{{ bos_token }}\n{%- if messages[0]['role'] == 'system' -%}\n {%- if messages[0]['content'] is string -%}\n {%- set first_user_prefix = messages[0]['content'] + '\n\n' -%}\n {%- else -%}\n {%- set first_user_prefix = messages[0]['content'][0]['text'] + '\n\n' -%}\n {%- endif -%}\n {%- set loop_messages = messages[1:] -%}\n{%- else -%}\n {%- set first_user_prefix = \"\" -%}\n {%- set loop_messages = messages -%}\n{%- endif -%}\n{%- for message in loop_messages -%}\n {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%}\n {{ raise_exception(\"Conversation roles must alternate user/assistant/user/assistant/...\") }}\n {%- endif -%}\n {%- if (message['role'] == 'assistant') -%}\n {%- set role = \"model\" -%}\n {%- else -%}\n {%- set role = message['role'] -%}\n {%- endif -%}\n {{ '<start_of_turn>' + role + '\n' + (first_user_prefix if loop.first else \"\") }}\n {%- if message['content'] is string -%}\n {{ message['content'] | trim }}\n {%- elif message['content'] is iterable -%}\n {%- for item in message['content'] -%}\n {%- if item['type'] == 'image' -%}\n {{ '<start_of_image>' }}\n {%- elif item['type'] == 'text' -%}\n {{ item['text'] | trim }}\n {%- endif -%}\n {%- endfor -%}\n {%- else -%}\n {{ raise_exception(\"Invalid content type\") }}\n {%- endif -%}\n {{ '<end_of_turn>\n' }}\n{%- endfor -%}\n{%- if add_generation_prompt -%}\n {{'<start_of_turn>model\n'}}\n{%- endif -%}\n"
3
+ }
gemma-3-12b-it-qat-q4_0-unquantized/config.json ADDED
@@ -0,0 +1,62 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Gemma3ForConditionalGeneration"
4
+ ],
5
+ "boi_token_index": 255999,
6
+ "eoi_token_index": 256000,
7
+ "eos_token_id": [
8
+ 1,
9
+ 106
10
+ ],
11
+ "image_token_index": 262144,
12
+ "initializer_range": 0.02,
13
+ "mm_tokens_per_image": 256,
14
+ "model_type": "gemma3",
15
+ "text_config": {
16
+ "attention_bias": false,
17
+ "attention_dropout": 0.0,
18
+ "attn_logit_softcapping": null,
19
+ "cache_implementation": "hybrid",
20
+ "final_logit_softcapping": null,
21
+ "head_dim": 256,
22
+ "hidden_activation": "gelu_pytorch_tanh",
23
+ "hidden_size": 3840,
24
+ "initializer_range": 0.02,
25
+ "intermediate_size": 15360,
26
+ "max_position_embeddings": 131072,
27
+ "model_type": "gemma3_text",
28
+ "num_attention_heads": 16,
29
+ "num_hidden_layers": 48,
30
+ "num_key_value_heads": 8,
31
+ "query_pre_attn_scalar": 256,
32
+ "rms_norm_eps": 1e-06,
33
+ "rope_local_base_freq": 10000,
34
+ "rope_scaling": {
35
+ "factor": 8.0,
36
+ "rope_type": "linear"
37
+ },
38
+ "rope_theta": 1000000,
39
+ "sliding_window": 1024,
40
+ "sliding_window_pattern": 6,
41
+ "torch_dtype": "bfloat16",
42
+ "use_cache": true,
43
+ "vocab_size": 262208
44
+ },
45
+ "torch_dtype": "bfloat16",
46
+ "transformers_version": "4.52.0.dev0",
47
+ "vision_config": {
48
+ "attention_dropout": 0.0,
49
+ "hidden_act": "gelu_pytorch_tanh",
50
+ "hidden_size": 1152,
51
+ "image_size": 896,
52
+ "intermediate_size": 4304,
53
+ "layer_norm_eps": 1e-06,
54
+ "model_type": "siglip_vision_model",
55
+ "num_attention_heads": 16,
56
+ "num_channels": 3,
57
+ "num_hidden_layers": 27,
58
+ "patch_size": 14,
59
+ "torch_dtype": "bfloat16",
60
+ "vision_use_head": false
61
+ }
62
+ }
gemma-3-12b-it-qat-q4_0-unquantized/config_light.json ADDED
@@ -0,0 +1,38 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Gemma3ForCausalLM"
4
+ ],
5
+ "eos_token_id": [
6
+ 1,
7
+ 106
8
+ ],
9
+ "attention_bias": false,
10
+ "attention_dropout": 0.0,
11
+ "attn_logit_softcapping": null,
12
+ "cache_implementation": "hybrid",
13
+ "final_logit_softcapping": null,
14
+ "head_dim": 256,
15
+ "hidden_activation": "gelu_pytorch_tanh",
16
+ "hidden_size": 3840,
17
+ "initializer_range": 0.02,
18
+ "intermediate_size": 15360,
19
+ "max_position_embeddings": 131072,
20
+ "model_type": "gemma3_text",
21
+ "num_attention_heads": 16,
22
+ "num_hidden_layers": 48,
23
+ "num_key_value_heads": 8,
24
+ "query_pre_attn_scalar": 256,
25
+ "rms_norm_eps": 1e-06,
26
+ "rope_local_base_freq": 10000,
27
+ "rope_scaling": {
28
+ "factor": 8.0,
29
+ "rope_type": "linear"
30
+ },
31
+ "rope_theta": 1000000,
32
+ "sliding_window": 1024,
33
+ "sliding_window_pattern": 6,
34
+ "torch_dtype": "bfloat16",
35
+ "transformers_version": "4.52.0.dev0",
36
+ "use_cache": true,
37
+ "vocab_size": 262208
38
+ }
gemma-3-12b-it-qat-q4_0-unquantized/gemma-3-12b-it-qat-q4_0-unquantized.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:073717ab1964fb2d5a7a410cecb5b76de72525bf578cc4f99432cd231d06997a
3
+ size 24374806648
gemma-3-12b-it-qat-q4_0-unquantized/gemma-3-12b-it-qat-q4_0-unquantized_quanto_bf16_int8.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1fad55b5df6c660c7982985c8ced76369114b1d31bbec93fbecc66a8bf30f36a
3
+ size 13210647730
gemma-3-12b-it-qat-q4_0-unquantized/generation_config.json ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "cache_implementation": "hybrid",
3
+ "do_sample": true,
4
+ "eos_token_id": [
5
+ 1,
6
+ 106
7
+ ],
8
+ "top_k": 64,
9
+ "top_p": 0.95,
10
+ "transformers_version": "4.52.0.dev0"
11
+ }
gemma-3-12b-it-qat-q4_0-unquantized/model.safetensors.index.json ADDED
The diff for this file is too large to render. See raw diff
 
gemma-3-12b-it-qat-q4_0-unquantized/preprocessor_config.json ADDED
@@ -0,0 +1,29 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "do_convert_rgb": null,
3
+ "do_normalize": true,
4
+ "do_pan_and_scan": null,
5
+ "do_rescale": true,
6
+ "do_resize": true,
7
+ "image_mean": [
8
+ 0.5,
9
+ 0.5,
10
+ 0.5
11
+ ],
12
+ "image_processor_type": "Gemma3ImageProcessor",
13
+ "image_seq_length": 256,
14
+ "image_std": [
15
+ 0.5,
16
+ 0.5,
17
+ 0.5
18
+ ],
19
+ "pan_and_scan_max_num_crops": null,
20
+ "pan_and_scan_min_crop_size": null,
21
+ "pan_and_scan_min_ratio_to_activate": null,
22
+ "processor_class": "Gemma3Processor",
23
+ "resample": 2,
24
+ "rescale_factor": 0.00392156862745098,
25
+ "size": {
26
+ "height": 896,
27
+ "width": 896
28
+ }
29
+ }
gemma-3-12b-it-qat-q4_0-unquantized/processor_config.json ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ {
2
+ "image_seq_length": 256,
3
+ "processor_class": "Gemma3Processor"
4
+ }
gemma-3-12b-it-qat-q4_0-unquantized/readme.md ADDED
File without changes
gemma-3-12b-it-qat-q4_0-unquantized/special_tokens_map.json ADDED
@@ -0,0 +1,33 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "boi_token": "<start_of_image>",
3
+ "bos_token": {
4
+ "content": "<bos>",
5
+ "lstrip": false,
6
+ "normalized": false,
7
+ "rstrip": false,
8
+ "single_word": false
9
+ },
10
+ "eoi_token": "<end_of_image>",
11
+ "eos_token": {
12
+ "content": "<eos>",
13
+ "lstrip": false,
14
+ "normalized": false,
15
+ "rstrip": false,
16
+ "single_word": false
17
+ },
18
+ "image_token": "<image_soft_token>",
19
+ "pad_token": {
20
+ "content": "<pad>",
21
+ "lstrip": false,
22
+ "normalized": false,
23
+ "rstrip": false,
24
+ "single_word": false
25
+ },
26
+ "unk_token": {
27
+ "content": "<unk>",
28
+ "lstrip": false,
29
+ "normalized": false,
30
+ "rstrip": false,
31
+ "single_word": false
32
+ }
33
+ }
gemma-3-12b-it-qat-q4_0-unquantized/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7d4046bf0505a327dd5a0abbb427ecd4fc82f99c2ceaa170bc61ecde12809b0c
3
+ size 33384570
gemma-3-12b-it-qat-q4_0-unquantized/tokenizer.model ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1299c11d7cf632ef3b4e11937501358ada021bbdf7c47638d13c0ee982f2e79c
3
+ size 4689074
gemma-3-12b-it-qat-q4_0-unquantized/tokenizer_config.json ADDED
The diff for this file is too large to render. See raw diff
 
gemma4-12b-ltx-v1/chat_template.jinja ADDED
@@ -0,0 +1,390 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {#
2
+ Template: Google Gemma 4 Canonical Chat Template
3
+ Author: Google Gemma Engineering Team
4
+ Published: 2026-07-09
5
+ Context: Fixed tool-calling loops, turn closures, and thinking content-ordering.
6
+ #}
7
+ {%- macro format_parameters(properties, required, filter_keys=false) -%}
8
+ {%- set standard_keys = ['description', 'type', 'properties', 'required', 'nullable'] -%}
9
+ {%- set ns = namespace(found_first=false) -%}
10
+ {%- for key, value in properties | dictsort -%}
11
+ {%- set add_comma = false -%}
12
+ {%- if not filter_keys or key not in standard_keys -%}
13
+ {%- if ns.found_first %},{% endif -%}
14
+ {%- set ns.found_first = true -%}
15
+ {{ key }}:{
16
+ {%- if value['description'] -%}
17
+ description:<|"|>{{ value['description'] }}<|"|>
18
+ {%- set add_comma = true -%}
19
+ {%- endif -%}
20
+ {%- if value['type'] | upper == 'STRING' -%}
21
+ {%- if value['enum'] -%}
22
+ {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%}
23
+ enum:{{ format_argument(value['enum']) }}
24
+ {%- endif -%}
25
+ {%- elif value['type'] | upper == 'ARRAY' -%}
26
+ {%- if value['items'] is mapping and value['items'] -%}
27
+ {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%}
28
+ items:{
29
+ {%- set ns_items = namespace(found_first=false) -%}
30
+ {%- for item_key, item_value in value['items'] | dictsort -%}
31
+ {%- if item_value is not none -%}
32
+ {%- if ns_items.found_first %},{% endif -%}
33
+ {%- set ns_items.found_first = true -%}
34
+ {%- if item_key == 'properties' -%}
35
+ properties:{
36
+ {%- if item_value is mapping -%}
37
+ {{- format_parameters(item_value, value['items']['required'] | default([])) -}}
38
+ {%- endif -%}
39
+ }
40
+ {%- elif item_key == 'required' -%}
41
+ required:[
42
+ {%- for req_item in item_value -%}
43
+ <|"|>{{- req_item -}}<|"|>
44
+ {%- if not loop.last %},{% endif -%}
45
+ {%- endfor -%}
46
+ ]
47
+ {%- elif item_key == 'type' -%}
48
+ {%- if item_value is string -%}
49
+ type:{{ format_argument(item_value | upper) }}
50
+ {%- else -%}
51
+ type:{{ format_argument(item_value | map('upper') | list) }}
52
+ {%- endif -%}
53
+ {%- else -%}
54
+ {{ item_key }}:{{ format_argument(item_value) }}
55
+ {%- endif -%}
56
+ {%- endif -%}
57
+ {%- endfor -%}
58
+ }
59
+ {%- endif -%}
60
+ {%- endif -%}
61
+ {%- if value['nullable'] %}
62
+ {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%}
63
+ nullable:true
64
+ {%- endif -%}
65
+ {%- if value['type'] | upper == 'OBJECT' -%}
66
+ {%- if value['properties'] is defined and value['properties'] is mapping -%}
67
+ {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%}
68
+ properties:{
69
+ {{- format_parameters(value['properties'], value['required'] | default([])) -}}
70
+ }
71
+ {%- elif value is mapping -%}
72
+ {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%}
73
+ properties:{
74
+ {{- format_parameters(value, value['required'] | default([]), filter_keys=true) -}}
75
+ }
76
+ {%- endif -%}
77
+ {%- if value['required'] -%}
78
+ {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%}
79
+ required:[
80
+ {%- for item in value['required'] | default([]) -%}
81
+ <|"|>{{- item -}}<|"|>
82
+ {%- if not loop.last %},{% endif -%}
83
+ {%- endfor -%}
84
+ ]
85
+ {%- endif -%}
86
+ {%- endif -%}
87
+ {%- if add_comma %},{%- else -%} {%- set add_comma = true -%} {% endif -%}
88
+ type:<|"|>{{ value['type'] | upper }}<|"|>}
89
+ {%- endif -%}
90
+ {%- endfor -%}
91
+ {%- endmacro -%}
92
+ {%- macro format_function_declaration(tool_data) -%}
93
+ declaration:{{- tool_data['function']['name'] -}}{description:<|"|>{{- tool_data['function']['description'] -}}<|"|>
94
+ {%- set params = tool_data['function']['parameters'] -%}
95
+ {%- if params -%}
96
+ ,parameters:{
97
+ {%- if params['properties'] -%}
98
+ properties:{ {{- format_parameters(params['properties'], params['required']) -}} },
99
+ {%- endif -%}
100
+ {%- if params['required'] -%}
101
+ required:[
102
+ {%- for item in params['required'] -%}
103
+ <|"|>{{- item -}}<|"|>
104
+ {{- ',' if not loop.last -}}
105
+ {%- endfor -%}
106
+ ],
107
+ {%- endif -%}
108
+ {%- if params['type'] -%}
109
+ type:<|"|>{{- params['type'] | upper -}}<|"|>}
110
+ {%- endif -%}
111
+ {%- endif -%}
112
+ {%- if 'response' in tool_data['function'] -%}
113
+ {%- set response_declaration = tool_data['function']['response'] -%}
114
+ ,response:{
115
+ {%- if response_declaration['description'] -%}
116
+ description:<|"|>{{- response_declaration['description'] -}}<|"|>,
117
+ {%- endif -%}
118
+ {%- if response_declaration['type'] | upper == 'OBJECT' -%}
119
+ type:<|"|>{{- response_declaration['type'] | upper -}}<|"|>}
120
+ {%- endif -%}
121
+ {%- endif -%}
122
+ }
123
+ {%- endmacro -%}
124
+ {%- macro format_argument(argument, escape_keys=True) -%}
125
+ {%- if argument is none -%}
126
+ {{- 'null' -}}
127
+ {%- elif argument is string -%}
128
+ {{- '<|"|>' + argument + '<|"|>' -}}
129
+ {%- elif argument is boolean -%}
130
+ {{- 'true' if argument else 'false' -}}
131
+ {%- elif argument is mapping -%}
132
+ {{- '{' -}}
133
+ {%- set ns = namespace(found_first=false) -%}
134
+ {%- for key, value in argument | dictsort -%}
135
+ {%- if ns.found_first %},{% endif -%}
136
+ {%- set ns.found_first = true -%}
137
+ {%- if escape_keys -%}
138
+ {{- '<|"|>' + key + '<|"|>' -}}
139
+ {%- else -%}
140
+ {{- key -}}
141
+ {%- endif -%}
142
+ :{{- format_argument(value, escape_keys=escape_keys) -}}
143
+ {%- endfor -%}
144
+ {{- '}' -}}
145
+ {%- elif argument is sequence -%}
146
+ {{- '[' -}}
147
+ {%- for item in argument -%}
148
+ {{- format_argument(item, escape_keys=escape_keys) -}}
149
+ {%- if not loop.last %},{% endif -%}
150
+ {%- endfor -%}
151
+ {{- ']' -}}
152
+ {%- else -%}
153
+ {{- argument -}}
154
+ {%- endif -%}
155
+ {%- endmacro -%}
156
+ {%- macro strip_thinking(text) -%}
157
+ {%- set ns = namespace(result='') -%}
158
+ {%- for part in text.split('<channel|>') -%}
159
+ {%- if '<|channel>' in part -%}
160
+ {%- set ns.result = ns.result + part.split('<|channel>')[0] -%}
161
+ {%- else -%}
162
+ {%- set ns.result = ns.result + part -%}
163
+ {%- endif -%}
164
+ {%- endfor -%}
165
+ {{- ns.result | trim -}}
166
+ {%- endmacro -%}
167
+
168
+ {%- macro format_tool_response_block(tool_name, response) -%}
169
+ {{- '<|tool_response>' -}}
170
+ {%- if response is mapping -%}
171
+ {{- 'response:' + tool_name + '{' -}}
172
+ {%- for key, value in response | dictsort -%}
173
+ {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
174
+ {%- if not loop.last %},{% endif -%}
175
+ {%- endfor -%}
176
+ {{- '}' -}}
177
+ {%- else -%}
178
+ {{- 'response:' + tool_name + '{value:' + format_argument(response, escape_keys=False) + '}' -}}
179
+ {%- endif -%}
180
+ {{- '<tool_response|>' -}}
181
+ {%- endmacro -%}
182
+
183
+ {#- ===== SETUP ===== -#}
184
+ {%- set ns = namespace(prev_message_type=None, prev_non_tool_role=None) -%}
185
+ {%- set loop_messages = messages -%}
186
+ {%- set enable_thinking = enable_thinking | default(false) -%}
187
+ {%- set preserve_thinking = preserve_thinking | default(false) -%}
188
+ {{- bos_token -}}
189
+ {#- Handle System/Tool Definitions Block -#}
190
+ {%- if enable_thinking or tools or (messages and messages[0]['role'] in ['system', 'developer']) -%}
191
+ {{- '<|turn>system\n' -}}
192
+ {#- Inject Thinking token at the very top of the FIRST system turn -#}
193
+ {%- if enable_thinking -%}
194
+ {{- '<|think|>\n' -}}
195
+ {%- set ns.prev_message_type = 'think' -%}
196
+ {%- endif -%}
197
+ {%- if messages and messages[0]['role'] in ['system', 'developer'] -%}
198
+ {%- if messages[0]['content'] is string -%}
199
+ {{- messages[0]['content'] | trim -}}
200
+ {%- elif messages[0]['content'] is sequence -%}
201
+ {%- for item in messages[0]['content'] -%}
202
+ {{- item['text'] | trim + ' '-}}
203
+ {%- endfor -%}
204
+ {%- endif -%}
205
+ {%- set loop_messages = messages[1:] -%}
206
+ {%- endif -%}
207
+ {%- if tools -%}
208
+ {%- for tool in tools %}
209
+ {{- '<|tool>' -}}
210
+ {{- format_function_declaration(tool) | trim -}}
211
+ {{- '<tool|>' -}}
212
+ {%- endfor %}
213
+ {%- set ns.prev_message_type = 'tool' -%}
214
+ {%- endif -%}
215
+ {{- '<turn|>\n' -}}
216
+ {%- endif %}
217
+
218
+ {#- Pre-scan: find last user message index for reasoning guard -#}
219
+ {%- set ns_turn = namespace(last_user_idx=-1) -%}
220
+ {%- for i in range(loop_messages | length) -%}
221
+ {%- if loop_messages[i]['role'] == 'user' -%}
222
+ {%- set ns_turn.last_user_idx = i -%}
223
+ {%- endif -%}
224
+ {%- endfor -%}
225
+
226
+ {#- Loop through messages -#}
227
+ {%- for message in loop_messages -%}
228
+ {%- if message['role'] != 'tool' -%}
229
+ {%- set ns.prev_message_type = None -%}
230
+ {%- set role = 'model' if message['role'] == 'assistant' else message['role'] -%}
231
+ {#- Detect continuation using tracked state — O(1) instead of O(n) backward scan -#}
232
+ {%- set continue_same_model_turn = (role == 'model' and ns.prev_non_tool_role == 'assistant') -%}
233
+ {%- if not continue_same_model_turn -%}
234
+ {{- '<|turn>' + role + '\n' }}
235
+
236
+ {%- endif -%}
237
+
238
+ {#- Render reasoning/reasoning_content as thinking channel -#}
239
+ {%- set thinking_text = message.get('reasoning') or message.get('reasoning_content') -%}
240
+ {%- set thinking_gate = (loop.index0 > ns_turn.last_user_idx) or (preserve_thinking and message.get('tool_calls')) -%}
241
+ {%- if thinking_text and thinking_gate -%}
242
+ {{- '<|channel>thought\n' + thinking_text + '\n<channel|>' -}}
243
+ {%- endif -%}
244
+
245
+ {%- if message.get('tool_calls') -%}
246
+ {%- for tool_call in message.get('tool_calls') -%}
247
+ {%- set function = tool_call['function'] -%}
248
+ {{- '<|tool_call>call:' + function['name'] + '{' -}}
249
+ {%- if function['arguments'] is mapping -%}
250
+ {%- set ns_args = namespace(found_first=false) -%}
251
+ {%- for key, value in function['arguments'] | dictsort -%}
252
+ {%- if ns_args.found_first %},{% endif -%}
253
+ {%- set ns_args.found_first = true -%}
254
+ {{- key -}}:{{- format_argument(value, escape_keys=False) -}}
255
+ {%- endfor -%}
256
+ {%- elif function['arguments'] is none -%}
257
+ {%- else -%}
258
+ {{- raise_exception(
259
+ "chat_template: tool_calls[].function.arguments must be a "
260
+ "JSON object (mapping), not a string. Deserialize arguments "
261
+ "before passing to the template."
262
+ ) -}}
263
+ {%- endif -%}
264
+ {{- '}<tool_call|>' -}}
265
+ {%- endfor -%}
266
+ {%- set ns.prev_message_type = 'tool_call' -%}
267
+ {%- endif -%}
268
+
269
+ {%- set ns_tr_out = namespace(flag=false) -%}
270
+ {%- if message.get('tool_responses') -%}
271
+ {#- Legacy: tool_responses embedded on the assistant message (Google/Gemma native) -#}
272
+ {%- for tool_response in message.get('tool_responses') -%}
273
+ {{- format_tool_response_block(tool_response['name'] | default('unknown', true), tool_response['response']) -}}
274
+ {%- set ns_tr_out.flag = true -%}
275
+ {%- set ns.prev_message_type = 'tool_response' -%}
276
+ {%- endfor -%}
277
+ {%- elif message.get('tool_calls') -%}
278
+ {#- OpenAI Chat Completions: forward-scan consecutive role:tool messages -#}
279
+ {%- set ns_tool_scan = namespace(stopped=false) -%}
280
+ {%- for k in range(loop.index0 + 1, loop_messages | length) -%}
281
+ {%- if ns_tool_scan.stopped -%}
282
+ {%- elif loop_messages[k]['role'] != 'tool' -%}
283
+ {%- set ns_tool_scan.stopped = true -%}
284
+ {%- else -%}
285
+ {%- set follow = loop_messages[k] -%}
286
+ {#- Resolve tool_call_id to function name -#}
287
+ {%- set ns_tname = namespace(name=follow.get('name') or 'unknown') -%}
288
+ {%- for tc in message.get('tool_calls') -%}
289
+ {%- if tc.get('id') == follow.get('tool_call_id') -%}
290
+ {%- set ns_tname.name = tc['function']['name'] -%}
291
+ {%- endif -%}
292
+ {%- endfor -%}
293
+ {#- Handle content as string or content-parts array -#}
294
+ {%- set tool_body = follow.get('content') -%}
295
+ {%- if tool_body is string -%}
296
+ {{- format_tool_response_block(ns_tname.name, tool_body) -}}
297
+ {%- elif tool_body is sequence and tool_body is not string -%}
298
+ {%- set ns_txt = namespace(s='') -%}
299
+ {%- for part in tool_body -%}
300
+ {%- if part.get('type') == 'text' -%}
301
+ {%- set ns_txt.s = ns_txt.s + (part.get('text') | default('')) -%}
302
+ {%- endif -%}
303
+ {%- endfor -%}
304
+ {{- format_tool_response_block(ns_tname.name, ns_txt.s) -}}
305
+ {%- for part in tool_body -%}
306
+ {%- if part.get('type') in ['image', 'image_url'] -%}
307
+ {{- '<|image|>' -}}
308
+ {%- elif part.get('type') in ['audio', 'input_audio'] -%}
309
+ {{- '<|audio|>' -}}
310
+ {%- elif part.get('type') == 'video' -%}
311
+ {{- '<|video|>' -}}
312
+ {%- endif -%}
313
+ {%- endfor -%}
314
+ {%- else -%}
315
+ {{- format_tool_response_block(ns_tname.name, tool_body) -}}
316
+ {%- endif -%}
317
+ {%- set ns_tr_out.flag = true -%}
318
+ {%- set ns.prev_message_type = 'tool_response' -%}
319
+ {%- endif -%}
320
+ {%- endfor -%}
321
+ {%- endif -%}
322
+
323
+ {%- set captured_content -%}
324
+ {%- if message.get('content') is string -%}
325
+ {%- if role == 'model' -%}
326
+ {{- strip_thinking(message['content']) -}}
327
+ {%- else -%}
328
+ {{- message['content'] | trim -}}
329
+ {%- endif -%}
330
+ {%- elif message.get('content') is sequence -%}
331
+ {%- for item in message['content'] -%}
332
+ {%- if item.get('type') == 'text' -%}
333
+ {%- if role == 'model' -%}
334
+ {{- strip_thinking(item['text']) -}}
335
+ {%- else -%}
336
+ {{- item['text'] | trim -}}
337
+ {%- endif -%}
338
+ {%- elif item.get('type') in ['image', 'image_url'] -%}
339
+ {{- '<|image|>' -}}
340
+ {%- elif item.get('type') in ['audio', 'input_audio'] -%}
341
+ {{- '<|audio|>' -}}
342
+ {%- elif item.get('type') == 'video' -%}
343
+ {{- '<|video|>' -}}
344
+ {%- endif -%}
345
+ {%- endfor -%}
346
+ {%- endif -%}
347
+ {%- endset -%}
348
+
349
+ {{- captured_content -}}
350
+ {%- set has_content = captured_content | trim | length > 0 -%}
351
+
352
+ {#- Forward-scan: find next non-tool message role for continuation detection -#}
353
+ {%- set next_nt = namespace(role=None, found=false) -%}
354
+ {%- for j in range(loop.index0 + 1, loop_messages | length) -%}
355
+ {%- if not next_nt.found -%}
356
+ {%- if loop_messages[j]['role'] != 'tool' -%}
357
+ {%- set next_nt.role = loop_messages[j]['role'] -%}
358
+ {%- set next_nt.found = true -%}
359
+ {%- endif -%}
360
+ {%- endif -%}
361
+ {%- endfor -%}
362
+
363
+ {%- set continues_into_next = (
364
+ role == 'model'
365
+ and next_nt.role == 'assistant'
366
+ and (not message.get('tool_calls') or ns_tr_out.flag)
367
+ ) -%}
368
+
369
+ {%- if ns.prev_message_type == 'tool_call' and not ns_tr_out.flag -%}
370
+ {{- '<|tool_response>' -}}
371
+ {%- elif continues_into_next -%}
372
+ {%- elif not (ns_tr_out.flag and not has_content and not next_nt.found) -%}
373
+ {{- '<turn|>\n' -}}
374
+ {%- endif -%}
375
+
376
+ {#- Track previous non-tool role for next iteration (avoids O(n) backward scan) -#}
377
+ {%- set ns.prev_non_tool_role = message['role'] -%}
378
+ {%- endif -%}
379
+ {%- endfor -%}
380
+
381
+ {%- if add_generation_prompt -%}
382
+ {%- if ns.prev_message_type != 'tool_response' and ns.prev_message_type != 'tool_call' -%}
383
+ {{- '<|turn>model\n' -}}
384
+ {%- if not enable_thinking -%}
385
+ {{- '<|channel>thought\n<channel|>' -}}
386
+ {%- endif -%}
387
+ {%- elif ns.prev_message_type == 'tool_response' and enable_thinking -%}
388
+ {{- '<|channel>thought\n' -}}
389
+ {%- endif -%}
390
+ {%- endif -%}
gemma4-12b-ltx-v1/config.json ADDED
@@ -0,0 +1,100 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "attention_bias": false,
3
+ "attention_dropout": 0.0,
4
+ "attention_k_eq_v": true,
5
+ "bos_token_id": 2,
6
+ "dtype": "bfloat16",
7
+ "enable_moe_block": false,
8
+ "eos_token_id": 1,
9
+ "final_logit_softcapping": 30.0,
10
+ "global_head_dim": 512,
11
+ "head_dim": 256,
12
+ "hidden_activation": "gelu_pytorch_tanh",
13
+ "hidden_size": 3840,
14
+ "hidden_size_per_layer_input": 0,
15
+ "initializer_range": 0.02,
16
+ "intermediate_size": 15360,
17
+ "layer_types": [
18
+ "sliding_attention",
19
+ "sliding_attention",
20
+ "sliding_attention",
21
+ "sliding_attention",
22
+ "sliding_attention",
23
+ "full_attention",
24
+ "sliding_attention",
25
+ "sliding_attention",
26
+ "sliding_attention",
27
+ "sliding_attention",
28
+ "sliding_attention",
29
+ "full_attention",
30
+ "sliding_attention",
31
+ "sliding_attention",
32
+ "sliding_attention",
33
+ "sliding_attention",
34
+ "sliding_attention",
35
+ "full_attention",
36
+ "sliding_attention",
37
+ "sliding_attention",
38
+ "sliding_attention",
39
+ "sliding_attention",
40
+ "sliding_attention",
41
+ "full_attention",
42
+ "sliding_attention",
43
+ "sliding_attention",
44
+ "sliding_attention",
45
+ "sliding_attention",
46
+ "sliding_attention",
47
+ "full_attention",
48
+ "sliding_attention",
49
+ "sliding_attention",
50
+ "sliding_attention",
51
+ "sliding_attention",
52
+ "sliding_attention",
53
+ "full_attention",
54
+ "sliding_attention",
55
+ "sliding_attention",
56
+ "sliding_attention",
57
+ "sliding_attention",
58
+ "sliding_attention",
59
+ "full_attention",
60
+ "sliding_attention",
61
+ "sliding_attention",
62
+ "sliding_attention",
63
+ "sliding_attention",
64
+ "sliding_attention",
65
+ "full_attention"
66
+ ],
67
+ "max_position_embeddings": 262144,
68
+ "model_type": "gemma4_unified_text",
69
+ "moe_intermediate_size": null,
70
+ "num_attention_heads": 16,
71
+ "num_experts": null,
72
+ "num_global_key_value_heads": 1,
73
+ "num_hidden_layers": 48,
74
+ "num_key_value_heads": 8,
75
+ "num_kv_shared_layers": 0,
76
+ "pad_token_id": 0,
77
+ "rms_norm_eps": 1e-06,
78
+ "rope_parameters": {
79
+ "full_attention": {
80
+ "partial_rotary_factor": 0.25,
81
+ "rope_theta": 1000000.0,
82
+ "rope_type": "proportional"
83
+ },
84
+ "sliding_attention": {
85
+ "rope_theta": 10000.0,
86
+ "rope_type": "default"
87
+ }
88
+ },
89
+ "sliding_window": 1024,
90
+ "tie_word_embeddings": true,
91
+ "top_k_experts": null,
92
+ "use_bidirectional_attention": "vision",
93
+ "use_cache": true,
94
+ "use_double_wide_mlp": false,
95
+ "vocab_size": 262144,
96
+ "vocab_size_per_layer_input": 262144,
97
+ "architectures": [
98
+ "Gemma4UnifiedForCausalLM"
99
+ ]
100
+ }
gemma4-12b-ltx-v1/gemma4-12b-ltx-v1_bf16.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5bee560374b6556c499bf162c45c773b0dc87dba4fc84999627f2f3f92a3d872
3
+ size 23814782152
gemma4-12b-ltx-v1/gemma4-12b-ltx-v1_int8_convrot.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6a23b673266b65a318e26cad27fabd5c67609f4ecf51aa13996feb07d0060903
3
+ size 12923893224
gemma4-12b-ltx-v1/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cc8d3a0ce36466ccc1278bf987df5f71db1719b9ca6b4118264f45cb627bfe0f
3
+ size 32169626
gemma4-12b-ltx-v1/tokenizer_config.json ADDED
@@ -0,0 +1,142 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "audio_token": "<|audio|>",
3
+ "backend": "tokenizers",
4
+ "boa_token": "<|audio>",
5
+ "boi_token": "<|image>",
6
+ "bos_token": "<bos>",
7
+ "eoa_token": "<audio|>",
8
+ "eoc_token": "<channel|>",
9
+ "eoi_token": "<image|>",
10
+ "eos_token": "<eos>",
11
+ "eot_token": "<turn|>",
12
+ "escape_token": "<|\"|>",
13
+ "etc_token": "<tool_call|>",
14
+ "etd_token": "<tool|>",
15
+ "etr_token": "<tool_response|>",
16
+ "extra_special_tokens": [
17
+ "<|video|>"
18
+ ],
19
+ "image_token": "<|image|>",
20
+ "is_local": false,
21
+ "local_files_only": false,
22
+ "mask_token": "<mask>",
23
+ "model_max_length": 1000000000000000019884624838656,
24
+ "model_specific_special_tokens": {
25
+ "audio_token": "<|audio|>",
26
+ "boa_token": "<|audio>",
27
+ "boi_token": "<|image>",
28
+ "eoa_token": "<audio|>",
29
+ "eoc_token": "<channel|>",
30
+ "eoi_token": "<image|>",
31
+ "eot_token": "<turn|>",
32
+ "escape_token": "<|\"|>",
33
+ "etc_token": "<tool_call|>",
34
+ "etd_token": "<tool|>",
35
+ "etr_token": "<tool_response|>",
36
+ "image_token": "<|image|>",
37
+ "soc_token": "<|channel>",
38
+ "sot_token": "<|turn>",
39
+ "stc_token": "<|tool_call>",
40
+ "std_token": "<|tool>",
41
+ "str_token": "<|tool_response>",
42
+ "think_token": "<|think|>"
43
+ },
44
+ "pad_token": "<pad>",
45
+ "padding_side": "left",
46
+ "processor_class": "Gemma4UnifiedProcessor",
47
+ "response_schema": {
48
+ "properties": {
49
+ "content": {
50
+ "type": "string"
51
+ },
52
+ "role": {
53
+ "const": "assistant"
54
+ },
55
+ "thinking": {
56
+ "type": "string"
57
+ },
58
+ "tool_calls": {
59
+ "items": {
60
+ "properties": {
61
+ "function": {
62
+ "properties": {
63
+ "arguments": {
64
+ "additionalProperties": {},
65
+ "type": "object",
66
+ "x-parser": "gemma4-tool-call"
67
+ },
68
+ "name": {
69
+ "type": "string"
70
+ }
71
+ },
72
+ "type": "object",
73
+ "x-regex": "call\\:(?P<name>\\w+)(?P<arguments>\\{.*\\})"
74
+ },
75
+ "type": {
76
+ "const": "function"
77
+ }
78
+ },
79
+ "type": "object"
80
+ },
81
+ "type": "array",
82
+ "x-regex-iterator": "<\\|tool_call>(.*?)<tool_call\\|>"
83
+ }
84
+ },
85
+ "type": "object",
86
+ "x-regex": "(\\<\\|channel\\>thought\\n(?P<thinking>.*?)\\<channel\\|\\>)?(?P<tool_calls>\\<\\|tool_call\\>.*\\<tool_call\\|\\>)?(?P<content>(?:(?!\\<turn\\|\\>)(?!\\<\\|tool_response\\>).)+)?(?:\\<turn\\|\\>|\\<\\|tool_response\\>)?"
87
+ },
88
+ "response_template": {
89
+ "defaults": {
90
+ "role": "assistant"
91
+ },
92
+ "fields": {
93
+ "content": {
94
+ "close": [
95
+ "<turn|>",
96
+ "<|tool_response>",
97
+ "<eos>"
98
+ ],
99
+ "content": "text"
100
+ },
101
+ "thinking": {
102
+ "close": "<channel|>",
103
+ "content": "text",
104
+ "open": "<|channel>thought\n"
105
+ },
106
+ "tool_calls": {
107
+ "close": "<tool_call|>",
108
+ "content": "json",
109
+ "content_args": {
110
+ "string_delims": [
111
+ [
112
+ "<|\"|>",
113
+ "<|\"|>"
114
+ ]
115
+ ],
116
+ "unquoted_keys": true
117
+ },
118
+ "open_pattern": "<\\|tool_call>call:(?P<name>\\w+)",
119
+ "repeats": true,
120
+ "transform": {
121
+ "function": {
122
+ "arguments": "{content}",
123
+ "name": "{name}"
124
+ },
125
+ "type": "function"
126
+ }
127
+ }
128
+ },
129
+ "start_anchor": [
130
+ "<|turn>model\n",
131
+ "<tool_response|>"
132
+ ]
133
+ },
134
+ "soc_token": "<|channel>",
135
+ "sot_token": "<|turn>",
136
+ "stc_token": "<|tool_call>",
137
+ "std_token": "<|tool>",
138
+ "str_token": "<|tool_response>",
139
+ "think_token": "<|think|>",
140
+ "tokenizer_class": "GemmaTokenizer",
141
+ "unk_token": "<unk>"
142
+ }
hubert-large-ll60k/config.json ADDED
@@ -0,0 +1,77 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_name_or_path": "facebook/hubert-large-ll60k",
3
+ "activation_dropout": 0.0,
4
+ "apply_spec_augment": true,
5
+ "architectures": [
6
+ "HubertModel"
7
+ ],
8
+ "attention_dropout": 0.1,
9
+ "bos_token_id": 1,
10
+ "conv_bias": true,
11
+ "conv_dim": [
12
+ 512,
13
+ 512,
14
+ 512,
15
+ 512,
16
+ 512,
17
+ 512,
18
+ 512
19
+ ],
20
+ "conv_kernel": [
21
+ 10,
22
+ 3,
23
+ 3,
24
+ 3,
25
+ 3,
26
+ 2,
27
+ 2
28
+ ],
29
+ "conv_stride": [
30
+ 5,
31
+ 2,
32
+ 2,
33
+ 2,
34
+ 2,
35
+ 2,
36
+ 2
37
+ ],
38
+ "ctc_loss_reduction": "sum",
39
+ "ctc_zero_infinity": false,
40
+ "do_stable_layer_norm": true,
41
+ "eos_token_id": 2,
42
+ "feat_extract_activation": "gelu",
43
+ "feat_extract_dropout": 0.0,
44
+ "feat_extract_norm": "layer",
45
+ "feat_proj_dropout": 0.1,
46
+ "final_dropout": 0.0,
47
+ "gradient_checkpointing": false,
48
+ "hidden_act": "gelu",
49
+ "hidden_dropout": 0.1,
50
+ "hidden_size": 1024,
51
+ "initializer_range": 0.02,
52
+ "intermediate_size": 4096,
53
+ "layer_norm_eps": 1e-05,
54
+ "layerdrop": 0.1,
55
+ "mask_channel_length": 10,
56
+ "mask_channel_min_space": 1,
57
+ "mask_channel_other": 0.0,
58
+ "mask_channel_prob": 0.0,
59
+ "mask_channel_selection": "static",
60
+ "mask_feature_length": 10,
61
+ "mask_feature_prob": 0.0,
62
+ "mask_time_length": 10,
63
+ "mask_time_min_space": 1,
64
+ "mask_time_other": 0.0,
65
+ "mask_time_prob": 0.075,
66
+ "mask_time_selection": "static",
67
+ "model_type": "hubert",
68
+ "num_attention_heads": 16,
69
+ "num_conv_pos_embedding_groups": 16,
70
+ "num_conv_pos_embeddings": 128,
71
+ "num_feat_extract_layers": 7,
72
+ "num_hidden_layers": 24,
73
+ "pad_token_id": 0,
74
+ "transformers_version": "4.10.0.dev0",
75
+ "vocab_size": 32,
76
+ "tokenizer_class": "Wav2Vec2CTCTokenizer"
77
+ }
hubert-large-ll60k/preprocessor_config.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "do_normalize": true,
3
+ "feature_extractor_type": "Wav2Vec2FeatureExtractor",
4
+ "feature_size": 1,
5
+ "padding_side": "right",
6
+ "padding_value": 0,
7
+ "return_attention_mask": true,
8
+ "sampling_rate": 16000
9
+ }
hubert-large-ll60k/pytorch_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:45a299050945479a68cffe2ab7a63fd08931718ba18f05a42dbb86a5164178e0
3
+ size 1261920069
id-lora-celebvhq-ltx2.3.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:12e6be9c52c83047cb71470c1c63b64ebc8b13dc0dfa34400f6576ba2cd88ecf
3
+ size 1157884304
id-lora-celebvhq-ltx2.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2369a6adaba1fd2c37da07d8d56a762c3d95bf14c8f19f57b3150ccfb7804d78
3
+ size 1157884288
kokoro/config.json ADDED
@@ -0,0 +1,150 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "istftnet": {
3
+ "upsample_kernel_sizes": [20, 12],
4
+ "upsample_rates": [10, 6],
5
+ "gen_istft_hop_size": 5,
6
+ "gen_istft_n_fft": 20,
7
+ "resblock_dilation_sizes": [
8
+ [1, 3, 5],
9
+ [1, 3, 5],
10
+ [1, 3, 5]
11
+ ],
12
+ "resblock_kernel_sizes": [3, 7, 11],
13
+ "upsample_initial_channel": 512
14
+ },
15
+ "dim_in": 64,
16
+ "dropout": 0.2,
17
+ "hidden_dim": 512,
18
+ "max_conv_dim": 512,
19
+ "max_dur": 50,
20
+ "multispeaker": true,
21
+ "n_layer": 3,
22
+ "n_mels": 80,
23
+ "n_token": 178,
24
+ "style_dim": 128,
25
+ "text_encoder_kernel_size": 5,
26
+ "plbert": {
27
+ "hidden_size": 768,
28
+ "num_attention_heads": 12,
29
+ "intermediate_size": 2048,
30
+ "max_position_embeddings": 512,
31
+ "num_hidden_layers": 12,
32
+ "dropout": 0.1
33
+ },
34
+ "vocab": {
35
+ ";": 1,
36
+ ":": 2,
37
+ ",": 3,
38
+ ".": 4,
39
+ "!": 5,
40
+ "?": 6,
41
+ "—": 9,
42
+ "…": 10,
43
+ "\"": 11,
44
+ "(": 12,
45
+ ")": 13,
46
+ "“": 14,
47
+ "”": 15,
48
+ " ": 16,
49
+ "\u0303": 17,
50
+ "ʣ": 18,
51
+ "ʥ": 19,
52
+ "ʦ": 20,
53
+ "ʨ": 21,
54
+ "ᵝ": 22,
55
+ "\uAB67": 23,
56
+ "A": 24,
57
+ "I": 25,
58
+ "O": 31,
59
+ "Q": 33,
60
+ "S": 35,
61
+ "T": 36,
62
+ "W": 39,
63
+ "Y": 41,
64
+ "ᵊ": 42,
65
+ "a": 43,
66
+ "b": 44,
67
+ "c": 45,
68
+ "d": 46,
69
+ "e": 47,
70
+ "f": 48,
71
+ "h": 50,
72
+ "i": 51,
73
+ "j": 52,
74
+ "k": 53,
75
+ "l": 54,
76
+ "m": 55,
77
+ "n": 56,
78
+ "o": 57,
79
+ "p": 58,
80
+ "q": 59,
81
+ "r": 60,
82
+ "s": 61,
83
+ "t": 62,
84
+ "u": 63,
85
+ "v": 64,
86
+ "w": 65,
87
+ "x": 66,
88
+ "y": 67,
89
+ "z": 68,
90
+ "ɑ": 69,
91
+ "ɐ": 70,
92
+ "ɒ": 71,
93
+ "æ": 72,
94
+ "β": 75,
95
+ "ɔ": 76,
96
+ "ɕ": 77,
97
+ "ç": 78,
98
+ "ɖ": 80,
99
+ "ð": 81,
100
+ "ʤ": 82,
101
+ "ə": 83,
102
+ "ɚ": 85,
103
+ "ɛ": 86,
104
+ "ɜ": 87,
105
+ "ɟ": 90,
106
+ "ɡ": 92,
107
+ "ɥ": 99,
108
+ "ɨ": 101,
109
+ "ɪ": 102,
110
+ "ʝ": 103,
111
+ "ɯ": 110,
112
+ "ɰ": 111,
113
+ "ŋ": 112,
114
+ "ɳ": 113,
115
+ "ɲ": 114,
116
+ "ɴ": 115,
117
+ "ø": 116,
118
+ "ɸ": 118,
119
+ "θ": 119,
120
+ "œ": 120,
121
+ "ɹ": 123,
122
+ "ɾ": 125,
123
+ "ɻ": 126,
124
+ "ʁ": 128,
125
+ "ɽ": 129,
126
+ "ʂ": 130,
127
+ "ʃ": 131,
128
+ "ʈ": 132,
129
+ "ʧ": 133,
130
+ "ʊ": 135,
131
+ "ʋ": 136,
132
+ "ʌ": 138,
133
+ "ɣ": 139,
134
+ "ɤ": 140,
135
+ "χ": 142,
136
+ "ʎ": 143,
137
+ "ʒ": 147,
138
+ "ʔ": 148,
139
+ "ˈ": 156,
140
+ "ˌ": 157,
141
+ "ː": 158,
142
+ "ʰ": 162,
143
+ "ʲ": 164,
144
+ "↓": 169,
145
+ "→": 171,
146
+ "↗": 172,
147
+ "↘": 173,
148
+ "ᵻ": 177
149
+ }
150
+ }
kokoro/kokoro-v1_0.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:496dba118d1a58f5f3db2efc88dbdc216e0483fc89fe6e47ee1f2c53f18ad1e4
3
+ size 327212226
kokoro/voices/af_heart.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0ab5709b8ffab19bfd849cd11d98f75b60af7733253ad0d67b12382a102cb4ff
3
+ size 523425
loras/LTX-2.3-Licon-MSR-V1.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4008d341c58791659a9b3840ac8d76f08d8941f82670dc2d3dad0f71876bbae5
3
+ size 654443424
loras/LTX-2.3-Licon-MSR-V2.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6f61d3b5c61b160c409b45ebaa72fd7ab9bb38bf3bf7f09edaddc87762d5fa98
3
+ size 654443392
loras/Ltx2.3-Licon-VBVR-I2V-96000-R32.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ec5ff61d3e2959babf01112a8d7f00776273415573aa29a82f2af00070fd1408
3
+ size 554006432
loras/omninft-ltx2-19b-rl-lora-r32.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f7927a790774bb21c0a90490c775b37f88275d52e04ce6d5c6d267b0dbdb0a87
3
+ size 1233687664
loras/omninft-ltx2.3-22b-rl-lora-r32.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d4e266785d020e23dfe7b3d3a486e41d76d58b0b3a8923711095b9badcd71bc1
3
+ size 1233687664
loras/readme.txt ADDED
File without changes