DannyJun commited on Mar 7

Commit

a2af795

verified ·

1 Parent(s): 5ad4e1a

Upload folder using huggingface_hub

Browse files

Files changed (22) hide show

.gitattributes +1 -0
added_tokens.json +429 -0
chat_template.jinja +1 -0
config.json +275 -0
configuration_molmoact.py +355 -0
generation_config.json +6 -0
image_processing_molmoact.py +951 -0
merges.txt +0 -0
model-00001-of-00004.safetensors +3 -0
model-00002-of-00004.safetensors +3 -0
model-00003-of-00004.safetensors +3 -0
model-00004-of-00004.safetensors +3 -0
model.safetensors.index.json +621 -0
model.yaml +195 -0
modeling_molmoact.py +2124 -0
preprocessor_config.json +27 -0
processing_molmoact.py +463 -0
processor_config.json +14 -0
special_tokens_map.json +1944 -0
tokenizer.json +3 -0
tokenizer_config.json +3713 -0
vocab.json +0 -0

.gitattributes CHANGED Viewed

@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+tokenizer.json filter=lfs diff=lfs merge=lfs -text

added_tokens.json ADDED Viewed

	@@ -0,0 +1,429 @@

+{
+  "</tool_call>": 151658,
+  "<DEPTH_0>": 151667,
+  "<DEPTH_100>": 151767,
+  "<DEPTH_101>": 151768,
+  "<DEPTH_102>": 151769,
+  "<DEPTH_103>": 151770,
+  "<DEPTH_104>": 151771,
+  "<DEPTH_105>": 151772,
+  "<DEPTH_106>": 151773,
+  "<DEPTH_107>": 151774,
+  "<DEPTH_108>": 151775,
+  "<DEPTH_109>": 151776,
+  "<DEPTH_10>": 151677,
+  "<DEPTH_110>": 151777,
+  "<DEPTH_111>": 151778,
+  "<DEPTH_112>": 151779,
+  "<DEPTH_113>": 151780,
+  "<DEPTH_114>": 151781,
+  "<DEPTH_115>": 151782,
+  "<DEPTH_116>": 151783,
+  "<DEPTH_117>": 151784,
+  "<DEPTH_118>": 151785,
+  "<DEPTH_119>": 151786,
+  "<DEPTH_11>": 151678,
+  "<DEPTH_120>": 151787,
+  "<DEPTH_121>": 151788,
+  "<DEPTH_122>": 151789,
+  "<DEPTH_123>": 151790,
+  "<DEPTH_124>": 151791,
+  "<DEPTH_125>": 151792,
+  "<DEPTH_126>": 151793,
+  "<DEPTH_127>": 151794,
+  "<DEPTH_12>": 151679,
+  "<DEPTH_13>": 151680,
+  "<DEPTH_14>": 151681,
+  "<DEPTH_15>": 151682,
+  "<DEPTH_16>": 151683,
+  "<DEPTH_17>": 151684,
+  "<DEPTH_18>": 151685,
+  "<DEPTH_19>": 151686,
+  "<DEPTH_1>": 151668,
+  "<DEPTH_20>": 151687,
+  "<DEPTH_21>": 151688,
+  "<DEPTH_22>": 151689,
+  "<DEPTH_23>": 151690,
+  "<DEPTH_24>": 151691,
+  "<DEPTH_25>": 151692,
+  "<DEPTH_26>": 151693,
+  "<DEPTH_27>": 151694,
+  "<DEPTH_28>": 151695,
+  "<DEPTH_29>": 151696,
+  "<DEPTH_2>": 151669,
+  "<DEPTH_30>": 151697,
+  "<DEPTH_31>": 151698,
+  "<DEPTH_32>": 151699,
+  "<DEPTH_33>": 151700,
+  "<DEPTH_34>": 151701,
+  "<DEPTH_35>": 151702,
+  "<DEPTH_36>": 151703,
+  "<DEPTH_37>": 151704,
+  "<DEPTH_38>": 151705,
+  "<DEPTH_39>": 151706,
+  "<DEPTH_3>": 151670,
+  "<DEPTH_40>": 151707,
+  "<DEPTH_41>": 151708,
+  "<DEPTH_42>": 151709,
+  "<DEPTH_43>": 151710,
+  "<DEPTH_44>": 151711,
+  "<DEPTH_45>": 151712,
+  "<DEPTH_46>": 151713,
+  "<DEPTH_47>": 151714,
+  "<DEPTH_48>": 151715,
+  "<DEPTH_49>": 151716,
+  "<DEPTH_4>": 151671,
+  "<DEPTH_50>": 151717,
+  "<DEPTH_51>": 151718,
+  "<DEPTH_52>": 151719,
+  "<DEPTH_53>": 151720,
+  "<DEPTH_54>": 151721,
+  "<DEPTH_55>": 151722,
+  "<DEPTH_56>": 151723,
+  "<DEPTH_57>": 151724,
+  "<DEPTH_58>": 151725,
+  "<DEPTH_59>": 151726,
+  "<DEPTH_5>": 151672,
+  "<DEPTH_60>": 151727,
+  "<DEPTH_61>": 151728,
+  "<DEPTH_62>": 151729,
+  "<DEPTH_63>": 151730,
+  "<DEPTH_64>": 151731,
+  "<DEPTH_65>": 151732,
+  "<DEPTH_66>": 151733,
+  "<DEPTH_67>": 151734,
+  "<DEPTH_68>": 151735,
+  "<DEPTH_69>": 151736,
+  "<DEPTH_6>": 151673,
+  "<DEPTH_70>": 151737,
+  "<DEPTH_71>": 151738,
+  "<DEPTH_72>": 151739,
+  "<DEPTH_73>": 151740,
+  "<DEPTH_74>": 151741,
+  "<DEPTH_75>": 151742,
+  "<DEPTH_76>": 151743,
+  "<DEPTH_77>": 151744,
+  "<DEPTH_78>": 151745,
+  "<DEPTH_79>": 151746,
+  "<DEPTH_7>": 151674,
+  "<DEPTH_80>": 151747,
+  "<DEPTH_81>": 151748,
+  "<DEPTH_82>": 151749,
+  "<DEPTH_83>": 151750,
+  "<DEPTH_84>": 151751,
+  "<DEPTH_85>": 151752,
+  "<DEPTH_86>": 151753,
+  "<DEPTH_87>": 151754,
+  "<DEPTH_88>": 151755,
+  "<DEPTH_89>": 151756,
+  "<DEPTH_8>": 151675,
+  "<DEPTH_90>": 151757,
+  "<DEPTH_91>": 151758,
+  "<DEPTH_92>": 151759,
+  "<DEPTH_93>": 151760,
+  "<DEPTH_94>": 151761,
+  "<DEPTH_95>": 151762,
+  "<DEPTH_96>": 151763,
+  "<DEPTH_97>": 151764,
+  "<DEPTH_98>": 151765,
+  "<DEPTH_99>": 151766,
+  "<DEPTH_9>": 151676,
+  "<DEPTH_END>": 151666,
+  "<DEPTH_START>": 151665,
+  "<im_col>": 152067,
+  "<im_end>": 152065,
+  "<im_low>": 152069,
+  "<im_patch>": 152066,
+  "<im_start>": 152064,
+  "<tool_call>": 151657,
+  "<|box_end|>": 151649,
+  "<|box_start|>": 151648,
+  "<|endoftext|>": 151643,
+  "<|file_sep|>": 151664,
+  "<|fim_middle|>": 151660,
+  "<|fim_pad|>": 151662,
+  "<|fim_prefix|>": 151659,
+  "<|fim_suffix|>": 151661,
+  "<|im_end|>": 151645,
+  "<|im_start|>": 151644,
+  "<|image_pad|>": 151655,
+  "<|image|>": 152068,
+  "<|object_ref_end|>": 151647,
+  "<|object_ref_start|>": 151646,
+  "<|quad_end|>": 151651,
+  "<|quad_start|>": 151650,
+  "<|repo_name|>": 151663,
+  "<|video_pad|>": 151656,
+  "<|vision_end|>": 151653,
+  "<|vision_pad|>": 151654,
+  "<|vision_start|>": 151652,
+  "|<EXTRA_TOKENS_0>|": 151795,
+  "|<EXTRA_TOKENS_100>|": 151895,
+  "|<EXTRA_TOKENS_101>|": 151896,
+  "|<EXTRA_TOKENS_102>|": 151897,
+  "|<EXTRA_TOKENS_103>|": 151898,
+  "|<EXTRA_TOKENS_104>|": 151899,
+  "|<EXTRA_TOKENS_105>|": 151900,
+  "|<EXTRA_TOKENS_106>|": 151901,
+  "|<EXTRA_TOKENS_107>|": 151902,
+  "|<EXTRA_TOKENS_108>|": 151903,
+  "|<EXTRA_TOKENS_109>|": 151904,
+  "|<EXTRA_TOKENS_10>|": 151805,
+  "|<EXTRA_TOKENS_110>|": 151905,
+  "|<EXTRA_TOKENS_111>|": 151906,
+  "|<EXTRA_TOKENS_112>|": 151907,
+  "|<EXTRA_TOKENS_113>|": 151908,
+  "|<EXTRA_TOKENS_114>|": 151909,
+  "|<EXTRA_TOKENS_115>|": 151910,
+  "|<EXTRA_TOKENS_116>|": 151911,
+  "|<EXTRA_TOKENS_117>|": 151912,
+  "|<EXTRA_TOKENS_118>|": 151913,
+  "|<EXTRA_TOKENS_119>|": 151914,
+  "|<EXTRA_TOKENS_11>|": 151806,
+  "|<EXTRA_TOKENS_120>|": 151915,
+  "|<EXTRA_TOKENS_121>|": 151916,
+  "|<EXTRA_TOKENS_122>|": 151917,
+  "|<EXTRA_TOKENS_123>|": 151918,
+  "|<EXTRA_TOKENS_124>|": 151919,
+  "|<EXTRA_TOKENS_125>|": 151920,
+  "|<EXTRA_TOKENS_126>|": 151921,
+  "|<EXTRA_TOKENS_127>|": 151922,
+  "|<EXTRA_TOKENS_128>|": 151923,
+  "|<EXTRA_TOKENS_129>|": 151924,
+  "|<EXTRA_TOKENS_12>|": 151807,
+  "|<EXTRA_TOKENS_130>|": 151925,
+  "|<EXTRA_TOKENS_131>|": 151926,
+  "|<EXTRA_TOKENS_132>|": 151927,
+  "|<EXTRA_TOKENS_133>|": 151928,
+  "|<EXTRA_TOKENS_134>|": 151929,
+  "|<EXTRA_TOKENS_135>|": 151930,
+  "|<EXTRA_TOKENS_136>|": 151931,
+  "|<EXTRA_TOKENS_137>|": 151932,
+  "|<EXTRA_TOKENS_138>|": 151933,
+  "|<EXTRA_TOKENS_139>|": 151934,
+  "|<EXTRA_TOKENS_13>|": 151808,
+  "|<EXTRA_TOKENS_140>|": 151935,
+  "|<EXTRA_TOKENS_141>|": 151936,
+  "|<EXTRA_TOKENS_142>|": 151937,
+  "|<EXTRA_TOKENS_143>|": 151938,
+  "|<EXTRA_TOKENS_144>|": 151939,
+  "|<EXTRA_TOKENS_145>|": 151940,
+  "|<EXTRA_TOKENS_146>|": 151941,
+  "|<EXTRA_TOKENS_147>|": 151942,
+  "|<EXTRA_TOKENS_148>|": 151943,
+  "|<EXTRA_TOKENS_149>|": 151944,
+  "|<EXTRA_TOKENS_14>|": 151809,
+  "|<EXTRA_TOKENS_150>|": 151945,
+  "|<EXTRA_TOKENS_151>|": 151946,
+  "|<EXTRA_TOKENS_152>|": 151947,
+  "|<EXTRA_TOKENS_153>|": 151948,
+  "|<EXTRA_TOKENS_154>|": 151949,
+  "|<EXTRA_TOKENS_155>|": 151950,
+  "|<EXTRA_TOKENS_156>|": 151951,
+  "|<EXTRA_TOKENS_157>|": 151952,
+  "|<EXTRA_TOKENS_158>|": 151953,
+  "|<EXTRA_TOKENS_159>|": 151954,
+  "|<EXTRA_TOKENS_15>|": 151810,
+  "|<EXTRA_TOKENS_160>|": 151955,
+  "|<EXTRA_TOKENS_161>|": 151956,
+  "|<EXTRA_TOKENS_162>|": 151957,
+  "|<EXTRA_TOKENS_163>|": 151958,
+  "|<EXTRA_TOKENS_164>|": 151959,
+  "|<EXTRA_TOKENS_165>|": 151960,
+  "|<EXTRA_TOKENS_166>|": 151961,
+  "|<EXTRA_TOKENS_167>|": 151962,
+  "|<EXTRA_TOKENS_168>|": 151963,
+  "|<EXTRA_TOKENS_169>|": 151964,
+  "|<EXTRA_TOKENS_16>|": 151811,
+  "|<EXTRA_TOKENS_170>|": 151965,
+  "|<EXTRA_TOKENS_171>|": 151966,
+  "|<EXTRA_TOKENS_172>|": 151967,
+  "|<EXTRA_TOKENS_173>|": 151968,
+  "|<EXTRA_TOKENS_174>|": 151969,
+  "|<EXTRA_TOKENS_175>|": 151970,
+  "|<EXTRA_TOKENS_176>|": 151971,
+  "|<EXTRA_TOKENS_177>|": 151972,
+  "|<EXTRA_TOKENS_178>|": 151973,
+  "|<EXTRA_TOKENS_179>|": 151974,
+  "|<EXTRA_TOKENS_17>|": 151812,
+  "|<EXTRA_TOKENS_180>|": 151975,
+  "|<EXTRA_TOKENS_181>|": 151976,
+  "|<EXTRA_TOKENS_182>|": 151977,
+  "|<EXTRA_TOKENS_183>|": 151978,
+  "|<EXTRA_TOKENS_184>|": 151979,
+  "|<EXTRA_TOKENS_185>|": 151980,
+  "|<EXTRA_TOKENS_186>|": 151981,
+  "|<EXTRA_TOKENS_187>|": 151982,
+  "|<EXTRA_TOKENS_188>|": 151983,
+  "|<EXTRA_TOKENS_189>|": 151984,
+  "|<EXTRA_TOKENS_18>|": 151813,
+  "|<EXTRA_TOKENS_190>|": 151985,
+  "|<EXTRA_TOKENS_191>|": 151986,
+  "|<EXTRA_TOKENS_192>|": 151987,
+  "|<EXTRA_TOKENS_193>|": 151988,
+  "|<EXTRA_TOKENS_194>|": 151989,
+  "|<EXTRA_TOKENS_195>|": 151990,
+  "|<EXTRA_TOKENS_196>|": 151991,
+  "|<EXTRA_TOKENS_197>|": 151992,
+  "|<EXTRA_TOKENS_198>|": 151993,
+  "|<EXTRA_TOKENS_199>|": 151994,
+  "|<EXTRA_TOKENS_19>|": 151814,
+  "|<EXTRA_TOKENS_1>|": 151796,
+  "|<EXTRA_TOKENS_200>|": 151995,
+  "|<EXTRA_TOKENS_201>|": 151996,
+  "|<EXTRA_TOKENS_202>|": 151997,
+  "|<EXTRA_TOKENS_203>|": 151998,
+  "|<EXTRA_TOKENS_204>|": 151999,
+  "|<EXTRA_TOKENS_205>|": 152000,
+  "|<EXTRA_TOKENS_206>|": 152001,
+  "|<EXTRA_TOKENS_207>|": 152002,
+  "|<EXTRA_TOKENS_208>|": 152003,
+  "|<EXTRA_TOKENS_209>|": 152004,
+  "|<EXTRA_TOKENS_20>|": 151815,
+  "|<EXTRA_TOKENS_210>|": 152005,
+  "|<EXTRA_TOKENS_211>|": 152006,
+  "|<EXTRA_TOKENS_212>|": 152007,
+  "|<EXTRA_TOKENS_213>|": 152008,
+  "|<EXTRA_TOKENS_214>|": 152009,
+  "|<EXTRA_TOKENS_215>|": 152010,
+  "|<EXTRA_TOKENS_216>|": 152011,
+  "|<EXTRA_TOKENS_217>|": 152012,
+  "|<EXTRA_TOKENS_218>|": 152013,
+  "|<EXTRA_TOKENS_219>|": 152014,
+  "|<EXTRA_TOKENS_21>|": 151816,
+  "|<EXTRA_TOKENS_220>|": 152015,
+  "|<EXTRA_TOKENS_221>|": 152016,
+  "|<EXTRA_TOKENS_222>|": 152017,
+  "|<EXTRA_TOKENS_223>|": 152018,
+  "|<EXTRA_TOKENS_224>|": 152019,
+  "|<EXTRA_TOKENS_225>|": 152020,
+  "|<EXTRA_TOKENS_226>|": 152021,
+  "|<EXTRA_TOKENS_227>|": 152022,
+  "|<EXTRA_TOKENS_228>|": 152023,
+  "|<EXTRA_TOKENS_229>|": 152024,
+  "|<EXTRA_TOKENS_22>|": 151817,
+  "|<EXTRA_TOKENS_230>|": 152025,
+  "|<EXTRA_TOKENS_231>|": 152026,
+  "|<EXTRA_TOKENS_232>|": 152027,
+  "|<EXTRA_TOKENS_233>|": 152028,
+  "|<EXTRA_TOKENS_234>|": 152029,
+  "|<EXTRA_TOKENS_235>|": 152030,
+  "|<EXTRA_TOKENS_236>|": 152031,
+  "|<EXTRA_TOKENS_237>|": 152032,
+  "|<EXTRA_TOKENS_238>|": 152033,
+  "|<EXTRA_TOKENS_239>|": 152034,
+  "|<EXTRA_TOKENS_23>|": 151818,
+  "|<EXTRA_TOKENS_240>|": 152035,
+  "|<EXTRA_TOKENS_241>|": 152036,
+  "|<EXTRA_TOKENS_242>|": 152037,
+  "|<EXTRA_TOKENS_243>|": 152038,
+  "|<EXTRA_TOKENS_244>|": 152039,
+  "|<EXTRA_TOKENS_245>|": 152040,
+  "|<EXTRA_TOKENS_246>|": 152041,
+  "|<EXTRA_TOKENS_247>|": 152042,
+  "|<EXTRA_TOKENS_248>|": 152043,
+  "|<EXTRA_TOKENS_249>|": 152044,
+  "|<EXTRA_TOKENS_24>|": 151819,
+  "|<EXTRA_TOKENS_250>|": 152045,
+  "|<EXTRA_TOKENS_251>|": 152046,
+  "|<EXTRA_TOKENS_252>|": 152047,
+  "|<EXTRA_TOKENS_253>|": 152048,
+  "|<EXTRA_TOKENS_254>|": 152049,
+  "|<EXTRA_TOKENS_255>|": 152050,
+  "|<EXTRA_TOKENS_256>|": 152051,
+  "|<EXTRA_TOKENS_257>|": 152052,
+  "|<EXTRA_TOKENS_258>|": 152053,
+  "|<EXTRA_TOKENS_259>|": 152054,
+  "|<EXTRA_TOKENS_25>|": 151820,
+  "|<EXTRA_TOKENS_260>|": 152055,
+  "|<EXTRA_TOKENS_261>|": 152056,
+  "|<EXTRA_TOKENS_262>|": 152057,
+  "|<EXTRA_TOKENS_263>|": 152058,
+  "|<EXTRA_TOKENS_264>|": 152059,
+  "|<EXTRA_TOKENS_265>|": 152060,
+  "|<EXTRA_TOKENS_266>|": 152061,
+  "|<EXTRA_TOKENS_267>|": 152062,
+  "|<EXTRA_TOKENS_268>|": 152063,
+  "|<EXTRA_TOKENS_26>|": 151821,
+  "|<EXTRA_TOKENS_27>|": 151822,
+  "|<EXTRA_TOKENS_28>|": 151823,
+  "|<EXTRA_TOKENS_29>|": 151824,
+  "|<EXTRA_TOKENS_2>|": 151797,
+  "|<EXTRA_TOKENS_30>|": 151825,
+  "|<EXTRA_TOKENS_31>|": 151826,
+  "|<EXTRA_TOKENS_32>|": 151827,
+  "|<EXTRA_TOKENS_33>|": 151828,
+  "|<EXTRA_TOKENS_34>|": 151829,
+  "|<EXTRA_TOKENS_35>|": 151830,
+  "|<EXTRA_TOKENS_36>|": 151831,
+  "|<EXTRA_TOKENS_37>|": 151832,
+  "|<EXTRA_TOKENS_38>|": 151833,
+  "|<EXTRA_TOKENS_39>|": 151834,
+  "|<EXTRA_TOKENS_3>|": 151798,
+  "|<EXTRA_TOKENS_40>|": 151835,
+  "|<EXTRA_TOKENS_41>|": 151836,
+  "|<EXTRA_TOKENS_42>|": 151837,
+  "|<EXTRA_TOKENS_43>|": 151838,
+  "|<EXTRA_TOKENS_44>|": 151839,
+  "|<EXTRA_TOKENS_45>|": 151840,
+  "|<EXTRA_TOKENS_46>|": 151841,
+  "|<EXTRA_TOKENS_47>|": 151842,
+  "|<EXTRA_TOKENS_48>|": 151843,
+  "|<EXTRA_TOKENS_49>|": 151844,
+  "|<EXTRA_TOKENS_4>|": 151799,
+  "|<EXTRA_TOKENS_50>|": 151845,
+  "|<EXTRA_TOKENS_51>|": 151846,
+  "|<EXTRA_TOKENS_52>|": 151847,
+  "|<EXTRA_TOKENS_53>|": 151848,
+  "|<EXTRA_TOKENS_54>|": 151849,
+  "|<EXTRA_TOKENS_55>|": 151850,
+  "|<EXTRA_TOKENS_56>|": 151851,
+  "|<EXTRA_TOKENS_57>|": 151852,
+  "|<EXTRA_TOKENS_58>|": 151853,
+  "|<EXTRA_TOKENS_59>|": 151854,
+  "|<EXTRA_TOKENS_5>|": 151800,
+  "|<EXTRA_TOKENS_60>|": 151855,
+  "|<EXTRA_TOKENS_61>|": 151856,
+  "|<EXTRA_TOKENS_62>|": 151857,
+  "|<EXTRA_TOKENS_63>|": 151858,
+  "|<EXTRA_TOKENS_64>|": 151859,
+  "|<EXTRA_TOKENS_65>|": 151860,
+  "|<EXTRA_TOKENS_66>|": 151861,
+  "|<EXTRA_TOKENS_67>|": 151862,
+  "|<EXTRA_TOKENS_68>|": 151863,
+  "|<EXTRA_TOKENS_69>|": 151864,
+  "|<EXTRA_TOKENS_6>|": 151801,
+  "|<EXTRA_TOKENS_70>|": 151865,
+  "|<EXTRA_TOKENS_71>|": 151866,
+  "|<EXTRA_TOKENS_72>|": 151867,
+  "|<EXTRA_TOKENS_73>|": 151868,
+  "|<EXTRA_TOKENS_74>|": 151869,
+  "|<EXTRA_TOKENS_75>|": 151870,
+  "|<EXTRA_TOKENS_76>|": 151871,
+  "|<EXTRA_TOKENS_77>|": 151872,
+  "|<EXTRA_TOKENS_78>|": 151873,
+  "|<EXTRA_TOKENS_79>|": 151874,
+  "|<EXTRA_TOKENS_7>|": 151802,
+  "|<EXTRA_TOKENS_80>|": 151875,
+  "|<EXTRA_TOKENS_81>|": 151876,
+  "|<EXTRA_TOKENS_82>|": 151877,
+  "|<EXTRA_TOKENS_83>|": 151878,
+  "|<EXTRA_TOKENS_84>|": 151879,
+  "|<EXTRA_TOKENS_85>|": 151880,
+  "|<EXTRA_TOKENS_86>|": 151881,
+  "|<EXTRA_TOKENS_87>|": 151882,
+  "|<EXTRA_TOKENS_88>|": 151883,
+  "|<EXTRA_TOKENS_89>|": 151884,
+  "|<EXTRA_TOKENS_8>|": 151803,
+  "|<EXTRA_TOKENS_90>|": 151885,
+  "|<EXTRA_TOKENS_91>|": 151886,
+  "|<EXTRA_TOKENS_92>|": 151887,
+  "|<EXTRA_TOKENS_93>|": 151888,
+  "|<EXTRA_TOKENS_94>|": 151889,
+  "|<EXTRA_TOKENS_95>|": 151890,
+  "|<EXTRA_TOKENS_96>|": 151891,
+  "|<EXTRA_TOKENS_97>|": 151892,
+  "|<EXTRA_TOKENS_98>|": 151893,
+  "|<EXTRA_TOKENS_99>|": 151894,
+  "|<EXTRA_TOKENS_9>|": 151804
+}

chat_template.jinja ADDED Viewed

	@@ -0,0 +1 @@

+ {% for message in messages %}{%- if (loop.index % 2 == 1 and message['role'].lower() != 'user') or (loop.index % 2 == 0 and message['role'].lower() != 'assistant') -%}{{ raise_exception('Conversation roles must alternate user/assistant/user/assistant/...') }}{%- endif -%}{{ message['role'].capitalize() + ': ' }}{% if message['content'] is string %}{{ message['content'] }}{% else %}{% for content in message['content'] %}{% if content['type'] == 'text' %}{{ content['text'] }}{%- if not loop.last -%}{{ ' ' }}{%- endif -%}{% endif %}{% endfor %}{% endif %}{%- if not loop.last -%}{{ ' ' }}{%- endif -%}{% endfor %}{% if add_generation_prompt %}{{ ' Assistant:' }}{% endif %}

config.json ADDED Viewed

	@@ -0,0 +1,275 @@

+{
+  "adapter_config": {
+    "attention_dropout": 0.0,
+    "float32_attention": true,
+    "head_dim": 72,
+    "hidden_act": "silu",
+    "hidden_size": 1152,
+    "image_feature_dropout": 0.0,
+    "image_padding_embed": null,
+    "initializer_range": 0.02,
+    "intermediate_size": 18944,
+    "model_type": "",
+    "num_attention_heads": 16,
+    "num_key_value_heads": 16,
+    "residual_dropout": 0.0,
+    "text_hidden_size": 3584,
+    "vit_layers": [
+      -3,
+      -9
+    ]
+  },
+  "architectures": [
+    "MolmoActForActionReasoning"
+  ],
+  "auto_map": {
+    "AutoConfig": "configuration_molmoact.MolmoActConfig",
+    "AutoModelForImageTextToText": "modeling_molmoact.MolmoActForActionReasoning"
+  },
+  "image_patch_id": 152066,
+  "initializer_range": 0.02,
+  "llm_config": {
+    "additional_vocab_size": 128,
+    "attention_dropout": 0.0,
+    "embedding_dropout": 0.0,
+    "head_dim": 128,
+    "hidden_act": "silu",
+    "hidden_size": 3584,
+    "initializer_range": 0.02,
+    "intermediate_size": 18944,
+    "layer_norm_eps": 1e-06,
+    "max_position_embeddings": 4096,
+    "model_type": "molmoact_llm",
+    "norm_after": false,
+    "num_attention_heads": 28,
+    "num_hidden_layers": 28,
+    "num_key_value_heads": 4,
+    "qk_norm_type": "olmo",
+    "qkv_bias": true,
+    "residual_dropout": 0.0,
+    "rope_scaling": null,
+    "rope_theta": 1000000.0,
+    "use_cache": true,
+    "use_qk_norm": false,
+    "vocab_size": 152064
+  },
+  "model_type": "molmoact",
+  "n_action_bins": 256,
+  "norm_stats": {
+    "molmoact": {
+      "action": {
+        "max": [
+          0.06042003631591797,
+          0.09417290985584259,
+          0.07019275426864624,
+          0.2616892158985138,
+          0.11751057207584381,
+          0.16968433558940887,
+          1.0
+        ],
+        "mean": [
+          0.0005706787342205644,
+          0.0002448957529850304,
+          -3.5987635783385485e-05,
+          0.00021597897284664214,
+          -0.0004896928439848125,
+          -0.000241481073317118,
+          0.5570635199546814
+        ],
+        "min": [
+          -0.07434078305959702,
+          -0.07339745759963989,
+          -0.06539416313171387,
+          -0.1688285619020462,
+          -0.10289879888296127,
+          -0.2667275667190552,
+          0.0
+        ],
+        "q01": [
+          -0.01538565568625927,
+          -0.021047022193670273,
+          -0.01688069850206375,
+          -0.044314172118902206,
+          -0.03890235349535942,
+          -0.04788423702120781,
+          0.0
+        ],
+        "q99": [
+          0.014661382883787155,
+          0.026515591889619827,
+          0.021398313343524933,
+          0.04216696694493294,
+          0.03401297703385353,
+          0.04957397282123566,
+          1.0
+        ],
+        "std": [
+          0.005207270849496126,
+          0.007506529800593853,
+          0.006415561307221651,
+          0.013248044066131115,
+          0.010928540490567684,
+          0.014873150736093521,
+          0.49715080857276917
+        ]
+      },
+      "num_entries": 1560068
+    },
+    "libero_object_no_noops_modified": {
+      "action": {
+        "max": [
+          0.9375,
+          0.8919642567634583,
+          0.9375,
+          0.17678570747375488,
+          0.35035714507102966,
+          0.1810714304447174,
+          1.0
+        ],
+        "mean": [
+          0.07096529006958008,
+          0.13498851656913757,
+          -0.04601382836699486,
+          0.00123520044144243,
+          0.006998839322477579,
+          -0.015027612447738647,
+          0.46428999304771423
+        ],
+        "min": [
+          -0.8839285969734192,
+          -0.9375,
+          -0.9375,
+          -0.15000000596046448,
+          -0.29035714268684387,
+          -0.32892856001853943,
+          0.0
+        ],
+        "q01": [
+          -0.5383928418159485,
+          -0.8758928775787354,
+          -0.9375,
+          -0.06964285671710968,
+          -0.11678571254014969,
+          -0.15964286029338837,
+          0.0
+        ],
+        "q99": [
+          0.8464285731315613,
+          0.84375,
+          0.9375,
+          0.08142857253551483,
+          0.14892856776714325,
+          0.0867857113480568,
+          1.0
+        ],
+        "std": [
+          0.2681235373020172,
+          0.43846824765205383,
+          0.4474974274635315,
+          0.024446550756692886,
+          0.049355510622262955,
+          0.042107198387384415,
+          0.49879148602485657
+        ]
+      },
+      "num_trajectories": 454,
+      "num_transitions": 66984,
+      "proprio": {
+        "max": [
+          0.14580604434013367,
+          0.33216384053230286,
+          0.3857804834842682,
+          3.4003844261169434,
+          0.7954911589622498,
+          0.6642207503318787,
+          0.0,
+          0.04104341194033623,
+          -0.00018117300351150334
+        ],
+        "mean": [
+          -0.02999030612409115,
+          -0.007947085425257683,
+          0.20293472707271576,
+          3.1086409091949463,
+          -0.21404768526554108,
+          -0.11307074874639511,
+          0.0,
+          0.029380427673459053,
+          -0.030556727200746536
+        ],
+        "min": [
+          -0.1765444278717041,
+          -0.29457300901412964,
+          0.008128180168569088,
+          2.2890501022338867,
+          -1.883241891860962,
+          -1.0600427389144897,
+          0.0,
+          0.0006495157140307128,
+          -0.041782498359680176
+        ],
+        "q01": [
+          -0.14911890715360643,
+          -0.25978428691625594,
+          0.009925739830359817,
+          2.7545341420173646,
+          -1.3996034812927245,
+          -0.6867720144987106,
+          0.0,
+          0.008197814421728254,
+          -0.04015838988125324
+        ],
+        "q99": [
+          0.09063626825809479,
+          0.29066365867853167,
+          0.3370887073874472,
+          3.2611824750900267,
+          0.32092821151018125,
+          0.4037663781642913,
+          0.0,
+          0.039891827926039694,
+          -0.009106044843792932
+        ],
+        "std": [
+          0.06694897264242172,
+          0.17608462274074554,
+          0.07807064801454544,
+          0.0868484303355217,
+          0.33540457487106323,
+          0.20728276669979095,
+          0.0,
+          0.00956575945019722,
+          0.009197483770549297
+        ]
+      }
+    }
+  },
+  "tie_word_embeddings": false,
+  "torch_dtype": "bfloat16",
+  "transformers_version": "4.52.1",
+  "use_cache": true,
+  "vit_config": {
+    "attention_dropout": 0.0,
+    "float32_attention": true,
+    "head_dim": 72,
+    "hidden_act": "gelu_pytorch_tanh",
+    "hidden_size": 1152,
+    "image_default_input_size": [
+      378,
+      378
+    ],
+    "image_num_pos": 729,
+    "image_patch_size": 14,
+    "initializer_range": 0.02,
+    "intermediate_size": 4304,
+    "layer_norm_eps": 1e-06,
+    "model_type": "molmoact_vit",
+    "num_attention_heads": 16,
+    "num_hidden_layers": 27,
+    "num_key_value_heads": 16,
+    "patch_bias": true,
+    "pre_layernorm": false,
+    "residual_dropout": 0.0,
+    "use_cls_token": false
+  }
+}

configuration_molmoact.py ADDED Viewed

	@@ -0,0 +1,355 @@

+"""
+MolmoAct configuration
+"""
+from typing import Tuple, Optional, Dict, Any
+from transformers import PretrainedConfig
+from transformers.modeling_rope_utils import rope_config_validation
+from transformers.utils import logging
+logger = logging.get_logger(__name__)
+class MolmoActVitConfig(PretrainedConfig):
+    r"""
+    This is the configuration class to store the configuration of a [`MolmoActVisionTransformer`].
+    It is used to instantiate a `MolmoActVisionTransformer` according to the specified arguments,
+    defining the model architecture.
+    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
+    documentation from [`PretrainedConfig`] for more information.
+    Example:
+    ```python
+    >>> from transformers import MolmoActVitConfig, MolmoActVisionTransformer
+    >>> # Initializing a MolmoActVitConfig
+    >>> configuration = MolmoActVitConfig()
+    >>> # Initializing a MolmoActVisionTransformer (with random weights)
+    >>> model = MolmoActVisionTransformer(configuration)
+    >>> # Accessing the model configuration
+    >>> configuration = model.config
+    ```"""
+    model_type = "molmoact_vit"
+    def __init__(
+        self,
+        hidden_size: int = 1152,
+        intermediate_size: int = 4304,
+        num_hidden_layers: int = 27,
+        num_attention_heads: int = 16,
+        num_key_value_heads: int = 16,
+        head_dim: int = 72,
+        hidden_act: str = "gelu_pytorch_tanh",
+        layer_norm_eps: float = 1e-6,
+        image_default_input_size: Tuple[int, int] = (378, 378),
+        image_patch_size: int = 14,
+        image_num_pos: int = 577,
+        attention_dropout: float = 0.0,
+        residual_dropout: float = 0.0,
+        initializer_range: float = 0.02,
+        float32_attention: bool = True,
+        use_cls_token: bool = False,      # True for OpenCLIP
+        patch_bias: bool = True,          # False for OpenCLIP
+        pre_layernorm: bool = False,      # True for OpenCLIP
+        **kwargs,
+    ):
+        super().__init__(**kwargs)
+        self.hidden_size = hidden_size
+        self.intermediate_size = intermediate_size
+        self.num_hidden_layers = num_hidden_layers
+        self.num_attention_heads = num_attention_heads
+        self.num_key_value_heads = num_key_value_heads
+        self.head_dim = head_dim
+        self.hidden_act = hidden_act
+        self.layer_norm_eps = layer_norm_eps
+        self.image_default_input_size = image_default_input_size
+        self.image_patch_size = image_patch_size
+        self.image_num_pos = image_num_pos
+        self.attention_dropout = attention_dropout
+        self.residual_dropout = residual_dropout
+        self.initializer_range = initializer_range
+        self.float32_attention = float32_attention
+        self.use_cls_token = use_cls_token
+        self.patch_bias = patch_bias
+        self.pre_layernorm = pre_layernorm
+    @property
+    def image_num_patch(self):
+        h, w = self.image_default_input_size
+        return h // self.image_patch_size, w // self.image_patch_size
+class MolmoActAdapterConfig(PretrainedConfig):
+    r"""
+    This is the configuration class to store the configuration of MolmoActAdapter. With MolmoActVitConfig,
+    It is used to instantiate an MolmoActVisionBackbone according to the specified arguments,
+    defining the model architecture.
+    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
+    documentation from [`PretrainedConfig`] for more information.
+    Example:
+    ```python
+    >>> from transformers import MolmoActVitConfig, MolmoActAdapterConfig, MolmoActVisionBackbone
+    >>> # Initializing a MolmoActVitConfig and a MolmoActAdapterConfig
+    >>> vit_config = MolmoActVitConfig()
+    >>> adapter_config = MolmoPoolingConfig()
+    >>> # Initializing a MolmoActVisionBackbone (with random weights)
+    >>> model = MolmoActVisionBackbone(vit_config, adapter_config)
+    >>> # Accessing the model configuration
+    >>> vit_configuration = model.vit_config
+    >>> adapter_configuration = model.adapter_config
+    ```"""
+    def __init__(
+        self,
+        vit_layers: Tuple = (-3, -9),
+        hidden_size: int = 1152,
+        num_attention_heads: int = 16,
+        num_key_value_heads: int = 16,
+        head_dim: int = 72,
+        float32_attention: bool = True,
+        attention_dropout: float = 0.0,
+        residual_dropout: float = 0.0,
+        hidden_act: str = "silu",
+        intermediate_size: int = 18944,
+        text_hidden_size: int = 3584,
+        image_feature_dropout: float = 0.0,
+        initializer_range: float = 0.02,
+        # pooling_mode: str = "indices",            # "indices" (SigLIP) or "2x2_attention" (OpenCLIP)
+        image_padding_embed: Optional[str] = None,  # e.g. "pad_and_partial_pad"
+        **kwargs,
+    ):
+        super().__init__(**kwargs)
+        self.vit_layers = vit_layers
+        self.hidden_size = hidden_size
+        self.num_attention_heads = num_attention_heads
+        self.num_key_value_heads = num_key_value_heads
+        self.head_dim = head_dim
+        self.float32_attention = float32_attention
+        self.attention_dropout = attention_dropout
+        self.residual_dropout = residual_dropout
+        self.hidden_act = hidden_act
+        self.intermediate_size = intermediate_size
+        self.text_hidden_size = text_hidden_size
+        self.image_feature_dropout = image_feature_dropout
+        self.initializer_range = initializer_range
+        # self.pooling_mode = pooling_mode
+        self.image_padding_embed = image_padding_embed
+class MolmoActLlmConfig(PretrainedConfig):
+    r"""
+    This is the configuration class to store the configuration of a [`MolmoActLlm`]. It is used to instantiate a
+    `MolmoActLlm` according to the specified arguments, defining the model architecture.
+    Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
+    documentation from [`PretrainedConfig`] for more information.
+    Example:
+    ```python
+    >>> from transformers import MolmoActLlmConfig, MolmoActLlm
+    >>> # Initializing a MolmoActLlmConfig
+    >>> configuration = MolmoActLlmConfig()
+    >>> # Initializing a MolmoActLlm (with random weights)
+    >>> model = MolmoActLlm(configuration)
+    >>> # Accessing the model configuration
+    >>> configuration = model.config
+    ```"""
+    model_type = "molmoact_llm"
+    keys_to_ignore_at_inference = ["past_key_values"]
+    base_model_tp_plan = {
+        "blocks.*.self_attn.att_proj": "colwise",
+        "blocks.*.self_attn.attn_out": "rowwise",
+        "blocks.*.mlp.ff_proj": "colwise",
+        "blocks.*.mlp.ff_out": "rowwise",
+    }
+    base_model_pp_plan = {
+        "wte": (["input_ids"], ["inputs_embeds"]),
+        "blocks": (["hidden_states", "attention_mask"], ["hidden_states"]),
+        "ln_f": (["hidden_states"], ["hidden_states"]),
+    }
+    def __init__(
+        self,
+        hidden_size: int = 3584,
+        num_attention_heads: int = 28,
+        num_key_value_heads: Optional[int] = 4,
+        head_dim: int = 128,
+        vocab_size: int = 152064,
+        additional_vocab_size: int = 128,
+        qkv_bias: bool = True,
+        num_hidden_layers: int = 48,
+        intermediate_size: int = 18944,
+        hidden_act: str = "silu",
+        embedding_dropout: float=0.0,
+        attention_dropout: float=0.0,
+        residual_dropout: float = 0.0,
+        max_position_embeddings: int = 4096,
+        rope_theta: float = 1000000.0,
+        rope_scaling: Dict[str, Any] = None,
+        use_qk_norm: bool = False,
+        qk_norm_type: str = "olmo",
+        layer_norm_eps: int = 1e-6,
+        norm_after: bool = False,
+        initializer_range: float = 0.02,
+        use_cache=True,
+        tie_word_embeddings=False,
+        **kwargs,
+    ):
+        super().__init__(
+            tie_word_embeddings=tie_word_embeddings,
+            **kwargs
+        )
+        self.hidden_size = hidden_size
+        self.num_attention_heads = num_attention_heads
+        if num_key_value_heads is None:
+            num_key_value_heads = num_attention_heads
+        self.num_key_value_heads = num_key_value_heads
+        self.head_dim = head_dim
+        self.vocab_size = vocab_size
+        self.additional_vocab_size = additional_vocab_size
+        self.qkv_bias = qkv_bias
+        self.num_hidden_layers = num_hidden_layers
+        self.intermediate_size = intermediate_size
+        self.hidden_act = hidden_act
+        self.embedding_dropout = embedding_dropout
+        self.attention_dropout = attention_dropout
+        self.residual_dropout = residual_dropout
+        self.max_position_embeddings = max_position_embeddings
+        self.rope_theta = rope_theta
+        self.rope_scaling = rope_scaling
+        self.use_qk_norm = use_qk_norm
+        self.qk_norm_type = qk_norm_type
+        self.layer_norm_eps = layer_norm_eps
+        self.norm_after = norm_after
+        self.initializer_range = initializer_range
+        self.use_cache = use_cache
+        # Validate the correctness of rotary position embeddings parameters
+        rope_config_validation(self)
+class MolmoActConfig(PretrainedConfig):
+    r"""
+    This is the configuration class to store the configuration of a [`MolmoActForActionReasoning`].
+    It is used to instantiate an MolmoAct model according to the specified arguments, defining the model architecture.
+    Example:
+    ```python
+    >>> from transformers import MolmoActConfig, MolmoActVitConfig, MolmoActAdapterConfig, MolmoActLlmConfig
+    >>> # Initializing a MolmoActVitConfig
+    >>> vit_config = MolmoActVitConfig()
+    >>> # Initializing a MolmoActAdapterConfig
+    >>> adapter_config = MolmoActAdapterConfig()
+    >>> # Initializing a MolmoActLlmConfig
+    >>> llm_config = MolmoActLlmConfig()
+    >>> # Initializing a MolmoActConfig
+    >>> configuration = MolmoActConfig(vit_config, adapter_config, llm_config, image_patch_id=152069)
+    >>> # Initializing a model
+    >>> model = MolmoActForActionReasoning(configuration)
+    >>> # Accessing the model configuration
+    >>> configuration = model.config
+    ```"""
+    model_type = "molmoact"
+    sub_configs = {
+        "llm_config": MolmoActLlmConfig,
+        "vit_config": MolmoActVitConfig,
+        "adapter_config": MolmoActAdapterConfig,
+    }
+    def __init__(
+        self,
+        vit_config: MolmoActVitConfig = None,
+        adapter_config: MolmoActAdapterConfig = None,
+        llm_config: MolmoActLlmConfig = None,
+        image_patch_id: int = None,
+        initializer_range: float = 0.02,
+        n_action_bins: int = 256,
+        norm_stats: dict = {},
+        **kwargs,
+    ):
+        super().__init__(**kwargs)
+        if vit_config is None:
+            self.vit_config = MolmoActVitConfig()
+        elif isinstance(vit_config, dict):
+            self.vit_config = MolmoActVitConfig(**vit_config)
+        else:
+            self.vit_config = vit_config
+        if adapter_config is None:
+            self.adapter_config = MolmoActAdapterConfig()
+        elif isinstance(adapter_config, dict):
+            self.adapter_config = MolmoActAdapterConfig(**adapter_config)
+        else:
+            self.adapter_config = adapter_config
+        if llm_config is None:
+            self.llm_config = MolmoActLlmConfig()
+        elif isinstance(llm_config, dict):
+            self.llm_config = MolmoActLlmConfig(**llm_config)
+        else:
+            self.llm_config = llm_config
+        self.image_patch_id = image_patch_id
+        self.initializer_range = initializer_range
+        self.n_action_bins = n_action_bins
+        self.norm_stats = norm_stats
+    @property
+    def image_num_patch(self):
+        assert self.vit_config is not None
+        return self.vit_config.image_num_patch
+    @property
+    def num_attention_heads(self):
+        return self.llm_config.num_attention_heads
+    @property
+    def num_key_value_heads(self):
+        return self.llm_config.num_key_value_heads
+    @property
+    def head_dim(self):
+        return self.llm_config.head_dim
+    @property
+    def num_hidden_layers(self):
+        return self.llm_config.num_hidden_layers
+    @property
+    def hidden_size(self):
+        return self.llm_config.hidden_size
+    @property
+    def vocab_size(self):
+        return self.llm_config.vocab_size
+    @property
+    def max_position_embeddings(self):
+        return self.llm_config.max_position_embeddings
+MolmoActVitConfig.register_for_auto_class()
+MolmoActAdapterConfig.register_for_auto_class()
+MolmoActLlmConfig.register_for_auto_class()
+MolmoActConfig.register_for_auto_class()

generation_config.json ADDED Viewed

	@@ -0,0 +1,6 @@

+{
+  "bos_token_id": 151643,
+  "eos_token_id": 151643,
+  "pad_token_id": 151643,
+  "transformers_version": "4.52.1"
+}

image_processing_molmoact.py ADDED Viewed

	@@ -0,0 +1,951 @@

+"""Image processor class for MolmoAct"""
+from typing import TYPE_CHECKING, Tuple, List, Optional, Union, Dict, Any
+import numpy as np
+import einops
+import torch
+import torchvision.transforms
+from torchvision.transforms import InterpolationMode
+from torchvision.transforms.functional import convert_image_dtype
+from transformers.image_utils import (
+    OPENAI_CLIP_MEAN,
+    OPENAI_CLIP_STD,
+    ChannelDimension,
+    ImageInput,
+    is_valid_image,
+    valid_images,
+    to_numpy_array,
+)
+from transformers.image_transforms import convert_to_rgb, to_channel_dimension_format
+from transformers.processing_utils import ImagesKwargs
+from transformers.image_processing_utils import BaseImageProcessor
+from transformers.utils import logging
+from transformers.feature_extraction_utils import BatchFeature
+from transformers.utils import TensorType, logging
+if TYPE_CHECKING:
+    from transformers.utils import TensorType, logging
+logger = logging.get_logger(__name__)
+def is_multi_image(image: Union[ImageInput, List[ImageInput]]) -> bool:
+    return isinstance(image, (list, tuple))
+def make_batched_images(images) -> List[ImageInput]:
+    """
+    Accepts images in list or nested list format.
+    Args:
+        images (`Union[List[List[ImageInput]], List[ImageInput], ImageInput]`):
+            The input image.
+    Returns:
+        list: A list of images or a list of lists of images.
+    """
+    if isinstance(images, (list, tuple)) and isinstance(images[0], (list, tuple)) and is_valid_image(images[0][0]):
+        return images
+    elif isinstance(images, (list, tuple)) and is_valid_image(images[0]):
+        return images
+    elif is_valid_image(images):
+        return [images]
+    raise ValueError(f"Could not make batched images from {images}")
+def normalize_image(image: np.ndarray, normalize_mode: str) -> np.ndarray:
+    if normalize_mode == "openai":
+        image -= np.array(OPENAI_CLIP_MEAN, dtype=np.float32)[None, None, :]
+        image /= np.array(OPENAI_CLIP_STD, dtype=np.float32)[None, None, :]
+    elif normalize_mode == "siglip":
+        image = np.asarray(-1.0, dtype=np.float32) + image * np.asarray(2.0, dtype=np.float32)
+    elif normalize_mode == "dino":
+        image -= np.array([0.485, 0.456, 0.406], dtype=np.float32)[None, None, :]
+        image /= np.array([0.229, 0.224, 0.225], dtype=np.float32)[None, None, :]
+    else:
+        raise NotImplementedError(normalize_mode)
+    return image
+# Helper to ensure output_size is a 2-tuple of built-in Python ints
+def _ensure_pyint_size2(size):
+    """
+    Ensure `size` is a 2-tuple of built-in Python ints.
+    Accepts int, list/tuple, or numpy array of length 1 or 2.
+    """
+    import numpy as np
+    # If it's an array-like, normalize to length-2 tuple
+    if isinstance(size, (list, tuple, np.ndarray)):
+        if len(size) == 2:
+            return (int(size[0]), int(size[1]))
+        elif len(size) == 1:
+            s = int(size[0])
+            return (s, s)
+        else:
+            # Fallback: try to interpret as square size using first element
+            s = int(size[0])
+            return (s, s)
+    # Scalar → square size
+    s = int(size)
+    return (s, s)
+def resize_and_pad(
+    image,
+    desired_output_size,
+    resize_method="torch-bilinear",
+    pad_value=0,
+):
+    """Resize an image while padding to preserve uts aspect ratio."""
+    desired_output_size = _ensure_pyint_size2(desired_output_size)
+    desired_height, desired_width = desired_output_size
+    height, width = image.shape[:2]
+    # Cast into float32 since the training code did this in float32 and it (very rarely) effects
+    # the results after rounding.
+    image_scale_y = np.array(desired_height, np.float32) / np.array(height, np.float32)
+    image_scale_x = np.array(desired_width, np.float32) / np.array(width, np.float32)
+    image_scale = min(image_scale_x, image_scale_y)
+    scaled_height = int(np.array(height, np.float32) * image_scale)
+    scaled_width = int(np.array(width, np.float32) * image_scale)
+    if resize_method in ["torch-bilinear"]:
+        image = torch.permute(torch.from_numpy(image), [2, 0, 1])
+        image = convert_image_dtype(image)  # resize in float32 to match the training code
+        mode = InterpolationMode.BILINEAR
+        image = torchvision.transforms.Resize([scaled_height, scaled_width], mode, antialias=True)(image)
+        image = torch.clip(image, 0.0, 1.0)
+        image = torch.permute(image, [1, 2, 0]).numpy()
+    else:
+        raise NotImplementedError(resize_method)
+    top_pad = (desired_height - scaled_height) // 2
+    left_pad = (desired_width - scaled_width) // 2
+    padding = [
+        [top_pad, desired_height - scaled_height - top_pad],
+        [left_pad, desired_width - scaled_width - left_pad],
+        [0, 0]
+    ]
+    image_mask = np.pad(np.ones_like(image[:, :, 0], dtype=bool), padding[:2])
+    image = np.pad(image, padding, constant_values=pad_value)
+    return image, image_mask
+def metaclip_resize(image, desired_output_size):
+    desired_output_size = _ensure_pyint_size2(desired_output_size)
+    image = torch.permute(torch.from_numpy(image), [2, 0, 1])
+    if torch.is_floating_point(image):
+        image = torchvision.transforms.Resize(
+            desired_output_size, InterpolationMode.BICUBIC, antialias=True)(image)
+        image = torch.clip(image, 0.0, 1.0)
+    else:
+        assert image.dtype == torch.uint8, "Expected float images or uint8 images, but got {}".format(image.dtype)
+        image = torchvision.transforms.Resize(
+            desired_output_size, InterpolationMode.BICUBIC, antialias=True)(image)
+        image = image.to(torch.float32)
+        image = torch.clip(image, 0, 255)
+        image = image / 255.0
+    resized = torch.permute(image, [1, 2, 0]).numpy()
+    image_mask = np.ones_like(resized[:, :, 0], dtype=np.bool_)
+    return resized, image_mask
+def siglip_resize_and_pad(
+    image: np.ndarray,
+    desired_output_size: Tuple[int, int],
+) -> Tuple[np.ndarray, np.ndarray]:
+    desired_output_size = _ensure_pyint_size2(desired_output_size)
+    # by default, image is a single image
+    image = torch.permute(torch.from_numpy(image), [2, 0, 1])
+    dtype = image.dtype
+    if torch.is_floating_point(image):
+        in_min = 0.0
+        in_max = 1.0
+        resized = torchvision.transforms.Resize(
+            desired_output_size,
+            InterpolationMode.BILINEAR,
+            antialias=False,
+        )(image)
+        resized = torch.clip(resized, 0.0, 1.0).to(dtype)
+    else:
+        assert image.dtype == torch.uint8, "SigLIP expects float images or uint8 images, but got {}".format(image.dtype)
+        in_min = 0.0
+        in_max = 255.0
+        resized = torchvision.transforms.Resize(
+            desired_output_size,
+            InterpolationMode.BILINEAR,
+            antialias=False,
+        )(image)
+        resized = torch.clip(resized, 0, 255).to(dtype)
+    resized = resized.to(torch.float32)
+    resized = (resized - in_min) / (in_max - in_min)
+    resized = torch.permute(resized, [1, 2, 0]).numpy()
+    image_mask = np.ones_like(resized[:, :, 0], dtype=np.bool_)
+    return resized, image_mask
+def dino_resize_and_pad(
+    image: np.ndarray,
+    desired_output_size: Tuple[int, int],
+) -> Tuple[np.ndarray, np.ndarray]:
+    desired_output_size = _ensure_pyint_size2(desired_output_size)
+    image = torch.permute(torch.from_numpy(image), [2, 0, 1])
+    dtype = image.dtype
+    if torch.is_floating_point(image):
+        resized = torchvision.transforms.Resize(
+            desired_output_size,
+            InterpolationMode.BICUBIC,
+            antialias=True,
+        )(image)
+        resized = torch.clip(resized, 0.0, 1.0).to(torch.float32)
+    else:
+        assert image.dtype == torch.uint8, "DINOv2 expects float images or uint8 images, but got {}".format(image.dtype)
+        resized = torchvision.transforms.Resize(
+            desired_output_size,
+            InterpolationMode.BICUBIC,
+            antialias=True,
+        )(image)
+        resized = torch.clip(resized, 0, 255).to(torch.float32)
+        resized = resized / 255.0
+    resized = torch.permute(resized, [1, 2, 0]).numpy()
+    image_mask = np.ones_like(resized[:, :, 0], dtype=np.bool_)
+    return resized, image_mask
+def resize_image(
+    image: np.ndarray,
+    resize_mode: str,
+    output_size: Tuple[int, int],
+    pad_value: float,
+) -> Tuple[np.ndarray, np.ndarray]:
+    if resize_mode == "siglip":
+        return siglip_resize_and_pad(image, output_size)
+    elif resize_mode == "dino":
+        return dino_resize_and_pad(image, output_size)
+    elif resize_mode == "metaclip":
+        return metaclip_resize(image, output_size)
+    else:
+        resize = "torch-bilinear" if resize_mode == "default" else resize_mode
+        return resize_and_pad(
+            image, output_size, resize_method=resize, pad_value=pad_value,
+        )
+def select_tiling(h, w, patch_size, max_num_crops):
+    """Divide in image of size [w, h] in up to max_num_patches of size patch_size"""
+    original_size = np.stack([h, w])  # [1, 2]
+    original_res = h * w
+    tilings = []
+    for i in range(1, max_num_crops + 1):
+        for j in range(1, max_num_crops + 1):
+            if i*j <= max_num_crops:
+                tilings.append((i, j))
+    # sort so argmin and argmax favour smaller tilings in the event of a tie
+    tilings.sort(key=lambda x: (x[0]*x[1], x[0]))
+    candidate_tilings = np.array(tilings, dtype=np.int32)  # [n_resolutions, 2]
+    candidate_resolutions = candidate_tilings * patch_size  # [n_resolutions, 2]
+    # How much we would need to scale the image to fit exactly in each tiling
+    original_size = np.stack([h, w], dtype=np.float32)  # [1, 2]
+    # The original size can be zero in rare cases if the image is smaller than the margin
+    # In those cases letting the scale become infinite means the tiling is based on the
+    # other side, or falls back to the smallest tiling
+    with np.errstate(divide='ignore'):
+        required_scale_d = candidate_resolutions.astype(np.float32) / original_size,
+    required_scale = np.min(required_scale_d, axis=-1, keepdims=True)  # [n_resolutions, 1]
+    if np.all(required_scale < 1):
+        # We are forced to downscale, so try to minimize the amount of downscaling
+        ix = np.argmax(required_scale)
+    else:
+        # Pick the resolution that required the least upscaling so that it most closely fits the image
+        required_scale = np.where(required_scale < 1.0, 10e9, required_scale)
+        ix = np.argmin(required_scale)
+    return candidate_tilings[ix]
+def build_resized_image(
+    image: np.ndarray,
+    resize_mode: str,
+    normalized_mode: str,
+    base_image_input_size: List[int],
+    pad_value: float,
+    image_patch_size: int,
+) -> Tuple[np.ndarray, np.ndarray, np.ndarray]:
+    resized, resized_mask = resize_image(
+        image, resize_mode, base_image_input_size, pad_value,
+    )
+    resized = normalize_image(resized, normalized_mode)
+    if len(resized.shape) == 3:
+        resized = np.expand_dims(resized, 0)
+    resized_mask = np.expand_dims(resized_mask, 0)
+    crop_patch_w = base_image_input_size[1] // image_patch_size
+    crop_patch_h = base_image_input_size[0] // image_patch_size
+    resize_idx = np.arange(crop_patch_w*crop_patch_h).reshape([crop_patch_h, crop_patch_w])
+    return resized, resized_mask, resize_idx
+def build_overlapping_crops(
+    image: np.ndarray,
+    resize_mode: str,
+    normalize_mode: str,
+    max_crops: int,
+    overlap_margins: List[int],
+    base_image_input_size: List[int],
+    pad_value: float,
+    image_patch_size: int,
+) -> Tuple[np.ndarray, np.ndarray, np.ndarray]:
+    """Decompose an image into a set of overlapping crops
+    :return crop_arr: [n_crops, h, w, 3] The crops
+    :return mask_arr: [n_crops, h, w] The padding masks
+    :return patch_idx: [overlap_patch_h, overlap_patch_w] For each patch in the resized image
+                        the crops were extracted from, what patch in `crop_arr` it corresponds to
+    """
+    original_image_h, original_image_w = image.shape[:2]
+    crop_size = base_image_input_size[0]
+    assert base_image_input_size[0] == base_image_input_size[1]
+    left_margin, right_margin = overlap_margins
+    total_margin_pixels = image_patch_size * (right_margin + left_margin)  # pixels removed per dim
+    crop_patches = base_image_input_size[0] // image_patch_size  # patches per crop dim
+    crop_window_patches = crop_patches - (right_margin + left_margin)  # usable patches
+    crop_window_size = crop_window_patches * image_patch_size
+    crop_patch_w = base_image_input_size[1] // image_patch_size
+    crop_patch_h = base_image_input_size[0] // image_patch_size
+    original_image_h, original_image_w = image.shape[:2]
+    crop_size = base_image_input_size[0]
+    # Decide how to tile the image, to account for the overlap margins we compute the tiling
+    # as if we had an image without the margins and were using a crop size without the margins
+    tiling = select_tiling(
+        original_image_h - total_margin_pixels,
+        original_image_w - total_margin_pixels,
+        crop_window_size,
+        max_crops,
+    )
+    src, img_mask = resize_image(
+        image,
+        resize_mode,
+        [tiling[0]*crop_window_size+total_margin_pixels, tiling[1]*crop_window_size+total_margin_pixels],
+        pad_value,
+    )
+    src = normalize_image(src, normalize_mode)
+    # Now we have to split the image into crops, and track what patches came from
+    # where in `patch_idx_arr`
+    n_crops = tiling[0] * tiling[1]
+    crop_arr = np.zeros([n_crops, crop_size, crop_size, 3], dtype=src.dtype)
+    mask_arr = np.zeros([n_crops, crop_size, crop_size], dtype=img_mask.dtype)
+    patch_idx_arr = np.zeros([n_crops, crop_patch_h, crop_patch_w], dtype=np.int32)
+    on = 0
+    on_crop = 0
+    for i in range(tiling[0]):
+        # Slide over `src` by `crop_window_size` steps, but extract crops of size `crops_size`
+        # which results in overlapping crop windows
+        y0 = i*crop_window_size
+        for j in range(tiling[1]):
+            x0 = j*crop_window_size
+            crop_arr[on_crop] = src[y0:y0+crop_size, x0:x0+crop_size]
+            mask_arr[on_crop] = img_mask[y0:y0+crop_size, x0:x0+crop_size]
+            patch_idx = np.arange(crop_patch_w*crop_patch_h).reshape(crop_patch_h, crop_patch_w)
+            patch_idx += on_crop * crop_patch_h * crop_patch_w
+            # Mask out idx that are in the overlap region
+            if i != 0:
+                patch_idx[:left_margin, :] = -1
+            if j != 0:
+                patch_idx[:, :left_margin] = -1
+            if i != tiling[0]-1:
+                patch_idx[-right_margin:, :] = -1
+            if j != tiling[1]-1:
+                patch_idx[:, -right_margin:] = -1
+            patch_idx_arr[on_crop] = patch_idx
+            on_crop += 1
+    # `patch_idx_arr` is ordered crop-by-crop, here we transpose `patch_idx_arr`
+    # so it is ordered left-to-right order
+    patch_idx_arr = np.reshape(
+        patch_idx_arr,
+        [tiling[0], tiling[1], crop_patch_h, crop_patch_w]
+    )
+    patch_idx_arr = np.transpose(patch_idx_arr, [0, 2, 1, 3])
+    patch_idx_arr = np.reshape(patch_idx_arr, [-1])
+    # Now get the parts not in the overlap region, so it should map each patch in `src`
+    # to the correct patch it should come from in `crop_arr`
+    patch_idx_arr = patch_idx_arr[patch_idx_arr >= 0].reshape(
+        src.shape[0]//image_patch_size,
+        src.shape[1]//image_patch_size,
+    )
+    return crop_arr, mask_arr, patch_idx_arr
+def batch_pixels_to_patches(array: np.ndarray, patch_size: int) -> np.ndarray:
+    """Reshape images of [n_images, h, w, 3] -> [n_images, n_patches, pixels_per_patch]"""
+    if len(array.shape) == 3:
+        n_crops, h, w = array.shape
+        h_patches = h//patch_size
+        w_patches = w//patch_size
+        array = np.reshape(array, [n_crops, h_patches, patch_size, w_patches, patch_size])
+        array = np.transpose(array, [0, 1, 3, 2, 4])
+        array = np.reshape(array, [n_crops, h_patches*w_patches, patch_size*patch_size])
+        return array
+    else:
+        n_crops, h, w, c = array.shape
+        h_patches = h//patch_size
+        w_patches = w//patch_size
+        array = np.reshape(array, [n_crops, h_patches, patch_size, w_patches, patch_size, c])
+        array = np.transpose(array, [0, 1, 3, 2, 4, 5])
+        array = np.reshape(array, [n_crops, h_patches*w_patches, patch_size*patch_size*c])
+        return array
+def arange_for_pooling(
+    idx_arr: np.ndarray,
+    pool_h: int,
+    pool_w: int,
+) -> np.ndarray:
+    h_pad = pool_h * ((idx_arr.shape[0] + pool_h - 1) // pool_h) - idx_arr.shape[0]
+    w_pad = pool_w * ((idx_arr.shape[1] + pool_w - 1) // pool_w) - idx_arr.shape[1]
+    idx_arr = np.pad(idx_arr, [[h_pad//2, (h_pad+1)//2], [w_pad//2, (w_pad+1)//2]],
+                     mode='constant',constant_values=-1)
+    return einops.rearrange(
+        idx_arr, "(h dh) (w dw) -> h w (dh dw)", dh=pool_h, dw=pool_w)
+def image_to_patches_and_grids(
+    image: ImageInput,
+    crop_mode: str,
+    resize_mode: str,
+    normalize_mode: str,
+    max_crops: int,
+    overlap_margins: List[int],
+    base_image_input_size: List[int],
+    pad_value: float,
+    image_patch_size: int,
+    image_pooling_w: int,
+    image_pooling_h: int,
+) -> Tuple[np.ndarray, np.ndarray, np.ndarray, np.ndarray]:
+    """
+    :return image_grids, the shape of each (low-res, high-res) image after pooling
+    :return crops, the image crops to processes with the ViT
+    :return mask, the padding mask for each crop
+    :return pooled_patch_idx, for each patch_id tokens in `image_tokens`, the indices of the
+                                patches in `crops` to pool for that token, masked with -1
+    """
+    if isinstance(base_image_input_size, int):
+        base_image_input_size = (base_image_input_size, base_image_input_size)
+    base_image_input_d = image_patch_size
+    pooling_w = image_pooling_w
+    pooling_h = image_pooling_h
+    crop_patch_w = base_image_input_size[1] // base_image_input_d
+    crop_patch_h = base_image_input_size[0] // base_image_input_d
+    if crop_mode == "resize":
+        resized, resized_mask, resize_idx = build_resized_image(
+            image,
+            resize_mode,
+            normalize_mode,
+            base_image_input_size,
+            pad_value,
+            image_patch_size
+        )
+        pooling_idx = arange_for_pooling(resize_idx, pooling_h, pooling_w)
+        h, w = pooling_idx.shape[:2]
+        pooling_idx = pooling_idx.reshape([-1, pooling_h*pooling_w])
+        image_grid = [np.array([h, w])]
+        return (
+            np.stack(image_grid, 0),
+            batch_pixels_to_patches(resized, image_patch_size),
+            batch_pixels_to_patches(resized_mask, image_patch_size).mean(-1),
+            pooling_idx,
+        )
+    if crop_mode in ["overlap-and-resize-c2", "overlap-and-resize"]:
+        crop_arr, mask_arr, patch_idx_arr = build_overlapping_crops(
+            image,
+            resize_mode,
+            normalize_mode,
+            max_crops,
+            overlap_margins,
+            base_image_input_size,
+            pad_value,
+            image_patch_size,
+        )
+        pooling_idx = arange_for_pooling(patch_idx_arr, pooling_h, pooling_w)
+        h, w = pooling_idx.shape[:2]
+        pooling_idx = pooling_idx.reshape([-1, pooling_h*pooling_w])
+        image_grid = [np.array([h, w])]
+        if crop_mode == "overlap-and-resize":
+            crop_arr = batch_pixels_to_patches(crop_arr, image_patch_size)
+            mask_arr = batch_pixels_to_patches(mask_arr, image_patch_size).astype(np.float32).mean(axis=-1)
+            return np.stack(image_grid, 0), crop_arr, mask_arr, pooling_idx
+        # Finally do the same for the global image
+        resized, resized_mask, resize_idx = build_resized_image(
+            image,
+            resize_mode,
+            normalize_mode,
+            base_image_input_size,
+            pad_value,
+            image_patch_size
+        )
+        crop_arr = np.concatenate([resized, crop_arr], 0)
+        mask_arr = np.concatenate([resized_mask, mask_arr], 0)
+        resize_idx = arange_for_pooling(resize_idx, pooling_h, pooling_w)
+        h, w = resize_idx.shape[:2]
+        resize_idx = resize_idx.reshape([-1, pooling_h*pooling_w])
+        # Global image goes first, so the order of patches in previous crops gets increased
+        pooling_idx = np.where(
+            pooling_idx >= 0,
+            pooling_idx + crop_patch_h*crop_patch_w,
+            -1
+        )
+        pooling_idx = np.concatenate([resize_idx, pooling_idx])
+        image_grid = [
+            np.array([h, w]),
+        ] + image_grid
+        mask_arr = batch_pixels_to_patches(mask_arr, image_patch_size).astype(np.float32).mean(axis=-1)
+        return (
+            np.stack(image_grid, 0),
+            batch_pixels_to_patches(crop_arr, image_patch_size),
+            mask_arr,
+            pooling_idx
+        )
+    else:
+        raise NotImplementedError(crop_mode)
+def image_to_patches_and_tokens(
+    image: ImageInput,
+    crop_mode: str,
+    use_col_tokens: bool,
+    resize_mode: str,
+    normalize_mode: str,
+    max_crops: int,
+    overlap_margins: List[int],
+    base_image_input_size: List[int],
+    pad_value: float,
+    image_patch_size: int,
+    image_pooling_w: int,
+    image_pooling_h: int,
+    image_patch_token_id: int,
+    image_col_token_id: int,
+    image_start_token_id: int,
+    image_end_token_id: int,
+) -> Tuple[np.ndarray, np.ndarray, np.ndarray, np.ndarray]:
+    """
+    :return image_tokens, the token IDS for this image, including special tokens
+    :return crops, the image crops to processes with the ViT
+    :return mask, the padding mask for each crop
+    :return pooled_patch_idx, for each patch_id tokens in `image_tokens`, the indices of the
+                                patches in `crops` to pool for that token, masked with -1
+    """
+    if isinstance(base_image_input_size, int):
+        base_image_input_size = (base_image_input_size, base_image_input_size)
+    base_image_input_d = image_patch_size
+    pooling_w = image_pooling_w
+    pooling_h = image_pooling_h
+    patch_id = image_patch_token_id
+    col_id = image_col_token_id
+    start_id = image_start_token_id
+    end_id = image_end_token_id
+    crop_patch_w = base_image_input_size[1] // base_image_input_d
+    crop_patch_h = base_image_input_size[0] // base_image_input_d
+    if crop_mode == "resize":
+        resized, resized_mask, resize_idx = build_resized_image(
+            image,
+            resize_mode,
+            normalize_mode,
+            base_image_input_size,
+            pad_value,
+            image_patch_size
+        )
+        pooling_idx = arange_for_pooling(resize_idx, pooling_h, pooling_w)
+        h, w = pooling_idx.shape[:2]
+        pooling_idx = pooling_idx.reshape([-1, pooling_h*pooling_w])
+        per_row = np.full(
+            (w,),
+            patch_id,
+            dtype=np.int32
+        )
+        if use_col_tokens:
+            per_row = np.concatenate([per_row, [col_id]], 0)
+        extra_tokens = np.tile(per_row, [h])
+        joint = [
+            [start_id],
+            extra_tokens,
+            [end_id],
+        ]
+        return (
+            np.concatenate(joint, 0),
+            batch_pixels_to_patches(resized, image_patch_size),
+            batch_pixels_to_patches(resized_mask, image_patch_size).mean(-1),
+            pooling_idx,
+        )
+    if crop_mode in ["overlap-and-resize-c2", "overlap-and-resize"]:
+        crop_arr, mask_arr, patch_idx_arr = build_overlapping_crops(
+            image,
+            resize_mode,
+            normalize_mode,
+            max_crops,
+            overlap_margins,
+            base_image_input_size,
+            pad_value,
+            image_patch_size,
+        )
+        pooling_idx = arange_for_pooling(patch_idx_arr, pooling_h, pooling_w)
+        h, w = pooling_idx.shape[:2]
+        pooling_idx = pooling_idx.reshape([-1, pooling_h*pooling_w])
+        # Now build the output tokens
+        per_row = np.full(w, patch_id, dtype=np.int32)
+        if use_col_tokens:
+            per_row = np.concatenate([per_row, [col_id]], 0)
+        joint = np.tile(per_row, [h])
+        joint = [
+            [start_id],
+            joint,
+            [end_id]
+        ]
+        if crop_mode == "overlap-and-resize":
+            crop_arr = batch_pixels_to_patches(crop_arr, image_patch_size)
+            mask_arr = batch_pixels_to_patches(mask_arr, image_patch_size).astype(np.float32).mean(axis=-1)
+            return np.concatenate(joint, 0), crop_arr, mask_arr, pooling_idx
+        # Finally do the same for the global image
+        resized, resized_mask, resize_idx = build_resized_image(
+            image,
+            resize_mode,
+            normalize_mode,
+            base_image_input_size,
+            pad_value,
+            image_patch_size
+        )
+        crop_arr = np.concatenate([resized, crop_arr], 0)
+        mask_arr = np.concatenate([resized_mask, mask_arr], 0)
+        resize_idx = arange_for_pooling(resize_idx, pooling_h, pooling_w)
+        h, w = resize_idx.shape[:2]
+        resize_idx = resize_idx.reshape([-1, pooling_h*pooling_w])
+        # Global image goes first, so the order of patches in previous crops gets increased
+        pooling_idx = np.where(
+            pooling_idx >= 0,
+            pooling_idx + crop_patch_h*crop_patch_w,
+            -1
+        )
+        pooling_idx = np.concatenate([resize_idx, pooling_idx])
+        per_row = np.full(
+            (w,),
+            patch_id,
+            dtype=np.int32
+        )
+        if use_col_tokens:
+            per_row = np.concatenate([per_row, [col_id]], 0)
+        extra_tokens = np.tile(per_row, [h])
+        joint = [
+            [start_id],
+            extra_tokens,
+            [end_id],
+        ] + joint
+        mask_arr = batch_pixels_to_patches(mask_arr, image_patch_size).astype(np.float32).mean(axis=-1)
+        return (
+            np.concatenate(joint, 0),
+            batch_pixels_to_patches(crop_arr, image_patch_size),
+            mask_arr,
+            pooling_idx
+        )
+    else:
+        raise NotImplementedError(crop_mode)
+class MolmoActImagesKwargs(ImagesKwargs, total=False):
+    crop_mode: Optional[str]
+    resize_mode: Optional[str]
+    normalize_mode: Optional[str]
+    max_crops: Optional[int]
+    max_multi_image_crops: Optional[int]
+    overlap_margins: Optional[List[int]]
+    base_image_input_size: Optional[List[int]]
+    pad_value: Optional[float]
+    image_patch_size: Optional[int]
+    image_pooling_w: Optional[int]
+    image_pooling_h: Optional[int]
+class MolmoActImageProcessor(BaseImageProcessor):
+    model_input_names = ["images", "pooled_patches_idx", "image_masks"]
+    def __init__(
+        self,
+        crop_mode: str = "overlap-and-resize-c2",
+        resize_mode: str = "siglip",
+        normalize_mode: str = "siglip",
+        max_crops: int = 8,
+        max_multi_image_crops: int = 4,
+        overlap_margins: List[int] = [4, 4],
+        base_image_input_size: List[int] = (378, 378),
+        pad_value: float = 0.0,
+        image_patch_size: int = 14,
+        image_pooling_w: int = 2,
+        image_pooling_h: int = 2,
+        do_convert_rgb: bool = True,
+        do_pad: Optional[bool] = True,
+        **kwargs,
+    ) -> None:
+        super().__init__(**kwargs)
+        self.crop_mode = crop_mode
+        self.resize_mode = resize_mode
+        self.normalize_mode = normalize_mode
+        self.overlap_margins = overlap_margins
+        self.max_crops = max_crops
+        self.max_multi_image_crops = max_multi_image_crops
+        self.overlap_margins = overlap_margins
+        self.base_image_input_size = base_image_input_size
+        self.pad_value = pad_value
+        self.image_patch_size = image_patch_size
+        self.image_pooling_w = image_pooling_w
+        self.image_pooling_h = image_pooling_h
+        self.do_convert_rgb = do_convert_rgb
+        self.do_pad = do_pad
+    def to_channel_dimension_last(
+        self,
+        images: List[ImageInput],
+    ) -> List[ImageInput]:
+        """
+        Convert images to channel dimension last.
+        """
+        new_images = []
+        for image in images:
+            if is_multi_image(image):
+                new_images.append([to_channel_dimension_format(img, ChannelDimension.LAST) for img in image])
+            else:
+                new_images.append(to_channel_dimension_format(image, ChannelDimension.LAST))
+        return new_images
+    def to_numpy_array(
+        self,
+        images: List[ImageInput],
+    ) -> List[np.ndarray]:
+        """
+        Convert images to numpy array.
+        """
+        new_images = []
+        for image in images:
+            if is_multi_image(image):
+                new_images.append([to_numpy_array(img) for img in image])
+            else:
+                new_images.append(to_numpy_array(image))
+        return new_images
+    def to_rgb(
+        self,
+        images: List[ImageInput],
+    ) -> List[ImageInput]:
+        """
+        Convert images to RGB.
+        """
+        new_images = []
+        for image in images:
+            if is_multi_image(image):
+                new_images.append([convert_to_rgb(img) for img in image])
+            else:
+                new_images.append(convert_to_rgb(image))
+        return new_images
+    def pad_arrays(self, arrays: List[np.ndarray], pad_value: float = -1) -> np.ndarray:
+        max_len = max(arr.shape[0] for arr in arrays)
+        padded_arr = np.full(
+            [len(arrays), max_len] + list(arrays[0].shape[1:]), pad_value, dtype=arrays[0].dtype
+        )
+        for ix, arr in enumerate(arrays):
+            padded_arr[ix, :len(arr)] = arr[:max_len]
+        return padded_arr
+    def pad_for_batching(self, data: Dict[str, Any]) -> Dict[str, Any]:
+        """
+        Pad the data for batching.
+        """
+        images = self.pad_arrays(data["images"])
+        pooled_patches_idx = self.pad_arrays(data["pooled_patches_idx"])
+        image_masks = self.pad_arrays(data["image_masks"])
+        image_grids = self.pad_arrays(data["image_grids"])
+        new_data = dict(
+            images=images,
+            pooled_patches_idx=pooled_patches_idx,
+            image_masks=image_masks,
+            image_grids=image_grids,
+        )
+        return new_data
+    def preprocess(
+        self,
+        images: Union[ImageInput, List[ImageInput]],
+        crop_mode: Optional[str] = None,
+        resize_mode: Optional[str] = None,
+        normalize_mode: Optional[str] = None,
+        max_crops: Optional[int] = None,
+        max_multi_image_crops: Optional[int] = None,
+        overlap_margins: Optional[List[int]] = None,
+        base_image_input_size: Optional[List[int]] = None,
+        pad_value: Optional[float] = None,
+        image_patch_size: Optional[int] = None,
+        image_pooling_w: Optional[int] = None,
+        image_pooling_h: Optional[int] = None,
+        do_convert_rgb: Optional[bool] = None,
+        do_pad: Optional[bool] = None,
+        return_tensors: Optional[Union[str, TensorType]] = None,
+        **kwargs,
+    ) -> BatchFeature:
+        """
+        Preprocess an image for the model.
+        Args:
+            image: The image to preprocess.
+            crop_mode: The crop mode to use. If None, use the default crop mode.
+            resize_mode: The resize mode to use. If None, use the default resize mode.
+            normalize_mode: The normalization mode to use. If None, use the default normalization mode.
+            max_crops: The maximum number of crops to use. If None, use the default value.
+            max_multi_image_crops: The maximum number of crops to use for multi-image inputs.
+            overlap_margins: The overlap margins to use. If None, use the default values.
+            base_image_input_size: The base image input size to use. If None, use the default size.
+            pad_value: The padding value to use. If None, use the default value.
+            image_patch_size: The size of the image patches. If None, use the default size.
+            image_pooling_h: The height of the image pooling. If None, use the default height.
+            image_pooling_w: The width of the image pooling. If None, use the default width.
+            do_convert_rgb: Whether to convert the image to RGB. If None, use the default value.
+            do_pad: Whether to pad image features. If None, use the default value.
+        Returns:
+            A tuple containing:
+                - The image grids
+                - The preprocessed images
+                - The padding masks
+                - The pooling indices
+        """
+        images = make_batched_images(images)
+        if not valid_images(images):
+            raise ValueError("Invalid image input")
+        crop_mode = crop_mode or self.crop_mode
+        normalize_mode = normalize_mode or self.normalize_mode
+        resize_mode = resize_mode or self.resize_mode
+        max_crops = max_crops or self.max_crops
+        max_multi_image_crops = max_multi_image_crops or self.max_multi_image_crops
+        overlap_margins = overlap_margins or self.overlap_margins
+        base_image_input_size = base_image_input_size or self.base_image_input_size
+        pad_value = pad_value or self.pad_value
+        image_patch_size = image_patch_size or self.image_patch_size
+        image_pooling_w = image_pooling_w or self.image_pooling_w
+        image_pooling_h = image_pooling_h or self.image_pooling_h
+        do_convert_rgb = do_convert_rgb or self.do_convert_rgb
+        do_pad = do_pad or self.do_pad
+        if do_convert_rgb:
+            images = self.to_rgb(images)
+        # All transformations expect numpy arrays.
+        images = self.to_numpy_array(images)
+        # All transformations expect channel dimension last.
+        images = self.to_channel_dimension_last(images)
+        batch_image_grids = []
+        batch_crops = []
+        batch_crop_masks = []
+        batch_pooled_patches_idx = []
+        for image in images:
+            if is_multi_image(image):
+                all_image_grids = []
+                all_crops = []
+                all_crop_masks = []
+                pooled_patches_idx = []
+                for img in image:
+                    image_grid, crops, img_mask, pooled_idx = image_to_patches_and_grids(
+                        img,
+                        crop_mode,
+                        resize_mode,
+                        normalize_mode,
+                        max_multi_image_crops,
+                        overlap_margins,
+                        base_image_input_size,
+                        pad_value,
+                        image_patch_size,
+                        image_pooling_w,
+                        image_pooling_h,
+                    )
+                    pooled_patches_idx.append(pooled_idx + sum(np.prod(x.shape[:2]) for x in all_crops))
+                    all_crops.append(crops)
+                    all_crop_masks.append(img_mask)
+                    all_image_grids.append(image_grid)
+                all_image_grids = np.concatenate(all_image_grids, 0)
+                all_crops = np.concatenate(all_crops, 0)
+                all_crop_masks = np.concatenate(all_crop_masks, 0)
+                pooled_patches_idx = np.concatenate(pooled_patches_idx, 0)
+                batch_image_grids.append(all_image_grids)
+                batch_crops.append(all_crops)
+                batch_crop_masks.append(all_crop_masks)
+                batch_pooled_patches_idx.append(pooled_patches_idx)
+            else:
+                image_grid, crops, img_mask, pooled_idx = image_to_patches_and_grids(
+                    image,
+                    crop_mode,
+                    resize_mode,
+                    normalize_mode,
+                    max_crops,
+                    overlap_margins,
+                    base_image_input_size,
+                    pad_value,
+                    image_patch_size,
+                    image_pooling_w,
+                    image_pooling_h,
+                )
+                batch_image_grids.append(image_grid)
+                batch_crops.append(crops)
+                batch_crop_masks.append(img_mask)
+                batch_pooled_patches_idx.append(pooled_idx)
+        data =dict(
+            images=batch_crops,
+            pooled_patches_idx=batch_pooled_patches_idx,
+            image_masks=batch_crop_masks,
+            image_grids=batch_image_grids,
+        )
+        if do_pad:
+            data = self.pad_for_batching(data)
+        return BatchFeature(data, tensor_type=return_tensors)
+MolmoActImageProcessor.register_for_auto_class()

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

model-00001-of-00004.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:b260e2559d00f63d1258700e2888f431b044516c3b247e821b3a30b680669158
+size 4878581216

model-00002-of-00004.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:dbb097d41e3cc93a614b198e1e03303bb3be52aad034eb092e03d7f3f34f5827
+size 4932745864

model-00003-of-00004.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:1521db237396dde6648744022225535d3e2d5c898fd097716fc79259ca600d5e
+size 4994552920

model-00004-of-00004.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:4d9ae4ccf21dba50b95426ccfd3c08ebaa1c1a31b6c44b67dab8c7990afb9e82
+size 1433042592

model.safetensors.index.json ADDED Viewed

	@@ -0,0 +1,621 @@

+{
+  "metadata": {
+    "total_size": 16238835616
+  },
+  "weight_map": {
+    "lm_head.weight": "model-00004-of-00004.safetensors",
+    "model.transformer.blocks.0.attn_norm.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.0.ff_norm.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.0.mlp.ff_out.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.0.mlp.ff_proj.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.0.self_attn.att_proj.bias": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.0.self_attn.att_proj.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.0.self_attn.attn_out.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.1.attn_norm.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.1.ff_norm.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.1.mlp.ff_out.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.1.mlp.ff_proj.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.1.self_attn.att_proj.bias": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.1.self_attn.att_proj.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.1.self_attn.attn_out.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.10.attn_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.10.ff_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.10.mlp.ff_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.10.mlp.ff_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.10.self_attn.att_proj.bias": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.10.self_attn.att_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.10.self_attn.attn_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.11.attn_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.11.ff_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.11.mlp.ff_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.11.mlp.ff_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.11.self_attn.att_proj.bias": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.11.self_attn.att_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.11.self_attn.attn_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.12.attn_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.12.ff_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.12.mlp.ff_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.12.mlp.ff_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.12.self_attn.att_proj.bias": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.12.self_attn.att_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.12.self_attn.attn_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.13.attn_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.13.ff_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.13.mlp.ff_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.13.mlp.ff_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.13.self_attn.att_proj.bias": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.13.self_attn.att_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.13.self_attn.attn_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.14.attn_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.14.ff_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.14.mlp.ff_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.14.mlp.ff_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.14.self_attn.att_proj.bias": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.14.self_attn.att_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.14.self_attn.attn_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.15.attn_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.15.ff_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.15.mlp.ff_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.15.mlp.ff_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.15.self_attn.att_proj.bias": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.15.self_attn.att_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.15.self_attn.attn_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.16.attn_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.16.ff_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.16.mlp.ff_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.16.mlp.ff_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.16.self_attn.att_proj.bias": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.16.self_attn.att_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.16.self_attn.attn_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.17.attn_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.17.ff_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.17.mlp.ff_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.17.mlp.ff_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.17.self_attn.att_proj.bias": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.17.self_attn.att_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.17.self_attn.attn_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.18.attn_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.18.ff_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.18.mlp.ff_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.18.mlp.ff_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.18.self_attn.att_proj.bias": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.18.self_attn.att_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.18.self_attn.attn_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.19.attn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.19.ff_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.19.mlp.ff_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.19.mlp.ff_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.19.self_attn.att_proj.bias": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.19.self_attn.att_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.19.self_attn.attn_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.2.attn_norm.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.2.ff_norm.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.2.mlp.ff_out.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.2.mlp.ff_proj.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.2.self_attn.att_proj.bias": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.2.self_attn.att_proj.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.2.self_attn.attn_out.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.20.attn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.20.ff_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.20.mlp.ff_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.20.mlp.ff_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.20.self_attn.att_proj.bias": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.20.self_attn.att_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.20.self_attn.attn_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.21.attn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.21.ff_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.21.mlp.ff_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.21.mlp.ff_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.21.self_attn.att_proj.bias": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.21.self_attn.att_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.21.self_attn.attn_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.22.attn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.22.ff_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.22.mlp.ff_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.22.mlp.ff_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.22.self_attn.att_proj.bias": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.22.self_attn.att_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.22.self_attn.attn_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.23.attn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.23.ff_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.23.mlp.ff_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.23.mlp.ff_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.23.self_attn.att_proj.bias": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.23.self_attn.att_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.23.self_attn.attn_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.24.attn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.24.ff_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.24.mlp.ff_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.24.mlp.ff_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.24.self_attn.att_proj.bias": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.24.self_attn.att_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.24.self_attn.attn_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.25.attn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.25.ff_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.25.mlp.ff_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.25.mlp.ff_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.25.self_attn.att_proj.bias": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.25.self_attn.att_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.25.self_attn.attn_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.26.attn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.26.ff_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.26.mlp.ff_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.26.mlp.ff_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.26.self_attn.att_proj.bias": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.26.self_attn.att_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.26.self_attn.attn_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.27.attn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.27.ff_norm.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.27.mlp.ff_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.27.mlp.ff_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.27.self_attn.att_proj.bias": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.27.self_attn.att_proj.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.27.self_attn.attn_out.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.blocks.3.attn_norm.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.3.ff_norm.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.3.mlp.ff_out.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.3.mlp.ff_proj.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.3.self_attn.att_proj.bias": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.3.self_attn.att_proj.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.3.self_attn.attn_out.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.4.attn_norm.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.4.ff_norm.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.4.mlp.ff_out.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.4.mlp.ff_proj.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.4.self_attn.att_proj.bias": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.4.self_attn.att_proj.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.4.self_attn.attn_out.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.5.attn_norm.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.5.ff_norm.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.5.mlp.ff_out.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.5.mlp.ff_proj.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.5.self_attn.att_proj.bias": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.5.self_attn.att_proj.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.5.self_attn.attn_out.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.6.attn_norm.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.6.ff_norm.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.6.mlp.ff_out.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.6.mlp.ff_proj.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.6.self_attn.att_proj.bias": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.6.self_attn.att_proj.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.6.self_attn.attn_out.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.7.attn_norm.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.7.ff_norm.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.7.mlp.ff_out.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.7.mlp.ff_proj.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.7.self_attn.att_proj.bias": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.7.self_attn.att_proj.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.7.self_attn.attn_out.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.8.attn_norm.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.8.ff_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.8.mlp.ff_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.8.mlp.ff_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.8.self_attn.att_proj.bias": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.8.self_attn.att_proj.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.8.self_attn.attn_out.weight": "model-00001-of-00004.safetensors",
+    "model.transformer.blocks.9.attn_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.9.ff_norm.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.9.mlp.ff_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.9.mlp.ff_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.9.self_attn.att_proj.bias": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.9.self_attn.att_proj.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.blocks.9.self_attn.attn_out.weight": "model-00002-of-00004.safetensors",
+    "model.transformer.ln_f.weight": "model-00003-of-00004.safetensors",
+    "model.transformer.wte.embedding": "model-00001-of-00004.safetensors",
+    "model.transformer.wte.new_embedding": "model-00001-of-00004.safetensors",
+    "model.vision_backbone.image_pooling_2d.wk.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_pooling_2d.wk.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_pooling_2d.wo.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_pooling_2d.wo.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_pooling_2d.wq.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_pooling_2d.wq.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_pooling_2d.wv.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_pooling_2d.wv.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_projector.w1.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_projector.w2.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_projector.w3.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.patch_embedding.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.patch_embedding.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.positional_embedding": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.0.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.0.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.0.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.0.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.0.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.0.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.0.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.0.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.0.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.0.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.0.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.0.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.0.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.0.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.0.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.0.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.1.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.1.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.1.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.1.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.1.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.1.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.1.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.1.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.1.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.1.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.1.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.1.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.1.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.1.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.1.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.1.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.10.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.10.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.10.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.10.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.10.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.10.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.10.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.10.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.10.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.10.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.10.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.10.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.10.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.10.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.10.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.10.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.11.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.11.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.11.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.11.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.11.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.11.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.11.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.11.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.11.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.11.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.11.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.11.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.11.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.11.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.11.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.11.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.12.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.12.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.12.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.12.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.12.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.12.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.12.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.12.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.12.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.12.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.12.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.12.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.12.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.12.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.12.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.12.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.13.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.13.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.13.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.13.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.13.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.13.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.13.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.13.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.13.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.13.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.13.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.13.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.13.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.13.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.13.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.13.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.14.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.14.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.14.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.14.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.14.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.14.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.14.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.14.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.14.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.14.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.14.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.14.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.14.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.14.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.14.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.14.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.15.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.15.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.15.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.15.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.15.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.15.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.15.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.15.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.15.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.15.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.15.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.15.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.15.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.15.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.15.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.15.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.16.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.16.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.16.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.16.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.16.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.16.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.16.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.16.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.16.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.16.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.16.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.16.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.16.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.16.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.16.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.16.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.17.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.17.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.17.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.17.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.17.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.17.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.17.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.17.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.17.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.17.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.17.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.17.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.17.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.17.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.17.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.17.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.18.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.18.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.18.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.18.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.18.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.18.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.18.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.18.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.18.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.18.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.18.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.18.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.18.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.18.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.18.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.18.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.19.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.19.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.19.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.19.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.19.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.19.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.19.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.19.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.19.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.19.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.19.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.19.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.19.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.19.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.19.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.19.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.2.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.2.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.2.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.2.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.2.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.2.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.2.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.2.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.2.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.2.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.2.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.2.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.2.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.2.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.2.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.2.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.20.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.20.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.20.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.20.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.20.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.20.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.20.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.20.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.20.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.20.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.20.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.20.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.20.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.20.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.20.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.20.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.21.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.21.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.21.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.21.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.21.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.21.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.21.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.21.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.21.attention_norm.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.21.attention_norm.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.21.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.21.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.21.feed_forward.w2.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.21.feed_forward.w2.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.21.ffn_norm.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.21.ffn_norm.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.22.attention.wk.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.22.attention.wk.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.22.attention.wo.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.22.attention.wo.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.22.attention.wq.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.22.attention.wq.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.22.attention.wv.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.22.attention.wv.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.22.attention_norm.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.22.attention_norm.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.22.feed_forward.w1.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.22.feed_forward.w1.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.22.feed_forward.w2.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.22.feed_forward.w2.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.22.ffn_norm.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.22.ffn_norm.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.23.attention.wk.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.23.attention.wk.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.23.attention.wo.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.23.attention.wo.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.23.attention.wq.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.23.attention.wq.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.23.attention.wv.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.23.attention.wv.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.23.attention_norm.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.23.attention_norm.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.23.feed_forward.w1.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.23.feed_forward.w1.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.23.feed_forward.w2.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.23.feed_forward.w2.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.23.ffn_norm.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.23.ffn_norm.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.24.attention.wk.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.24.attention.wk.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.24.attention.wo.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.24.attention.wo.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.24.attention.wq.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.24.attention.wq.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.24.attention.wv.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.24.attention.wv.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.24.attention_norm.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.24.attention_norm.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.24.feed_forward.w1.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.24.feed_forward.w1.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.24.feed_forward.w2.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.24.feed_forward.w2.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.24.ffn_norm.bias": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.24.ffn_norm.weight": "model-00004-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.3.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.3.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.3.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.3.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.3.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.3.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.3.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.3.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.3.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.3.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.3.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.3.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.3.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.3.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.3.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.3.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.4.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.4.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.4.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.4.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.4.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.4.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.4.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.4.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.4.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.4.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.4.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.4.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.4.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.4.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.4.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.4.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.5.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.5.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.5.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.5.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.5.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.5.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.5.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.5.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.5.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.5.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.5.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.5.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.5.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.5.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.5.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.5.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.6.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.6.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.6.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.6.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.6.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.6.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.6.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.6.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.6.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.6.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.6.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.6.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.6.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.6.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.6.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.6.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.7.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.7.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.7.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.7.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.7.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.7.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.7.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.7.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.7.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.7.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.7.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.7.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.7.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.7.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.7.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.7.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.8.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.8.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.8.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.8.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.8.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.8.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.8.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.8.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.8.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.8.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.8.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.8.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.8.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.8.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.8.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.8.ffn_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.9.attention.wk.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.9.attention.wk.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.9.attention.wo.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.9.attention.wo.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.9.attention.wq.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.9.attention.wq.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.9.attention.wv.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.9.attention.wv.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.9.attention_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.9.attention_norm.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.9.feed_forward.w1.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.9.feed_forward.w1.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.9.feed_forward.w2.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.9.feed_forward.w2.weight": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.9.ffn_norm.bias": "model-00003-of-00004.safetensors",
+    "model.vision_backbone.image_vit.transformer.resblocks.9.ffn_norm.weight": "model-00003-of-00004.safetensors"
+  }
+}

model.yaml ADDED Viewed

	@@ -0,0 +1,195 @@

+model_name: molmo
+llm:
+  d_model: 3584
+  n_heads: 28
+  n_kv_heads: 4
+  head_dim: null
+  qkv_bias: true
+  clip_qkv: null
+  n_layers: 28
+  mlp_ratio: 4
+  mlp_hidden_size: 37888
+  activation_type: swiglu
+  block_type: sequential
+  rope: true
+  rope_full_precision: true
+  rope_theta: 1000000.0
+  rope_type: default
+  rope_factor: null
+  rope_high_freq_factor: null
+  rope_low_freq_factor: null
+  rope_original_max_position_embeddings: null
+  attention_type: sdpa
+  float32_attention: true
+  attention_dropout: 0.0
+  attention_layer_norm: false
+  attention_layer_norm_type: olmo
+  residual_dropout: 0.1
+  response_residual_dropout: 0.0
+  layer_norm_type: rms
+  layer_norm_with_affine: true
+  layer_norm_eps: 1.0e-06
+  attention_layer_norm_with_affine: true
+  max_sequence_length: 4096
+  max_position_embeddings: null
+  include_bias: false
+  bias_for_layer_norm: null
+  norm_after: false
+  moe_num_experts: 8
+  moe_top_k: 2
+  moe_mlp_impl: sparse
+  moe_log_expert_assignment: false
+  moe_shared_expert: false
+  moe_lbl_in_fp32: false
+  moe_interleave: false
+  moe_loss_weight: 0.1
+  moe_zloss_weight: null
+  moe_dropless: true
+  moe_capacity_factor: 1.25
+  embedding_dropout: 0.0
+  scale_logits: false
+  vocab_size: 152064
+  additional_vocab_size: 128
+  weight_tying: false
+  embedding_size: 152064
+  use_position_ids: true
+  tokenizer:
+    identifier: Qwen/Qwen2.5-7B
+    tokenizer_dir: null
+    depth_tokens: true
+  init_path: gs://mm-olmo/pretrained_llms/qwen2.5-7b.pt
+  init_incremental: null
+  new_embedding_init_range: 0.02
+  initializer_range: 0.02
+  normalize_input_embeds: false
+  activation_checkpoint: whole_layer
+  compile: blocks
+  fix_pad_tokenizer: false
+  resize_vocab: false
+  init_std: 0.02
+  init_fn: normal
+  init_cutoff_factor: null
+vision_backbone:
+  vit:
+    image_model_type: siglip
+    image_default_input_size:
+    - 378
+    - 378
+    image_patch_size: 14
+    image_pos_patch_size: 14
+    image_emb_dim: 1152
+    image_num_heads: 16
+    image_num_key_value_heads: 16
+    image_num_layers: 27
+    image_head_dim: 72
+    image_mlp_dim: 4304
+    image_mlp_activations: gelu_pytorch_tanh
+    image_dropout_rate: 0.0
+    image_num_pos: 729
+    image_norm_eps: 1.0e-06
+    attention_dropout: 0.0
+    residual_dropout: 0.0
+    initializer_range: 0.02
+    float32_attention: true
+    attention_type: sdpa
+    activation_checkpointing: true
+    init_path: gs://mm-olmo/pretrained_image_encoders/siglip2-so400m-14-384.pt
+    resize_mode: siglip
+    pad_value: 0.0
+    normalize: siglip
+  image_pooling_2d: attention_meanq
+  pooling_attention_mask: false
+  image_projector: mlp
+  image_padding_embed: null
+  vit_layers:
+  - -3
+  - -9
+  skip_unused_layers: true
+  image_feature_dropout: 0.0
+  connector_activation_checkpointing: true
+  compile_vit: blocks
+data_formatter:
+  prompt_templates: uber_model
+  message_format: role
+  system_prompt: demo_or_style
+  always_start_with_space: false
+  default_inference_len: 65
+  select_answer: best
+  debug: false
+  image_last: false
+  format_message_list: null
+  p_one_message: 0.0
+mm_preprocessor:
+  crop_mode: overlap-and-resize-c2
+  max_crops: 8
+  max_images: 2
+  max_multi_image_crops: 8
+  pooling_w: 2
+  pooling_h: 2
+  overlap_margins:
+  - 4
+  - 4
+  use_col_tokens: true
+  loss_token_weighting: root_subsegments
+  legacy_image_mask: false
+  max_answer_len: null
+  img_aug: true
+bi_directional_attn: null
+lora_enable: true
+lora_rank: 32
+lora_alpha: 16
+lora_dropout: 0.0
+lora_bias: none
+n_action_bins: 256
+norm_stats:
+  molmoact:
+    action:
+      mean:
+      - 0.0005706787342205644
+      - 0.0002448957529850304
+      - -3.5987635783385485e-05
+      - 0.00021597897284664214
+      - -0.0004896928439848125
+      - -0.000241481073317118
+      - 0.5570635199546814
+      std:
+      - 0.005207270849496126
+      - 0.007506529800593853
+      - 0.006415561307221651
+      - 0.013248044066131115
+      - 0.010928540490567684
+      - 0.014873150736093521
+      - 0.49715080857276917
+      min:
+      - -0.07434078305959702
+      - -0.07339745759963989
+      - -0.06539416313171387
+      - -0.1688285619020462
+      - -0.10289879888296127
+      - -0.2667275667190552
+      - 0.0
+      max:
+      - 0.06042003631591797
+      - 0.09417290985584259
+      - 0.07019275426864624
+      - 0.2616892158985138
+      - 0.11751057207584381
+      - 0.16968433558940887
+      - 1.0
+      q01:
+      - -0.01538565568625927
+      - -0.021047022193670273
+      - -0.01688069850206375
+      - -0.044314172118902206
+      - -0.03890235349535942
+      - -0.04788423702120781
+      - 0.0
+      q99:
+      - 0.014661382883787155
+      - 0.026515591889619827
+      - 0.021398313343524933
+      - 0.04216696694493294
+      - 0.03401297703385353
+      - 0.04957397282123566
+      - 1.0
+    num_entries: 1560068

modeling_molmoact.py ADDED Viewed

	@@ -0,0 +1,2124 @@

+import math
+from copy import deepcopy
+from dataclasses import dataclass
+from typing import List, Optional, Tuple, Union, Dict, Any, Sequence, Callable
+import torch
+from torch import nn
+from torch.nn import functional as F
+from contextlib import nullcontext
+from transformers.models.auto import AutoModelForCausalLM, AutoModelForImageTextToText
+from transformers.activations import ACT2FN
+from transformers.cache_utils import Cache, DynamicCache
+from transformers.generation import GenerationMixin
+from transformers.generation.configuration_utils import GenerationConfig
+from transformers.generation.utils import GenerateOutput
+from transformers.integrations import use_kernel_forward_from_hub
+from transformers.modeling_attn_mask_utils import AttentionMaskConverter
+from transformers.modeling_flash_attention_utils import _flash_attention_forward, FlashAttentionKwargs
+from transformers import GradientCheckpointingLayer
+from transformers.modeling_outputs import (
+    BaseModelOutput,
+    BaseModelOutputWithPast,
+    BaseModelOutputWithPooling,
+    CausalLMOutputWithPast,
+)
+from transformers.modeling_rope_utils import ROPE_INIT_FUNCTIONS, dynamic_rope_update
+from transformers.modeling_utils import ALL_ATTENTION_FUNCTIONS, PreTrainedModel
+from transformers.processing_utils import Unpack
+from transformers.utils import (
+    ModelOutput,
+    can_return_tuple,
+    is_torch_flex_attn_available,
+    logging,
+    add_start_docstrings,
+    add_start_docstrings_to_model_forward,
+)
+from .configuration_molmoact import MolmoActConfig, MolmoActVitConfig, MolmoActAdapterConfig, MolmoActLlmConfig
+import re
+import numpy as np
+from transformers import Qwen2Tokenizer
+if is_torch_flex_attn_available():
+    from torch.nn.attention.flex_attention import BlockMask
+    from transformers.integrations.flex_attention import make_flex_block_causal_mask
+logger = logging.get_logger(__name__)
+MOLMO_START_DOCSTRING = r"""
+    This model inherits from [`PreTrainedModel`]. Check the superclass documentation for the generic methods the
+    library implements for all its model (such as downloading or saving, resizing the input embeddings, pruning heads
+    etc.)
+    This model is also a PyTorch [torch.nn.Module](https://pytorch.org/docs/stable/nn.html#torch.nn.Module) subclass.
+    Use it as a regular PyTorch Module and refer to the PyTorch documentation for all matter related to general usage
+    and behavior.
+    Parameters:
+        config ([`MolmoActConfig`]):
+            Model configuration class with all the parameters of the model. Initializing with a config file does not
+            load the weights associated with the model, only the configuration. Check out the
+            [`~PreTrainedModel.from_pretrained`] method to load the model weights.
+"""
+NUM_RE = re.compile(r'[+-]?(?:\d+(?:\.\d+)?|\.\d+)(?:[eE][+-]?\d+)?$')
+DEPTH_RE = re.compile(r'<DEPTH_START>(.*?)<DEPTH_END>', re.DOTALL)
+# One-level-nested [...] matcher: outer block that may contain inner [ ... ] lists
+OUTER_BLOCK_RE = re.compile(r'\[(?:[^\[\]]|\[[^\[\]]*\])+\]')
+def _is_number(s: str) -> bool:
+    return bool(NUM_RE.match(s))
+def _has_non_ascii(s: str) -> bool:
+    return any(ord(ch) > 127 for ch in s)
+def _to_number(s: str):
+    """Parse string number to int when possible, else float."""
+    v = float(s)
+    return int(v) if v.is_integer() else v
+def extract_depth_string(text: str, include_tags: bool = False) -> list[str]:
+    """
+    Return all occurrences of depth strings.
+    If include_tags=True, each item is '<DEPTH_START>...<DEPTH_END>';
+    otherwise each item is just the inner '...'.
+    """
+    matches = list(DEPTH_RE.finditer(text))
+    if include_tags:
+        return [m.group(0) for m in matches]
+    return [m.group(1) for m in matches]
+def extract_trace_lists(
+    text: str,
+    point_len: int | None = 2,     # e.g., 2 for [x,y], 3 for [x,y,z]; None = any length ≥1
+    min_points: int = 1
+) -> list[list[list[float]]]:
+    """
+    Extract *numeric* lists-of-lists like [[140,225],[130,212],...].
+    Returns a list of traces; each trace is a list of points (lists of numbers).
+    Heuristic:
+      - Find outer [ ... ] blocks that may contain inner lists
+      - Keep blocks where every inner list is fully numeric
+      - Enforce per-point length (point_len) and a minimum number of points (min_points)
+    """
+    traces: list[list[list[float]]] = []
+    # Find outer blocks that can contain nested lists
+    for block in OUTER_BLOCK_RE.findall(text):
+        inner_strs = re.findall(r'\[([^\[\]]+)\]', block)  # contents of each inner [...]
+        if len(inner_strs) < min_points:
+            continue
+        rows: list[list[float]] = []
+        ok = True
+        for row in inner_strs:
+            parts = [p.strip().strip('"').strip("'") for p in row.split(',')]
+            if point_len is not None and len(parts) != point_len:
+                ok = False
+                break
+            if not all(_is_number(p) for p in parts):
+                ok = False
+                break
+            rows.append([_to_number(p) for p in parts])
+        if ok:
+            traces.append(rows)
+    return traces
+def extract_action_token_lists(
+    text: str,
+    only_len: int | None = None,         # e.g., 7 if you expect 7-D actions
+    require_non_ascii: bool = True       # set False if your tokens can be pure ASCII
+) -> list[list[str]]:
+    """
+    Extract all [ ... ] groups split by commas, discard numeric lists,
+    and return token lists (quotes stripped, whitespace trimmed).
+    """
+    lists = []
+    # Match NON-nested bracketed groups: [ ... ] without inner [ or ]
+    for inner in re.findall(r'\[([^\[\]]+)\]', text):
+        parts = [p.strip().strip('"').strip("'") for p in inner.split(',')]
+        if only_len is not None and len(parts) != only_len:
+            continue
+        # If *all* items are numeric -> not action tokens (like coordinates)
+        if all(_is_number(p) for p in parts):
+            continue
+        # Optionally require at least one non-ASCII char across tokens (helps exclude plain words/numbers)
+        if require_non_ascii and not any(_has_non_ascii(p) for p in parts):
+            continue
+        lists.append(parts)
+    return lists
+@dataclass
+class MolmoActCausalLMOutputWithPast(ModelOutput):
+    """
+    Base class for MolmoAct causal language model (or autoregressive) outputs.
+    Args:
+        loss (`torch.FloatTensor` of shape `(1,)`, *optional*, returned when `labels` is provided):
+            Language modeling loss (for next-token prediction).
+        logits (`torch.FloatTensor` of shape `(batch_size, sequence_length, config.vocab_size)`):
+            Prediction scores of the language modeling head (scores for each vocabulary token before SoftMax).
+        past_key_values (`tuple(tuple(torch.FloatTensor))`, *optional*, returned when `use_cache=True` is passed or when `config.use_cache=True`):
+            Tuple of `tuple(torch.FloatTensor)` of length `config.n_layers`, with each tuple having 2 tensors of shape
+            `(batch_size, num_heads, sequence_length, embed_size_per_head)`)
+            Contains pre-computed hidden-states (key and values in the self-attention blocks) that can be used (see
+            `past_key_values` input) to speed up sequential decoding.
+        hidden_states (`tuple(torch.FloatTensor)`, *optional*, returned when `output_hidden_states=True` is passed or when `config.output_hidden_states=True`):
+            Tuple of `torch.FloatTensor` (one for the output of the embeddings, if the model has an embedding layer, +
+            one for the output of each layer) of shape `(batch_size, sequence_length, hidden_size)`.
+            Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.
+        attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
+            Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length,
+            sequence_length)`.
+            Attentions weights after the attention softmax, used to compute the weighted average in the self-attention
+            heads.
+        image_hidden_states (`torch.FloatTensor`, *optional*):
+            A `torch.FloatTensor` of size `(batch_size, num_images, sequence_length, hidden_size)`.
+            image_hidden_states of the model produced by the vision encoder and after projecting the last hidden state.
+    """
+    loss: Optional[torch.FloatTensor] = None
+    logits: Optional[torch.FloatTensor] = None
+    past_key_values: Optional[List[torch.FloatTensor]] = None
+    hidden_states: Optional[Tuple[torch.FloatTensor]] = None
+    attentions: Optional[Tuple[torch.FloatTensor]] = None
+    image_hidden_states: Optional[torch.FloatTensor] = None
+@dataclass
+class MolmoActModelOutputWithPast(BaseModelOutputWithPast):
+    """
+    Base class for MolmoAct outputs, with hidden states and attentions.
+    Args:
+        last_hidden_state (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`):
+            Sequence of hidden-states at the output of the last layer of the model.
+        past_key_values (`tuple(tuple(torch.FloatTensor))`, *optional*, returned when `use_cache=True` is passed or when `config.use_cache=True`):
+            Tuple of `tuple(torch.FloatTensor)` of length `config.n_layers`, with each tuple having 2 tensors of shape
+            `(batch_size, num_heads, sequence_length, embed_size_per_head)`)
+            Contains pre-computed hidden-states (key and values in the self-attention blocks) that can be used (see
+            `past_key_values` input) to speed up sequential decoding.
+        hidden_states (`tuple(torch.FloatTensor)`, *optional*, returned when `output_hidden_states=True` is passed or when `config.output_hidden_states=True`):
+            Tuple of `torch.FloatTensor` (one for the output of the embeddings, if the model has an embedding layer, +
+            one for the output of each layer) of shape `(batch_size, sequence_length, hidden_size)`.
+            Hidden-states of the model at the output of each layer plus the optional initial embedding outputs.
+        attentions (`tuple(torch.FloatTensor)`, *optional*, returned when `output_attentions=True` is passed or when `config.output_attentions=True`):
+            Tuple of `torch.FloatTensor` (one for each layer) of shape `(batch_size, num_heads, sequence_length,
+            sequence_length)`.
+            Attentions weights after the attention softmax, used to compute the weighted average in the self-attention
+            heads.
+        image_hidden_states (`torch.FloatTensor`, *optional*):
+            A `torch.FloatTensor` of size `(batch_num_patches, hidden_size)`.
+            image_hidden_states of the model produced by the vision backbone
+    """
+    image_hidden_states: Optional[torch.FloatTensor] = None
+    logits: Optional[torch.FloatTensor] = None
+class MolmoActPreTrainedModel(PreTrainedModel):
+    config_class = MolmoActLlmConfig
+    base_model_prefix = "model"
+    supports_gradient_checkpointing = True
+    _no_split_modules = ["MolmoActDecoderLayer", "MolmoActPostNormDecoderLayer"]
+    _skip_keys_device_placement = ["past_key_values"]
+    _supports_flash_attn_2 = True
+    _supports_sdpa = True
+    _supports_flex_attn = False
+    _supports_cache_class = True
+    _supports_quantized_cache = True
+    _supports_static_cache = True
+    _supports_attention_backend = True
+    def _init_weights(self, module):
+        std = self.config.initializer_range
+        if isinstance(module, (nn.Linear,)):
+            module.weight.data.normal_(mean=0.0, std=std)
+            if module.bias is not None:
+                module.bias.data.zero_()
+        elif isinstance(module, MolmoActEmbedding):
+            module.embedding.data.normal_(mean=0.0, std=std)
+            module.new_embedding.data.normal_(mean=0.0, std=std)
+        elif isinstance(module, nn.Embedding):
+            module.weight.data.normal_(mean=0.0, std=std)
+            if module.padding_idx is not None:
+                module.weight.data[module.padding_idx].zero_()
+        elif isinstance(module, MolmoActRMSNorm):
+            module.weight.data.fill_(1.0)
+        elif isinstance(module, nn.LayerNorm):
+            module.weight.data.fill_(1.0)
+            if module.bias is not None:
+                module.bias.data.zero_()
+class ViTMLP(nn.Module):
+    def __init__(self, dim: int, hidden_dim: int, hidden_act: str, device: Union[str, torch.device] = None):
+        super().__init__()
+        self.w1 = nn.Linear(dim, hidden_dim, bias=True, device=device)
+        self.act = ACT2FN[hidden_act]
+        self.w2 = nn.Linear(hidden_dim, dim, bias=True, device=device)
+    def forward(self, x: torch.Tensor) -> torch.Tensor:
+        return self.w2(self.act(self.w1(x)))
+class ViTMultiHeadDotProductAttention(nn.Module):
+    def __init__(
+        self,
+        hidden_size: int,
+        num_heads: int,
+        num_key_value_heads: int,
+        head_dim: int,
+        use_bias: bool = True,
+        input_dim: Optional[int] = None,
+        float32_attention: bool = True,
+        attention_dropout: float = 0.0,
+        residual_dropout: float = 0.0,
+        device: Union[str, torch.device] = None,
+        attn_implementation: str = "eager",
+    ):
+        super().__init__()
+        self.hidden_size = hidden_size
+        self.num_heads = num_heads
+        self.head_dim = head_dim
+        self.num_key_value_heads = num_key_value_heads
+        self.num_key_value_groups = self.num_heads // self.num_key_value_heads
+        self.attn_implementation = attn_implementation
+        self.is_causal = False
+        input_dim = input_dim or hidden_size
+        self.wq = nn.Linear(
+            input_dim,
+            self.num_heads * self.head_dim,
+            bias=use_bias,
+            device=device,
+        )
+        self.wk = nn.Linear(
+            input_dim,
+            self.num_key_value_heads * self.head_dim,
+            bias=use_bias,
+            device=device,
+        )
+        self.wv = nn.Linear(
+            input_dim,
+            self.num_key_value_heads * self.head_dim,
+            bias=use_bias,
+            device=device,
+        )
+        self.wo = nn.Linear(
+            self.num_heads * self.head_dim,
+            self.hidden_size,
+        )
+        self.float32_attention = float32_attention
+        self.attention_dropout = attention_dropout
+        self.residual_dropout = nn.Dropout(residual_dropout)
+    def _split_heads(self, hidden_states, num_heads) -> torch.Tensor:
+        return hidden_states.reshape(hidden_states.shape[:2] + (num_heads, self.head_dim))
+    def _merge_heads(self, hidden_states) -> torch.Tensor:
+        return hidden_states.reshape(hidden_states.shape[:2] + (self.hidden_size,))
+    def forward(
+        self,
+        inputs_q: torch.Tensor,
+        inputs_kv: Optional[torch.Tensor] = None,
+        attn_mask: Optional[torch.Tensor] = None,
+    ) -> torch.Tensor:
+        if inputs_kv is not None:
+            inputs_k = inputs_kv
+            inputs_v = inputs_kv
+        else:
+            inputs_k = inputs_q
+            inputs_v = inputs_q
+        xq, xk, xv = self.wq(inputs_q), self.wk(inputs_k), self.wv(inputs_v)
+        xq = self._split_heads(xq, self.num_heads)
+        xk = self._split_heads(xk, self.num_key_value_heads)
+        xv = self._split_heads(xv, self.num_key_value_heads)
+        if self.num_heads != self.num_key_value_heads:
+            xk = xk.repeat_interleave(self.num_key_value_groups, dim=2, output_size=self.num_heads)
+            xv = xv.repeat_interleave(self.num_key_value_groups, dim=2, output_size=self.num_heads)
+        og_dtype = xq.dtype
+        if self.float32_attention:
+            xq = xq.to(torch.float)
+            xk = xk.to(torch.float)
+            xv = xv.to(torch.float)
+        elif self.attn_implementation == "sdpa" and not torch.is_autocast_enabled():
+            xv = xv.to(torch.float)
+        dropout_p = 0.0 if not self.training else self.attention_dropout
+        if self.attn_implementation == "eager":
+            attn_weights = torch.einsum("...qhd,...khd->...hqk", xq / math.sqrt(xq.size(-1)), xk)
+            attn_weights = F.softmax(attn_weights, dim=-1)
+            attn_weights = F.dropout(
+                attn_weights,
+                p=dropout_p,
+                training=self.training
+            )
+            attn_output = torch.einsum("...hqk,...khd->...qhd", attn_weights.to(xv.dtype), xv)
+        elif self.attn_implementation == "sdpa":
+            if not torch.is_autocast_enabled():
+                xv = xv.to(torch.float)
+            flash_ok = (
+                attn_mask is None
+                and xq.dtype in (torch.float16, torch.bfloat16)
+                and xk.dtype == xq.dtype
+                and xv.dtype == xq.dtype
+            )
+            sdp_ctx = (
+                torch.backends.cuda.sdp_kernel(
+                    enable_flash=flash_ok,
+                    enable_mem_efficient=True,
+                    enable_math=True,
+                    enable_cudnn=True,
+                )
+                if hasattr(torch.backends.cuda, "sdp_kernel")
+                else nullcontext()
+            )
+            with sdp_ctx:
+                attn_output = F.scaled_dot_product_attention(
+                    xq.transpose(1, 2).contiguous(),
+                    xk.transpose(1, 2).contiguous(),
+                    xv.transpose(1, 2).contiguous(),
+                    attn_mask=attn_mask,
+                    is_causal=False,
+                    dropout_p=dropout_p,
+                ).transpose(1, 2)
+        elif self.attn_implementation == "flash_attention_2":
+            assert not self.config.float32_attention
+            # Downcast in case we are running with fp32 hidden states
+            attn_output = _flash_attention_forward(
+                xq.transpose(1, 2).to(torch.bfloat16),
+                xk.transpose(1, 2).to(torch.bfloat16),
+                xv.transpose(1, 2).to(torch.bfloat16),
+                attention_mask=None,
+                query_length=inputs_q.shape[1],
+                is_causal=False,
+                dropout=dropout_p,
+            )
+        else:
+            raise ValueError(f"Attention implementation {self.attn_implementation} not supported")
+        attn_output = attn_output.to(og_dtype)
+        attn_output = self._merge_heads(attn_output)
+        attn_output = self.wo(attn_output)
+        attn_output = self.residual_dropout(attn_output)
+        return attn_output
+class MolmoActVisionBlock(nn.Module):
+    def __init__(self, config: MolmoActVitConfig, device: Union[str, torch.device] = None):
+        super().__init__()
+        self.attention = ViTMultiHeadDotProductAttention(
+            hidden_size=config.hidden_size,
+            num_heads=config.num_attention_heads,
+            num_key_value_heads=config.num_key_value_heads,
+            head_dim=config.head_dim,
+            float32_attention=config.float32_attention,
+            attention_dropout=config.attention_dropout,
+            residual_dropout=config.residual_dropout,
+            device=device,
+            attn_implementation=config._attn_implementation,
+        )
+        self.feed_forward = ViTMLP(config.hidden_size, config.intermediate_size, config.hidden_act, device=device)
+        self.attention_norm = nn.LayerNorm(config.hidden_size, eps=config.layer_norm_eps, device=device)
+        self.ffn_norm = nn.LayerNorm(config.hidden_size, eps=config.layer_norm_eps, device=device)
+    def forward(self, x: torch.Tensor) -> torch.Tensor:
+        x = x + self.attention(self.attention_norm(x))
+        x = x + self.feed_forward(self.ffn_norm(x))
+        return x
+class MolmoActVisionBlockCollection(nn.Module):
+    def __init__(self, config: MolmoActVitConfig, device: Union[str, torch.device] = None):
+        super().__init__()
+        self.conifg = config
+        self.resblocks = nn.ModuleList([
+            MolmoActVisionBlock(config, device) for _ in range(config.num_hidden_layers)
+        ])
+    def forward(self, x: torch.Tensor) -> List[torch.Tensor]:
+        hidden_states = []
+        for r in self.resblocks:
+            x = r(x)
+            hidden_states.append(x)
+        return hidden_states
+def _expand_token(token, batch_size: int):
+    return token.view(1, 1, -1).expand(batch_size, -1, -1)
+class MolmoActVisionTransformer(nn.Module):
+    def __init__(self, config: MolmoActVitConfig, device: Union[str, torch.device] = None):
+        super().__init__()
+        self.config = config
+        self.scale = config.hidden_size ** -0.5
+        # optional CLS
+        self.num_prefix_tokens: int = 1 if config.use_cls_token else 0
+        if config.use_cls_token:
+            self.class_embedding = nn.Parameter(
+                torch.zeros(config.hidden_size, device=device)
+            )
+        # positional embeddings
+        self.positional_embedding = nn.Parameter(
+            torch.zeros(config.image_num_pos, config.hidden_size, device=device),
+        )
+        image_patch_size = config.image_patch_size
+        self.patch_embedding = nn.Linear(
+            image_patch_size * image_patch_size * 3,
+            config.hidden_size,
+            bias=config.patch_bias,
+            device=device,
+        )
+        # optional pre-LN
+        self.pre_ln = nn.LayerNorm(config.hidden_size, eps=config.layer_norm_eps, device=device) \
+                      if config.pre_layernorm else None
+        self.transformer = MolmoActVisionBlockCollection(config, device)
+    def add_pos_emb(self, x: torch.Tensor, patch_num: int) -> torch.Tensor:
+        pos_emb = self.positional_embedding
+        if self.config.use_cls_token:
+            cls_pos, pos_emb = pos_emb[:1], pos_emb[1:]   # split out CLS
+        pos_emb = pos_emb.reshape(
+            (int(math.sqrt(pos_emb.shape[0])), int(math.sqrt(pos_emb.shape[0])), pos_emb.shape[1])
+        )
+        (patch_num_0, patch_num_1) = patch_num
+        if pos_emb.shape[0] != patch_num_0 or pos_emb.shape[1] != patch_num_1:
+            # Dervied from https://github.com/facebookresearch/mae/blob/main/util/pos_embed.py
+            # antialias: default True in jax.image.resize
+            pos_emb = pos_emb.unsqueeze(0).permute(0, 3, 1, 2)
+            pos_emb = F.interpolate(
+                pos_emb, size=(patch_num_0, patch_num_1), mode="bicubic", align_corners=False, antialias=True,
+            )
+            pos_emb = pos_emb.permute(0, 2, 3, 1).squeeze(0)
+        pos_emb = pos_emb.reshape(-1, pos_emb.shape[-1])
+        if self.config.use_cls_token:
+            x = x + torch.cat([cls_pos[None, :, :], pos_emb[None, :, :]], dim=1).to(x.dtype)
+        else:
+            x = x + pos_emb[None, :, :].to(x.dtype)
+        return x
+    def forward(self, x: torch.Tensor, patch_num: int = None) -> List[torch.Tensor]:
+        """
+        : param x: (batch_size, num_patch, n_pixels)
+        """
+        if patch_num is None:
+            patch_num = self.config.image_num_patch
+        B, N, D = x.shape
+        x = self.patch_embedding(x)
+        if self.config.use_cls_token:
+            x = torch.cat([_expand_token(self.class_embedding, x.size(0)).to(x.dtype), x], dim=1)
+        # class embeddings and positional embeddings
+        x = self.add_pos_emb(x, patch_num)
+        if self.pre_ln is not None:
+            x = self.pre_ln(x)
+        hidden_states = self.transformer(x)
+        return hidden_states
+class ImageProjectorMLP(nn.Module):
+    def __init__(
+        self,
+        input_dim: int,
+        hidden_dim: int,
+        output_dim: int,
+        hidden_act: str,
+        device: Union[str, torch.device] = None,
+    ):
+        super().__init__()
+        self.w1 = nn.Linear(input_dim, hidden_dim, bias=False, device=device)
+        self.w2 = nn.Linear(hidden_dim, output_dim, bias=False, device=device)
+        self.w3 = nn.Linear(input_dim, hidden_dim, bias=False, device=device)
+        self.act = ACT2FN[hidden_act]
+    def forward(self, x: torch.Tensor) -> torch.Tensor:
+        return self.w2(self.act(self.w1(x)) * self.w3(x))
+class MolmoActVisionBackbone(nn.Module):
+    def __init__(self, vit_config: MolmoActVitConfig, adapter_config: MolmoActAdapterConfig):
+        super().__init__()
+        self.vit_config = vit_config
+        self.adapter_config = adapter_config
+        self.vit_layers = []
+        for layer in adapter_config.vit_layers:
+            if layer >= 0:
+                self.vit_layers.append(layer)
+            else:
+                self.vit_layers.append(layer + vit_config.num_hidden_layers)
+        last_layer_needed = max(self.vit_layers) + 1
+        if last_layer_needed < vit_config.num_hidden_layers:
+            new_vit_config = deepcopy(vit_config)
+            new_vit_config.num_hidden_layers = last_layer_needed
+            self.image_vit = MolmoActVisionTransformer(new_vit_config)
+        else:
+            self.image_vit = MolmoActVisionTransformer(vit_config)
+        self.num_prefix_tokens: int = self.image_vit.num_prefix_tokens
+        # optional pad_embed
+        self.pad_embed = None
+        if adapter_config.image_padding_embed == "pad_and_partial_pad":
+            pool_dim = vit_config.hidden_size * len(adapter_config.vit_layers)
+            self.pad_embed = nn.Parameter(torch.zeros((2, pool_dim)))
+        pool_dim = vit_config.hidden_size * len(adapter_config.vit_layers)
+        self.image_pooling_2d = ViTMultiHeadDotProductAttention(
+            hidden_size=adapter_config.hidden_size,
+            num_heads=adapter_config.num_attention_heads,
+            num_key_value_heads=adapter_config.num_key_value_heads,
+            head_dim=adapter_config.head_dim,
+            input_dim=pool_dim,
+            float32_attention=adapter_config.float32_attention,
+            attention_dropout=adapter_config.attention_dropout,
+            residual_dropout=adapter_config.residual_dropout,
+            attn_implementation=adapter_config._attn_implementation,
+        )
+        self.image_projector = ImageProjectorMLP(
+            adapter_config.hidden_size,
+            adapter_config.intermediate_size,
+            adapter_config.text_hidden_size,
+            adapter_config.hidden_act,
+        )
+        self.image_feature_dropout = nn.Dropout(adapter_config.image_feature_dropout)
+    def encode_image(self, images: torch.Tensor) -> torch.Tensor:
+        """
+        : param images: (batch_size, num_crops, num_patch, n_pixels)
+        """
+        B, T, N, D = images.shape
+        images = images.view(B * T, N, D)
+        image_features = self.image_vit(images)
+        features = []
+        for layer in self.vit_layers:
+            features.append(image_features[layer])
+        image_features = torch.cat(features, dim=-1)
+        if self.num_prefix_tokens > 0:
+            image_features = image_features[:, 1:]
+        image_features = image_features.view(B, T, N, -1)
+        return image_features
+    @property
+    def dtype(self) -> torch.dtype:
+        return self.image_vit.patch_embedding.weight.dtype
+    @property
+    def device(self) -> torch.device:
+        return self.image_vit.patch_embedding.weight.device
+    def forward(
+        self,
+        images: torch.Tensor,
+        pooled_patches_idx: torch.Tensor,
+        image_masks: torch.Tensor = None,
+    ) -> Tuple[torch.Tensor, Optional[torch.Tensor]]:
+        # image_features: (batch_size, num_crops(=num_image), num_patch, nximage_emb_dim)
+        batch_size, num_image = images.shape[:2]
+        images = images.to(device=self.device, dtype=self.dtype)
+        image_features = self.encode_image(images)
+        # optional padding embeddings
+        if self.pad_embed is not None and image_masks is not None:
+            image_masks = image_masks.to(device=self.device)
+            all_pad = (image_masks == 0).to(image_features.dtype)
+            partial = torch.logical_and(image_masks < 1, ~ (image_masks == 0)).to(image_features.dtype)
+            image_features = image_features + self.pad_embed[0][None,None,None,:] * all_pad[...,None] \
+                            + self.pad_embed[1][None,None,None,:] * partial[...,None]
+        image_features = self.image_feature_dropout(image_features)
+        dim = image_features.shape[-1]
+        valid = pooled_patches_idx >= 0
+        valid_token = torch.any(valid, -1)
+        # Use `pooled_patches_idx` to arange the features for image pooling
+        batch_idx = torch.arange(pooled_patches_idx.shape[0], dtype=torch.long, device=pooled_patches_idx.device)
+        batch_idx = torch.tile(batch_idx.view(batch_size, 1, 1), [1, pooled_patches_idx.shape[1], pooled_patches_idx.shape[2]])
+        # Now [batch, num_high_res_features, pool_dim, dim]
+        to_pool = image_features.reshape(batch_size, -1, dim)[batch_idx, torch.clip(pooled_patches_idx, 0)]
+        to_pool = to_pool * valid.to(self.dtype)[:, :, :, None]
+        to_pool = to_pool.reshape([-1, pooled_patches_idx.shape[-1], dim])
+        query = to_pool.mean(-2, keepdim=True)
+        pooled_features = self.image_pooling_2d(query, to_pool)
+        pooled_features = pooled_features.reshape([batch_size, -1, pooled_features.shape[-1]])
+        # MLP layer to map the feature.
+        pooled_features = self.image_projector(pooled_features)
+        return pooled_features.view(-1, pooled_features.shape[-1])[valid_token.flatten()]
+# Copied from transformers.models.llama.modeling_llama.rotate_half
+def rotate_half(x):
+    """Rotates half the hidden dims of the input."""
+    x1 = x[..., : x.shape[-1] // 2]
+    x2 = x[..., x.shape[-1] // 2 :]
+    return torch.cat((-x2, x1), dim=-1)
+# Copied from transformers.models.llama.modeling_llama.apply_rotary_pos_emb
+def apply_rotary_pos_emb(q, k, cos, sin, position_ids=None, unsqueeze_dim=1):
+    """Applies Rotary Position Embedding to the query and key tensors.
+    Args:
+        q (`torch.Tensor`): The query tensor.
+        k (`torch.Tensor`): The key tensor.
+        cos (`torch.Tensor`): The cosine part of the rotary embedding.
+        sin (`torch.Tensor`): The sine part of the rotary embedding.
+        position_ids (`torch.Tensor`, *optional*):
+            Deprecated and unused.
+        unsqueeze_dim (`int`, *optional*, defaults to 1):
+            The 'unsqueeze_dim' argument specifies the dimension along which to unsqueeze cos[position_ids] and
+            sin[position_ids] so that they can be properly broadcasted to the dimensions of q and k. For example, note
+            that cos[position_ids] and sin[position_ids] have the shape [batch_size, seq_len, head_dim]. Then, if q and
+            k have the shape [batch_size, heads, seq_len, head_dim], then setting unsqueeze_dim=1 makes
+            cos[position_ids] and sin[position_ids] broadcastable to the shapes of q and k. Similarly, if q and k have
+            the shape [batch_size, seq_len, heads, head_dim], then set unsqueeze_dim=2.
+    Returns:
+        `tuple(torch.Tensor)` comprising of the query and key tensors rotated using the Rotary Position Embedding.
+    """
+    cos = cos.unsqueeze(unsqueeze_dim)
+    sin = sin.unsqueeze(unsqueeze_dim)
+    q_embed = (q * cos) + (rotate_half(q) * sin)
+    k_embed = (k * cos) + (rotate_half(k) * sin)
+    return q_embed, k_embed
+# Copied from transformers.models.llama.modeling_llama.LlamaRotaryEmbedding
+class MolmoActRotaryEmbedding(nn.Module):
+    def __init__(self, config: MolmoActLlmConfig, device: Union[str, torch.device] = None):
+        super().__init__()
+        # BC: "rope_type" was originally "type"
+        if hasattr(config, "rope_scaling") and config.rope_scaling is not None:
+            self.rope_type = config.rope_scaling.get("rope_type", config.rope_scaling.get("type"))
+        else:
+            self.rope_type = "default"
+        self.max_seq_len_cached = config.max_position_embeddings
+        self.original_max_seq_len = config.max_position_embeddings
+        self.config = config
+        self.rope_init_fn = ROPE_INIT_FUNCTIONS[self.rope_type]
+        inv_freq, self.attention_scaling = self.rope_init_fn(self.config, device)
+        self.register_buffer("inv_freq", inv_freq, persistent=False)
+        self.original_inv_freq = self.inv_freq
+    @torch.no_grad()
+    @dynamic_rope_update  # power user: used with advanced RoPE types (e.g. dynamic rope)
+    def forward(self, x, position_ids: torch.Tensor) -> Tuple[torch.Tensor, torch.Tensor]:
+        inv_freq_expanded = self.inv_freq[None, :, None].float().expand(position_ids.shape[0], -1, 1).to(x.device)
+        position_ids_expanded = position_ids[:, None, :].float()
+        device_type = x.device.type if isinstance(x.device.type, str) and x.device.type != "mps" else "cpu"
+        with torch.autocast(device_type=device_type, enabled=False):  # Force float32
+            freqs = (inv_freq_expanded.float() @ position_ids_expanded.float()).transpose(1, 2)
+            emb = torch.cat((freqs, freqs), dim=-1)
+            cos = emb.cos() * self.attention_scaling
+            sin = emb.sin() * self.attention_scaling
+        return cos.to(dtype=x.dtype), sin.to(dtype=x.dtype)
+@use_kernel_forward_from_hub("RMSNorm")
+class MolmoActRMSNorm(nn.Module):
+    def __init__(
+        self,
+        size: int,
+        eps: float = 1e-6,
+        device: Union[str, torch.device] = None,
+    ):
+        super().__init__()
+        self.weight = nn.Parameter(torch.ones(size, device=device))
+        self.eps = eps
+    def forward(self, x: torch.Tensor) -> torch.Tensor:
+        with torch.autocast(enabled=False, device_type=x.device.type):
+            og_dtype = x.dtype
+            x = x.to(torch.float32)
+            variance = x.pow(2).mean(-1, keepdim=True)
+            x = x * torch.rsqrt(variance + self.eps)
+            x = x.to(og_dtype)
+        return self.weight * x
+    def extra_repr(self):
+        return f"{tuple(self.weight.shape)}, eps={self.eps}"
+# Copied from transformers.models.llama.modeling_llama.repeat_kv
+def repeat_kv(hidden_states: torch.Tensor, n_rep: int) -> torch.Tensor:
+    """
+    This is the equivalent of torch.repeat_interleave(x, dim=1, repeats=n_rep). The hidden states go from (batch,
+    num_key_value_heads, seqlen, head_dim) to (batch, num_attention_heads, seqlen, head_dim)
+    """
+    batch, num_key_value_heads, slen, head_dim = hidden_states.shape
+    if n_rep == 1:
+        return hidden_states
+    hidden_states = hidden_states[:, :, None, :, :].expand(batch, num_key_value_heads, n_rep, slen, head_dim)
+    return hidden_states.reshape(batch, num_key_value_heads * n_rep, slen, head_dim)
+def eager_attention_forward(
+    module: nn.Module,
+    query: torch.Tensor,
+    key: torch.Tensor,
+    value: torch.Tensor,
+    attention_mask: Optional[torch.Tensor],
+    scaling: float,
+    dropout: float = 0.0,
+    **kwargs,
+) -> Tuple[torch.Tensor, Optional[torch.Tensor]]:
+    key_states = repeat_kv(key, module.num_key_value_groups)
+    value_states = repeat_kv(value, module.num_key_value_groups)
+    attn_weights = torch.matmul(query, key_states.transpose(2, 3)) * scaling
+    if attention_mask is not None:
+        causal_mask = attention_mask[:, :, :, : key_states.shape[-2]]
+        attn_weights = attn_weights + causal_mask
+    attn_weights = nn.functional.softmax(attn_weights, dim=-1, dtype=torch.float32).to(query.dtype)
+    attn_weights = nn.functional.dropout(attn_weights, p=dropout, training=module.training)
+    attn_output = torch.matmul(attn_weights, value_states)
+    attn_output = attn_output.transpose(1, 2).contiguous()
+    return attn_output, attn_weights
+class MolmoActAttention(nn.Module):
+    """Multi-headed attention from 'Attention Is All You Need' paper"""
+    # copied from transformers.models.llama.modeling_llama.LlamaAttention.__init__ with Llama->MolmoAct
+    def __init__(self, config: MolmoActLlmConfig, layer_idx: Optional[int] = None) -> None:
+        super().__init__()
+        self.config = config
+        self.layer_idx = layer_idx
+        if layer_idx is None:
+            logger.warning_once(
+                f"Instantiating {self.__class__.__name__} without passing a `layer_idx` is not recommended and will "
+                "lead to errors during the forward call if caching is used. Please make sure to provide a `layer_idx` "
+                "when creating this class."
+            )
+        self.num_heads = config.num_attention_heads
+        self.num_key_value_heads = config.num_key_value_heads
+        self.num_key_value_groups = config.num_attention_heads // config.num_key_value_heads
+        self.head_dim = config.head_dim
+        self.scaling = self.head_dim**-0.5
+        self.is_causal = True
+        if (config.head_dim * config.num_attention_heads) != config.hidden_size:
+            raise ValueError(
+                f"hidden_size must be divisible by num_heads (got `hidden_size`: {config.hidden_size}"
+                f" and `num_attention_heads`: {config.num_attention_heads})."
+            )
+        self.fused_dims = (
+            config.hidden_size,
+            config.head_dim * config.num_key_value_heads,
+            config.head_dim * config.num_key_value_heads,
+        )
+        self.att_proj = nn.Linear(
+            config.hidden_size,
+            sum(self.fused_dims),
+            bias=config.qkv_bias,
+        )
+        # Layer norms.
+        self.k_norm: Optional[MolmoActRMSNorm] = None
+        self.q_norm: Optional[MolmoActRMSNorm] = None
+        self.qk_norm_type: Optional[str] = None
+        if config.use_qk_norm:
+            k_norm_size = (
+                config.head_dim
+                if config.qk_norm_type == "qwen3" else
+                config.num_key_value_heads * config.head_dim
+            )
+            self.k_norm = MolmoActRMSNorm(k_norm_size, eps=config.layer_norm_eps)
+            q_norm_size = (
+                config.head_dim
+                if config.qk_norm_type == "qwen3" else
+                config.num_attention_heads * config.head_dim
+            )
+            self.q_norm = MolmoActRMSNorm(q_norm_size, eps=config.layer_norm_eps)
+            self.qk_norm_type = config.qk_norm_type
+        self.attention_dropout = config.attention_dropout
+        self.attn_out = nn.Linear(
+            config.hidden_size,
+            config.hidden_size,
+            bias=False,
+        )
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        position_embeddings: Tuple[torch.Tensor, torch.Tensor],
+        attention_mask: Optional[torch.Tensor],
+        past_key_value: Optional[Cache] = None,
+        cache_position: Optional[torch.LongTensor] = None,
+        **kwargs: Unpack[FlashAttentionKwargs],
+    ) -> Tuple[torch.Tensor, Optional[torch.Tensor], Optional[Tuple[torch.Tensor]]]:
+        input_shape = hidden_states.shape[:-1]
+        hidden_shape = (*input_shape, -1, self.head_dim)
+        qkv = self.att_proj(hidden_states)
+        query_states, key_states, value_states = qkv.split(self.fused_dims, dim=-1)
+        value_states = value_states.view(hidden_shape)
+        # Optionally apply layer norm to keys and queries.
+        if self.q_norm is not None and self.k_norm is not None and self.qk_norm_type != "qwen3":
+            query_states = self.q_norm(query_states)
+            key_states = self.k_norm(key_states)
+        query_states = query_states.view(hidden_shape)
+        key_states = key_states.view(hidden_shape)
+        if self.q_norm is not None and self.k_norm is not None and self.qk_norm_type == "qwen3":
+            query_states = self.q_norm(query_states)
+            key_states = self.k_norm(key_states)
+        query_states = query_states.transpose(1, 2)
+        key_states = key_states.transpose(1, 2)
+        value_states = value_states.transpose(1, 2)
+        cos, sin = position_embeddings
+        query_states, key_states = apply_rotary_pos_emb(query_states, key_states, cos, sin)
+        if past_key_value is not None:
+            # sin and cos are specific to RoPE models; cache_position needed for the static cache
+            cache_kwargs = {"sin": sin, "cos": cos, "cache_position": cache_position}
+            key_states, value_states = past_key_value.update(key_states, value_states, self.layer_idx, cache_kwargs)
+        attention_interface: Callable = eager_attention_forward
+        if self.config._attn_implementation != "eager":
+            if self.config._attn_implementation == "sdpa" and kwargs.get("output_attentions", False):
+                logger.warning_once(
+                    "`torch.nn.functional.scaled_dot_product_attention` does not support `output_attentions=True`. Falling back to "
+                    'eager attention. This warning can be removed using the argument `attn_implementation="eager"` when loading the model.'
+                )
+            else:
+                attention_interface = ALL_ATTENTION_FUNCTIONS[self.config._attn_implementation]
+        attn_output, attn_weights = attention_interface(
+            self,
+            query_states,
+            key_states,
+            value_states,
+            attention_mask,
+            dropout=0.0 if not self.training else self.attention_dropout,
+            scaling=self.scaling,
+            **kwargs,
+        )
+        attn_output = attn_output.reshape(*input_shape, -1).contiguous()
+        attn_output = self.attn_out(attn_output)
+        return attn_output, attn_weights
+class LanguageModelMLP(nn.Module):
+    def __init__(
+        self,
+        input_dim: int,
+        intermediate_size: int,
+        hidden_act: str,
+        device: Union[str, torch.device] = None,
+    ):
+        super().__init__()
+        self.ff_proj = nn.Linear(input_dim, intermediate_size * 2, bias=False, device=device)
+        self.ff_out = nn.Linear(intermediate_size, input_dim, bias=False, device=device)
+        self.act = ACT2FN[hidden_act]
+    def forward(self, x: torch.Tensor) -> torch.Tensor:
+        x = self.ff_proj(x)
+        x, gate = x.chunk(2, dim=-1)
+        x = self.act(gate) * x
+        x = self.ff_out(x)
+        return x
+class MolmoActDecoderLayer(GradientCheckpointingLayer):
+    def __init__(
+        self,
+        config: MolmoActLlmConfig,
+        layer_idx: Optional[int] = None,
+        device: Union[str, torch.device] = None
+    ):
+        super().__init__()
+        self.config = config
+        self.self_attn = MolmoActAttention(config, layer_idx)
+        self.attn_norm = MolmoActRMSNorm(
+            config.hidden_size, eps=config.layer_norm_eps, device=device)
+        self.dropout = nn.Dropout(config.residual_dropout)
+        self.mlp = LanguageModelMLP(
+            config.hidden_size, config.intermediate_size, config.hidden_act, device=device)
+        self.ff_norm = MolmoActRMSNorm(
+            config.hidden_size, eps=config.layer_norm_eps, device=device)
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        attention_mask: Optional[torch.Tensor] = None,
+        position_ids: Optional[torch.LongTensor] = None,
+        past_key_value: Optional[Tuple[torch.Tensor]] = None,
+        output_attentions: Optional[bool] = False,
+        use_cache: Optional[bool] = False,
+        cache_position: Optional[torch.LongTensor] = None,
+        position_embeddings: Optional[Tuple[torch.Tensor, torch.Tensor]] = None,  # will become mandatory in v4.46
+        **kwargs,
+    ) -> Tuple[torch.FloatTensor, Optional[Tuple[torch.FloatTensor, torch.FloatTensor]]]:
+        """
+        Args:
+            hidden_states (`torch.FloatTensor`): input to the layer of shape `(batch, seq_len, embed_dim)`
+            attention_mask (`torch.FloatTensor`, *optional*): attention mask of size
+                `(batch, sequence_length)` where padding elements are indicated by 0.
+            output_attentions (`bool`, *optional*):
+                Whether or not to return the attentions tensors of all attention layers. See `attentions` under
+                returned tensors for more detail.
+            use_cache (`bool`, *optional*):
+                If set to `True`, `past_key_values` key value states are returned and can be used to speed up decoding
+                (see `past_key_values`).
+            past_key_value (`Tuple(torch.FloatTensor)`, *optional*): cached past key and value projection states
+            cache_position (`torch.LongTensor` of shape `(sequence_length)`, *optional*):
+                Indices depicting the position of the input sequence tokens in the sequence.
+            position_embeddings (`Tuple[torch.FloatTensor, torch.FloatTensor]`, *optional*):
+                Tuple containing the cosine and sine positional embeddings of shape `(batch_size, seq_len, head_dim)`,
+                with `head_dim` being the embedding dimension of each attention head.
+            kwargs (`dict`, *optional*):
+                Arbitrary kwargs to be ignored, used for FSDP and other methods that injects code
+                into the model
+        """
+        residual = hidden_states
+        hidden_states = self.attn_norm(hidden_states)
+        # Self Attention
+        hidden_states, self_attn_weights = self.self_attn(
+            hidden_states=hidden_states,
+            attention_mask=attention_mask,
+            position_ids=position_ids,
+            past_key_value=past_key_value,
+            output_attentions=output_attentions,
+            use_cache=use_cache,
+            cache_position=cache_position,
+            position_embeddings=position_embeddings,
+        )
+        hidden_states = residual + self.dropout(hidden_states)
+        # Fully Connected
+        residual = hidden_states
+        hidden_states = self.ff_norm(hidden_states)
+        hidden_states = self.mlp(hidden_states)
+        hidden_states = residual + self.dropout(hidden_states)
+        outputs = (hidden_states,)
+        if output_attentions:
+            outputs += (self_attn_weights,)
+        return outputs
+class MolmoActPostNormDecoderLayer(MolmoActDecoderLayer):
+    def forward(
+        self,
+        hidden_states: torch.Tensor,
+        attention_mask: Optional[torch.Tensor] = None,
+        position_ids: Optional[torch.LongTensor] = None,
+        past_key_value: Optional[Tuple[torch.Tensor]] = None,
+        output_attentions: Optional[bool] = False,
+        use_cache: Optional[bool] = False,
+        cache_position: Optional[torch.LongTensor] = None,
+        position_embeddings: Optional[Tuple[torch.Tensor, torch.Tensor]] = None,  # will become mandatory in v4.46
+        **kwargs,
+    ) -> Tuple[torch.FloatTensor, Optional[Tuple[torch.FloatTensor, torch.FloatTensor]]]:
+        """
+        Args:
+            hidden_states (`torch.FloatTensor`): input to the layer of shape `(batch, seq_len, embed_dim)`
+            attention_mask (`torch.FloatTensor`, *optional*): attention mask of size
+                `(batch, sequence_length)` where padding elements are indicated by 0.
+            output_attentions (`bool`, *optional*):
+                Whether or not to return the attentions tensors of all attention layers. See `attentions` under
+                returned tensors for more detail.
+            use_cache (`bool`, *optional*):
+                If set to `True`, `past_key_values` key value states are returned and can be used to speed up decoding
+                (see `past_key_values`).
+            past_key_value (`Tuple(torch.FloatTensor)`, *optional*): cached past key and value projection states
+            cache_position (`torch.LongTensor` of shape `(sequence_length)`, *optional*):
+                Indices depicting the position of the input sequence tokens in the sequence.
+            position_embeddings (`Tuple[torch.FloatTensor, torch.FloatTensor]`, *optional*):
+                Tuple containing the cosine and sine positional embeddings of shape `(batch_size, seq_len, head_dim)`,
+                with `head_dim` being the embedding dimension of each attention head.
+            kwargs (`dict`, *optional*):
+                Arbitrary kwargs to be ignored, used for FSDP and other methods that injects code
+                into the model
+        """
+        residual = hidden_states
+        # Self Attention
+        hidden_states, self_attn_weights = self.self_attn(
+            hidden_states=hidden_states,
+            attention_mask=attention_mask,
+            position_ids=position_ids,
+            past_key_value=past_key_value,
+            output_attentions=output_attentions,
+            use_cache=use_cache,
+            cache_position=cache_position,
+            position_embeddings=position_embeddings,
+        )
+        hidden_states = self.attn_norm(hidden_states)
+        hidden_states = residual + self.dropout(hidden_states)
+        # Fully Connected
+        residual = hidden_states
+        hidden_states = self.mlp(hidden_states)
+        hidden_states = self.ff_norm(hidden_states)
+        hidden_states = residual + self.dropout(hidden_states)
+        outputs = (hidden_states,)
+        if output_attentions:
+            outputs += (self_attn_weights,)
+        return outputs
+class MolmoActEmbedding(nn.Module):
+    def __init__(
+        self,
+        num_embeddings: int,
+        num_new_embeddings: int,
+        features: int,
+        device: Union[str, torch.device] = None,
+    ):
+        super().__init__()
+        self.embedding = nn.Parameter(
+            torch.zeros(num_embeddings, features, device=device),
+        )
+        self.new_embedding = nn.Parameter(
+            torch.zeros(num_new_embeddings, features, device=device),
+        )
+    def forward(self, x: torch.Tensor) -> torch.Tensor:
+        return F.embedding(x, torch.cat([self.embedding, self.new_embedding], dim=0))
+MOLMO2_TEXT_ONLY_INPUTS_DOCSTRING = r"""
+    Args:
+        input_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
+            Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you provide
+            it.
+            Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
+            [`PreTrainedTokenizer.__call__`] for details.
+            [What are input IDs?](../glossary#input-ids)
+        attention_mask (`torch.Tensor` of shape `(batch_size, sequence_length)`, *optional*):
+            Mask to avoid performing attention on padding token indices. Mask values selected in `[0, 1]`:
+            - 1 for tokens that are **not masked**,
+            - 0 for tokens that are **masked**.
+            [What are attention masks?](../glossary#attention-mask)
+            Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
+            [`PreTrainedTokenizer.__call__`] for details.
+            If `past_key_values` is used, optionally only the last `input_ids` have to be input (see
+            `past_key_values`).
+            If you want to change padding behavior, you should read [`modeling_opt._prepare_decoder_attention_mask`]
+            and modify to your needs. See diagram 1 in [the paper](https://arxiv.org/abs/1910.13461) for more
+            information on the default strategy.
+            - 1 indicates the head is **not masked**,
+            - 0 indicates the head is **masked**.
+        position_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
+            Indices of positions of each input sequence tokens in the position embeddings. Selected in the range `[0,
+            config.n_positions - 1]`.
+            [What are position IDs?](../glossary#position-ids)
+        past_key_values (`Cache` or `tuple(tuple(torch.FloatTensor))`, *optional*):
+            Pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention
+            blocks) that can be used to speed up sequential decoding. This typically consists in the `past_key_values`
+            returned by the model at a previous stage of decoding, when `use_cache=True` or `config.use_cache=True`.
+            Two formats are allowed:
+            - a [`~cache_utils.Cache`] instance, see our
+            [kv cache guide](https://huggingface.co/docs/transformers/en/kv_cache);
+            - Tuple of `tuple(torch.FloatTensor)` of length `config.n_layers`, with each tuple having 2 tensors of
+            shape `(batch_size, num_heads, sequence_length, embed_size_per_head)`). This is also known as the legacy
+            cache format.
+            The model will output the same cache format that is fed as input. If no `past_key_values` are passed, the
+            legacy cache format will be returned.
+            If `past_key_values` are used, the user can optionally input only the last `input_ids` (those that don't
+            have their past key value states given to this model) of shape `(batch_size, 1)` instead of all `input_ids`
+            of shape `(batch_size, sequence_length)`.
+        inputs_embeds (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`, *optional*):
+            Optionally, instead of passing `input_ids` you can choose to directly pass an embedded representation. This
+            is useful if you want more control over how to convert `input_ids` indices into associated vectors than the
+            model's internal embedding lookup matrix.
+        use_cache (`bool`, *optional*):
+            If set to `True`, `past_key_values` key value states are returned and can be used to speed up decoding (see
+            `past_key_values`).
+        output_attentions (`bool`, *optional*):
+            Whether or not to return the attentions tensors of all attention layers. See `attentions` under returned
+            tensors for more detail.
+        output_hidden_states (`bool`, *optional*):
+            Whether or not to return the hidden states of all layers. See `hidden_states` under returned tensors for
+            more detail.
+        return_dict (`bool`, *optional*):
+            Whether or not to return a [`CausalLMOutputWithPast`] instead of a plain tuple.
+        cache_position (`torch.LongTensor` of shape `(sequence_length)`, *optional*):
+            Indices depicting the position of the input sequence tokens in the sequence. Contrarily to `position_ids`,
+            this tensor is not affected by padding. It is used to update the cache in the correct position and to infer
+            the complete sequence length.
+"""
+@add_start_docstrings(
+    "The bare MolmoAct text-only model outputting raw hidden-states without any specific head on top.",
+    MOLMO_START_DOCSTRING,
+)
+class MolmoActLlm(MolmoActPreTrainedModel):
+    def __init__(self, config: MolmoActLlmConfig):
+        super().__init__(config)
+        self.config = config
+        if config.additional_vocab_size is not None:
+            self.wte = MolmoActEmbedding(
+                config.vocab_size,
+                config.additional_vocab_size,
+                config.hidden_size,
+            )
+        else:
+            self.wte = nn.Embedding(config.vocab_size, config.hidden_size)
+        self.emb_drop = nn.Dropout(config.embedding_dropout)
+        decoder_layer = MolmoActPostNormDecoderLayer if config.norm_after else MolmoActDecoderLayer
+        self.blocks = nn.ModuleList(
+            [decoder_layer(config, layer_idx) for layer_idx in range(config.num_hidden_layers)]
+        )
+        self.ln_f = MolmoActRMSNorm(config.hidden_size, eps=config.layer_norm_eps)
+        self.rotary_emb = MolmoActRotaryEmbedding(config)
+        self.gradient_checkpointing = False
+        # Initialize weights and apply final processing
+        self.post_init()
+    def get_input_embeddings(self) -> torch.nn.Module:
+        return self.wte
+    def set_input_embeddings(self, value: torch.nn.Module) -> None:
+        self.wte = value
+    @can_return_tuple
+    def forward(
+        self,
+        input_ids: Optional[torch.LongTensor] = None,
+        attention_mask: Optional[torch.Tensor] = None,
+        position_ids: Optional[torch.LongTensor] = None,
+        past_key_values: Optional[Cache] = None,
+        inputs_embeds: Optional[torch.FloatTensor] = None,
+        use_cache: Optional[bool] = None,
+        output_attentions: Optional[bool] = None,
+        output_hidden_states: Optional[bool] = None,
+        cache_position: Optional[torch.LongTensor] = None,
+        **flash_attn_kwargs: Unpack[FlashAttentionKwargs],
+    ) -> BaseModelOutputWithPast:
+        output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        use_cache = use_cache if use_cache is not None else self.config.use_cache
+        if (input_ids is None) ^ (inputs_embeds is not None):
+            raise ValueError("You must specify exactly one of input_ids or inputs_embeds")
+        if self.gradient_checkpointing and self.training and use_cache:
+            logger.warning_once(
+                "`use_cache=True` is incompatible with gradient checkpointing. Setting `use_cache=False`."
+            )
+            use_cache = False
+        # TODO (joao): remove this exception in v4.56 -- it exists for users that try to pass a legacy cache
+        if not isinstance(past_key_values, (type(None), Cache)):
+            raise ValueError("The `past_key_values` should be either a `Cache` object or `None`.")
+        if inputs_embeds is None:
+            input_ids = input_ids * (input_ids != -1).to(input_ids.dtype)
+            inputs_embeds = self.wte(input_ids)
+        if use_cache and past_key_values is None:
+            past_key_values = DynamicCache()
+        if cache_position is None:
+            past_seen_tokens = past_key_values.get_seq_length() if past_key_values is not None else 0
+            cache_position = torch.arange(
+                past_seen_tokens, past_seen_tokens + inputs_embeds.shape[1], device=inputs_embeds.device
+            )
+        if position_ids is None:
+            position_ids = cache_position.unsqueeze(0)
+        causal_mask = self._update_causal_mask(
+            attention_mask, inputs_embeds, cache_position, past_key_values, output_attentions
+        )
+        hidden_states = inputs_embeds
+        # create position embeddings to be shared across the decoder layers
+        position_embeddings = self.rotary_emb(hidden_states, position_ids)
+        # decoder layers
+        all_hidden_states = () if output_hidden_states else None
+        all_self_attns = () if output_attentions else None
+        for decoder_block in self.blocks[: self.config.num_hidden_layers]:
+            if output_hidden_states:
+                all_hidden_states += (hidden_states,)
+            layer_outputs = decoder_block(
+                hidden_states,
+                attention_mask=causal_mask,
+                position_ids=position_ids,
+                past_key_value=past_key_values,
+                output_attentions=output_attentions,
+                use_cache=use_cache,
+                cache_position=cache_position,
+                position_embeddings=position_embeddings,
+                **flash_attn_kwargs,
+            )
+            hidden_states = layer_outputs[0]
+            if output_attentions:
+                all_self_attns += (layer_outputs[1],)
+        hidden_states = self.ln_f(hidden_states)
+        # add hidden states from the last decoder layer
+        if output_hidden_states:
+            all_hidden_states += (hidden_states,)
+        return BaseModelOutputWithPast(
+            last_hidden_state=hidden_states,
+            past_key_values=past_key_values if use_cache else None,
+            hidden_states=all_hidden_states,
+            attentions=all_self_attns,
+        )
+    def _update_causal_mask(
+        self,
+        attention_mask: Union[torch.Tensor, "BlockMask"],
+        input_tensor: torch.Tensor,
+        cache_position: torch.Tensor,
+        past_key_values: Cache,
+        output_attentions: bool = False,
+    ):
+        if self.config._attn_implementation == "flash_attention_2":
+            if attention_mask is not None and (attention_mask == 0.0).any():
+                return attention_mask
+            return None
+        if self.config._attn_implementation == "flex_attention":
+            if isinstance(attention_mask, torch.Tensor):
+                attention_mask = make_flex_block_causal_mask(attention_mask)
+            return attention_mask
+        # For SDPA, when possible, we will rely on its `is_causal` argument instead of its `attn_mask` argument, in
+        # order to dispatch on Flash Attention 2. This feature is not compatible with static cache, as SDPA will fail
+        # to infer the attention mask.
+        past_seen_tokens = past_key_values.get_seq_length() if past_key_values is not None else 0
+        using_compilable_cache = past_key_values.is_compileable if past_key_values is not None else False
+        # When output attentions is True, sdpa implementation's forward method calls the eager implementation's forward
+        if self.config._attn_implementation == "sdpa" and not using_compilable_cache and not output_attentions:
+            if AttentionMaskConverter._ignore_causal_mask_sdpa(
+                attention_mask,
+                inputs_embeds=input_tensor,
+                past_key_values_length=past_seen_tokens,
+                is_training=self.training,
+            ):
+                return None
+        dtype = input_tensor.dtype
+        sequence_length = input_tensor.shape[1]
+        if using_compilable_cache:
+            target_length = past_key_values.get_max_cache_shape()
+        else:
+            target_length = (
+                attention_mask.shape[-1]
+                if isinstance(attention_mask, torch.Tensor)
+                else past_seen_tokens + sequence_length + 1
+            )
+        # In case the provided `attention` mask is 2D, we generate a causal mask here (4D).
+        causal_mask = self._prepare_4d_causal_attention_mask_with_cache_position(
+            attention_mask,
+            sequence_length=sequence_length,
+            target_length=target_length,
+            dtype=dtype,
+            cache_position=cache_position,
+            batch_size=input_tensor.shape[0],
+        )
+        if (
+            self.config._attn_implementation == "sdpa"
+            and attention_mask is not None
+            and attention_mask.device.type in ["cuda", "xpu", "npu"]
+            and not output_attentions
+        ):
+            # Attend to all tokens in fully masked rows in the causal_mask, for example the relevant first rows when
+            # using left padding. This is required by F.scaled_dot_product_attention memory-efficient attention path.
+            # Details: https://github.com/pytorch/pytorch/issues/110213
+            min_dtype = torch.finfo(dtype).min
+            causal_mask = AttentionMaskConverter._unmask_unattended(causal_mask, min_dtype)
+        return causal_mask
+    @staticmethod
+    def _prepare_4d_causal_attention_mask_with_cache_position(
+        attention_mask: torch.Tensor,
+        sequence_length: int,
+        target_length: int,
+        dtype: torch.dtype,
+        cache_position: torch.Tensor,
+        batch_size: int,
+        **kwargs,
+    ):
+        """
+        Creates a causal 4D mask of shape `(batch_size, 1, query_length, key_value_length)` from a 2D mask of shape
+        `(batch_size, key_value_length)`, or if the input `attention_mask` is already 4D, do nothing.
+        Args:
+            attention_mask (`torch.Tensor`):
+                A 2D attention mask of shape `(batch_size, key_value_length)` or a 4D attention mask of shape
+                `(batch_size, 1, query_length, key_value_length)`.
+            sequence_length (`int`):
+                The sequence length being processed.
+            target_length (`int`):
+                The target length: when generating with static cache, the mask should be as long as the static cache,
+                to account for the 0 padding, the part of the cache that is not filled yet.
+            dtype (`torch.dtype`):
+                The dtype to use for the 4D attention mask.
+            cache_position (`torch.Tensor`):
+                Indices depicting the position of the input sequence tokens in the sequence.
+            batch_size (`torch.Tensor`):
+                Batch size.
+        """
+        if attention_mask is not None and attention_mask.dim() == 4:
+            # In this case we assume that the mask comes already in inverted form and requires no inversion or slicing.
+            causal_mask = attention_mask
+        else:
+            min_dtype = torch.finfo(dtype).min
+            causal_mask = torch.full(
+                (sequence_length, target_length), fill_value=min_dtype, dtype=dtype, device=cache_position.device
+            )
+            if sequence_length != 1:
+                causal_mask = torch.triu(causal_mask, diagonal=1)
+            causal_mask *= torch.arange(target_length, device=cache_position.device) > cache_position.reshape(-1, 1)
+            causal_mask = causal_mask[None, None, :, :].expand(batch_size, 1, -1, -1)
+            if attention_mask is not None:
+                causal_mask = causal_mask.clone()  # copy to contiguous memory for in-place edit
+                mask_length = attention_mask.shape[-1]
+                padding_mask = causal_mask[:, :, :, :mask_length] + attention_mask[:, None, None, :].to(
+                    causal_mask.device
+                )
+                padding_mask = padding_mask == 0
+                causal_mask[:, :, :, :mask_length] = causal_mask[:, :, :, :mask_length].masked_fill(
+                    padding_mask, min_dtype
+                )
+        return causal_mask
+@add_start_docstrings(
+    "The MolmoAct text-only model which consists of a language model + lm head.",
+    MOLMO_START_DOCSTRING,
+)
+class MolmoActForCausalLM(MolmoActPreTrainedModel, GenerationMixin):
+    _tied_weights_keys = []  # Weights are not tied
+    _tp_plan = {"lm_head": "colwise_rep"}
+    _pp_plan = {"lm_head": (["hidden_states"], ["logits"])}
+    base_model_prefix = "model"
+    def __init__(self, config: MolmoActLlmConfig):
+        super().__init__(config)
+        self.model = MolmoActLlm(config)
+        self.vocab_size = config.vocab_size
+        self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
+        # Initialize weights and apply final processing
+        self.post_init()
+    def get_input_embeddings(self) -> torch.nn.Module:
+        return self.model.wte
+    def set_input_embeddings(self, value: torch.nn.Module) -> None:
+        self.model.wte = value
+    def get_output_embeddings(self):
+        return self.lm_head
+    def set_output_embeddings(self, value: torch.nn.Module) -> None:
+        self.lm_head = value
+    def set_decoder(self, decoder: torch.nn.Module) -> None:
+        self.model = decoder
+    def get_decoder(self) -> torch.nn.Module:
+        return self.model
+    @can_return_tuple
+    @add_start_docstrings_to_model_forward(MOLMO2_TEXT_ONLY_INPUTS_DOCSTRING)
+    def forward(
+        self,
+        input_ids: Optional[torch.LongTensor] = None,
+        attention_mask: Optional[torch.Tensor] = None,
+        position_ids: Optional[torch.LongTensor] = None,
+        past_key_values: Optional[Cache] = None,
+        inputs_embeds: Optional[torch.FloatTensor] = None,
+        labels: Optional[torch.LongTensor] = None,
+        use_cache: Optional[bool] = None,
+        output_attentions: Optional[bool] = None,
+        output_hidden_states: Optional[bool] = None,
+        cache_position: Optional[torch.LongTensor] = None,
+        logits_to_keep: Union[int, torch.Tensor] = 0,
+        **kwargs,
+    ) -> CausalLMOutputWithPast:
+        r"""
+        ```python
+        >>> from transformers import AutoTokenizer, MolmoActForCausalLM
+        >>> model = MolmoActForCausalLM.from_pretrained("...")
+        >>> tokenizer = AutoTokenizer.from_pretrained("...")
+        >>> prompt = "Hey, are you conscious? Can you talk to me?"
+        >>> inputs = tokenizer(prompt, return_tensors="pt")
+        >>> # Generate
+        >>> generate_ids = model.generate(inputs.input_ids, max_length=30)
+        >>> tokenizer.batch_decode(generate_ids, skip_special_tokens=True, clean_up_tokenization_spaces=False)[0]
+        "Hey, are you conscious? Can you talk to me?\nI'm not conscious, but I can talk to you."
+        ```"""
+        output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        # decoder outputs consists of (dec_features, layer_state, dec_hidden, dec_attn)
+        outputs: BaseModelOutputWithPast = self.model(
+            input_ids=input_ids,
+            attention_mask=attention_mask,
+            position_ids=position_ids,
+            past_key_values=past_key_values,
+            inputs_embeds=inputs_embeds,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            cache_position=cache_position,
+            **kwargs,
+        )
+        hidden_states = outputs.last_hidden_state
+        # Only compute necessary logits, and do not upcast them to float if we are not computing the loss
+        slice_indices = slice(-logits_to_keep, None) if isinstance(logits_to_keep, int) else logits_to_keep
+        logits = self.lm_head(hidden_states[:, slice_indices, :])
+        loss = None
+        if labels is not None:
+            loss = self.loss_function(logits=logits, labels=labels, vocab_size=self.config.vocab_size, **kwargs)
+        return CausalLMOutputWithPast(
+            loss=loss,
+            logits=logits,
+            past_key_values=outputs.past_key_values,
+            hidden_states=outputs.hidden_states,
+            attentions=outputs.attentions,
+        )
+MOLMO2_INPUTS_DOCSTRING = r"""
+    Args:
+        input_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`):
+            Indices of input sequence tokens in the vocabulary. Padding will be ignored by default should you provide
+            it.
+            Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
+            [`PreTrainedTokenizer.__call__`] for details.
+            [What are input IDs?](../glossary#input-ids)
+        images (`torch.FloatTensor` of shape `(batch_size, n_crops, 27*27, 3*14*14)`, *optional*):
+            The input crops in with pixel values between 0 and 1 and normalized with SigLIP2 mean/std
+            Each crop contains 27x27 patches with 14*14*3 pixel values
+        image_masks  (`torch.FloatTensor` of shape `(batch_size, n_crops, n_patches, n_features)`, *optional*):
+            Image masks showing what percent of each patch is paddding
+        pooled_patches_idx (`torch.LongTensor` of shape `(batch_size, n_image_tokens, n_pooled_patches)`):
+            For each patch_id tokens in `input_ids`, the indices of the patches in `images`
+            to pool for that token, masked with -1
+            means ignore the patch.
+        attention_mask (`torch.Tensor` of shape `(batch_size, sequence_length)`, *optional*):
+            Mask to avoid performing attention on padding token indices. Mask values selected in `[0, 1]`:
+            - 1 for tokens that are **not masked**,
+            - 0 for tokens that are **masked**.
+            [What are attention masks?](../glossary#attention-mask)
+            Indices can be obtained using [`AutoTokenizer`]. See [`PreTrainedTokenizer.encode`] and
+            [`PreTrainedTokenizer.__call__`] for details.
+            If `past_key_values` is used, optionally only the last `input_ids` have to be input (see
+            `past_key_values`).
+            If you want to change padding behavior, you should read [`modeling_opt._prepare_decoder_attention_mask`]
+            and modify to your needs. See diagram 1 in [the paper](https://arxiv.org/abs/1910.13461) for more
+            information on the default strategy.
+            - 1 indicates the head is **not masked**,
+            - 0 indicates the head is **masked**.
+        position_ids (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
+            Indices of positions of each input sequence tokens in the position embeddings. Selected in the range `[0,
+            config.n_positions - 1]`.
+            [What are position IDs?](../glossary#position-ids)
+        past_key_values (`Cache` or `tuple(tuple(torch.FloatTensor))`, *optional*):
+            Pre-computed hidden-states (key and values in the self-attention blocks and in the cross-attention
+            blocks) that can be used to speed up sequential decoding. This typically consists in the `past_key_values`
+            returned by the model at a previous stage of decoding, when `use_cache=True` or `config.use_cache=True`.
+            Two formats are allowed:
+            - a [`~cache_utils.Cache`] instance, see our
+            [kv cache guide](https://huggingface.co/docs/transformers/en/kv_cache);
+            - Tuple of `tuple(torch.FloatTensor)` of length `config.n_layers`, with each tuple having 2 tensors of
+            shape `(batch_size, num_heads, sequence_length, embed_size_per_head)`). This is also known as the legacy
+            cache format.
+            The model will output the same cache format that is fed as input. If no `past_key_values` are passed, the
+            legacy cache format will be returned.
+            If `past_key_values` are used, the user can optionally input only the last `input_ids` (those that don't
+            have their past key value states given to this model) of shape `(batch_size, 1)` instead of all `input_ids`
+            of shape `(batch_size, sequence_length)`.
+        inputs_embeds (`torch.FloatTensor` of shape `(batch_size, sequence_length, hidden_size)`, *optional*):
+            Optionally, instead of passing `input_ids` you can choose to directly pass an embedded representation. This
+            is useful if you want more control over how to convert `input_ids` indices into associated vectors than the
+            model's internal embedding lookup matrix.
+        use_cache (`bool`, *optional*):
+            If set to `True`, `past_key_values` key value states are returned and can be used to speed up decoding (see
+            `past_key_values`).
+        output_attentions (`bool`, *optional*):
+            Whether or not to return the attentions tensors of all attention layers. See `attentions` under returned
+            tensors for more detail.
+        output_hidden_states (`bool`, *optional*):
+            Whether or not to return the hidden states of all layers. See `hidden_states` under returned tensors for
+            more detail.
+        return_dict (`bool`, *optional*):
+            Whether or not to return a [`MolmoActCausalLMOutputWithPast`] instead of a plain tuple.
+        cache_position (`torch.LongTensor` of shape `(sequence_length)`, *optional*):
+            Indices depicting the position of the input sequence tokens in the sequence. Contrarily to `position_ids`,
+            this tensor is not affected by padding. It is used to update the cache in the correct position and to infer
+            the complete sequence length.
+"""
+@add_start_docstrings(
+    "The bare MolmoAct model outputting raw hidden-states without any specific head on top.",
+    MOLMO_START_DOCSTRING,
+)
+class MolmoActModel(MolmoActPreTrainedModel):
+    _checkpoint_conversion_mapping = {}
+    def __init__(self, config: MolmoActConfig):
+        super().__init__(config)
+        self.transformer: MolmoActLlm = MolmoActLlm(config.llm_config)
+        self.vision_backbone: Optional[MolmoActVisionBackbone] = None
+        if config.vit_config is not None and config.adapter_config is not None:
+            self.vision_backbone = MolmoActVisionBackbone(config.vit_config, config.adapter_config)
+        # Initialize weights and apply final processing
+        self.post_init()
+    def get_input_embeddings(self) -> torch.nn.Module:
+        return self.transformer.wte
+    def set_input_embeddings(self, value: torch.nn.Module) -> None:
+        self.transformer.wte = value
+    @property
+    def device(self) -> torch.device:
+        return self.transformer.ln_f.weight.device
+    def build_input_embeddings(
+        self,
+        input_ids: torch.LongTensor,
+        images: Optional[torch.FloatTensor] = None,  # image inputs
+        image_masks: Optional[torch.Tensor] = None,
+        pooled_patches_idx: Optional[torch.LongTensor] = None,
+    ) -> Tuple[torch.Tensor, Optional[torch.Tensor]]:
+        # Get embeddings of input.
+        # shape: (batch_size, seq_len, d_model)
+        input_ids = input_ids * (input_ids != -1).to(input_ids.dtype)
+        x = self.transformer.wte(input_ids)
+        image_features: Optional[torch.FloatTensor] = None
+        if images is not None:
+            image_features = self.vision_backbone(images, pooled_patches_idx)
+            is_image_patch = input_ids.view(-1) == self.config.image_patch_id
+            assert is_image_patch.sum() == len(image_features)
+            x.view(-1, x.shape[-1])[is_image_patch] += image_features
+        # shape: (batch_size, seq_len, d_model)
+        x = self.transformer.emb_drop(x)  # type: ignore
+        return x, image_features
+    @can_return_tuple
+    def forward(
+        self,
+        input_ids: Optional[torch.LongTensor] = None,
+        images: Optional[torch.FloatTensor] = None,
+        image_masks: Optional[torch.Tensor] = None,
+        pooled_patches_idx: Optional[torch.Tensor] = None,
+        attention_mask: Optional[torch.Tensor] = None,
+        position_ids: Optional[torch.Tensor] = None,
+        past_key_values: Optional[Union[Cache, List[torch.FloatTensor]]] = None,
+        inputs_embeds: Optional[torch.FloatTensor] = None,
+        use_cache: Optional[bool] = None,
+        output_attentions: Optional[bool] = None,
+        output_hidden_states: Optional[bool] = None,
+        cache_position: Optional[torch.LongTensor] = None,
+    ) -> Union[Tuple, MolmoActModelOutputWithPast]:
+        output_attentions = output_attentions if output_attentions is not None else self.config.output_attentions
+        output_hidden_states = (
+            output_hidden_states if output_hidden_states is not None else self.config.output_hidden_states
+        )
+        use_cache = use_cache if use_cache is not None else self.config.use_cache
+        if (input_ids is None) ^ (inputs_embeds is not None):
+            raise ValueError("You must specify exactly one of input_ids or inputs_embeds")
+        if images is not None and inputs_embeds is not None:
+            raise ValueError(
+                "You cannot specify both images and inputs_embeds at the same time."
+            )
+        if inputs_embeds is None:
+            inputs_embeds, image_features = self.build_input_embeddings(
+                input_ids, images, image_masks, pooled_patches_idx)
+        outputs = self.transformer(
+            attention_mask=attention_mask,
+            position_ids=position_ids,
+            past_key_values=past_key_values,
+            inputs_embeds=inputs_embeds,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            cache_position=cache_position,
+        )
+        return MolmoActModelOutputWithPast(
+            last_hidden_state=outputs.last_hidden_state,
+            past_key_values=outputs.past_key_values,
+            hidden_states=outputs.hidden_states,
+            attentions=outputs.attentions,
+            image_hidden_states=image_features if images is not None else None,
+        )
+@add_start_docstrings(
+    "The MolmoAct model which consists of a vision backbone and a language model + lm head.",
+    MOLMO_START_DOCSTRING,
+)
+class MolmoActForActionReasoning(MolmoActPreTrainedModel, GenerationMixin):
+    _checkpoint_conversion_mapping = {}
+    _tied_weights_keys = []  # Weights are not tied
+    config_class = MolmoActConfig
+    def __init__(self, config: MolmoActConfig):
+        super().__init__(config)
+        self.model = MolmoActModel(config)
+        self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
+        self.vocab_size = config.vocab_size
+        # Initialize weights and apply final processing
+        self.post_init()
+        # --- Action parsing / de-tokenization setup ---
+        # Stats dict expected under config.norm_stats (per-dataset key). If missing, default to empty.
+        self.norm_stats = getattr(config, "norm_stats", None) or {}
+        # Number of discretization bins used for action tokens, defaults to 256.
+        self.n_action_bins = getattr(config, "n_action_bins", 256)
+        # Precompute bin centers in [-1, 1] for inverse token to value mapping.
+        self.bins = np.linspace(-1.0, 1.0, self.n_action_bins)
+        self.bin_centers = (self.bins[:-1] + self.bins[1:]) / 2.0
+        # Lazily constructed tokenizer for converting token strings to ids
+        self._qwen_tokenizer = None
+    def get_input_embeddings(self) -> torch.nn.Module:
+        return self.model.transformer.wte
+    def set_input_embeddings(self, value: torch.nn.Module) -> None:
+        self.model.transformer.wte = value
+    def get_output_embeddings(self):
+        self.lm_head
+    def set_output_embeddings(self, value: torch.nn.Module) -> None:
+        self.lm_head = value
+    # Make modules available throught conditional class for BC
+    @property
+    def language_model(self) -> torch.nn.Module:
+        return self.model.transformer
+    @property
+    def vision_backbone(self) -> torch.nn.Module:
+        return self.model.vision_backbone
+    @can_return_tuple
+    @add_start_docstrings_to_model_forward(MOLMO2_INPUTS_DOCSTRING)
+    def forward(
+        self,
+        input_ids: torch.LongTensor = None,
+        images: Optional[torch.Tensor] = None,
+        image_masks: Optional[torch.Tensor] = None,
+        pooled_patches_idx: Optional[torch.Tensor] = None,
+        attention_mask: Optional[torch.Tensor] = None,
+        position_ids: Optional[torch.LongTensor] = None,
+        past_key_values: Optional[List[torch.FloatTensor]] = None,
+        inputs_embeds: Optional[torch.FloatTensor] = None,
+        labels: Optional[torch.LongTensor] = None,
+        use_cache: Optional[bool] = None,
+        output_attentions: Optional[bool] = None,
+        output_hidden_states: Optional[bool] = None,
+        cache_position: Optional[torch.LongTensor] = None,
+        logits_to_keep: Union[int, torch.Tensor] = 0,
+        **kwargs,
+    ) -> Union[Tuple, MolmoActCausalLMOutputWithPast]:
+        r"""
+        ```python
+        >>> from PIL import Image
+        >>> import requests
+        >>> from transformers import AutoProcessor, MolmoActForActionReasoning
+        >>> model = MolmoActForActionReasoning.from_pretrained("...")
+        >>> processor = AutoProcessor.from_pretrained("...")
+        >>> prompt = "What's the content of the image?"
+        >>> url = "https://www.ilankelman.org/stopsigns/australia.jpg"
+        >>> image = Image.open(requests.get(url, stream=True).raw)
+        >>> inputs = processor(images=image, text=prompt, apply_chat_template=True, return_tensors="pt")
+        >>> # Generate
+        >>> generated_ids = model.generate(**inputs, max_new_tokens=15)
+        >>> generated_tokens = generated_ids[:, inputs['input_ids'].size(1):]
+        >>> processor.batch_decode(generated_tokens, skip_special_tokens=True, clean_up_tokenization_spaces=False)[0]
+        "The image features a busy city street with a stop sign prominently displayed"
+        ```"""
+        outputs = self.model(
+            input_ids=input_ids,
+            images=images,
+            image_masks=image_masks,
+            pooled_patches_idx=pooled_patches_idx,
+            attention_mask=attention_mask,
+            position_ids=position_ids,
+            past_key_values=past_key_values,
+            inputs_embeds=inputs_embeds,
+            use_cache=use_cache,
+            output_attentions=output_attentions,
+            output_hidden_states=output_hidden_states,
+            cache_position=cache_position,
+        )
+        hidden_states = outputs.last_hidden_state
+        slice_indices = slice(-logits_to_keep, None) if isinstance(logits_to_keep, int) else logits_to_keep
+        logits = self.lm_head(hidden_states[:, slice_indices, :])
+        loss = None
+        if labels is not None:
+            loss = self.loss_function(logits=logits, labels=labels, vocab_size=self.vocab_size)
+        return MolmoActCausalLMOutputWithPast(
+            loss=loss,
+            logits=logits,
+            past_key_values=outputs.past_key_values,
+            hidden_states=outputs.hidden_states,
+            attentions=outputs.attentions,
+            image_hidden_states=outputs.image_hidden_states,
+        )
+    # ===== Utilities for action parsing / un-normalization =====
+    def _check_unnorm_key(self, unnorm_key: Optional[str]) -> str:
+        """Validate and resolve which dataset key to use from self.norm_stats."""
+        if not self.norm_stats:
+            raise ValueError("No norm_stats found in config; cannot unnormalize actions.")
+        if unnorm_key is None:
+            if len(self.norm_stats) != 1:
+                raise ValueError(
+                    f"Model has multiple dataset stats; please pass `unnorm_key` from {list(self.norm_stats.keys())}"
+                )
+            return next(iter(self.norm_stats.keys()))
+        if unnorm_key not in self.norm_stats:
+            raise ValueError(f"`unnorm_key`={unnorm_key!r} not in {list(self.norm_stats.keys())}")
+        return unnorm_key
+    def get_action_dim(self, unnorm_key: Optional[str] = None) -> int:
+        """Return action dimensionality from q01 stats length for the dataset key."""
+        key = self._check_unnorm_key(unnorm_key)
+        return len(self.norm_stats[key]["action"]["q01"])
+    def get_action_stats(self, unnorm_key: Optional[str] = None) -> Dict[str, Any]:
+        """Return the full action stats dict for a given dataset key."""
+        key = self._check_unnorm_key(unnorm_key)
+        return self.norm_stats[key]["action"]
+    @torch.no_grad()
+    def parse_action(self, text: str, unnorm_key: Optional[str] = None) -> list:
+        """
+        Parse a generated text to extract one 1×D action token list, decode to continuous values,
+        and unnormalize using dataset-specific stats from `config.norm_stats`.
+        This follows the pipeline used in `experiments/robot/libero/main_libero_10_evaluation.py`:
+        - Find bracketed token lists following the phrase "the action that the robot should take is" (case-insensitive),
+          falling back to any bracketed list in the text.
+        - Convert token strings → ids via Qwen2Tokenizer.
+        - Map ids → discretized bin indices using: `discretized = vocab_size - token_id - 1` (clipped to bins)
+        - Convert bins → normalized actions in [-1, 1] using precomputed `bin_centers`.
+        - Unnormalize with q01/q99 and optional `mask` from norm_stats.
+        Returns:
+            List[float]: unnormalized action vector of length D.
+        """
+        # Resolve action dimension and stats
+        action_dim = self.get_action_dim(unnorm_key)
+        stats = self.get_action_stats(unnorm_key)
+        q01 = np.asarray(stats["q01"], dtype=np.float32)
+        q99 = np.asarray(stats["q99"], dtype=np.float32)
+        mask = np.asarray(stats.get("mask", np.ones_like(q01, dtype=bool)), dtype=bool)
+        # the gripper state should not be normalized
+        mask[-1] = False
+        # Lazily load the tokenizer (shared across calls)
+        if self._qwen_tokenizer is None:
+            self._qwen_tokenizer = Qwen2Tokenizer.from_pretrained("Qwen/Qwen2-7B")
+        token_lists = extract_action_token_lists(text, only_len=action_dim)
+        action_lists = []
+        # Choose the first list (temporal aggregation, if any, should be done by the caller)
+        for tokens in token_lists:
+            # Convert tokens → ids (replace None with vocab_size to avoid negatives)
+            ids = self._qwen_tokenizer.convert_tokens_to_ids(tokens)
+            ids = [self._qwen_tokenizer.vocab_size if i is None else int(i) for i in ids]
+            ids = np.asarray(ids, dtype=np.int64)
+            # ids → discretized bin indices → normalized actions in [-1, 1]
+            discretized = self._qwen_tokenizer.vocab_size - ids
+            discretized = np.clip(discretized - 1, a_min=0, a_max=self.bin_centers.shape[0] - 1)
+            normalized = self.bin_centers[discretized]
+            # Unnormalize using per-dimension statistics
+            unnorm = 0.5 * (normalized + 1.0) * (q99 - q01) + q01
+            actions = np.where(mask, unnorm, normalized)
+            action_lists.append([float(x) for x in actions])
+        # Return a Python list of float actions
+        return action_lists
+    @torch.no_grad()
+    def parse_trace(self, text: str) -> list:
+        return extract_trace_lists(text, point_len=2, min_points=1)
+    @torch.no_grad()
+    def parse_depth(self, text: str) -> list:
+        return extract_depth_string(text, include_tags=True)
+    def prepare_inputs_for_generation(
+        self,
+        input_ids: torch.LongTensor,
+        past_key_values: Optional[List[torch.FloatTensor]] = None,
+        inputs_embeds: Optional[torch.FloatTensor] = None,
+        images: Optional[torch.FloatTensor] = None,
+        image_masks: Optional[torch.Tensor] = None,
+        pooled_patches_idx: Optional[torch.Tensor] = None,
+        attention_mask: Optional[torch.Tensor] = None,
+        cache_position: Optional[torch.LongTensor] = None,
+        logits_to_keep: Optional[Union[int, torch.Tensor]] = None,
+        **kwargs,
+    ):
+        model_inputs = super().prepare_inputs_for_generation(
+            input_ids,
+            past_key_values=past_key_values,
+            inputs_embeds=inputs_embeds,
+            attention_mask=attention_mask,
+            cache_position=cache_position,
+            logits_to_keep=logits_to_keep,
+            **kwargs,
+        )
+        if cache_position[0] == 0:
+            model_inputs["images"] = images
+            model_inputs["pooled_patches_idx"] = pooled_patches_idx
+            model_inputs["image_masks"] = image_masks
+        return model_inputs
+    def _update_model_kwargs_for_generation(
+        self,
+        outputs: ModelOutput,
+        model_kwargs: Dict[str, Any],
+        is_encoder_decoder: bool = False,
+        num_new_tokens: int = 1,
+    ) -> Dict[str, Any]:
+        if model_kwargs["use_cache"] and "images" in model_kwargs:
+            # After the first step, no long pass the images into forward since the images tokens
+            # are already cached
+            for k in ["images", "image_masks", "pooled_patches_idx"]:
+                del model_kwargs[k]
+        return super()._update_model_kwargs_for_generation(outputs, model_kwargs, is_encoder_decoder, num_new_tokens)
+    @staticmethod
+    def _prepare_4d_causal_attention_mask_with_cache_position(
+        attention_mask: torch.Tensor,
+        sequence_length: int,
+        target_length: int,
+        dtype: torch.dtype,
+        cache_position: torch.Tensor,
+        batch_size: int,
+        **kwargs,
+    ):
+        """
+        Creates a causal 4D mask of shape `(batch_size, 1, query_length, key_value_length)` from a 2D mask of shape
+        `(batch_size, key_value_length)`, or if the input `attention_mask` is already 4D, do nothing.
+        Args:
+            attention_mask (`torch.Tensor`):
+                A 2D attention mask of shape `(batch_size, key_value_length)` or a 4D attention mask of shape
+                `(batch_size, 1, query_length, key_value_length)`.
+            sequence_length (`int`):
+                The sequence length being processed.
+            target_length (`int`):
+                The target length: when generating with static cache, the mask should be as long as the static cache,
+                to account for the 0 padding, the part of the cache that is not filled yet.
+            dtype (`torch.dtype`):
+                The dtype to use for the 4D attention mask.
+            cache_position (`torch.Tensor`):
+                Indices depicting the position of the input sequence tokens in the sequence.
+            batch_size (`torch.Tensor`):
+                Batch size.
+        """
+        if attention_mask is not None and attention_mask.dim() == 4:
+            # In this case we assume that the mask comes already in inverted form and requires no inversion or slicing.
+            causal_mask = attention_mask
+        else:
+            min_dtype = torch.finfo(dtype).min
+            causal_mask = torch.full(
+                (sequence_length, target_length), fill_value=min_dtype, dtype=dtype, device=cache_position.device
+            )
+            if sequence_length != 1:
+                causal_mask = torch.triu(causal_mask, diagonal=1)
+            causal_mask *= torch.arange(target_length, device=cache_position.device) > cache_position.reshape(-1, 1)
+            causal_mask = causal_mask[None, None, :, :].expand(batch_size, 1, -1, -1)
+            if attention_mask is not None:
+                causal_mask = causal_mask.clone()  # copy to contiguous memory for in-place edit
+                mask_length = attention_mask.shape[-1]
+                padding_mask = causal_mask[:, :, :, :mask_length] + attention_mask[:, None, None, :].to(
+                    causal_mask.device
+                )
+                padding_mask = padding_mask == 0
+                causal_mask[:, :, :, :mask_length] = causal_mask[:, :, :, :mask_length].masked_fill(
+                    padding_mask, min_dtype
+                )
+        return causal_mask
+# Always register for multi-modal features
+AutoModelForImageTextToText.register(MolmoActConfig, MolmoActForActionReasoning)
+AutoModelForCausalLM.register(MolmoActLlmConfig, MolmoActForCausalLM)

preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,27 @@

+{
+  "auto_map": {
+    "AutoImageProcessor": "image_processing_molmoact.MolmoActImageProcessor",
+    "AutoProcessor": "processing_molmoact.MolmoActProcessor"
+  },
+  "base_image_input_size": [
+    378,
+    378
+  ],
+  "crop_mode": "overlap-and-resize-c2",
+  "do_convert_rgb": true,
+  "do_pad": true,
+  "image_patch_size": 14,
+  "image_pooling_h": 2,
+  "image_pooling_w": 2,
+  "image_processor_type": "MolmoActImageProcessor",
+  "max_crops": 8,
+  "max_multi_image_crops": 8,
+  "normalize_mode": "siglip",
+  "overlap_margins": [
+    4,
+    4
+  ],
+  "pad_value": 0.0,
+  "processor_class": "MolmoActProcessor",
+  "resize_mode": "siglip"
+}

processing_molmoact.py ADDED Viewed

	@@ -0,0 +1,463 @@

+"""
+Processor class for MolmoAct.
+"""
+from typing import List, Optional, Union, Dict, Tuple
+import PIL
+from PIL import ImageFile, ImageOps
+try:
+    from typing import Unpack
+except ImportError:
+    from typing_extensions import Unpack
+import numpy as np
+import torch
+from transformers.image_utils import ImageInput
+from transformers.processing_utils import (
+    ProcessingKwargs,
+    ProcessorMixin,
+)
+from transformers.feature_extraction_utils import BatchFeature
+from transformers.tokenization_utils_base import TextInput, PreTokenizedInput
+from transformers.utils import logging
+from transformers import AutoTokenizer
+from .image_processing_molmoact import MolmoActImagesKwargs, MolmoActImageProcessor
+logger = logging.get_logger(__name__)
+# Special tokens, these should be present in any tokenizer we use since the preprocessor uses them
+IMAGE_PATCH_TOKEN = f"<im_patch>"  # Where to insert high-res tokens
+IMAGE_LOW_RES_TOKEN = f"<im_low>"  # Where to insert low-res tokens
+IM_START_TOKEN = f"<im_start>"
+IM_END_TOKEN = f"<im_end>"
+IM_COL_TOKEN = f"<im_col>"
+IMAGE_PROMPT = "<|image|>"
+EXTRA_TOKENS = (IM_START_TOKEN, IM_END_TOKEN, IMAGE_PATCH_TOKEN,
+                IM_COL_TOKEN, IMAGE_PROMPT, IMAGE_LOW_RES_TOKEN)
+DEMO_STYLES = [
+    "point_count",
+    "pointing",
+    "cosyn_point",
+    "user_qa",
+    "long_caption",
+    "short_caption",
+    "correction_qa",
+    "demo",
+    "android_control",
+]
+def setup_pil():
+    PIL.Image.MAX_IMAGE_PIXELS = None
+    ImageFile.LOAD_TRUNCATED_IMAGES = True
+def get_special_token_ids(tokenizer: AutoTokenizer) -> Dict[str, int]:
+    ids = tokenizer.encode("".join(EXTRA_TOKENS), add_special_tokens=False)
+    assert len(ids) == len(EXTRA_TOKENS)
+    return {k: i for k, i in zip(EXTRA_TOKENS, ids)}
+def load_image(image: Union[PIL.Image.Image, np.ndarray]) -> np.ndarray:
+    """Load image"""
+    setup_pil()
+    if isinstance(image, PIL.Image.Image):
+        image = image.convert("RGB")
+        image = ImageOps.exif_transpose(image)
+        return np.array(image)
+    elif isinstance(image, np.ndarray):
+        assert len(image.shape) == 3, "Image should have 3 dimensions"
+        assert image.shape[2] == 3, "Image should have 3 channels"
+        assert image.dtype == np.uint8, "Image should have uint8 type"
+        return image
+    else:
+        raise ValueError("Image should be PIL.Image or np.ndarray")
+class MolmoActProcessorKwargs(ProcessingKwargs, total=False):
+    """MolmoAct processor kwargs"""
+    images_kwargs: MolmoActImagesKwargs
+    _defaults = {
+        "text_kwargs": {
+            "padding": False,
+        },
+    }
+class MolmoActProcessor(ProcessorMixin):
+    attributes = ["image_processor", "tokenizer"]
+    optional_attributes = [
+        "chat_template",
+        "prompt_templates",
+        "message_format",
+        "system_prompt",
+        "style",
+        "always_start_with_space",
+        "default_inference_len",
+        "use_col_tokens",
+        "image_padding_mask",
+    ]
+    image_processor_class = "AutoImageProcessor"
+    tokenizer_class = "AutoTokenizer"
+    def __init__(
+        self,
+        image_processor: MolmoActImageProcessor = None,
+        tokenizer: AutoTokenizer = None,
+        chat_template: Optional[str] = None,
+        prompt_templates: Optional[str] = "uber_model",
+        message_format: Optional[str] = "role",
+        system_prompt: Optional[str] = "demo_or_style",
+        style: Optional[str] = "demo",
+        always_start_with_space: Optional[bool] = False,
+        default_inference_len: Optional[int] = 65,
+        use_col_tokens: Optional[bool] = True,
+        image_padding_mask: bool = False,
+        **kwargs
+    ) -> None:
+        if tokenizer.padding_side != "left":
+            logger.warning(f"Tokenizer {tokenizer.name_or_path} is not left-padded, padding side will be set to left")
+            tokenizer.padding_side = "left"  # type: ignore
+        super().__init__(
+            image_processor,
+            tokenizer,
+            chat_template=chat_template,
+            prompt_templates=prompt_templates,
+            message_format=message_format,
+            system_prompt=system_prompt,
+            style=style,
+            always_start_with_space=always_start_with_space,
+            default_inference_len=default_inference_len,
+            use_col_tokens=use_col_tokens,
+            image_padding_mask=image_padding_mask,
+        )
+        self._special_tokens = None
+    @property
+    def special_token_ids(self):
+        if self._special_tokens is None:
+            self._special_tokens = get_special_token_ids(self.tokenizer)
+        return self._special_tokens
+    def get_user_prompt(self, text: TextInput) -> str:
+        """Get user prompt"""
+        if self.prompt_templates == "none":
+            return ""
+        elif self.prompt_templates == "uber_model":
+            return text
+        else:
+            raise NotImplementedError(self.prompt_templates)
+    def get_prefix(self) -> str:
+        """Get prefix"""
+        if self.system_prompt == "style_and_length":  # captioner
+            assert self.style in ["long_caption"]
+            style = self.style
+            n = None if self.default_inference_len is None else str(self.default_inference_len)
+            if n is not None and len(n) > 0:  # allow empty string to signal unconditioned
+                prefix = style + " " + n + ":"
+            else:
+                prefix = style + " :"
+        elif self.system_prompt == "demo_or_style":  # demo model
+            if self.style in DEMO_STYLES:
+                prefix = ""
+            else:
+                prefix = self.style + ":"
+        else:
+            raise NotImplementedError(self.system_prompt)
+        return prefix
+    def format_prompt(self, prompt: str) -> str:
+        """Format prompt"""
+        if self.message_format == "none":
+            pass
+        elif self.message_format == "role":
+            prompt = "User: " + prompt + " Assistant:"
+        else:
+            raise NotImplementedError(self.message_format)
+        if self.always_start_with_space:
+            prompt = " " + prompt
+        return prompt
+    def get_prompt(self, text: TextInput) -> str:
+        prompt = self.get_user_prompt(text)
+        if self.system_prompt and self.system_prompt != "none":
+            prefix = self.get_prefix()
+            if len(prefix) > 0 and len(prompt) > 0:
+                prompt = prefix + " " + prompt
+            elif len(prefix) > 0:
+                prompt = prefix
+        prompt = self.format_prompt(prompt)
+        return prompt
+    def get_image_tokens(self, image_grid: np.ndarray):
+        joint = []
+        for h, w in image_grid:
+            per_row = np.full(w, IMAGE_PATCH_TOKEN)
+            if self.use_col_tokens:
+                per_row = np.concatenate([per_row, [IM_COL_TOKEN]], 0)
+            extra_tokens = np.tile(per_row, [h])
+            joint += [
+                [IM_START_TOKEN],
+                extra_tokens,
+                [IM_END_TOKEN],
+            ]
+        return np.concatenate(joint)
+    def insert_bos_numpy(
+        self,
+        input_ids: np.ndarray,
+        attention_mask: np.ndarray,
+        bos_token_id: int,
+        pad_token_id: int,
+    ):
+        """
+        Args:
+            input_ids: [B, S] array with left padding
+            attention_mask: [B, S] array (0 for pad, 1 for valid)
+            bos_token_id: int
+            pad_token_id: int
+        Returns:
+            input_ids_out: [B, S] or [B, S+1] array with bos inserted if needed
+            attention_mask_out: same shape as input_ids_out
+        """
+        need_to_expand = len(input_ids.shape) == 1
+        if need_to_expand:
+            input_ids = input_ids[None, :]
+            attention_mask = attention_mask[None, :]
+        B, S = input_ids.shape
+        # Handle zero-length sequence
+        if S == 0:
+            new_input_ids = np.full((B, 1), bos_token_id, dtype=input_ids.dtype)
+            new_attention_mask = np.ones((B, 1), dtype=attention_mask.dtype)
+            if need_to_expand:
+                new_input_ids = new_input_ids[0]
+                new_attention_mask = new_attention_mask[0]
+            return new_input_ids, new_attention_mask
+        first_valid_index = (attention_mask == 1).argmax(axis=-1)  # [B]
+        bos_already_present = np.all(input_ids[np.arange(B), first_valid_index] == bos_token_id)
+        if bos_already_present:
+            if need_to_expand:
+                input_ids = input_ids[0]
+                attention_mask = attention_mask[0]
+            return input_ids, attention_mask
+        else:
+            new_input_ids = np.full((B, S+1), pad_token_id, dtype=input_ids.dtype)
+            new_attention_mask = np.zeros((B, S+1), dtype=attention_mask.dtype)
+            src_idx = np.tile(np.arange(S), (B, 1))  # [B, S]
+            valid_mask = src_idx >= first_valid_index[:, None]  # [B, S]
+            tgt_idx = src_idx + 1  # shit right
+            batch_idx = np.tile(np.arange(B)[:, None], (1, S))  # [B, S]
+            # flatten valid_positions
+            flat_vals = input_ids[valid_mask]
+            flat_batch = batch_idx[valid_mask]
+            flat_tgt = tgt_idx[valid_mask]
+            new_input_ids[flat_batch, flat_tgt] = flat_vals
+            new_attention_mask[flat_batch, flat_tgt] = 1
+            insert_pos = first_valid_index
+            new_input_ids[np.arange(B), insert_pos] = bos_token_id
+            new_attention_mask[np.arange(B), insert_pos] = 1
+            if need_to_expand:
+                new_input_ids = new_input_ids[0]
+                new_attention_mask = new_attention_mask[0]
+            return new_input_ids, new_attention_mask
+    def insert_bos_torch(
+        self,
+        input_ids: torch.Tensor,
+        attention_mask: torch.Tensor,
+        bos_token_id: int,
+        pad_token_id: int,
+    ):
+        """
+        Args:
+            input_ids: [B, S] tensor with left padding
+            attention_mask: [B, S] tensor (0 for pad, 1 for valid)
+            bos_token_id: int
+            pad_token_id: int
+        Returns:
+            input_ids_out: [B, S] or [B, S+1] tensor with bos inserted if needed
+            attention_mask_out: same shape as input_ids_out
+        """
+        B, S = input_ids.shape
+        device = input_ids.device
+        # Handle zero-length sequence
+        if S == 0:
+            new_input_ids = torch.full((B, 1), bos_token_id, dtype=input_ids.dtype, device=device)
+            new_attention_mask = torch.ones((B, 1), dtype=attention_mask.dtype, device=device)
+            return new_input_ids, new_attention_mask
+        first_valid_index = (attention_mask == 1).long().argmax(dim=-1)  # [B]
+        bos_already_present = (input_ids[torch.arange(B), first_valid_index] == bos_token_id).all()
+        if bos_already_present:
+            return input_ids, attention_mask
+        else:
+            new_input_ids = torch.full((B, S+1), pad_token_id, dtype=input_ids.dtype, device=device)
+            new_attention_mask = torch.zeros((B, S+1), dtype=attention_mask.dtype, device=device)
+            src_idx = torch.arange(S, device=device).expand(B, S)  # [B, S]
+            valid_mask = src_idx >= first_valid_index.unsqueeze(1)  # [B, S]
+            tgt_idx = src_idx + 1  # shift right
+            batch_idx = torch.arange(B, device=device).unsqueeze(1).expand_as(src_idx)
+            flat_vals = input_ids[valid_mask]
+            flat_batch = batch_idx[valid_mask]
+            flat_tgt = tgt_idx[valid_mask]
+            new_input_ids[flat_batch, flat_tgt] = flat_vals
+            new_attention_mask[flat_batch, flat_tgt] = 1
+            insert_pos = first_valid_index
+            batch_indices = torch.arange(B, device=device)
+            new_input_ids[batch_indices, insert_pos] = bos_token_id
+            new_attention_mask[batch_indices, insert_pos] = 1
+            return new_input_ids, new_attention_mask
+    def __call__(
+        self,
+        text: Union[TextInput, PreTokenizedInput, List[TextInput], List[PreTokenizedInput]] = None,
+        images: Union[ImageInput, List[ImageInput]] = None,
+        apply_chat_template: bool = False,
+        **kwargs: Unpack[MolmoActProcessorKwargs],
+    ) -> BatchFeature:
+        if images is None and text is None:
+            raise ValueError("You have to specify at least one of `images` or `text`.")
+        output_kwargs = self._merge_kwargs(
+            MolmoActProcessorKwargs,
+            tokenizer_init_kwargs=self.tokenizer.init_kwargs,
+            **kwargs,
+        )
+        if isinstance(text, (list, tuple)) and isinstance(images, (list, tuple)):
+            if len(text) != len(images):
+                raise ValueError("You have to provide the same number of text and images")
+            if len(text) > 1 and not output_kwargs["text_kwargs"].get("padding", False):
+                raise ValueError("You have to specify padding when you have multiple text inputs")
+        if isinstance(text, str):
+            text = [text]
+        elif not isinstance(text, list) and not isinstance(text[0], str):
+            raise ValueError("Invalid input text. Please provide a string, or a list of strings")
+        if images is not None:
+            image_inputs = self.image_processor(images, **output_kwargs["images_kwargs"])
+        else:
+            image_inputs = {}
+        if apply_chat_template:
+            text = [self.get_prompt(t) for t in text]
+        prompt_strings = text
+        if image_inputs.get("images", None) is not None:
+            prompt_strings = []
+            for idx, image_grids in enumerate(image_inputs.pop("image_grids")):
+                if isinstance(image_grids, torch.Tensor):
+                    image_grids = image_grids.cpu().numpy()
+                if isinstance(images, (list, tuple)) and isinstance(images[idx], (list, tuple)):
+                    image_grids = image_grids[~np.all(image_grids == -1, axis=-1)]
+                    offset = 2 if len(images[idx]) < len(image_grids) else 1 # whether to use both low and high res images
+                    all_image_strings = []
+                    for i in range(0, len(image_grids), offset):
+                        image_grids_i = image_grids[i:i+offset]
+                        image_tokens = self.get_image_tokens(image_grids_i)
+                        img_ix = i // offset
+                        all_image_strings.append(f"Image {img_ix + 1}" + "".join(image_tokens))
+                    image_string = "".join(all_image_strings)
+                    prompt_strings.append(image_string + text[idx])
+                else:
+                    image_grids = image_grids[~np.all(image_grids == -1, axis=-1)]
+                    assert len(image_grids) in [1, 2], "Only one or two crops are supported for single image inputs"
+                    image_tokens = self.get_image_tokens(image_grids)
+                    image_string = "".join(image_tokens)
+                    prompt_strings.append(image_string + text[idx])
+        text_inputs = self.tokenizer(prompt_strings, **output_kwargs["text_kwargs"])
+        input_ids = text_inputs["input_ids"]
+        attention_mask = text_inputs["attention_mask"]
+        is_list = isinstance(input_ids, (list, tuple))
+        if is_list:
+            input_ids = np.array(input_ids)
+            attention_mask = np.array(attention_mask)
+        use_numpy = isinstance(attention_mask, np.ndarray)
+        if use_numpy and np.issubdtype(input_ids.dtype, np.floating):
+            input_ids = input_ids.astype(np.int64)
+            attention_mask = attention_mask.astype(np.int64)
+        elif not use_numpy and torch.is_floating_point(input_ids):
+            input_ids = input_ids.to(torch.int64)
+            attention_mask = attention_mask.to(torch.int64)
+        bos = self.tokenizer.bos_token_id or self.tokenizer.eos_token_id
+        if use_numpy:
+            input_ids, attention_mask = self.insert_bos_numpy(
+                input_ids, attention_mask, bos, self.tokenizer.pad_token_id
+            )
+        else:
+            input_ids, attention_mask = self.insert_bos_torch(
+                input_ids, attention_mask, bos, self.tokenizer.pad_token_id
+            )
+        if is_list:
+            input_ids = input_ids.tolist()  # type: ignore
+            attention_mask = attention_mask.tolist()  # type: ignore
+        text_inputs["input_ids"] = input_ids
+        text_inputs["attention_mask"] = attention_mask
+        if kwargs.get("device", None) is not None:
+            text_inputs = text_inputs.to(device=kwargs.get("device"), non_blocking=True)
+        # there is no bos token in Qwen tokenizer
+        return BatchFeature(
+            data={**text_inputs, **image_inputs}, tensor_type=output_kwargs["common_kwargs"]["return_tensors"]
+        )
+    def batch_decode(self, *args, **kwargs):
+        """
+        This method forwards all its arguments to LlamaTokenizerFast's [`~PreTrainedTokenizer.batch_decode`]. Please
+        refer to the docstring of this method for more information.
+        """
+        return self.tokenizer.batch_decode(*args, **kwargs)
+    def decode(self, *args, **kwargs):
+        """
+        This method forwards all its arguments to LlamaTokenizerFast's [`~PreTrainedTokenizer.decode`]. Please refer to
+        the docstring of this method for more information.
+        """
+        return self.tokenizer.decode(*args, **kwargs)
+    @property
+    def model_input_names(self):
+        tokenizer_input_names = self.tokenizer.model_input_names
+        image_processor_input_names = self.image_processor.model_input_names
+        return list(dict.fromkeys(tokenizer_input_names + image_processor_input_names))
+MolmoActProcessor.register_for_auto_class()

processor_config.json ADDED Viewed

	@@ -0,0 +1,14 @@

+{
+  "always_start_with_space": false,
+  "auto_map": {
+    "AutoProcessor": "processing_molmoact.MolmoActProcessor"
+  },
+  "default_inference_len": 65,
+  "image_padding_mask": false,
+  "message_format": "role",
+  "processor_class": "MolmoActProcessor",
+  "prompt_templates": "uber_model",
+  "style": "demo_role",
+  "system_prompt": "demo_or_style",
+  "use_col_tokens": true
+}

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,1944 @@

+{
+  "additional_special_tokens": [
+    {
+      "content": "|<EXTRA_TOKENS_0>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_1>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_2>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_3>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_4>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_5>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_6>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_7>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_8>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_9>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_10>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_11>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_12>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_13>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_14>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_15>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_16>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_17>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_18>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_19>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_20>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_21>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_22>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_23>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_24>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_25>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_26>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_27>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_28>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_29>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_30>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_31>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_32>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_33>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_34>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_35>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_36>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_37>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_38>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_39>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_40>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_41>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_42>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_43>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_44>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_45>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_46>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_47>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_48>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_49>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_50>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_51>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_52>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_53>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_54>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_55>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_56>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_57>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_58>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_59>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_60>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_61>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_62>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_63>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_64>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_65>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_66>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_67>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_68>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_69>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_70>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_71>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_72>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_73>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_74>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_75>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_76>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_77>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_78>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_79>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_80>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_81>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_82>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_83>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_84>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_85>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_86>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_87>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_88>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_89>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_90>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_91>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_92>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_93>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_94>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_95>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_96>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_97>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_98>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_99>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_100>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_101>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_102>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_103>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_104>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_105>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_106>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_107>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_108>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_109>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_110>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_111>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_112>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_113>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_114>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_115>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_116>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_117>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_118>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_119>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_120>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_121>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_122>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_123>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_124>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_125>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_126>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_127>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_128>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_129>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_130>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_131>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_132>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_133>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_134>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_135>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_136>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_137>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_138>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_139>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_140>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_141>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_142>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_143>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_144>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_145>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_146>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_147>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_148>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_149>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_150>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_151>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_152>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_153>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_154>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_155>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_156>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_157>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_158>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_159>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_160>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_161>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_162>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_163>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_164>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_165>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_166>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_167>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_168>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_169>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_170>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_171>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_172>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_173>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_174>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_175>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_176>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_177>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_178>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_179>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_180>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_181>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_182>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_183>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_184>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_185>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_186>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_187>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_188>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_189>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_190>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_191>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_192>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_193>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_194>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_195>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_196>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_197>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_198>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_199>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_200>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_201>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_202>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_203>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_204>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_205>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_206>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_207>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_208>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_209>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_210>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_211>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_212>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_213>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_214>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_215>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_216>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_217>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_218>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_219>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_220>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_221>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_222>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_223>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_224>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_225>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_226>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_227>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_228>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_229>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_230>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_231>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_232>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_233>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_234>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_235>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_236>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_237>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_238>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_239>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_240>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_241>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_242>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_243>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_244>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_245>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_246>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_247>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_248>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_249>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_250>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_251>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_252>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_253>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_254>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_255>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_256>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_257>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_258>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_259>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_260>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_261>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_262>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_263>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_264>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_265>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_266>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_267>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "|<EXTRA_TOKENS_268>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "<im_start>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "<im_end>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "<im_patch>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "<im_col>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "<|image|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "<im_low>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    }
+  ],
+  "bos_token": "<|endoftext|>",
+  "eos_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:70522ad61c51fe8b137105665e222eea81d787e5603c75641b02ba5480628ad6
+size 11500226

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,3713 @@

+{
+  "add_bos_token": false,
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "151643": {
+      "content": "<|endoftext|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151644": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151645": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151646": {
+      "content": "<|object_ref_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151647": {
+      "content": "<|object_ref_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151648": {
+      "content": "<|box_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151649": {
+      "content": "<|box_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151650": {
+      "content": "<|quad_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151651": {
+      "content": "<|quad_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151652": {
+      "content": "<|vision_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151653": {
+      "content": "<|vision_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151654": {
+      "content": "<|vision_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151655": {
+      "content": "<|image_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151656": {
+      "content": "<|video_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151657": {
+      "content": "<tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151658": {
+      "content": "</tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151659": {
+      "content": "<|fim_prefix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151660": {
+      "content": "<|fim_middle|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151661": {
+      "content": "<|fim_suffix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151662": {
+      "content": "<|fim_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151663": {
+      "content": "<|repo_name|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151664": {
+      "content": "<|file_sep|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151665": {
+      "content": "<DEPTH_START>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151666": {
+      "content": "<DEPTH_END>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151667": {
+      "content": "<DEPTH_0>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151668": {
+      "content": "<DEPTH_1>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151669": {
+      "content": "<DEPTH_2>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151670": {
+      "content": "<DEPTH_3>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151671": {
+      "content": "<DEPTH_4>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151672": {
+      "content": "<DEPTH_5>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151673": {
+      "content": "<DEPTH_6>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151674": {
+      "content": "<DEPTH_7>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151675": {
+      "content": "<DEPTH_8>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151676": {
+      "content": "<DEPTH_9>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151677": {
+      "content": "<DEPTH_10>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151678": {
+      "content": "<DEPTH_11>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151679": {
+      "content": "<DEPTH_12>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151680": {
+      "content": "<DEPTH_13>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151681": {
+      "content": "<DEPTH_14>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151682": {
+      "content": "<DEPTH_15>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151683": {
+      "content": "<DEPTH_16>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151684": {
+      "content": "<DEPTH_17>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151685": {
+      "content": "<DEPTH_18>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151686": {
+      "content": "<DEPTH_19>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151687": {
+      "content": "<DEPTH_20>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151688": {
+      "content": "<DEPTH_21>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151689": {
+      "content": "<DEPTH_22>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151690": {
+      "content": "<DEPTH_23>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151691": {
+      "content": "<DEPTH_24>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151692": {
+      "content": "<DEPTH_25>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151693": {
+      "content": "<DEPTH_26>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151694": {
+      "content": "<DEPTH_27>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151695": {
+      "content": "<DEPTH_28>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151696": {
+      "content": "<DEPTH_29>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151697": {
+      "content": "<DEPTH_30>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151698": {
+      "content": "<DEPTH_31>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151699": {
+      "content": "<DEPTH_32>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151700": {
+      "content": "<DEPTH_33>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151701": {
+      "content": "<DEPTH_34>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151702": {
+      "content": "<DEPTH_35>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151703": {
+      "content": "<DEPTH_36>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151704": {
+      "content": "<DEPTH_37>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151705": {
+      "content": "<DEPTH_38>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151706": {
+      "content": "<DEPTH_39>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151707": {
+      "content": "<DEPTH_40>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151708": {
+      "content": "<DEPTH_41>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151709": {
+      "content": "<DEPTH_42>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151710": {
+      "content": "<DEPTH_43>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151711": {
+      "content": "<DEPTH_44>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151712": {
+      "content": "<DEPTH_45>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151713": {
+      "content": "<DEPTH_46>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151714": {
+      "content": "<DEPTH_47>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151715": {
+      "content": "<DEPTH_48>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151716": {
+      "content": "<DEPTH_49>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151717": {
+      "content": "<DEPTH_50>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151718": {
+      "content": "<DEPTH_51>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151719": {
+      "content": "<DEPTH_52>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151720": {
+      "content": "<DEPTH_53>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151721": {
+      "content": "<DEPTH_54>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151722": {
+      "content": "<DEPTH_55>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151723": {
+      "content": "<DEPTH_56>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151724": {
+      "content": "<DEPTH_57>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151725": {
+      "content": "<DEPTH_58>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151726": {
+      "content": "<DEPTH_59>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151727": {
+      "content": "<DEPTH_60>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151728": {
+      "content": "<DEPTH_61>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151729": {
+      "content": "<DEPTH_62>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151730": {
+      "content": "<DEPTH_63>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151731": {
+      "content": "<DEPTH_64>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151732": {
+      "content": "<DEPTH_65>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151733": {
+      "content": "<DEPTH_66>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151734": {
+      "content": "<DEPTH_67>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151735": {
+      "content": "<DEPTH_68>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151736": {
+      "content": "<DEPTH_69>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151737": {
+      "content": "<DEPTH_70>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151738": {
+      "content": "<DEPTH_71>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151739": {
+      "content": "<DEPTH_72>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151740": {
+      "content": "<DEPTH_73>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151741": {
+      "content": "<DEPTH_74>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151742": {
+      "content": "<DEPTH_75>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151743": {
+      "content": "<DEPTH_76>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151744": {
+      "content": "<DEPTH_77>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151745": {
+      "content": "<DEPTH_78>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151746": {
+      "content": "<DEPTH_79>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151747": {
+      "content": "<DEPTH_80>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151748": {
+      "content": "<DEPTH_81>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151749": {
+      "content": "<DEPTH_82>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151750": {
+      "content": "<DEPTH_83>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151751": {
+      "content": "<DEPTH_84>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151752": {
+      "content": "<DEPTH_85>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151753": {
+      "content": "<DEPTH_86>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151754": {
+      "content": "<DEPTH_87>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151755": {
+      "content": "<DEPTH_88>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151756": {
+      "content": "<DEPTH_89>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151757": {
+      "content": "<DEPTH_90>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151758": {
+      "content": "<DEPTH_91>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151759": {
+      "content": "<DEPTH_92>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151760": {
+      "content": "<DEPTH_93>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151761": {
+      "content": "<DEPTH_94>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151762": {
+      "content": "<DEPTH_95>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151763": {
+      "content": "<DEPTH_96>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151764": {
+      "content": "<DEPTH_97>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151765": {
+      "content": "<DEPTH_98>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151766": {
+      "content": "<DEPTH_99>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151767": {
+      "content": "<DEPTH_100>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151768": {
+      "content": "<DEPTH_101>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151769": {
+      "content": "<DEPTH_102>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151770": {
+      "content": "<DEPTH_103>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151771": {
+      "content": "<DEPTH_104>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151772": {
+      "content": "<DEPTH_105>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151773": {
+      "content": "<DEPTH_106>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151774": {
+      "content": "<DEPTH_107>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151775": {
+      "content": "<DEPTH_108>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151776": {
+      "content": "<DEPTH_109>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151777": {
+      "content": "<DEPTH_110>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151778": {
+      "content": "<DEPTH_111>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151779": {
+      "content": "<DEPTH_112>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151780": {
+      "content": "<DEPTH_113>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151781": {
+      "content": "<DEPTH_114>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151782": {
+      "content": "<DEPTH_115>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151783": {
+      "content": "<DEPTH_116>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151784": {
+      "content": "<DEPTH_117>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151785": {
+      "content": "<DEPTH_118>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151786": {
+      "content": "<DEPTH_119>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151787": {
+      "content": "<DEPTH_120>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151788": {
+      "content": "<DEPTH_121>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151789": {
+      "content": "<DEPTH_122>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151790": {
+      "content": "<DEPTH_123>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151791": {
+      "content": "<DEPTH_124>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151792": {
+      "content": "<DEPTH_125>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151793": {
+      "content": "<DEPTH_126>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151794": {
+      "content": "<DEPTH_127>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151795": {
+      "content": "|<EXTRA_TOKENS_0>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151796": {
+      "content": "|<EXTRA_TOKENS_1>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151797": {
+      "content": "|<EXTRA_TOKENS_2>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151798": {
+      "content": "|<EXTRA_TOKENS_3>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151799": {
+      "content": "|<EXTRA_TOKENS_4>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151800": {
+      "content": "|<EXTRA_TOKENS_5>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151801": {
+      "content": "|<EXTRA_TOKENS_6>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151802": {
+      "content": "|<EXTRA_TOKENS_7>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151803": {
+      "content": "|<EXTRA_TOKENS_8>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151804": {
+      "content": "|<EXTRA_TOKENS_9>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151805": {
+      "content": "|<EXTRA_TOKENS_10>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151806": {
+      "content": "|<EXTRA_TOKENS_11>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151807": {
+      "content": "|<EXTRA_TOKENS_12>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151808": {
+      "content": "|<EXTRA_TOKENS_13>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151809": {
+      "content": "|<EXTRA_TOKENS_14>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151810": {
+      "content": "|<EXTRA_TOKENS_15>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151811": {
+      "content": "|<EXTRA_TOKENS_16>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151812": {
+      "content": "|<EXTRA_TOKENS_17>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151813": {
+      "content": "|<EXTRA_TOKENS_18>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151814": {
+      "content": "|<EXTRA_TOKENS_19>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151815": {
+      "content": "|<EXTRA_TOKENS_20>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151816": {
+      "content": "|<EXTRA_TOKENS_21>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151817": {
+      "content": "|<EXTRA_TOKENS_22>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151818": {
+      "content": "|<EXTRA_TOKENS_23>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151819": {
+      "content": "|<EXTRA_TOKENS_24>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151820": {
+      "content": "|<EXTRA_TOKENS_25>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151821": {
+      "content": "|<EXTRA_TOKENS_26>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151822": {
+      "content": "|<EXTRA_TOKENS_27>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151823": {
+      "content": "|<EXTRA_TOKENS_28>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151824": {
+      "content": "|<EXTRA_TOKENS_29>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151825": {
+      "content": "|<EXTRA_TOKENS_30>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151826": {
+      "content": "|<EXTRA_TOKENS_31>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151827": {
+      "content": "|<EXTRA_TOKENS_32>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151828": {
+      "content": "|<EXTRA_TOKENS_33>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151829": {
+      "content": "|<EXTRA_TOKENS_34>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151830": {
+      "content": "|<EXTRA_TOKENS_35>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151831": {
+      "content": "|<EXTRA_TOKENS_36>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151832": {
+      "content": "|<EXTRA_TOKENS_37>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151833": {
+      "content": "|<EXTRA_TOKENS_38>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151834": {
+      "content": "|<EXTRA_TOKENS_39>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151835": {
+      "content": "|<EXTRA_TOKENS_40>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151836": {
+      "content": "|<EXTRA_TOKENS_41>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151837": {
+      "content": "|<EXTRA_TOKENS_42>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151838": {
+      "content": "|<EXTRA_TOKENS_43>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151839": {
+      "content": "|<EXTRA_TOKENS_44>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151840": {
+      "content": "|<EXTRA_TOKENS_45>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151841": {
+      "content": "|<EXTRA_TOKENS_46>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151842": {
+      "content": "|<EXTRA_TOKENS_47>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151843": {
+      "content": "|<EXTRA_TOKENS_48>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151844": {
+      "content": "|<EXTRA_TOKENS_49>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151845": {
+      "content": "|<EXTRA_TOKENS_50>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151846": {
+      "content": "|<EXTRA_TOKENS_51>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151847": {
+      "content": "|<EXTRA_TOKENS_52>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151848": {
+      "content": "|<EXTRA_TOKENS_53>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151849": {
+      "content": "|<EXTRA_TOKENS_54>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151850": {
+      "content": "|<EXTRA_TOKENS_55>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151851": {
+      "content": "|<EXTRA_TOKENS_56>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151852": {
+      "content": "|<EXTRA_TOKENS_57>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151853": {
+      "content": "|<EXTRA_TOKENS_58>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151854": {
+      "content": "|<EXTRA_TOKENS_59>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151855": {
+      "content": "|<EXTRA_TOKENS_60>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151856": {
+      "content": "|<EXTRA_TOKENS_61>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151857": {
+      "content": "|<EXTRA_TOKENS_62>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151858": {
+      "content": "|<EXTRA_TOKENS_63>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151859": {
+      "content": "|<EXTRA_TOKENS_64>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151860": {
+      "content": "|<EXTRA_TOKENS_65>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151861": {
+      "content": "|<EXTRA_TOKENS_66>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151862": {
+      "content": "|<EXTRA_TOKENS_67>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151863": {
+      "content": "|<EXTRA_TOKENS_68>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151864": {
+      "content": "|<EXTRA_TOKENS_69>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151865": {
+      "content": "|<EXTRA_TOKENS_70>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151866": {
+      "content": "|<EXTRA_TOKENS_71>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151867": {
+      "content": "|<EXTRA_TOKENS_72>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151868": {
+      "content": "|<EXTRA_TOKENS_73>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151869": {
+      "content": "|<EXTRA_TOKENS_74>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151870": {
+      "content": "|<EXTRA_TOKENS_75>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151871": {
+      "content": "|<EXTRA_TOKENS_76>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151872": {
+      "content": "|<EXTRA_TOKENS_77>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151873": {
+      "content": "|<EXTRA_TOKENS_78>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151874": {
+      "content": "|<EXTRA_TOKENS_79>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151875": {
+      "content": "|<EXTRA_TOKENS_80>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151876": {
+      "content": "|<EXTRA_TOKENS_81>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151877": {
+      "content": "|<EXTRA_TOKENS_82>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151878": {
+      "content": "|<EXTRA_TOKENS_83>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151879": {
+      "content": "|<EXTRA_TOKENS_84>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151880": {
+      "content": "|<EXTRA_TOKENS_85>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151881": {
+      "content": "|<EXTRA_TOKENS_86>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151882": {
+      "content": "|<EXTRA_TOKENS_87>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151883": {
+      "content": "|<EXTRA_TOKENS_88>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151884": {
+      "content": "|<EXTRA_TOKENS_89>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151885": {
+      "content": "|<EXTRA_TOKENS_90>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151886": {
+      "content": "|<EXTRA_TOKENS_91>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151887": {
+      "content": "|<EXTRA_TOKENS_92>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151888": {
+      "content": "|<EXTRA_TOKENS_93>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151889": {
+      "content": "|<EXTRA_TOKENS_94>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151890": {
+      "content": "|<EXTRA_TOKENS_95>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151891": {
+      "content": "|<EXTRA_TOKENS_96>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151892": {
+      "content": "|<EXTRA_TOKENS_97>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151893": {
+      "content": "|<EXTRA_TOKENS_98>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151894": {
+      "content": "|<EXTRA_TOKENS_99>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151895": {
+      "content": "|<EXTRA_TOKENS_100>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151896": {
+      "content": "|<EXTRA_TOKENS_101>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151897": {
+      "content": "|<EXTRA_TOKENS_102>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151898": {
+      "content": "|<EXTRA_TOKENS_103>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151899": {
+      "content": "|<EXTRA_TOKENS_104>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151900": {
+      "content": "|<EXTRA_TOKENS_105>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151901": {
+      "content": "|<EXTRA_TOKENS_106>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151902": {
+      "content": "|<EXTRA_TOKENS_107>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151903": {
+      "content": "|<EXTRA_TOKENS_108>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151904": {
+      "content": "|<EXTRA_TOKENS_109>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151905": {
+      "content": "|<EXTRA_TOKENS_110>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151906": {
+      "content": "|<EXTRA_TOKENS_111>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151907": {
+      "content": "|<EXTRA_TOKENS_112>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151908": {
+      "content": "|<EXTRA_TOKENS_113>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151909": {
+      "content": "|<EXTRA_TOKENS_114>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151910": {
+      "content": "|<EXTRA_TOKENS_115>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151911": {
+      "content": "|<EXTRA_TOKENS_116>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151912": {
+      "content": "|<EXTRA_TOKENS_117>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151913": {
+      "content": "|<EXTRA_TOKENS_118>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151914": {
+      "content": "|<EXTRA_TOKENS_119>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151915": {
+      "content": "|<EXTRA_TOKENS_120>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151916": {
+      "content": "|<EXTRA_TOKENS_121>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151917": {
+      "content": "|<EXTRA_TOKENS_122>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151918": {
+      "content": "|<EXTRA_TOKENS_123>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151919": {
+      "content": "|<EXTRA_TOKENS_124>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151920": {
+      "content": "|<EXTRA_TOKENS_125>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151921": {
+      "content": "|<EXTRA_TOKENS_126>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151922": {
+      "content": "|<EXTRA_TOKENS_127>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151923": {
+      "content": "|<EXTRA_TOKENS_128>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151924": {
+      "content": "|<EXTRA_TOKENS_129>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151925": {
+      "content": "|<EXTRA_TOKENS_130>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151926": {
+      "content": "|<EXTRA_TOKENS_131>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151927": {
+      "content": "|<EXTRA_TOKENS_132>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151928": {
+      "content": "|<EXTRA_TOKENS_133>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151929": {
+      "content": "|<EXTRA_TOKENS_134>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151930": {
+      "content": "|<EXTRA_TOKENS_135>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151931": {
+      "content": "|<EXTRA_TOKENS_136>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151932": {
+      "content": "|<EXTRA_TOKENS_137>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151933": {
+      "content": "|<EXTRA_TOKENS_138>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151934": {
+      "content": "|<EXTRA_TOKENS_139>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151935": {
+      "content": "|<EXTRA_TOKENS_140>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151936": {
+      "content": "|<EXTRA_TOKENS_141>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151937": {
+      "content": "|<EXTRA_TOKENS_142>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151938": {
+      "content": "|<EXTRA_TOKENS_143>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151939": {
+      "content": "|<EXTRA_TOKENS_144>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151940": {
+      "content": "|<EXTRA_TOKENS_145>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151941": {
+      "content": "|<EXTRA_TOKENS_146>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151942": {
+      "content": "|<EXTRA_TOKENS_147>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151943": {
+      "content": "|<EXTRA_TOKENS_148>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151944": {
+      "content": "|<EXTRA_TOKENS_149>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151945": {
+      "content": "|<EXTRA_TOKENS_150>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151946": {
+      "content": "|<EXTRA_TOKENS_151>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151947": {
+      "content": "|<EXTRA_TOKENS_152>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151948": {
+      "content": "|<EXTRA_TOKENS_153>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151949": {
+      "content": "|<EXTRA_TOKENS_154>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151950": {
+      "content": "|<EXTRA_TOKENS_155>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151951": {
+      "content": "|<EXTRA_TOKENS_156>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151952": {
+      "content": "|<EXTRA_TOKENS_157>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151953": {
+      "content": "|<EXTRA_TOKENS_158>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151954": {
+      "content": "|<EXTRA_TOKENS_159>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151955": {
+      "content": "|<EXTRA_TOKENS_160>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151956": {
+      "content": "|<EXTRA_TOKENS_161>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151957": {
+      "content": "|<EXTRA_TOKENS_162>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151958": {
+      "content": "|<EXTRA_TOKENS_163>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151959": {
+      "content": "|<EXTRA_TOKENS_164>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151960": {
+      "content": "|<EXTRA_TOKENS_165>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151961": {
+      "content": "|<EXTRA_TOKENS_166>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151962": {
+      "content": "|<EXTRA_TOKENS_167>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151963": {
+      "content": "|<EXTRA_TOKENS_168>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151964": {
+      "content": "|<EXTRA_TOKENS_169>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151965": {
+      "content": "|<EXTRA_TOKENS_170>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151966": {
+      "content": "|<EXTRA_TOKENS_171>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151967": {
+      "content": "|<EXTRA_TOKENS_172>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151968": {
+      "content": "|<EXTRA_TOKENS_173>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151969": {
+      "content": "|<EXTRA_TOKENS_174>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151970": {
+      "content": "|<EXTRA_TOKENS_175>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151971": {
+      "content": "|<EXTRA_TOKENS_176>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151972": {
+      "content": "|<EXTRA_TOKENS_177>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151973": {
+      "content": "|<EXTRA_TOKENS_178>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151974": {
+      "content": "|<EXTRA_TOKENS_179>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151975": {
+      "content": "|<EXTRA_TOKENS_180>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151976": {
+      "content": "|<EXTRA_TOKENS_181>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151977": {
+      "content": "|<EXTRA_TOKENS_182>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151978": {
+      "content": "|<EXTRA_TOKENS_183>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151979": {
+      "content": "|<EXTRA_TOKENS_184>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151980": {
+      "content": "|<EXTRA_TOKENS_185>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151981": {
+      "content": "|<EXTRA_TOKENS_186>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151982": {
+      "content": "|<EXTRA_TOKENS_187>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151983": {
+      "content": "|<EXTRA_TOKENS_188>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151984": {
+      "content": "|<EXTRA_TOKENS_189>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151985": {
+      "content": "|<EXTRA_TOKENS_190>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151986": {
+      "content": "|<EXTRA_TOKENS_191>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151987": {
+      "content": "|<EXTRA_TOKENS_192>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151988": {
+      "content": "|<EXTRA_TOKENS_193>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151989": {
+      "content": "|<EXTRA_TOKENS_194>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151990": {
+      "content": "|<EXTRA_TOKENS_195>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151991": {
+      "content": "|<EXTRA_TOKENS_196>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151992": {
+      "content": "|<EXTRA_TOKENS_197>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151993": {
+      "content": "|<EXTRA_TOKENS_198>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151994": {
+      "content": "|<EXTRA_TOKENS_199>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151995": {
+      "content": "|<EXTRA_TOKENS_200>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151996": {
+      "content": "|<EXTRA_TOKENS_201>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151997": {
+      "content": "|<EXTRA_TOKENS_202>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151998": {
+      "content": "|<EXTRA_TOKENS_203>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151999": {
+      "content": "|<EXTRA_TOKENS_204>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152000": {
+      "content": "|<EXTRA_TOKENS_205>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152001": {
+      "content": "|<EXTRA_TOKENS_206>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152002": {
+      "content": "|<EXTRA_TOKENS_207>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152003": {
+      "content": "|<EXTRA_TOKENS_208>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152004": {
+      "content": "|<EXTRA_TOKENS_209>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152005": {
+      "content": "|<EXTRA_TOKENS_210>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152006": {
+      "content": "|<EXTRA_TOKENS_211>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152007": {
+      "content": "|<EXTRA_TOKENS_212>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152008": {
+      "content": "|<EXTRA_TOKENS_213>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152009": {
+      "content": "|<EXTRA_TOKENS_214>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152010": {
+      "content": "|<EXTRA_TOKENS_215>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152011": {
+      "content": "|<EXTRA_TOKENS_216>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152012": {
+      "content": "|<EXTRA_TOKENS_217>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152013": {
+      "content": "|<EXTRA_TOKENS_218>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152014": {
+      "content": "|<EXTRA_TOKENS_219>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152015": {
+      "content": "|<EXTRA_TOKENS_220>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152016": {
+      "content": "|<EXTRA_TOKENS_221>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152017": {
+      "content": "|<EXTRA_TOKENS_222>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152018": {
+      "content": "|<EXTRA_TOKENS_223>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152019": {
+      "content": "|<EXTRA_TOKENS_224>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152020": {
+      "content": "|<EXTRA_TOKENS_225>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152021": {
+      "content": "|<EXTRA_TOKENS_226>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152022": {
+      "content": "|<EXTRA_TOKENS_227>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152023": {
+      "content": "|<EXTRA_TOKENS_228>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152024": {
+      "content": "|<EXTRA_TOKENS_229>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152025": {
+      "content": "|<EXTRA_TOKENS_230>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152026": {
+      "content": "|<EXTRA_TOKENS_231>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152027": {
+      "content": "|<EXTRA_TOKENS_232>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152028": {
+      "content": "|<EXTRA_TOKENS_233>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152029": {
+      "content": "|<EXTRA_TOKENS_234>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152030": {
+      "content": "|<EXTRA_TOKENS_235>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152031": {
+      "content": "|<EXTRA_TOKENS_236>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152032": {
+      "content": "|<EXTRA_TOKENS_237>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152033": {
+      "content": "|<EXTRA_TOKENS_238>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152034": {
+      "content": "|<EXTRA_TOKENS_239>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152035": {
+      "content": "|<EXTRA_TOKENS_240>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152036": {
+      "content": "|<EXTRA_TOKENS_241>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152037": {
+      "content": "|<EXTRA_TOKENS_242>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152038": {
+      "content": "|<EXTRA_TOKENS_243>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152039": {
+      "content": "|<EXTRA_TOKENS_244>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152040": {
+      "content": "|<EXTRA_TOKENS_245>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152041": {
+      "content": "|<EXTRA_TOKENS_246>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152042": {
+      "content": "|<EXTRA_TOKENS_247>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152043": {
+      "content": "|<EXTRA_TOKENS_248>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152044": {
+      "content": "|<EXTRA_TOKENS_249>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152045": {
+      "content": "|<EXTRA_TOKENS_250>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152046": {
+      "content": "|<EXTRA_TOKENS_251>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152047": {
+      "content": "|<EXTRA_TOKENS_252>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152048": {
+      "content": "|<EXTRA_TOKENS_253>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152049": {
+      "content": "|<EXTRA_TOKENS_254>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152050": {
+      "content": "|<EXTRA_TOKENS_255>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152051": {
+      "content": "|<EXTRA_TOKENS_256>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152052": {
+      "content": "|<EXTRA_TOKENS_257>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152053": {
+      "content": "|<EXTRA_TOKENS_258>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152054": {
+      "content": "|<EXTRA_TOKENS_259>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152055": {
+      "content": "|<EXTRA_TOKENS_260>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152056": {
+      "content": "|<EXTRA_TOKENS_261>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152057": {
+      "content": "|<EXTRA_TOKENS_262>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152058": {
+      "content": "|<EXTRA_TOKENS_263>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152059": {
+      "content": "|<EXTRA_TOKENS_264>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152060": {
+      "content": "|<EXTRA_TOKENS_265>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152061": {
+      "content": "|<EXTRA_TOKENS_266>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152062": {
+      "content": "|<EXTRA_TOKENS_267>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152063": {
+      "content": "|<EXTRA_TOKENS_268>|",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152064": {
+      "content": "<im_start>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152065": {
+      "content": "<im_end>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152066": {
+      "content": "<im_patch>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152067": {
+      "content": "<im_col>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152068": {
+      "content": "<|image|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "152069": {
+      "content": "<im_low>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "additional_special_tokens": [
+    "|<EXTRA_TOKENS_0>|",
+    "|<EXTRA_TOKENS_1>|",
+    "|<EXTRA_TOKENS_2>|",
+    "|<EXTRA_TOKENS_3>|",
+    "|<EXTRA_TOKENS_4>|",
+    "|<EXTRA_TOKENS_5>|",
+    "|<EXTRA_TOKENS_6>|",
+    "|<EXTRA_TOKENS_7>|",
+    "|<EXTRA_TOKENS_8>|",
+    "|<EXTRA_TOKENS_9>|",
+    "|<EXTRA_TOKENS_10>|",
+    "|<EXTRA_TOKENS_11>|",
+    "|<EXTRA_TOKENS_12>|",
+    "|<EXTRA_TOKENS_13>|",
+    "|<EXTRA_TOKENS_14>|",
+    "|<EXTRA_TOKENS_15>|",
+    "|<EXTRA_TOKENS_16>|",
+    "|<EXTRA_TOKENS_17>|",
+    "|<EXTRA_TOKENS_18>|",
+    "|<EXTRA_TOKENS_19>|",
+    "|<EXTRA_TOKENS_20>|",
+    "|<EXTRA_TOKENS_21>|",
+    "|<EXTRA_TOKENS_22>|",
+    "|<EXTRA_TOKENS_23>|",
+    "|<EXTRA_TOKENS_24>|",
+    "|<EXTRA_TOKENS_25>|",
+    "|<EXTRA_TOKENS_26>|",
+    "|<EXTRA_TOKENS_27>|",
+    "|<EXTRA_TOKENS_28>|",
+    "|<EXTRA_TOKENS_29>|",
+    "|<EXTRA_TOKENS_30>|",
+    "|<EXTRA_TOKENS_31>|",
+    "|<EXTRA_TOKENS_32>|",
+    "|<EXTRA_TOKENS_33>|",
+    "|<EXTRA_TOKENS_34>|",
+    "|<EXTRA_TOKENS_35>|",
+    "|<EXTRA_TOKENS_36>|",
+    "|<EXTRA_TOKENS_37>|",
+    "|<EXTRA_TOKENS_38>|",
+    "|<EXTRA_TOKENS_39>|",
+    "|<EXTRA_TOKENS_40>|",
+    "|<EXTRA_TOKENS_41>|",
+    "|<EXTRA_TOKENS_42>|",
+    "|<EXTRA_TOKENS_43>|",
+    "|<EXTRA_TOKENS_44>|",
+    "|<EXTRA_TOKENS_45>|",
+    "|<EXTRA_TOKENS_46>|",
+    "|<EXTRA_TOKENS_47>|",
+    "|<EXTRA_TOKENS_48>|",
+    "|<EXTRA_TOKENS_49>|",
+    "|<EXTRA_TOKENS_50>|",
+    "|<EXTRA_TOKENS_51>|",
+    "|<EXTRA_TOKENS_52>|",
+    "|<EXTRA_TOKENS_53>|",
+    "|<EXTRA_TOKENS_54>|",
+    "|<EXTRA_TOKENS_55>|",
+    "|<EXTRA_TOKENS_56>|",
+    "|<EXTRA_TOKENS_57>|",
+    "|<EXTRA_TOKENS_58>|",
+    "|<EXTRA_TOKENS_59>|",
+    "|<EXTRA_TOKENS_60>|",
+    "|<EXTRA_TOKENS_61>|",
+    "|<EXTRA_TOKENS_62>|",
+    "|<EXTRA_TOKENS_63>|",
+    "|<EXTRA_TOKENS_64>|",
+    "|<EXTRA_TOKENS_65>|",
+    "|<EXTRA_TOKENS_66>|",
+    "|<EXTRA_TOKENS_67>|",
+    "|<EXTRA_TOKENS_68>|",
+    "|<EXTRA_TOKENS_69>|",
+    "|<EXTRA_TOKENS_70>|",
+    "|<EXTRA_TOKENS_71>|",
+    "|<EXTRA_TOKENS_72>|",
+    "|<EXTRA_TOKENS_73>|",
+    "|<EXTRA_TOKENS_74>|",
+    "|<EXTRA_TOKENS_75>|",
+    "|<EXTRA_TOKENS_76>|",
+    "|<EXTRA_TOKENS_77>|",
+    "|<EXTRA_TOKENS_78>|",
+    "|<EXTRA_TOKENS_79>|",
+    "|<EXTRA_TOKENS_80>|",
+    "|<EXTRA_TOKENS_81>|",
+    "|<EXTRA_TOKENS_82>|",
+    "|<EXTRA_TOKENS_83>|",
+    "|<EXTRA_TOKENS_84>|",
+    "|<EXTRA_TOKENS_85>|",
+    "|<EXTRA_TOKENS_86>|",
+    "|<EXTRA_TOKENS_87>|",
+    "|<EXTRA_TOKENS_88>|",
+    "|<EXTRA_TOKENS_89>|",
+    "|<EXTRA_TOKENS_90>|",
+    "|<EXTRA_TOKENS_91>|",
+    "|<EXTRA_TOKENS_92>|",
+    "|<EXTRA_TOKENS_93>|",
+    "|<EXTRA_TOKENS_94>|",
+    "|<EXTRA_TOKENS_95>|",
+    "|<EXTRA_TOKENS_96>|",
+    "|<EXTRA_TOKENS_97>|",
+    "|<EXTRA_TOKENS_98>|",
+    "|<EXTRA_TOKENS_99>|",
+    "|<EXTRA_TOKENS_100>|",
+    "|<EXTRA_TOKENS_101>|",
+    "|<EXTRA_TOKENS_102>|",
+    "|<EXTRA_TOKENS_103>|",
+    "|<EXTRA_TOKENS_104>|",
+    "|<EXTRA_TOKENS_105>|",
+    "|<EXTRA_TOKENS_106>|",
+    "|<EXTRA_TOKENS_107>|",
+    "|<EXTRA_TOKENS_108>|",
+    "|<EXTRA_TOKENS_109>|",
+    "|<EXTRA_TOKENS_110>|",
+    "|<EXTRA_TOKENS_111>|",
+    "|<EXTRA_TOKENS_112>|",
+    "|<EXTRA_TOKENS_113>|",
+    "|<EXTRA_TOKENS_114>|",
+    "|<EXTRA_TOKENS_115>|",
+    "|<EXTRA_TOKENS_116>|",
+    "|<EXTRA_TOKENS_117>|",
+    "|<EXTRA_TOKENS_118>|",
+    "|<EXTRA_TOKENS_119>|",
+    "|<EXTRA_TOKENS_120>|",
+    "|<EXTRA_TOKENS_121>|",
+    "|<EXTRA_TOKENS_122>|",
+    "|<EXTRA_TOKENS_123>|",
+    "|<EXTRA_TOKENS_124>|",
+    "|<EXTRA_TOKENS_125>|",
+    "|<EXTRA_TOKENS_126>|",
+    "|<EXTRA_TOKENS_127>|",
+    "|<EXTRA_TOKENS_128>|",
+    "|<EXTRA_TOKENS_129>|",
+    "|<EXTRA_TOKENS_130>|",
+    "|<EXTRA_TOKENS_131>|",
+    "|<EXTRA_TOKENS_132>|",
+    "|<EXTRA_TOKENS_133>|",
+    "|<EXTRA_TOKENS_134>|",
+    "|<EXTRA_TOKENS_135>|",
+    "|<EXTRA_TOKENS_136>|",
+    "|<EXTRA_TOKENS_137>|",
+    "|<EXTRA_TOKENS_138>|",
+    "|<EXTRA_TOKENS_139>|",
+    "|<EXTRA_TOKENS_140>|",
+    "|<EXTRA_TOKENS_141>|",
+    "|<EXTRA_TOKENS_142>|",
+    "|<EXTRA_TOKENS_143>|",
+    "|<EXTRA_TOKENS_144>|",
+    "|<EXTRA_TOKENS_145>|",
+    "|<EXTRA_TOKENS_146>|",
+    "|<EXTRA_TOKENS_147>|",
+    "|<EXTRA_TOKENS_148>|",
+    "|<EXTRA_TOKENS_149>|",
+    "|<EXTRA_TOKENS_150>|",
+    "|<EXTRA_TOKENS_151>|",
+    "|<EXTRA_TOKENS_152>|",
+    "|<EXTRA_TOKENS_153>|",
+    "|<EXTRA_TOKENS_154>|",
+    "|<EXTRA_TOKENS_155>|",
+    "|<EXTRA_TOKENS_156>|",
+    "|<EXTRA_TOKENS_157>|",
+    "|<EXTRA_TOKENS_158>|",
+    "|<EXTRA_TOKENS_159>|",
+    "|<EXTRA_TOKENS_160>|",
+    "|<EXTRA_TOKENS_161>|",
+    "|<EXTRA_TOKENS_162>|",
+    "|<EXTRA_TOKENS_163>|",
+    "|<EXTRA_TOKENS_164>|",
+    "|<EXTRA_TOKENS_165>|",
+    "|<EXTRA_TOKENS_166>|",
+    "|<EXTRA_TOKENS_167>|",
+    "|<EXTRA_TOKENS_168>|",
+    "|<EXTRA_TOKENS_169>|",
+    "|<EXTRA_TOKENS_170>|",
+    "|<EXTRA_TOKENS_171>|",
+    "|<EXTRA_TOKENS_172>|",
+    "|<EXTRA_TOKENS_173>|",
+    "|<EXTRA_TOKENS_174>|",
+    "|<EXTRA_TOKENS_175>|",
+    "|<EXTRA_TOKENS_176>|",
+    "|<EXTRA_TOKENS_177>|",
+    "|<EXTRA_TOKENS_178>|",
+    "|<EXTRA_TOKENS_179>|",
+    "|<EXTRA_TOKENS_180>|",
+    "|<EXTRA_TOKENS_181>|",
+    "|<EXTRA_TOKENS_182>|",
+    "|<EXTRA_TOKENS_183>|",
+    "|<EXTRA_TOKENS_184>|",
+    "|<EXTRA_TOKENS_185>|",
+    "|<EXTRA_TOKENS_186>|",
+    "|<EXTRA_TOKENS_187>|",
+    "|<EXTRA_TOKENS_188>|",
+    "|<EXTRA_TOKENS_189>|",
+    "|<EXTRA_TOKENS_190>|",
+    "|<EXTRA_TOKENS_191>|",
+    "|<EXTRA_TOKENS_192>|",
+    "|<EXTRA_TOKENS_193>|",
+    "|<EXTRA_TOKENS_194>|",
+    "|<EXTRA_TOKENS_195>|",
+    "|<EXTRA_TOKENS_196>|",
+    "|<EXTRA_TOKENS_197>|",
+    "|<EXTRA_TOKENS_198>|",
+    "|<EXTRA_TOKENS_199>|",
+    "|<EXTRA_TOKENS_200>|",
+    "|<EXTRA_TOKENS_201>|",
+    "|<EXTRA_TOKENS_202>|",
+    "|<EXTRA_TOKENS_203>|",
+    "|<EXTRA_TOKENS_204>|",
+    "|<EXTRA_TOKENS_205>|",
+    "|<EXTRA_TOKENS_206>|",
+    "|<EXTRA_TOKENS_207>|",
+    "|<EXTRA_TOKENS_208>|",
+    "|<EXTRA_TOKENS_209>|",
+    "|<EXTRA_TOKENS_210>|",
+    "|<EXTRA_TOKENS_211>|",
+    "|<EXTRA_TOKENS_212>|",
+    "|<EXTRA_TOKENS_213>|",
+    "|<EXTRA_TOKENS_214>|",
+    "|<EXTRA_TOKENS_215>|",
+    "|<EXTRA_TOKENS_216>|",
+    "|<EXTRA_TOKENS_217>|",
+    "|<EXTRA_TOKENS_218>|",
+    "|<EXTRA_TOKENS_219>|",
+    "|<EXTRA_TOKENS_220>|",
+    "|<EXTRA_TOKENS_221>|",
+    "|<EXTRA_TOKENS_222>|",
+    "|<EXTRA_TOKENS_223>|",
+    "|<EXTRA_TOKENS_224>|",
+    "|<EXTRA_TOKENS_225>|",
+    "|<EXTRA_TOKENS_226>|",
+    "|<EXTRA_TOKENS_227>|",
+    "|<EXTRA_TOKENS_228>|",
+    "|<EXTRA_TOKENS_229>|",
+    "|<EXTRA_TOKENS_230>|",
+    "|<EXTRA_TOKENS_231>|",
+    "|<EXTRA_TOKENS_232>|",
+    "|<EXTRA_TOKENS_233>|",
+    "|<EXTRA_TOKENS_234>|",
+    "|<EXTRA_TOKENS_235>|",
+    "|<EXTRA_TOKENS_236>|",
+    "|<EXTRA_TOKENS_237>|",
+    "|<EXTRA_TOKENS_238>|",
+    "|<EXTRA_TOKENS_239>|",
+    "|<EXTRA_TOKENS_240>|",
+    "|<EXTRA_TOKENS_241>|",
+    "|<EXTRA_TOKENS_242>|",
+    "|<EXTRA_TOKENS_243>|",
+    "|<EXTRA_TOKENS_244>|",
+    "|<EXTRA_TOKENS_245>|",
+    "|<EXTRA_TOKENS_246>|",
+    "|<EXTRA_TOKENS_247>|",
+    "|<EXTRA_TOKENS_248>|",
+    "|<EXTRA_TOKENS_249>|",
+    "|<EXTRA_TOKENS_250>|",
+    "|<EXTRA_TOKENS_251>|",
+    "|<EXTRA_TOKENS_252>|",
+    "|<EXTRA_TOKENS_253>|",
+    "|<EXTRA_TOKENS_254>|",
+    "|<EXTRA_TOKENS_255>|",
+    "|<EXTRA_TOKENS_256>|",
+    "|<EXTRA_TOKENS_257>|",
+    "|<EXTRA_TOKENS_258>|",
+    "|<EXTRA_TOKENS_259>|",
+    "|<EXTRA_TOKENS_260>|",
+    "|<EXTRA_TOKENS_261>|",
+    "|<EXTRA_TOKENS_262>|",
+    "|<EXTRA_TOKENS_263>|",
+    "|<EXTRA_TOKENS_264>|",
+    "|<EXTRA_TOKENS_265>|",
+    "|<EXTRA_TOKENS_266>|",
+    "|<EXTRA_TOKENS_267>|",
+    "|<EXTRA_TOKENS_268>|",
+    "<im_start>",
+    "<im_end>",
+    "<im_patch>",
+    "<im_col>",
+    "<|image|>",
+    "<im_low>"
+  ],
+  "auto_map": {
+    "AutoProcessor": "processing_molmoact.MolmoActProcessor"
+  },
+  "bos_token": "<|endoftext|>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|endoftext|>",
+  "errors": "replace",
+  "extra_special_tokens": {},
+  "model_max_length": 131072,
+  "pad_token": "<|endoftext|>",
+  "processor_class": "MolmoActProcessor",
+  "split_special_tokens": false,
+  "tokenizer_class": "Qwen2Tokenizer",
+  "unk_token": null
+}

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff