hadikhamoud commited on Jan 14

Commit

b85ccb6

verified ·

1 Parent(s): 608ac40

Upload folder using huggingface_hub

Browse files

Files changed (19) hide show

.gitattributes +1 -0
.ipynb_checkpoints/README-checkpoint.md +69 -0
README.md +69 -0
adapter_config.json +46 -0
adapter_model.safetensors +3 -0
added_tokens.json +28 -0
chat_template.jinja +120 -0
merges.txt +0 -0
optimizer.pt +3 -0
preprocessor_config.json +39 -0
rng_state.pth +3 -0
scheduler.pt +3 -0
special_tokens_map.json +31 -0
tokenizer.json +3 -0
tokenizer_config.json +240 -0
trainer_state.json +294 -0
training_args.bin +3 -0
video_preprocessor_config.json +41 -0
vocab.json +0 -0

.gitattributes CHANGED Viewed

@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text

 *.zip filter=lfs diff=lfs merge=lfs -text
 *.zst filter=lfs diff=lfs merge=lfs -text
 *tfevents* filter=lfs diff=lfs merge=lfs -text
+tokenizer.json filter=lfs diff=lfs merge=lfs -text

.ipynb_checkpoints/README-checkpoint.md ADDED Viewed

	@@ -0,0 +1,69 @@

+---
+base_model: Qwen/Qwen3-VL-8B-Instruct
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen3-VL-8B-Instruct
+- lora
+- sft
+- transformers
+- trl
+---
+# Model Card for ar-ms-baseline
+## Model Summary
+This model is the baseline system for the NAKBA NLP 2026: Arabic Manuscript Understanding Shared Task (Systems Track). It fine-tunes Qwen3-VL-8B-Instruct with LoRA to transcribe Arabic manuscript line images into text.
+## Model Details
+### Description
+- **Model type:** Vision-language OCR/HTR model (LoRA-adapted)
+- **Finetuned from model:** Qwen/Qwen3-VL-8B-Instruct
+### Sources
+- **Repository:** https://github.com/U4RASD/ar-ms-baseline
+- **Shared Task:** https://acrps.ai/nakba-nlp-manu-understanding-2026
+## Training Details
+### Training Data
+- NAKBA NLP 2026 Shared Task (Subtask 2) training split from the Omar Al-Saleh memoir collection.
+- Dataset includes line images with gold transcriptions.
+### Training Procedure
+- Supervised fine-tuning with LoRA adapters on Qwen/Qwen3-VL-8B-Instruct.
+#### Training Hyperparameters
+- **Config reference:** Hyperparameters are listed in `configs/default.json`
+## Evaluation
+### Testing Data, Factors & Metrics
+#### Testing Data
+- NAKBA NLP 2026 Shared Task (Subtask 2) released test set of line images.
+#### Metrics
+- **CER (Character Error Rate)**
+- **WER (Word Error Rate)**
+### Results
+On released test set:
+- CER: 0.2297
+- WER: 0.4998
+- **Hardware:** NVIDIA H100 SXM
+## Contact
+- ar-ms@dohainstitute.edu.qa

README.md ADDED Viewed

	@@ -0,0 +1,69 @@

+---
+base_model: Qwen/Qwen3-VL-8B-Instruct
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen3-VL-8B-Instruct
+- lora
+- sft
+- transformers
+- trl
+---
+# Model Card for ar-ms-baseline
+## Model Summary
+This model is the baseline system for the NAKBA NLP 2026: Arabic Manuscript Understanding Shared Task (Systems Track). It fine-tunes Qwen3-VL-8B-Instruct with LoRA to transcribe Arabic manuscript line images into text.
+## Model Details
+### Description
+- **Model type:** Vision-language OCR/HTR model (LoRA-adapted)
+- **Finetuned from model:** Qwen/Qwen3-VL-8B-Instruct
+### Sources
+- **Repository:** https://github.com/U4RASD/ar-ms-baseline
+- **Shared Task:** https://acrps.ai/nakba-nlp-manu-understanding-2026
+## Training Details
+### Training Data
+- NAKBA NLP 2026 Shared Task (Subtask 2) training split from the Omar Al-Saleh memoir collection.
+- Dataset includes line images with gold transcriptions.
+### Training Procedure
+- Supervised fine-tuning with LoRA adapters on Qwen/Qwen3-VL-8B-Instruct.
+#### Training Hyperparameters
+- **Config reference:** Hyperparameters are listed in `configs/default.json`
+## Evaluation
+### Testing Data, Factors & Metrics
+#### Testing Data
+- NAKBA NLP 2026 Shared Task (Subtask 2) released test set of line images.
+#### Metrics
+- **CER (Character Error Rate)**
+- **WER (Word Error Rate)**
+### Results
+On released test set:
+- CER: 0.2297
+- WER: 0.4998
+- **Hardware:** NVIDIA H100 SXM
+## Contact
+- ar-ms@dohainstitute.edu.qa

adapter_config.json ADDED Viewed

	@@ -0,0 +1,46 @@

+{
+  "alora_invocation_tokens": null,
+  "alpha_pattern": {},
+  "arrow_config": null,
+  "auto_mapping": null,
+  "base_model_name_or_path": "Qwen/Qwen3-VL-8B-Instruct",
+  "bias": "none",
+  "corda_config": null,
+  "ensure_weight_tying": false,
+  "eva_config": null,
+  "exclude_modules": null,
+  "fan_in_fan_out": false,
+  "inference_mode": true,
+  "init_lora_weights": true,
+  "layer_replication": null,
+  "layers_pattern": null,
+  "layers_to_transform": null,
+  "loftq_config": {},
+  "lora_alpha": 64,
+  "lora_bias": false,
+  "lora_dropout": 0.05,
+  "megatron_config": null,
+  "megatron_core": "megatron.core",
+  "modules_to_save": null,
+  "peft_type": "LORA",
+  "peft_version": "0.18.1",
+  "qalora_group_size": 16,
+  "r": 32,
+  "rank_pattern": {},
+  "revision": null,
+  "target_modules": [
+    "q_proj",
+    "up_proj",
+    "o_proj",
+    "gate_proj",
+    "v_proj",
+    "down_proj",
+    "k_proj"
+  ],
+  "target_parameters": null,
+  "task_type": "CAUSAL_LM",
+  "trainable_token_indices": null,
+  "use_dora": false,
+  "use_qalora": false,
+  "use_rslora": false
+}

adapter_model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:6033882d518232335ec9b1f3f6e0e59c45640e309703dd6ab1af181b4440a407
+size 349251312

added_tokens.json ADDED Viewed

	@@ -0,0 +1,28 @@

+{
+  "</think>": 151668,
+  "</tool_call>": 151658,
+  "</tool_response>": 151666,
+  "<think>": 151667,
+  "<tool_call>": 151657,
+  "<tool_response>": 151665,
+  "<|box_end|>": 151649,
+  "<|box_start|>": 151648,
+  "<|endoftext|>": 151643,
+  "<|file_sep|>": 151664,
+  "<|fim_middle|>": 151660,
+  "<|fim_pad|>": 151662,
+  "<|fim_prefix|>": 151659,
+  "<|fim_suffix|>": 151661,
+  "<|im_end|>": 151645,
+  "<|im_start|>": 151644,
+  "<|image_pad|>": 151655,
+  "<|object_ref_end|>": 151647,
+  "<|object_ref_start|>": 151646,
+  "<|quad_end|>": 151651,
+  "<|quad_start|>": 151650,
+  "<|repo_name|>": 151663,
+  "<|video_pad|>": 151656,
+  "<|vision_end|>": 151653,
+  "<|vision_pad|>": 151654,
+  "<|vision_start|>": 151652
+}

chat_template.jinja ADDED Viewed

	@@ -0,0 +1,120 @@

+{%- if tools %}
+    {{- '<|im_start|>system\n' }}
+    {%- if messages[0].role == 'system' %}
+        {%- if messages[0].content is string %}
+            {{- messages[0].content }}
+        {%- else %}
+            {%- for content in messages[0].content %}
+                {%- if 'text' in content %}
+                    {{- content.text }}
+                {%- endif %}
+            {%- endfor %}
+        {%- endif %}
+        {{- '\n\n' }}
+    {%- endif %}
+    {{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
+    {%- for tool in tools %}
+        {{- "\n" }}
+        {{- tool | tojson }}
+    {%- endfor %}
+    {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
+{%- else %}
+    {%- if messages[0].role == 'system' %}
+        {{- '<|im_start|>system\n' }}
+        {%- if messages[0].content is string %}
+            {{- messages[0].content }}
+        {%- else %}
+            {%- for content in messages[0].content %}
+                {%- if 'text' in content %}
+                    {{- content.text }}
+                {%- endif %}
+            {%- endfor %}
+        {%- endif %}
+        {{- '<|im_end|>\n' }}
+    {%- endif %}
+{%- endif %}
+{%- set image_count = namespace(value=0) %}
+{%- set video_count = namespace(value=0) %}
+{%- for message in messages %}
+    {%- if message.role == "user" %}
+        {{- '<|im_start|>' + message.role + '\n' }}
+        {%- if message.content is string %}
+            {{- message.content }}
+        {%- else %}
+            {%- for content in message.content %}
+                {%- if content.type == 'image' or 'image' in content or 'image_url' in content %}
+                    {%- set image_count.value = image_count.value + 1 %}
+                    {%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}
+                    <|vision_start|><|image_pad|><|vision_end|>
+                {%- elif content.type == 'video' or 'video' in content %}
+                    {%- set video_count.value = video_count.value + 1 %}
+                    {%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}
+                    <|vision_start|><|video_pad|><|vision_end|>
+                {%- elif 'text' in content %}
+                    {{- content.text }}
+                {%- endif %}
+            {%- endfor %}
+        {%- endif %}
+        {{- '<|im_end|>\n' }}
+    {%- elif message.role == "assistant" %}
+        {{- '<|im_start|>' + message.role + '\n' }}
+        {%- if message.content is string %}
+            {{- message.content }}
+        {%- else %}
+            {%- for content_item in message.content %}
+                {%- if 'text' in content_item %}
+                    {{- content_item.text }}
+                {%- endif %}
+            {%- endfor %}
+        {%- endif %}
+        {%- if message.tool_calls %}
+            {%- for tool_call in message.tool_calls %}
+                {%- if (loop.first and message.content) or (not loop.first) %}
+                    {{- '\n' }}
+                {%- endif %}
+                {%- if tool_call.function %}
+                    {%- set tool_call = tool_call.function %}
+                {%- endif %}
+                {{- '<tool_call>\n{"name": "' }}
+                {{- tool_call.name }}
+                {{- '", "arguments": ' }}
+                {%- if tool_call.arguments is string %}
+                    {{- tool_call.arguments }}
+                {%- else %}
+                    {{- tool_call.arguments | tojson }}
+                {%- endif %}
+                {{- '}\n</tool_call>' }}
+            {%- endfor %}
+        {%- endif %}
+        {{- '<|im_end|>\n' }}
+    {%- elif message.role == "tool" %}
+        {%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
+            {{- '<|im_start|>user' }}
+        {%- endif %}
+        {{- '\n<tool_response>\n' }}
+        {%- if message.content is string %}
+            {{- message.content }}
+        {%- else %}
+            {%- for content in message.content %}
+                {%- if content.type == 'image' or 'image' in content or 'image_url' in content %}
+                    {%- set image_count.value = image_count.value + 1 %}
+                    {%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}
+                    <|vision_start|><|image_pad|><|vision_end|>
+                {%- elif content.type == 'video' or 'video' in content %}
+                    {%- set video_count.value = video_count.value + 1 %}
+                    {%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}
+                    <|vision_start|><|video_pad|><|vision_end|>
+                {%- elif 'text' in content %}
+                    {{- content.text }}
+                {%- endif %}
+            {%- endfor %}
+        {%- endif %}
+        {{- '\n</tool_response>' }}
+        {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+            {{- '<|im_end|>\n' }}
+        {%- endif %}
+    {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+    {{- '<|im_start|>assistant\n' }}
+{%- endif %}

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

optimizer.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:6878b338b264ad42af5b8a8053d5416e124650c5e856a50acf1a5568fe32ded6
+size 698784715

preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,39 @@

+{
+  "crop_size": null,
+  "data_format": "channels_first",
+  "default_to_square": true,
+  "device": null,
+  "disable_grouping": null,
+  "do_center_crop": null,
+  "do_convert_rgb": true,
+  "do_normalize": true,
+  "do_pad": null,
+  "do_rescale": true,
+  "do_resize": true,
+  "image_mean": [
+    0.5,
+    0.5,
+    0.5
+  ],
+  "image_processor_type": "Qwen2VLImageProcessorFast",
+  "image_std": [
+    0.5,
+    0.5,
+    0.5
+  ],
+  "input_data_format": null,
+  "max_pixels": null,
+  "merge_size": 2,
+  "min_pixels": null,
+  "pad_size": null,
+  "patch_size": 16,
+  "processor_class": "Qwen3VLProcessor",
+  "resample": 3,
+  "rescale_factor": 0.00392156862745098,
+  "return_tensors": null,
+  "size": {
+    "longest_edge": 16777216,
+    "shortest_edge": 65536
+  },
+  "temporal_patch_size": 2
+}

rng_state.pth ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:633ab8881b9391d323b0ee76e5ac535652025f3f3f909f941add3c4930d53f71
+size 14645

scheduler.pt ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:a63a41f0505afccda0cfa3e950bf0aa585a681cdb415bd2662ba75c7fa30bf02
+size 1465

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,31 @@

+{
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>"
+  ],
+  "eos_token": {
+    "content": "<|im_end|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:aeb13307a71acd8fe81861d94ad54ab689df773318809eed3cbe794b4492dae4
+size 11422654

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,240 @@

+{
+  "add_bos_token": false,
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "151643": {
+      "content": "<|endoftext|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151644": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151645": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151646": {
+      "content": "<|object_ref_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151647": {
+      "content": "<|object_ref_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151648": {
+      "content": "<|box_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151649": {
+      "content": "<|box_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151650": {
+      "content": "<|quad_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151651": {
+      "content": "<|quad_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151652": {
+      "content": "<|vision_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151653": {
+      "content": "<|vision_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151654": {
+      "content": "<|vision_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151655": {
+      "content": "<|image_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151656": {
+      "content": "<|video_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "151657": {
+      "content": "<tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151658": {
+      "content": "</tool_call>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151659": {
+      "content": "<|fim_prefix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151660": {
+      "content": "<|fim_middle|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151661": {
+      "content": "<|fim_suffix|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151662": {
+      "content": "<|fim_pad|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151663": {
+      "content": "<|repo_name|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151664": {
+      "content": "<|file_sep|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151665": {
+      "content": "<tool_response>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151666": {
+      "content": "</tool_response>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151667": {
+      "content": "<think>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    },
+    "151668": {
+      "content": "</think>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": false
+    }
+  },
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>",
+    "<|object_ref_start|>",
+    "<|object_ref_end|>",
+    "<|box_start|>",
+    "<|box_end|>",
+    "<|quad_start|>",
+    "<|quad_end|>",
+    "<|vision_start|>",
+    "<|vision_end|>",
+    "<|vision_pad|>",
+    "<|image_pad|>",
+    "<|video_pad|>"
+  ],
+  "bos_token": null,
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|im_end|>",
+  "errors": "replace",
+  "extra_special_tokens": {},
+  "model_max_length": 262144,
+  "pad_token": "<|endoftext|>",
+  "processor_class": "Qwen3VLProcessor",
+  "split_special_tokens": false,
+  "tokenizer_class": "Qwen2Tokenizer",
+  "unk_token": null
+}

trainer_state.json ADDED Viewed

	@@ -0,0 +1,294 @@

+{
+  "best_global_step": null,
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 2.0,
+  "eval_steps": 500,
+  "global_step": 1332,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "entropy": 2.4947163558006284,
+      "epoch": 0.07507507507507508,
+      "grad_norm": 2.2692031860351562,
+      "learning_rate": 1.99976055134129e-05,
+      "loss": 9.666,
+      "mean_token_accuracy": 0.3714864462614059,
+      "num_tokens": 175430.0,
+      "step": 50
+    },
+    {
+      "entropy": 4.957374696731567,
+      "epoch": 0.15015015015015015,
+      "grad_norm": 0.4004897177219391,
+      "learning_rate": 1.98972684724999e-05,
+      "loss": 3.687,
+      "mean_token_accuracy": 0.5028938430547715,
+      "num_tokens": 350404.0,
+      "step": 100
+    },
+    {
+      "entropy": 4.707881135940552,
+      "epoch": 0.22522522522522523,
+      "grad_norm": 0.4421725571155548,
+      "learning_rate": 1.9650816346189154e-05,
+      "loss": 3.4827,
+      "mean_token_accuracy": 0.5126789313554764,
+      "num_tokens": 526514.0,
+      "step": 150
+    },
+    {
+      "entropy": 4.64449327468872,
+      "epoch": 0.3003003003003003,
+      "grad_norm": 0.4718657433986664,
+      "learning_rate": 1.926188754982551e-05,
+      "loss": 3.4128,
+      "mean_token_accuracy": 0.5216181075572968,
+      "num_tokens": 701236.0,
+      "step": 200
+    },
+    {
+      "entropy": 4.598890199661255,
+      "epoch": 0.37537537537537535,
+      "grad_norm": 0.4660639762878418,
+      "learning_rate": 1.8736223906463698e-05,
+      "loss": 3.3675,
+      "mean_token_accuracy": 0.528342125415802,
+      "num_tokens": 876499.0,
+      "step": 250
+    },
+    {
+      "entropy": 4.603140239715576,
+      "epoch": 0.45045045045045046,
+      "grad_norm": 0.47321391105651855,
+      "learning_rate": 1.8081585879342008e-05,
+      "loss": 3.3821,
+      "mean_token_accuracy": 0.5266077440977096,
+      "num_tokens": 1051864.0,
+      "step": 300
+    },
+    {
+      "entropy": 4.587698755264282,
+      "epoch": 0.5255255255255256,
+      "grad_norm": 0.5947113633155823,
+      "learning_rate": 1.7307638002821942e-05,
+      "loss": 3.3509,
+      "mean_token_accuracy": 0.5320688891410827,
+      "num_tokens": 1226112.0,
+      "step": 350
+    },
+    {
+      "entropy": 4.596508531570435,
+      "epoch": 0.6006006006006006,
+      "grad_norm": 0.5403749942779541,
+      "learning_rate": 1.6425806203196734e-05,
+      "loss": 3.3672,
+      "mean_token_accuracy": 0.5314443480968475,
+      "num_tokens": 1401199.0,
+      "step": 400
+    },
+    {
+      "entropy": 4.5638793849945065,
+      "epoch": 0.6756756756756757,
+      "grad_norm": 0.5813198685646057,
+      "learning_rate": 1.544910911576629e-05,
+      "loss": 3.3328,
+      "mean_token_accuracy": 0.5359585750102996,
+      "num_tokens": 1576517.0,
+      "step": 450
+    },
+    {
+      "entropy": 4.571099786758423,
+      "epoch": 0.7507507507507507,
+      "grad_norm": 0.5679024457931519,
+      "learning_rate": 1.4391965888473705e-05,
+      "loss": 3.3438,
+      "mean_token_accuracy": 0.5345500099658966,
+      "num_tokens": 1751444.0,
+      "step": 500
+    },
+    {
+      "entropy": 4.5622615623474125,
+      "epoch": 0.8258258258258259,
+      "grad_norm": 0.6713469624519348,
+      "learning_rate": 1.3269983309531584e-05,
+      "loss": 3.3411,
+      "mean_token_accuracy": 0.5363436281681061,
+      "num_tokens": 1926907.0,
+      "step": 550
+    },
+    {
+      "entropy": 4.562127561569214,
+      "epoch": 0.9009009009009009,
+      "grad_norm": 0.6330661177635193,
+      "learning_rate": 1.2099725401709685e-05,
+      "loss": 3.3382,
+      "mean_token_accuracy": 0.5368253135681152,
+      "num_tokens": 2102132.0,
+      "step": 600
+    },
+    {
+      "entropy": 4.561514482498169,
+      "epoch": 0.975975975975976,
+      "grad_norm": 0.7525476813316345,
+      "learning_rate": 1.0898468884803366e-05,
+      "loss": 3.3411,
+      "mean_token_accuracy": 0.537065578699112,
+      "num_tokens": 2277537.0,
+      "step": 650
+    },
+    {
+      "entropy": 4.548706264495849,
+      "epoch": 1.0510510510510511,
+      "grad_norm": 0.6428335309028625,
+      "learning_rate": 9.683948116432609e-06,
+      "loss": 3.3205,
+      "mean_token_accuracy": 0.5413484466075897,
+      "num_tokens": 2449566.0,
+      "step": 700
+    },
+    {
+      "entropy": 4.5432649517059325,
+      "epoch": 1.1261261261261262,
+      "grad_norm": 0.6926820278167725,
+      "learning_rate": 8.474093276654764e-06,
+      "loss": 3.3025,
+      "mean_token_accuracy": 0.5445451009273529,
+      "num_tokens": 2623865.0,
+      "step": 750
+    },
+    {
+      "entropy": 4.537814807891846,
+      "epoch": 1.2012012012012012,
+      "grad_norm": 0.6473987698554993,
+      "learning_rate": 7.286765661616761e-06,
+      "loss": 3.3031,
+      "mean_token_accuracy": 0.5444674134254456,
+      "num_tokens": 2798793.0,
+      "step": 800
+    },
+    {
+      "entropy": 4.533882551193237,
+      "epoch": 1.2762762762762763,
+      "grad_norm": 0.7260850667953491,
+      "learning_rate": 6.139493994152428e-06,
+      "loss": 3.3086,
+      "mean_token_accuracy": 0.5432299053668976,
+      "num_tokens": 2975077.0,
+      "step": 850
+    },
+    {
+      "entropy": 4.509051198959351,
+      "epoch": 1.3513513513513513,
+      "grad_norm": 0.7649775743484497,
+      "learning_rate": 5.0492156442170914e-06,
+      "loss": 3.262,
+      "mean_token_accuracy": 0.5496996784210205,
+      "num_tokens": 3149867.0,
+      "step": 900
+    },
+    {
+      "entropy": 4.533229093551636,
+      "epoch": 1.4264264264264264,
+      "grad_norm": 0.7427539229393005,
+      "learning_rate": 4.0320265795669815e-06,
+      "loss": 3.2888,
+      "mean_token_accuracy": 0.5470581102371216,
+      "num_tokens": 3324230.0,
+      "step": 950
+    },
+    {
+      "entropy": 4.5299335289001466,
+      "epoch": 1.5015015015015014,
+      "grad_norm": 0.7631803154945374,
+      "learning_rate": 3.1029437382047368e-06,
+      "loss": 3.2959,
+      "mean_token_accuracy": 0.5465954422950745,
+      "num_tokens": 3499600.0,
+      "step": 1000
+    },
+    {
+      "entropy": 4.519344367980957,
+      "epoch": 1.5765765765765765,
+      "grad_norm": 0.7936219573020935,
+      "learning_rate": 2.275683330727697e-06,
+      "loss": 3.2927,
+      "mean_token_accuracy": 0.5452944958209991,
+      "num_tokens": 3676089.0,
+      "step": 1050
+    },
+    {
+      "entropy": 4.532338409423828,
+      "epoch": 1.6516516516516515,
+      "grad_norm": 0.735207200050354,
+      "learning_rate": 1.562458345539739e-06,
+      "loss": 3.3022,
+      "mean_token_accuracy": 0.5458044326305389,
+      "num_tokens": 3851878.0,
+      "step": 1100
+    },
+    {
+      "entropy": 4.519860811233521,
+      "epoch": 1.7267267267267268,
+      "grad_norm": 0.9440169930458069,
+      "learning_rate": 9.737982463922102e-07,
+      "loss": 3.2767,
+      "mean_token_accuracy": 0.5487184035778045,
+      "num_tokens": 4026785.0,
+      "step": 1150
+    },
+    {
+      "entropy": 4.5121307468414305,
+      "epoch": 1.8018018018018018,
+      "grad_norm": 0.8288739919662476,
+      "learning_rate": 5.183935240903415e-07,
+      "loss": 3.2591,
+      "mean_token_accuracy": 0.5512541651725769,
+      "num_tokens": 4200908.0,
+      "step": 1200
+    },
+    {
+      "entropy": 4.517954025268555,
+      "epoch": 1.8768768768768769,
+      "grad_norm": 0.7847370505332947,
+      "learning_rate": 2.0296739727517335e-07,
+      "loss": 3.2798,
+      "mean_token_accuracy": 0.5484513938426971,
+      "num_tokens": 4376428.0,
+      "step": 1250
+    },
+    {
+      "entropy": 4.50932373046875,
+      "epoch": 1.951951951951952,
+      "grad_norm": 0.7920858860015869,
+      "learning_rate": 3.217655638451112e-08,
+      "loss": 3.2683,
+      "mean_token_accuracy": 0.5490582883358002,
+      "num_tokens": 4551678.0,
+      "step": 1300
+    }
+  ],
+  "logging_steps": 50,
+  "max_steps": 1332,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 2,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": true
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 2.767706644103712e+17,
+  "train_batch_size": 24,
+  "trial_name": null,
+  "trial_params": null
+}

training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:d2808136ec74de9f08c136f96c3bb98ce16654f47a82d6a006a9f1de3bde0b28
+size 6289

video_preprocessor_config.json ADDED Viewed

	@@ -0,0 +1,41 @@

+{
+  "crop_size": null,
+  "data_format": "channels_first",
+  "default_to_square": true,
+  "device": null,
+  "do_center_crop": null,
+  "do_convert_rgb": true,
+  "do_normalize": true,
+  "do_rescale": true,
+  "do_resize": true,
+  "do_sample_frames": true,
+  "fps": 2,
+  "image_mean": [
+    0.5,
+    0.5,
+    0.5
+  ],
+  "image_std": [
+    0.5,
+    0.5,
+    0.5
+  ],
+  "input_data_format": null,
+  "max_frames": 768,
+  "merge_size": 2,
+  "min_frames": 4,
+  "num_frames": null,
+  "pad_size": null,
+  "patch_size": 16,
+  "processor_class": "Qwen3VLProcessor",
+  "resample": 3,
+  "rescale_factor": 0.00392156862745098,
+  "return_metadata": false,
+  "size": {
+    "longest_edge": 25165824,
+    "shortest_edge": 4096
+  },
+  "temporal_patch_size": 2,
+  "video_metadata": null,
+  "video_processor_type": "Qwen3VLVideoProcessor"
+}

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff