Upload folder using huggingface_hub
Browse files- v0-20250508-174810/args.json +371 -0
- v0-20250508-174810/logging.jsonl +1 -0
- v0-20250508-174810/runs/events.out.tfevents.1746697713.dsw-2138-644b4b995d-5j9mw.1220945.0 +3 -0
- v0-20250508-174810/val_dataset.jsonl +0 -0
- v1-20250508-175113/args.json +371 -0
- v1-20250508-175113/checkpoint-500/README.md +202 -0
- v1-20250508-175113/checkpoint-500/adapter_config.json +31 -0
- v1-20250508-175113/checkpoint-500/adapter_model.safetensors +3 -0
- v1-20250508-175113/checkpoint-500/additional_config.json +1 -0
- v1-20250508-175113/checkpoint-500/args.json +371 -0
- v1-20250508-175113/checkpoint-500/trainer_state.json +1089 -0
- v1-20250508-175113/checkpoint-500/training_args.bin +3 -0
- v1-20250508-175113/checkpoint-500/vit.safetensors +3 -0
- v1-20250508-175113/checkpoint-599/README.md +202 -0
- v1-20250508-175113/checkpoint-599/adapter_config.json +31 -0
- v1-20250508-175113/checkpoint-599/adapter_model.safetensors +3 -0
- v1-20250508-175113/checkpoint-599/additional_config.json +1 -0
- v1-20250508-175113/checkpoint-599/args.json +371 -0
- v1-20250508-175113/checkpoint-599/trainer_state.json +1288 -0
- v1-20250508-175113/checkpoint-599/training_args.bin +3 -0
- v1-20250508-175113/checkpoint-599/vit.safetensors +3 -0
- v1-20250508-175113/images/eval_loss.png +0 -0
- v1-20250508-175113/images/eval_runtime.png +0 -0
- v1-20250508-175113/images/eval_samples_per_second.png +0 -0
- v1-20250508-175113/images/eval_steps_per_second.png +0 -0
- v1-20250508-175113/images/eval_token_acc.png +0 -0
- v1-20250508-175113/images/train_epoch.png +0 -0
- v1-20250508-175113/images/train_grad_norm.png +0 -0
- v1-20250508-175113/images/train_learning_rate.png +0 -0
- v1-20250508-175113/images/train_loss.png +0 -0
- v1-20250508-175113/images/train_memory(GiB).png +0 -0
- v1-20250508-175113/images/train_token_acc.png +0 -0
- v1-20250508-175113/images/train_total_flos.png +0 -0
- v1-20250508-175113/images/train_train_loss.png +0 -0
- v1-20250508-175113/images/train_train_runtime.png +0 -0
- v1-20250508-175113/images/train_train_samples_per_second.png +0 -0
- v1-20250508-175113/images/train_train_speed(iter_s).png +0 -0
- v1-20250508-175113/images/train_train_steps_per_second.png +0 -0
- v1-20250508-175113/logging.jsonl +128 -0
- v1-20250508-175113/runs/events.out.tfevents.1746697894.dsw-2138-644b4b995d-5j9mw.1228873.0 +3 -0
- v1-20250508-175113/val_dataset.jsonl +0 -0
v0-20250508-174810/args.json
ADDED
|
@@ -0,0 +1,371 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model": "/cpfs01/shared/llm_ddd/tanghuanze/ckpts/hf_hub/Qwen/Qwen2.5-VL-3B-Instruct",
|
| 3 |
+
"model_type": "qwen2_5_vl",
|
| 4 |
+
"model_revision": null,
|
| 5 |
+
"task_type": "causal_lm",
|
| 6 |
+
"torch_dtype": "bfloat16",
|
| 7 |
+
"attn_impl": null,
|
| 8 |
+
"num_labels": null,
|
| 9 |
+
"problem_type": null,
|
| 10 |
+
"rope_scaling": null,
|
| 11 |
+
"device_map": null,
|
| 12 |
+
"max_memory": {},
|
| 13 |
+
"local_repo_path": null,
|
| 14 |
+
"template": "qwen2_5_vl",
|
| 15 |
+
"system": null,
|
| 16 |
+
"max_length": 8192,
|
| 17 |
+
"truncation_strategy": "delete",
|
| 18 |
+
"max_pixels": null,
|
| 19 |
+
"agent_template": null,
|
| 20 |
+
"norm_bbox": null,
|
| 21 |
+
"response_prefix": null,
|
| 22 |
+
"padding_side": "right",
|
| 23 |
+
"loss_scale": "default",
|
| 24 |
+
"sequence_parallel_size": 1,
|
| 25 |
+
"use_chat_template": true,
|
| 26 |
+
"template_backend": "swift",
|
| 27 |
+
"dataset": [
|
| 28 |
+
"/cpfs01/shared/llm_ddd/zhangyulong/sa_work/msdata/wei682/amazon-qwen-file/updated_merged_data.json"
|
| 29 |
+
],
|
| 30 |
+
"val_dataset": [],
|
| 31 |
+
"split_dataset_ratio": 0.01,
|
| 32 |
+
"data_seed": 42,
|
| 33 |
+
"dataset_num_proc": 2,
|
| 34 |
+
"dataset_shuffle": true,
|
| 35 |
+
"val_dataset_shuffle": false,
|
| 36 |
+
"streaming": false,
|
| 37 |
+
"interleave_prob": null,
|
| 38 |
+
"stopping_strategy": "first_exhausted",
|
| 39 |
+
"shuffle_buffer_size": 1000,
|
| 40 |
+
"enable_cache": false,
|
| 41 |
+
"download_mode": "reuse_dataset_if_exists",
|
| 42 |
+
"columns": {},
|
| 43 |
+
"strict": false,
|
| 44 |
+
"remove_unused_columns": true,
|
| 45 |
+
"model_name": [
|
| 46 |
+
null,
|
| 47 |
+
null
|
| 48 |
+
],
|
| 49 |
+
"model_author": [
|
| 50 |
+
null,
|
| 51 |
+
null
|
| 52 |
+
],
|
| 53 |
+
"custom_dataset_info": [],
|
| 54 |
+
"quant_method": null,
|
| 55 |
+
"quant_bits": null,
|
| 56 |
+
"hqq_axis": null,
|
| 57 |
+
"bnb_4bit_compute_dtype": "bfloat16",
|
| 58 |
+
"bnb_4bit_quant_type": "nf4",
|
| 59 |
+
"bnb_4bit_use_double_quant": true,
|
| 60 |
+
"bnb_4bit_quant_storage": null,
|
| 61 |
+
"max_new_tokens": 64,
|
| 62 |
+
"temperature": 0.0,
|
| 63 |
+
"top_k": null,
|
| 64 |
+
"top_p": null,
|
| 65 |
+
"repetition_penalty": null,
|
| 66 |
+
"num_beams": 1,
|
| 67 |
+
"stream": false,
|
| 68 |
+
"stop_words": [],
|
| 69 |
+
"logprobs": false,
|
| 70 |
+
"top_logprobs": null,
|
| 71 |
+
"ckpt_dir": null,
|
| 72 |
+
"load_dataset_config": null,
|
| 73 |
+
"lora_modules": [],
|
| 74 |
+
"tuner_backend": "peft",
|
| 75 |
+
"train_type": "custom",
|
| 76 |
+
"adapters": [],
|
| 77 |
+
"external_plugins": [
|
| 78 |
+
"examples/train/multimodal/lora_llm_full_vit/custom_plugin.py"
|
| 79 |
+
],
|
| 80 |
+
"seed": 42,
|
| 81 |
+
"model_kwargs": {},
|
| 82 |
+
"load_args": false,
|
| 83 |
+
"load_data_args": false,
|
| 84 |
+
"use_hf": false,
|
| 85 |
+
"hub_token": null,
|
| 86 |
+
"custom_register_path": [],
|
| 87 |
+
"ignore_args_error": false,
|
| 88 |
+
"use_swift_lora": false,
|
| 89 |
+
"output_dir": "/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v0-20250508-174810",
|
| 90 |
+
"overwrite_output_dir": false,
|
| 91 |
+
"do_train": false,
|
| 92 |
+
"do_eval": false,
|
| 93 |
+
"do_predict": false,
|
| 94 |
+
"eval_strategy": "steps",
|
| 95 |
+
"prediction_loss_only": false,
|
| 96 |
+
"per_device_train_batch_size": 4,
|
| 97 |
+
"per_device_eval_batch_size": 4,
|
| 98 |
+
"per_gpu_train_batch_size": null,
|
| 99 |
+
"per_gpu_eval_batch_size": null,
|
| 100 |
+
"gradient_accumulation_steps": 4,
|
| 101 |
+
"eval_accumulation_steps": null,
|
| 102 |
+
"eval_delay": 0,
|
| 103 |
+
"torch_empty_cache_steps": null,
|
| 104 |
+
"learning_rate": 0.001,
|
| 105 |
+
"weight_decay": 0.1,
|
| 106 |
+
"adam_beta1": 0.9,
|
| 107 |
+
"adam_beta2": 0.95,
|
| 108 |
+
"adam_epsilon": 1e-08,
|
| 109 |
+
"max_grad_norm": 1.0,
|
| 110 |
+
"num_train_epochs": 1.0,
|
| 111 |
+
"max_steps": -1,
|
| 112 |
+
"lr_scheduler_type": "cosine",
|
| 113 |
+
"lr_scheduler_kwargs": null,
|
| 114 |
+
"warmup_ratio": 0.05,
|
| 115 |
+
"warmup_steps": 0,
|
| 116 |
+
"log_level": "passive",
|
| 117 |
+
"log_level_replica": "warning",
|
| 118 |
+
"log_on_each_node": true,
|
| 119 |
+
"logging_dir": "/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v0-20250508-174810/runs",
|
| 120 |
+
"logging_strategy": "steps",
|
| 121 |
+
"logging_first_step": true,
|
| 122 |
+
"logging_steps": 5,
|
| 123 |
+
"logging_nan_inf_filter": true,
|
| 124 |
+
"save_strategy": "steps",
|
| 125 |
+
"save_steps": 100.0,
|
| 126 |
+
"save_total_limit": 2,
|
| 127 |
+
"save_safetensors": true,
|
| 128 |
+
"save_on_each_node": false,
|
| 129 |
+
"save_only_model": true,
|
| 130 |
+
"restore_callback_states_from_checkpoint": false,
|
| 131 |
+
"no_cuda": false,
|
| 132 |
+
"use_cpu": false,
|
| 133 |
+
"use_mps_device": false,
|
| 134 |
+
"jit_mode_eval": false,
|
| 135 |
+
"use_ipex": false,
|
| 136 |
+
"bf16": true,
|
| 137 |
+
"fp16": false,
|
| 138 |
+
"fp16_opt_level": "O1",
|
| 139 |
+
"half_precision_backend": "auto",
|
| 140 |
+
"bf16_full_eval": false,
|
| 141 |
+
"fp16_full_eval": false,
|
| 142 |
+
"tf32": null,
|
| 143 |
+
"local_rank": 0,
|
| 144 |
+
"ddp_backend": null,
|
| 145 |
+
"tpu_num_cores": null,
|
| 146 |
+
"tpu_metrics_debug": false,
|
| 147 |
+
"debug": null,
|
| 148 |
+
"dataloader_drop_last": false,
|
| 149 |
+
"eval_steps": 100.0,
|
| 150 |
+
"dataloader_num_workers": 2,
|
| 151 |
+
"dataloader_prefetch_factor": null,
|
| 152 |
+
"past_index": -1,
|
| 153 |
+
"run_name": "/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v0-20250508-174810",
|
| 154 |
+
"disable_tqdm": null,
|
| 155 |
+
"label_names": null,
|
| 156 |
+
"load_best_model_at_end": false,
|
| 157 |
+
"metric_for_best_model": "loss",
|
| 158 |
+
"greater_is_better": false,
|
| 159 |
+
"ignore_data_skip": false,
|
| 160 |
+
"fsdp": "",
|
| 161 |
+
"fsdp_min_num_params": 0,
|
| 162 |
+
"fsdp_config": null,
|
| 163 |
+
"tp_size": 0,
|
| 164 |
+
"fsdp_transformer_layer_cls_to_wrap": null,
|
| 165 |
+
"accelerator_config": {
|
| 166 |
+
"dispatch_batches": false
|
| 167 |
+
},
|
| 168 |
+
"deepspeed": {
|
| 169 |
+
"fp16": {
|
| 170 |
+
"enabled": "auto",
|
| 171 |
+
"loss_scale": 0,
|
| 172 |
+
"loss_scale_window": 1000,
|
| 173 |
+
"initial_scale_power": 16,
|
| 174 |
+
"hysteresis": 2,
|
| 175 |
+
"min_loss_scale": 1
|
| 176 |
+
},
|
| 177 |
+
"bf16": {
|
| 178 |
+
"enabled": "auto"
|
| 179 |
+
},
|
| 180 |
+
"zero_optimization": {
|
| 181 |
+
"stage": 3,
|
| 182 |
+
"offload_optimizer": {
|
| 183 |
+
"device": "none",
|
| 184 |
+
"pin_memory": true
|
| 185 |
+
},
|
| 186 |
+
"offload_param": {
|
| 187 |
+
"device": "none",
|
| 188 |
+
"pin_memory": true
|
| 189 |
+
},
|
| 190 |
+
"overlap_comm": false,
|
| 191 |
+
"contiguous_gradients": true,
|
| 192 |
+
"sub_group_size": 1000000000.0,
|
| 193 |
+
"reduce_bucket_size": "auto",
|
| 194 |
+
"zero_quantized_weights": false,
|
| 195 |
+
"zero_quantized_gradients": false,
|
| 196 |
+
"stage3_prefetch_bucket_size": "auto",
|
| 197 |
+
"stage3_param_persistence_threshold": "auto",
|
| 198 |
+
"stage3_max_live_parameters": 1000000000.0,
|
| 199 |
+
"stage3_max_reuse_distance": 1000000000.0,
|
| 200 |
+
"stage3_gather_16bit_weights_on_model_save": true
|
| 201 |
+
},
|
| 202 |
+
"gradient_accumulation_steps": "auto",
|
| 203 |
+
"gradient_clipping": "auto",
|
| 204 |
+
"steps_per_print": 2000,
|
| 205 |
+
"train_batch_size": "auto",
|
| 206 |
+
"train_micro_batch_size_per_gpu": "auto",
|
| 207 |
+
"wall_clock_breakdown": false
|
| 208 |
+
},
|
| 209 |
+
"label_smoothing_factor": 0.0,
|
| 210 |
+
"optim": "adamw_torch",
|
| 211 |
+
"optim_args": null,
|
| 212 |
+
"adafactor": false,
|
| 213 |
+
"group_by_length": false,
|
| 214 |
+
"length_column_name": "length",
|
| 215 |
+
"report_to": [
|
| 216 |
+
"tensorboard"
|
| 217 |
+
],
|
| 218 |
+
"ddp_find_unused_parameters": null,
|
| 219 |
+
"ddp_bucket_cap_mb": null,
|
| 220 |
+
"ddp_broadcast_buffers": null,
|
| 221 |
+
"dataloader_pin_memory": true,
|
| 222 |
+
"dataloader_persistent_workers": false,
|
| 223 |
+
"skip_memory_metrics": true,
|
| 224 |
+
"use_legacy_prediction_loop": false,
|
| 225 |
+
"push_to_hub": false,
|
| 226 |
+
"resume_from_checkpoint": null,
|
| 227 |
+
"hub_model_id": null,
|
| 228 |
+
"hub_strategy": "every_save",
|
| 229 |
+
"hub_private_repo": null,
|
| 230 |
+
"hub_always_push": false,
|
| 231 |
+
"gradient_checkpointing": true,
|
| 232 |
+
"gradient_checkpointing_kwargs": null,
|
| 233 |
+
"include_inputs_for_metrics": false,
|
| 234 |
+
"include_for_metrics": [],
|
| 235 |
+
"eval_do_concat_batches": true,
|
| 236 |
+
"fp16_backend": "auto",
|
| 237 |
+
"push_to_hub_model_id": null,
|
| 238 |
+
"push_to_hub_organization": null,
|
| 239 |
+
"push_to_hub_token": null,
|
| 240 |
+
"mp_parameters": "",
|
| 241 |
+
"auto_find_batch_size": false,
|
| 242 |
+
"full_determinism": false,
|
| 243 |
+
"torchdynamo": null,
|
| 244 |
+
"ray_scope": "last",
|
| 245 |
+
"ddp_timeout": 1800,
|
| 246 |
+
"torch_compile": false,
|
| 247 |
+
"torch_compile_backend": null,
|
| 248 |
+
"torch_compile_mode": null,
|
| 249 |
+
"include_tokens_per_second": false,
|
| 250 |
+
"include_num_input_tokens_seen": false,
|
| 251 |
+
"neftune_noise_alpha": null,
|
| 252 |
+
"optim_target_modules": null,
|
| 253 |
+
"batch_eval_metrics": false,
|
| 254 |
+
"eval_on_start": false,
|
| 255 |
+
"use_liger_kernel": false,
|
| 256 |
+
"eval_use_gather_object": false,
|
| 257 |
+
"average_tokens_across_devices": false,
|
| 258 |
+
"sortish_sampler": false,
|
| 259 |
+
"predict_with_generate": false,
|
| 260 |
+
"generation_max_length": null,
|
| 261 |
+
"generation_num_beams": null,
|
| 262 |
+
"generation_config": null,
|
| 263 |
+
"check_model": true,
|
| 264 |
+
"acc_strategy": "token",
|
| 265 |
+
"train_dataloader_shuffle": true,
|
| 266 |
+
"metric_warmup_step": 0,
|
| 267 |
+
"fsdp_num": 1,
|
| 268 |
+
"acc_steps": 1,
|
| 269 |
+
"eval_use_evalscope": false,
|
| 270 |
+
"eval_datasets": [],
|
| 271 |
+
"eval_limit": null,
|
| 272 |
+
"eval_datasets_args": null,
|
| 273 |
+
"eval_generation_config": null,
|
| 274 |
+
"freeze_parameters": [
|
| 275 |
+
"visual",
|
| 276 |
+
"visual.merger"
|
| 277 |
+
],
|
| 278 |
+
"freeze_parameters_ratio": 0.0,
|
| 279 |
+
"trainable_parameters": [],
|
| 280 |
+
"freeze_llm": false,
|
| 281 |
+
"freeze_vit": true,
|
| 282 |
+
"freeze_aligner": true,
|
| 283 |
+
"target_modules": [
|
| 284 |
+
"all-linear"
|
| 285 |
+
],
|
| 286 |
+
"target_regex": null,
|
| 287 |
+
"modules_to_save": [],
|
| 288 |
+
"lora_rank": 64,
|
| 289 |
+
"lora_alpha": 32,
|
| 290 |
+
"lora_dropout": 0.05,
|
| 291 |
+
"lora_bias": "none",
|
| 292 |
+
"lora_dtype": null,
|
| 293 |
+
"lorap_lr_ratio": null,
|
| 294 |
+
"use_rslora": false,
|
| 295 |
+
"use_dora": false,
|
| 296 |
+
"lora_ga_batch_size": 2,
|
| 297 |
+
"lora_ga_iters": 2,
|
| 298 |
+
"lora_ga_max_length": 1024,
|
| 299 |
+
"lora_ga_direction": "ArB2r",
|
| 300 |
+
"lora_ga_scale": "stable",
|
| 301 |
+
"lora_ga_stable_gamma": 16,
|
| 302 |
+
"init_weights": true,
|
| 303 |
+
"fourier_n_frequency": 2000,
|
| 304 |
+
"fourier_scaling": 300.0,
|
| 305 |
+
"boft_block_size": 4,
|
| 306 |
+
"boft_block_num": 0,
|
| 307 |
+
"boft_n_butterfly_factor": 1,
|
| 308 |
+
"boft_dropout": 0.0,
|
| 309 |
+
"vera_rank": 256,
|
| 310 |
+
"vera_projection_prng_key": 0,
|
| 311 |
+
"vera_dropout": 0.0,
|
| 312 |
+
"vera_d_initial": 0.1,
|
| 313 |
+
"adapter_act": "gelu",
|
| 314 |
+
"adapter_length": 128,
|
| 315 |
+
"use_galore": false,
|
| 316 |
+
"galore_target_modules": null,
|
| 317 |
+
"galore_rank": 128,
|
| 318 |
+
"galore_update_proj_gap": 50,
|
| 319 |
+
"galore_scale": 1.0,
|
| 320 |
+
"galore_proj_type": "std",
|
| 321 |
+
"galore_optim_per_parameter": false,
|
| 322 |
+
"galore_with_embedding": false,
|
| 323 |
+
"galore_quantization": false,
|
| 324 |
+
"galore_proj_quant": false,
|
| 325 |
+
"galore_proj_bits": 4,
|
| 326 |
+
"galore_proj_group_size": 256,
|
| 327 |
+
"galore_cos_threshold": 0.4,
|
| 328 |
+
"galore_gamma_proj": 2,
|
| 329 |
+
"galore_queue_size": 5,
|
| 330 |
+
"adalora_target_r": 8,
|
| 331 |
+
"adalora_init_r": 12,
|
| 332 |
+
"adalora_tinit": 0,
|
| 333 |
+
"adalora_tfinal": 0,
|
| 334 |
+
"adalora_deltaT": 1,
|
| 335 |
+
"adalora_beta1": 0.85,
|
| 336 |
+
"adalora_beta2": 0.85,
|
| 337 |
+
"adalora_orth_reg_weight": 0.5,
|
| 338 |
+
"llamapro_num_new_blocks": 4,
|
| 339 |
+
"llamapro_num_groups": null,
|
| 340 |
+
"lisa_activated_layers": 0,
|
| 341 |
+
"lisa_step_interval": 20,
|
| 342 |
+
"reft_layer_key": null,
|
| 343 |
+
"reft_layers": null,
|
| 344 |
+
"reft_rank": 4,
|
| 345 |
+
"reft_intervention_type": "LoreftIntervention",
|
| 346 |
+
"reft_args": null,
|
| 347 |
+
"swanlab_token": null,
|
| 348 |
+
"swanlab_project": null,
|
| 349 |
+
"swanlab_workspace": null,
|
| 350 |
+
"swanlab_exp_name": null,
|
| 351 |
+
"swanlab_mode": "cloud",
|
| 352 |
+
"add_version": true,
|
| 353 |
+
"resume_only_model": false,
|
| 354 |
+
"create_checkpoint_symlink": false,
|
| 355 |
+
"packing": false,
|
| 356 |
+
"lazy_tokenize": true,
|
| 357 |
+
"loss_type": null,
|
| 358 |
+
"optimizer": "custom",
|
| 359 |
+
"metric": null,
|
| 360 |
+
"zero_hpz_partition_size": null,
|
| 361 |
+
"rank": 0,
|
| 362 |
+
"global_world_size": 4,
|
| 363 |
+
"local_world_size": 4,
|
| 364 |
+
"model_suffix": "Qwen2.5-VL-3B-Instruct",
|
| 365 |
+
"model_info": "ModelInfo(model_type='qwen2_5_vl', model_dir='/cpfs01/shared/llm_ddd/tanghuanze/ckpts/hf_hub/Qwen/Qwen2.5-VL-3B-Instruct', torch_dtype=torch.bfloat16, max_model_len=128000, quant_method=None, quant_bits=None, rope_scaling={'type': 'default', 'mrope_section': [16, 24, 24], 'rope_type': 'default'}, config=None, task_type='causal_lm', num_labels=None)",
|
| 366 |
+
"model_meta": "ModelMeta(model_type='qwen2_5_vl', model_groups=[ModelGroup(models=[Model(ms_model_id='Qwen/Qwen2.5-VL-3B-Instruct', hf_model_id='Qwen/Qwen2.5-VL-3B-Instruct', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-7B-Instruct', hf_model_id='Qwen/Qwen2.5-VL-7B-Instruct', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-32B-Instruct', hf_model_id='Qwen/Qwen2.5-VL-32B-Instruct', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-72B-Instruct', hf_model_id='Qwen/Qwen2.5-VL-72B-Instruct', model_path=None, ms_revision=None, hf_revision=None)], ignore_patterns=None, requires=None, tags=[]), ModelGroup(models=[Model(ms_model_id='Qwen/Qwen2.5-VL-3B-Instruct-AWQ', hf_model_id='Qwen/Qwen2.5-VL-3B-Instruct-AWQ', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-7B-Instruct-AWQ', hf_model_id='Qwen/Qwen2.5-VL-7B-Instruct-AWQ', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-32B-Instruct-AWQ', hf_model_id='Qwen/Qwen2.5-VL-32B-Instruct-AWQ', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-72B-Instruct-AWQ', hf_model_id='Qwen/Qwen2.5-VL-72B-Instruct-AWQ', model_path=None, ms_revision=None, hf_revision=None)], ignore_patterns=None, requires=None, tags=[])], template='qwen2_5_vl', get_function=<function get_model_tokenizer_qwen2_5_vl at 0x7f10cd44dbd0>, model_arch='qwen2_vl', architectures=['Qwen2_5_VLForConditionalGeneration'], additional_saved_files=[], torch_dtype=None, is_multimodal=True, is_reward=False, task_type=None, ignore_patterns=None, requires=['transformers>=4.49', 'qwen_vl_utils>=0.0.6', 'decord'], tags=[])",
|
| 367 |
+
"model_dir": "/cpfs01/shared/llm_ddd/tanghuanze/ckpts/hf_hub/Qwen/Qwen2.5-VL-3B-Instruct",
|
| 368 |
+
"hub": "<class 'swift.hub.hub.MSHub'>",
|
| 369 |
+
"evaluation_strategy": "steps",
|
| 370 |
+
"training_args": "Seq2SeqTrainingArguments(output_dir='/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v0-20250508-174810', overwrite_output_dir=False, do_train=False, do_eval=True, do_predict=False, eval_strategy=<IntervalStrategy.STEPS: 'steps'>, prediction_loss_only=False, per_device_train_batch_size=4, per_device_eval_batch_size=4, per_gpu_train_batch_size=None, per_gpu_eval_batch_size=None, gradient_accumulation_steps=4, eval_accumulation_steps=None, eval_delay=0, torch_empty_cache_steps=None, learning_rate=0.001, weight_decay=0.1, adam_beta1=0.9, adam_beta2=0.95, adam_epsilon=1e-08, max_grad_norm=1.0, num_train_epochs=1.0, max_steps=-1, lr_scheduler_type=<SchedulerType.COSINE: 'cosine'>, lr_scheduler_kwargs=None, warmup_ratio=0.05, warmup_steps=0, log_level='passive', log_level_replica='warning', log_on_each_node=True, logging_dir='/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v0-20250508-174810/runs', logging_strategy=<IntervalStrategy.STEPS: 'steps'>, logging_first_step=True, logging_steps=5, logging_nan_inf_filter=True, save_strategy=<SaveStrategy.STEPS: 'steps'>, save_steps=100, save_total_limit=2, save_safetensors=True, save_on_each_node=False, save_only_model=True, restore_callback_states_from_checkpoint=False, no_cuda=False, use_cpu=False, use_mps_device=False, seed=42, data_seed=42, jit_mode_eval=False, use_ipex=False, bf16=True, fp16=False, fp16_opt_level='O1', half_precision_backend='auto', bf16_full_eval=False, fp16_full_eval=False, tf32=None, local_rank=0, ddp_backend=None, tpu_num_cores=None, tpu_metrics_debug=False, debug=[], dataloader_drop_last=False, eval_steps=100, dataloader_num_workers=2, dataloader_prefetch_factor=10, past_index=-1, run_name='/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v0-20250508-174810', disable_tqdm=False, remove_unused_columns=False, label_names=None, load_best_model_at_end=False, metric_for_best_model='loss', greater_is_better=False, ignore_data_skip=False, fsdp=[], fsdp_min_num_params=0, fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False}, tp_size=0, fsdp_transformer_layer_cls_to_wrap=None, accelerator_config=AcceleratorConfig(split_batches=False, dispatch_batches=False, even_batches=True, use_seedable_sampler=True, non_blocking=False, gradient_accumulation_kwargs=None, use_configured_state=False), deepspeed={'fp16': {'enabled': 'auto', 'loss_scale': 0, 'loss_scale_window': 1000, 'initial_scale_power': 16, 'hysteresis': 2, 'min_loss_scale': 1}, 'bf16': {'enabled': 'auto'}, 'zero_optimization': {'stage': 3, 'offload_optimizer': {'device': 'none', 'pin_memory': True}, 'offload_param': {'device': 'none', 'pin_memory': True}, 'overlap_comm': False, 'contiguous_gradients': True, 'sub_group_size': 1000000000.0, 'reduce_bucket_size': 'auto', 'zero_quantized_weights': False, 'zero_quantized_gradients': False, 'stage3_prefetch_bucket_size': 'auto', 'stage3_param_persistence_threshold': 'auto', 'stage3_max_live_parameters': 1000000000.0, 'stage3_max_reuse_distance': 1000000000.0, 'stage3_gather_16bit_weights_on_model_save': True}, 'gradient_accumulation_steps': 'auto', 'gradient_clipping': 'auto', 'steps_per_print': 2000, 'train_batch_size': 'auto', 'train_micro_batch_size_per_gpu': 'auto', 'wall_clock_breakdown': False}, label_smoothing_factor=0.0, optim=<OptimizerNames.ADAMW_TORCH: 'adamw_torch'>, optim_args=None, adafactor=False, group_by_length=False, length_column_name='length', report_to=['tensorboard'], ddp_find_unused_parameters=None, ddp_bucket_cap_mb=None, ddp_broadcast_buffers=None, dataloader_pin_memory=True, dataloader_persistent_workers=False, skip_memory_metrics=True, use_legacy_prediction_loop=False, push_to_hub=False, resume_from_checkpoint=None, hub_model_id=None, hub_strategy=<HubStrategy.EVERY_SAVE: 'every_save'>, hub_token=None, hub_private_repo=None, hub_always_push=False, gradient_checkpointing=True, gradient_checkpointing_kwargs=None, include_inputs_for_metrics=False, include_for_metrics=[], eval_do_concat_batches=True, fp16_backend='auto', push_to_hub_model_id=None, push_to_hub_organization=None, push_to_hub_token=None, mp_parameters='', auto_find_batch_size=False, full_determinism=False, torchdynamo=None, ray_scope='last', ddp_timeout=1800, torch_compile=False, torch_compile_backend=None, torch_compile_mode=None, include_tokens_per_second=None, include_num_input_tokens_seen=None, neftune_noise_alpha=None, optim_target_modules=None, batch_eval_metrics=False, eval_on_start=False, use_liger_kernel=False, eval_use_gather_object=False, average_tokens_across_devices=None, sortish_sampler=False, predict_with_generate=False, generation_max_length=None, generation_num_beams=None, generation_config=None, check_model=True, acc_strategy='token', train_dataloader_shuffle=True, metric_warmup_step=0, fsdp_num=1, acc_steps=1, eval_use_evalscope=False, eval_datasets=[], eval_limit=None, eval_datasets_args=None, eval_generation_config=None, train_type='custom', optimizer='custom', local_repo_path=None, galore_config=None)"
|
| 371 |
+
}
|
v0-20250508-174810/logging.jsonl
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"loss": 2.58850908, "token_acc": 0.59252336, "grad_norm": 66.09007212, "learning_rate": 3.33e-06, "memory(GiB)": 21.94, "train_speed(iter/s)": 0.015739, "epoch": 0.00166736, "global_step/max_steps": "1/599", "percentage": "0.17%", "elapsed_time": "58s", "remaining_time": "9h 47m 25s"}
|
v0-20250508-174810/runs/events.out.tfevents.1746697713.dsw-2138-644b4b995d-5j9mw.1220945.0
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e7a7eaff0ea03799bd878e6116f9649fdf516f9218c6f75451b43067ea546e9a
|
| 3 |
+
size 8072
|
v0-20250508-174810/val_dataset.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
v1-20250508-175113/args.json
ADDED
|
@@ -0,0 +1,371 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model": "/cpfs01/shared/llm_ddd/tanghuanze/ckpts/hf_hub/Qwen/Qwen2.5-VL-3B-Instruct",
|
| 3 |
+
"model_type": "qwen2_5_vl",
|
| 4 |
+
"model_revision": null,
|
| 5 |
+
"task_type": "causal_lm",
|
| 6 |
+
"torch_dtype": "bfloat16",
|
| 7 |
+
"attn_impl": null,
|
| 8 |
+
"num_labels": null,
|
| 9 |
+
"problem_type": null,
|
| 10 |
+
"rope_scaling": null,
|
| 11 |
+
"device_map": null,
|
| 12 |
+
"max_memory": {},
|
| 13 |
+
"local_repo_path": null,
|
| 14 |
+
"template": "qwen2_5_vl",
|
| 15 |
+
"system": null,
|
| 16 |
+
"max_length": 8192,
|
| 17 |
+
"truncation_strategy": "delete",
|
| 18 |
+
"max_pixels": null,
|
| 19 |
+
"agent_template": null,
|
| 20 |
+
"norm_bbox": null,
|
| 21 |
+
"response_prefix": null,
|
| 22 |
+
"padding_side": "right",
|
| 23 |
+
"loss_scale": "default",
|
| 24 |
+
"sequence_parallel_size": 1,
|
| 25 |
+
"use_chat_template": true,
|
| 26 |
+
"template_backend": "swift",
|
| 27 |
+
"dataset": [
|
| 28 |
+
"/cpfs01/shared/llm_ddd/zhangyulong/sa_work/msdata/wei682/amazon-qwen-file/updated_merged_data.json"
|
| 29 |
+
],
|
| 30 |
+
"val_dataset": [],
|
| 31 |
+
"split_dataset_ratio": 0.01,
|
| 32 |
+
"data_seed": 42,
|
| 33 |
+
"dataset_num_proc": 2,
|
| 34 |
+
"dataset_shuffle": true,
|
| 35 |
+
"val_dataset_shuffle": false,
|
| 36 |
+
"streaming": false,
|
| 37 |
+
"interleave_prob": null,
|
| 38 |
+
"stopping_strategy": "first_exhausted",
|
| 39 |
+
"shuffle_buffer_size": 1000,
|
| 40 |
+
"enable_cache": false,
|
| 41 |
+
"download_mode": "reuse_dataset_if_exists",
|
| 42 |
+
"columns": {},
|
| 43 |
+
"strict": false,
|
| 44 |
+
"remove_unused_columns": true,
|
| 45 |
+
"model_name": [
|
| 46 |
+
null,
|
| 47 |
+
null
|
| 48 |
+
],
|
| 49 |
+
"model_author": [
|
| 50 |
+
null,
|
| 51 |
+
null
|
| 52 |
+
],
|
| 53 |
+
"custom_dataset_info": [],
|
| 54 |
+
"quant_method": null,
|
| 55 |
+
"quant_bits": null,
|
| 56 |
+
"hqq_axis": null,
|
| 57 |
+
"bnb_4bit_compute_dtype": "bfloat16",
|
| 58 |
+
"bnb_4bit_quant_type": "nf4",
|
| 59 |
+
"bnb_4bit_use_double_quant": true,
|
| 60 |
+
"bnb_4bit_quant_storage": null,
|
| 61 |
+
"max_new_tokens": 64,
|
| 62 |
+
"temperature": 0.0,
|
| 63 |
+
"top_k": null,
|
| 64 |
+
"top_p": null,
|
| 65 |
+
"repetition_penalty": null,
|
| 66 |
+
"num_beams": 1,
|
| 67 |
+
"stream": false,
|
| 68 |
+
"stop_words": [],
|
| 69 |
+
"logprobs": false,
|
| 70 |
+
"top_logprobs": null,
|
| 71 |
+
"ckpt_dir": null,
|
| 72 |
+
"load_dataset_config": null,
|
| 73 |
+
"lora_modules": [],
|
| 74 |
+
"tuner_backend": "peft",
|
| 75 |
+
"train_type": "custom",
|
| 76 |
+
"adapters": [],
|
| 77 |
+
"external_plugins": [
|
| 78 |
+
"examples/train/multimodal/lora_llm_full_vit/custom_plugin.py"
|
| 79 |
+
],
|
| 80 |
+
"seed": 42,
|
| 81 |
+
"model_kwargs": {},
|
| 82 |
+
"load_args": false,
|
| 83 |
+
"load_data_args": false,
|
| 84 |
+
"use_hf": false,
|
| 85 |
+
"hub_token": null,
|
| 86 |
+
"custom_register_path": [],
|
| 87 |
+
"ignore_args_error": false,
|
| 88 |
+
"use_swift_lora": false,
|
| 89 |
+
"output_dir": "/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113",
|
| 90 |
+
"overwrite_output_dir": false,
|
| 91 |
+
"do_train": false,
|
| 92 |
+
"do_eval": false,
|
| 93 |
+
"do_predict": false,
|
| 94 |
+
"eval_strategy": "steps",
|
| 95 |
+
"prediction_loss_only": false,
|
| 96 |
+
"per_device_train_batch_size": 4,
|
| 97 |
+
"per_device_eval_batch_size": 4,
|
| 98 |
+
"per_gpu_train_batch_size": null,
|
| 99 |
+
"per_gpu_eval_batch_size": null,
|
| 100 |
+
"gradient_accumulation_steps": 4,
|
| 101 |
+
"eval_accumulation_steps": null,
|
| 102 |
+
"eval_delay": 0,
|
| 103 |
+
"torch_empty_cache_steps": null,
|
| 104 |
+
"learning_rate": 0.001,
|
| 105 |
+
"weight_decay": 0.1,
|
| 106 |
+
"adam_beta1": 0.9,
|
| 107 |
+
"adam_beta2": 0.95,
|
| 108 |
+
"adam_epsilon": 1e-08,
|
| 109 |
+
"max_grad_norm": 1.0,
|
| 110 |
+
"num_train_epochs": 1.0,
|
| 111 |
+
"max_steps": -1,
|
| 112 |
+
"lr_scheduler_type": "cosine",
|
| 113 |
+
"lr_scheduler_kwargs": null,
|
| 114 |
+
"warmup_ratio": 0.05,
|
| 115 |
+
"warmup_steps": 0,
|
| 116 |
+
"log_level": "passive",
|
| 117 |
+
"log_level_replica": "warning",
|
| 118 |
+
"log_on_each_node": true,
|
| 119 |
+
"logging_dir": "/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113/runs",
|
| 120 |
+
"logging_strategy": "steps",
|
| 121 |
+
"logging_first_step": true,
|
| 122 |
+
"logging_steps": 5,
|
| 123 |
+
"logging_nan_inf_filter": true,
|
| 124 |
+
"save_strategy": "steps",
|
| 125 |
+
"save_steps": 100.0,
|
| 126 |
+
"save_total_limit": 2,
|
| 127 |
+
"save_safetensors": true,
|
| 128 |
+
"save_on_each_node": false,
|
| 129 |
+
"save_only_model": true,
|
| 130 |
+
"restore_callback_states_from_checkpoint": false,
|
| 131 |
+
"no_cuda": false,
|
| 132 |
+
"use_cpu": false,
|
| 133 |
+
"use_mps_device": false,
|
| 134 |
+
"jit_mode_eval": false,
|
| 135 |
+
"use_ipex": false,
|
| 136 |
+
"bf16": true,
|
| 137 |
+
"fp16": false,
|
| 138 |
+
"fp16_opt_level": "O1",
|
| 139 |
+
"half_precision_backend": "auto",
|
| 140 |
+
"bf16_full_eval": false,
|
| 141 |
+
"fp16_full_eval": false,
|
| 142 |
+
"tf32": null,
|
| 143 |
+
"local_rank": 0,
|
| 144 |
+
"ddp_backend": null,
|
| 145 |
+
"tpu_num_cores": null,
|
| 146 |
+
"tpu_metrics_debug": false,
|
| 147 |
+
"debug": null,
|
| 148 |
+
"dataloader_drop_last": false,
|
| 149 |
+
"eval_steps": 100.0,
|
| 150 |
+
"dataloader_num_workers": 2,
|
| 151 |
+
"dataloader_prefetch_factor": null,
|
| 152 |
+
"past_index": -1,
|
| 153 |
+
"run_name": "/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113",
|
| 154 |
+
"disable_tqdm": null,
|
| 155 |
+
"label_names": null,
|
| 156 |
+
"load_best_model_at_end": false,
|
| 157 |
+
"metric_for_best_model": "loss",
|
| 158 |
+
"greater_is_better": false,
|
| 159 |
+
"ignore_data_skip": false,
|
| 160 |
+
"fsdp": "",
|
| 161 |
+
"fsdp_min_num_params": 0,
|
| 162 |
+
"fsdp_config": null,
|
| 163 |
+
"tp_size": 0,
|
| 164 |
+
"fsdp_transformer_layer_cls_to_wrap": null,
|
| 165 |
+
"accelerator_config": {
|
| 166 |
+
"dispatch_batches": false
|
| 167 |
+
},
|
| 168 |
+
"deepspeed": {
|
| 169 |
+
"fp16": {
|
| 170 |
+
"enabled": "auto",
|
| 171 |
+
"loss_scale": 0,
|
| 172 |
+
"loss_scale_window": 1000,
|
| 173 |
+
"initial_scale_power": 16,
|
| 174 |
+
"hysteresis": 2,
|
| 175 |
+
"min_loss_scale": 1
|
| 176 |
+
},
|
| 177 |
+
"bf16": {
|
| 178 |
+
"enabled": "auto"
|
| 179 |
+
},
|
| 180 |
+
"zero_optimization": {
|
| 181 |
+
"stage": 3,
|
| 182 |
+
"offload_optimizer": {
|
| 183 |
+
"device": "none",
|
| 184 |
+
"pin_memory": true
|
| 185 |
+
},
|
| 186 |
+
"offload_param": {
|
| 187 |
+
"device": "none",
|
| 188 |
+
"pin_memory": true
|
| 189 |
+
},
|
| 190 |
+
"overlap_comm": false,
|
| 191 |
+
"contiguous_gradients": true,
|
| 192 |
+
"sub_group_size": 1000000000.0,
|
| 193 |
+
"reduce_bucket_size": "auto",
|
| 194 |
+
"zero_quantized_weights": false,
|
| 195 |
+
"zero_quantized_gradients": false,
|
| 196 |
+
"stage3_prefetch_bucket_size": "auto",
|
| 197 |
+
"stage3_param_persistence_threshold": "auto",
|
| 198 |
+
"stage3_max_live_parameters": 1000000000.0,
|
| 199 |
+
"stage3_max_reuse_distance": 1000000000.0,
|
| 200 |
+
"stage3_gather_16bit_weights_on_model_save": true
|
| 201 |
+
},
|
| 202 |
+
"gradient_accumulation_steps": "auto",
|
| 203 |
+
"gradient_clipping": "auto",
|
| 204 |
+
"steps_per_print": 2000,
|
| 205 |
+
"train_batch_size": "auto",
|
| 206 |
+
"train_micro_batch_size_per_gpu": "auto",
|
| 207 |
+
"wall_clock_breakdown": false
|
| 208 |
+
},
|
| 209 |
+
"label_smoothing_factor": 0.0,
|
| 210 |
+
"optim": "adamw_torch",
|
| 211 |
+
"optim_args": null,
|
| 212 |
+
"adafactor": false,
|
| 213 |
+
"group_by_length": false,
|
| 214 |
+
"length_column_name": "length",
|
| 215 |
+
"report_to": [
|
| 216 |
+
"tensorboard"
|
| 217 |
+
],
|
| 218 |
+
"ddp_find_unused_parameters": null,
|
| 219 |
+
"ddp_bucket_cap_mb": null,
|
| 220 |
+
"ddp_broadcast_buffers": null,
|
| 221 |
+
"dataloader_pin_memory": true,
|
| 222 |
+
"dataloader_persistent_workers": false,
|
| 223 |
+
"skip_memory_metrics": true,
|
| 224 |
+
"use_legacy_prediction_loop": false,
|
| 225 |
+
"push_to_hub": false,
|
| 226 |
+
"resume_from_checkpoint": null,
|
| 227 |
+
"hub_model_id": null,
|
| 228 |
+
"hub_strategy": "every_save",
|
| 229 |
+
"hub_private_repo": null,
|
| 230 |
+
"hub_always_push": false,
|
| 231 |
+
"gradient_checkpointing": true,
|
| 232 |
+
"gradient_checkpointing_kwargs": null,
|
| 233 |
+
"include_inputs_for_metrics": false,
|
| 234 |
+
"include_for_metrics": [],
|
| 235 |
+
"eval_do_concat_batches": true,
|
| 236 |
+
"fp16_backend": "auto",
|
| 237 |
+
"push_to_hub_model_id": null,
|
| 238 |
+
"push_to_hub_organization": null,
|
| 239 |
+
"push_to_hub_token": null,
|
| 240 |
+
"mp_parameters": "",
|
| 241 |
+
"auto_find_batch_size": false,
|
| 242 |
+
"full_determinism": false,
|
| 243 |
+
"torchdynamo": null,
|
| 244 |
+
"ray_scope": "last",
|
| 245 |
+
"ddp_timeout": 1800,
|
| 246 |
+
"torch_compile": false,
|
| 247 |
+
"torch_compile_backend": null,
|
| 248 |
+
"torch_compile_mode": null,
|
| 249 |
+
"include_tokens_per_second": false,
|
| 250 |
+
"include_num_input_tokens_seen": false,
|
| 251 |
+
"neftune_noise_alpha": null,
|
| 252 |
+
"optim_target_modules": null,
|
| 253 |
+
"batch_eval_metrics": false,
|
| 254 |
+
"eval_on_start": false,
|
| 255 |
+
"use_liger_kernel": false,
|
| 256 |
+
"eval_use_gather_object": false,
|
| 257 |
+
"average_tokens_across_devices": false,
|
| 258 |
+
"sortish_sampler": false,
|
| 259 |
+
"predict_with_generate": false,
|
| 260 |
+
"generation_max_length": null,
|
| 261 |
+
"generation_num_beams": null,
|
| 262 |
+
"generation_config": null,
|
| 263 |
+
"check_model": true,
|
| 264 |
+
"acc_strategy": "token",
|
| 265 |
+
"train_dataloader_shuffle": true,
|
| 266 |
+
"metric_warmup_step": 0,
|
| 267 |
+
"fsdp_num": 1,
|
| 268 |
+
"acc_steps": 1,
|
| 269 |
+
"eval_use_evalscope": false,
|
| 270 |
+
"eval_datasets": [],
|
| 271 |
+
"eval_limit": null,
|
| 272 |
+
"eval_datasets_args": null,
|
| 273 |
+
"eval_generation_config": null,
|
| 274 |
+
"freeze_parameters": [
|
| 275 |
+
"visual",
|
| 276 |
+
"visual.merger"
|
| 277 |
+
],
|
| 278 |
+
"freeze_parameters_ratio": 0.0,
|
| 279 |
+
"trainable_parameters": [],
|
| 280 |
+
"freeze_llm": false,
|
| 281 |
+
"freeze_vit": true,
|
| 282 |
+
"freeze_aligner": true,
|
| 283 |
+
"target_modules": [
|
| 284 |
+
"all-linear"
|
| 285 |
+
],
|
| 286 |
+
"target_regex": null,
|
| 287 |
+
"modules_to_save": [],
|
| 288 |
+
"lora_rank": 64,
|
| 289 |
+
"lora_alpha": 32,
|
| 290 |
+
"lora_dropout": 0.05,
|
| 291 |
+
"lora_bias": "none",
|
| 292 |
+
"lora_dtype": null,
|
| 293 |
+
"lorap_lr_ratio": null,
|
| 294 |
+
"use_rslora": false,
|
| 295 |
+
"use_dora": false,
|
| 296 |
+
"lora_ga_batch_size": 2,
|
| 297 |
+
"lora_ga_iters": 2,
|
| 298 |
+
"lora_ga_max_length": 1024,
|
| 299 |
+
"lora_ga_direction": "ArB2r",
|
| 300 |
+
"lora_ga_scale": "stable",
|
| 301 |
+
"lora_ga_stable_gamma": 16,
|
| 302 |
+
"init_weights": true,
|
| 303 |
+
"fourier_n_frequency": 2000,
|
| 304 |
+
"fourier_scaling": 300.0,
|
| 305 |
+
"boft_block_size": 4,
|
| 306 |
+
"boft_block_num": 0,
|
| 307 |
+
"boft_n_butterfly_factor": 1,
|
| 308 |
+
"boft_dropout": 0.0,
|
| 309 |
+
"vera_rank": 256,
|
| 310 |
+
"vera_projection_prng_key": 0,
|
| 311 |
+
"vera_dropout": 0.0,
|
| 312 |
+
"vera_d_initial": 0.1,
|
| 313 |
+
"adapter_act": "gelu",
|
| 314 |
+
"adapter_length": 128,
|
| 315 |
+
"use_galore": false,
|
| 316 |
+
"galore_target_modules": null,
|
| 317 |
+
"galore_rank": 128,
|
| 318 |
+
"galore_update_proj_gap": 50,
|
| 319 |
+
"galore_scale": 1.0,
|
| 320 |
+
"galore_proj_type": "std",
|
| 321 |
+
"galore_optim_per_parameter": false,
|
| 322 |
+
"galore_with_embedding": false,
|
| 323 |
+
"galore_quantization": false,
|
| 324 |
+
"galore_proj_quant": false,
|
| 325 |
+
"galore_proj_bits": 4,
|
| 326 |
+
"galore_proj_group_size": 256,
|
| 327 |
+
"galore_cos_threshold": 0.4,
|
| 328 |
+
"galore_gamma_proj": 2,
|
| 329 |
+
"galore_queue_size": 5,
|
| 330 |
+
"adalora_target_r": 8,
|
| 331 |
+
"adalora_init_r": 12,
|
| 332 |
+
"adalora_tinit": 0,
|
| 333 |
+
"adalora_tfinal": 0,
|
| 334 |
+
"adalora_deltaT": 1,
|
| 335 |
+
"adalora_beta1": 0.85,
|
| 336 |
+
"adalora_beta2": 0.85,
|
| 337 |
+
"adalora_orth_reg_weight": 0.5,
|
| 338 |
+
"llamapro_num_new_blocks": 4,
|
| 339 |
+
"llamapro_num_groups": null,
|
| 340 |
+
"lisa_activated_layers": 0,
|
| 341 |
+
"lisa_step_interval": 20,
|
| 342 |
+
"reft_layer_key": null,
|
| 343 |
+
"reft_layers": null,
|
| 344 |
+
"reft_rank": 4,
|
| 345 |
+
"reft_intervention_type": "LoreftIntervention",
|
| 346 |
+
"reft_args": null,
|
| 347 |
+
"swanlab_token": null,
|
| 348 |
+
"swanlab_project": null,
|
| 349 |
+
"swanlab_workspace": null,
|
| 350 |
+
"swanlab_exp_name": null,
|
| 351 |
+
"swanlab_mode": "cloud",
|
| 352 |
+
"add_version": true,
|
| 353 |
+
"resume_only_model": false,
|
| 354 |
+
"create_checkpoint_symlink": false,
|
| 355 |
+
"packing": false,
|
| 356 |
+
"lazy_tokenize": true,
|
| 357 |
+
"loss_type": null,
|
| 358 |
+
"optimizer": "custom",
|
| 359 |
+
"metric": null,
|
| 360 |
+
"zero_hpz_partition_size": null,
|
| 361 |
+
"rank": 0,
|
| 362 |
+
"global_world_size": 4,
|
| 363 |
+
"local_world_size": 4,
|
| 364 |
+
"model_suffix": "Qwen2.5-VL-3B-Instruct",
|
| 365 |
+
"model_info": "ModelInfo(model_type='qwen2_5_vl', model_dir='/cpfs01/shared/llm_ddd/tanghuanze/ckpts/hf_hub/Qwen/Qwen2.5-VL-3B-Instruct', torch_dtype=torch.bfloat16, max_model_len=128000, quant_method=None, quant_bits=None, rope_scaling={'type': 'default', 'mrope_section': [16, 24, 24], 'rope_type': 'default'}, config=None, task_type='causal_lm', num_labels=None)",
|
| 366 |
+
"model_meta": "ModelMeta(model_type='qwen2_5_vl', model_groups=[ModelGroup(models=[Model(ms_model_id='Qwen/Qwen2.5-VL-3B-Instruct', hf_model_id='Qwen/Qwen2.5-VL-3B-Instruct', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-7B-Instruct', hf_model_id='Qwen/Qwen2.5-VL-7B-Instruct', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-32B-Instruct', hf_model_id='Qwen/Qwen2.5-VL-32B-Instruct', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-72B-Instruct', hf_model_id='Qwen/Qwen2.5-VL-72B-Instruct', model_path=None, ms_revision=None, hf_revision=None)], ignore_patterns=None, requires=None, tags=[]), ModelGroup(models=[Model(ms_model_id='Qwen/Qwen2.5-VL-3B-Instruct-AWQ', hf_model_id='Qwen/Qwen2.5-VL-3B-Instruct-AWQ', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-7B-Instruct-AWQ', hf_model_id='Qwen/Qwen2.5-VL-7B-Instruct-AWQ', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-32B-Instruct-AWQ', hf_model_id='Qwen/Qwen2.5-VL-32B-Instruct-AWQ', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-72B-Instruct-AWQ', hf_model_id='Qwen/Qwen2.5-VL-72B-Instruct-AWQ', model_path=None, ms_revision=None, hf_revision=None)], ignore_patterns=None, requires=None, tags=[])], template='qwen2_5_vl', get_function=<function get_model_tokenizer_qwen2_5_vl at 0x7fe3082e5bd0>, model_arch='qwen2_vl', architectures=['Qwen2_5_VLForConditionalGeneration'], additional_saved_files=[], torch_dtype=None, is_multimodal=True, is_reward=False, task_type=None, ignore_patterns=None, requires=['transformers>=4.49', 'qwen_vl_utils>=0.0.6', 'decord'], tags=[])",
|
| 367 |
+
"model_dir": "/cpfs01/shared/llm_ddd/tanghuanze/ckpts/hf_hub/Qwen/Qwen2.5-VL-3B-Instruct",
|
| 368 |
+
"hub": "<class 'swift.hub.hub.MSHub'>",
|
| 369 |
+
"evaluation_strategy": "steps",
|
| 370 |
+
"training_args": "Seq2SeqTrainingArguments(output_dir='/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113', overwrite_output_dir=False, do_train=False, do_eval=True, do_predict=False, eval_strategy=<IntervalStrategy.STEPS: 'steps'>, prediction_loss_only=False, per_device_train_batch_size=4, per_device_eval_batch_size=4, per_gpu_train_batch_size=None, per_gpu_eval_batch_size=None, gradient_accumulation_steps=4, eval_accumulation_steps=None, eval_delay=0, torch_empty_cache_steps=None, learning_rate=0.001, weight_decay=0.1, adam_beta1=0.9, adam_beta2=0.95, adam_epsilon=1e-08, max_grad_norm=1.0, num_train_epochs=1.0, max_steps=-1, lr_scheduler_type=<SchedulerType.COSINE: 'cosine'>, lr_scheduler_kwargs=None, warmup_ratio=0.05, warmup_steps=0, log_level='passive', log_level_replica='warning', log_on_each_node=True, logging_dir='/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113/runs', logging_strategy=<IntervalStrategy.STEPS: 'steps'>, logging_first_step=True, logging_steps=5, logging_nan_inf_filter=True, save_strategy=<SaveStrategy.STEPS: 'steps'>, save_steps=100, save_total_limit=2, save_safetensors=True, save_on_each_node=False, save_only_model=True, restore_callback_states_from_checkpoint=False, no_cuda=False, use_cpu=False, use_mps_device=False, seed=42, data_seed=42, jit_mode_eval=False, use_ipex=False, bf16=True, fp16=False, fp16_opt_level='O1', half_precision_backend='auto', bf16_full_eval=False, fp16_full_eval=False, tf32=None, local_rank=0, ddp_backend=None, tpu_num_cores=None, tpu_metrics_debug=False, debug=[], dataloader_drop_last=False, eval_steps=100, dataloader_num_workers=2, dataloader_prefetch_factor=10, past_index=-1, run_name='/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113', disable_tqdm=False, remove_unused_columns=False, label_names=None, load_best_model_at_end=False, metric_for_best_model='loss', greater_is_better=False, ignore_data_skip=False, fsdp=[], fsdp_min_num_params=0, fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False}, tp_size=0, fsdp_transformer_layer_cls_to_wrap=None, accelerator_config=AcceleratorConfig(split_batches=False, dispatch_batches=False, even_batches=True, use_seedable_sampler=True, non_blocking=False, gradient_accumulation_kwargs=None, use_configured_state=False), deepspeed={'fp16': {'enabled': 'auto', 'loss_scale': 0, 'loss_scale_window': 1000, 'initial_scale_power': 16, 'hysteresis': 2, 'min_loss_scale': 1}, 'bf16': {'enabled': 'auto'}, 'zero_optimization': {'stage': 3, 'offload_optimizer': {'device': 'none', 'pin_memory': True}, 'offload_param': {'device': 'none', 'pin_memory': True}, 'overlap_comm': False, 'contiguous_gradients': True, 'sub_group_size': 1000000000.0, 'reduce_bucket_size': 'auto', 'zero_quantized_weights': False, 'zero_quantized_gradients': False, 'stage3_prefetch_bucket_size': 'auto', 'stage3_param_persistence_threshold': 'auto', 'stage3_max_live_parameters': 1000000000.0, 'stage3_max_reuse_distance': 1000000000.0, 'stage3_gather_16bit_weights_on_model_save': True}, 'gradient_accumulation_steps': 'auto', 'gradient_clipping': 'auto', 'steps_per_print': 2000, 'train_batch_size': 'auto', 'train_micro_batch_size_per_gpu': 'auto', 'wall_clock_breakdown': False}, label_smoothing_factor=0.0, optim=<OptimizerNames.ADAMW_TORCH: 'adamw_torch'>, optim_args=None, adafactor=False, group_by_length=False, length_column_name='length', report_to=['tensorboard'], ddp_find_unused_parameters=None, ddp_bucket_cap_mb=None, ddp_broadcast_buffers=None, dataloader_pin_memory=True, dataloader_persistent_workers=False, skip_memory_metrics=True, use_legacy_prediction_loop=False, push_to_hub=False, resume_from_checkpoint=None, hub_model_id=None, hub_strategy=<HubStrategy.EVERY_SAVE: 'every_save'>, hub_token=None, hub_private_repo=None, hub_always_push=False, gradient_checkpointing=True, gradient_checkpointing_kwargs=None, include_inputs_for_metrics=False, include_for_metrics=[], eval_do_concat_batches=True, fp16_backend='auto', push_to_hub_model_id=None, push_to_hub_organization=None, push_to_hub_token=None, mp_parameters='', auto_find_batch_size=False, full_determinism=False, torchdynamo=None, ray_scope='last', ddp_timeout=1800, torch_compile=False, torch_compile_backend=None, torch_compile_mode=None, include_tokens_per_second=None, include_num_input_tokens_seen=None, neftune_noise_alpha=None, optim_target_modules=None, batch_eval_metrics=False, eval_on_start=False, use_liger_kernel=False, eval_use_gather_object=False, average_tokens_across_devices=None, sortish_sampler=False, predict_with_generate=False, generation_max_length=None, generation_num_beams=None, generation_config=None, check_model=True, acc_strategy='token', train_dataloader_shuffle=True, metric_warmup_step=0, fsdp_num=1, acc_steps=1, eval_use_evalscope=False, eval_datasets=[], eval_limit=None, eval_datasets_args=None, eval_generation_config=None, train_type='custom', optimizer='custom', local_repo_path=None, galore_config=None)"
|
| 371 |
+
}
|
v1-20250508-175113/checkpoint-500/README.md
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
base_model: /cpfs01/shared/llm_ddd/tanghuanze/ckpts/hf_hub/Qwen/Qwen2.5-VL-3B-Instruct
|
| 3 |
+
library_name: peft
|
| 4 |
+
---
|
| 5 |
+
|
| 6 |
+
# Model Card for Model ID
|
| 7 |
+
|
| 8 |
+
<!-- Provide a quick summary of what the model is/does. -->
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
## Model Details
|
| 13 |
+
|
| 14 |
+
### Model Description
|
| 15 |
+
|
| 16 |
+
<!-- Provide a longer summary of what this model is. -->
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
- **Developed by:** [More Information Needed]
|
| 21 |
+
- **Funded by [optional]:** [More Information Needed]
|
| 22 |
+
- **Shared by [optional]:** [More Information Needed]
|
| 23 |
+
- **Model type:** [More Information Needed]
|
| 24 |
+
- **Language(s) (NLP):** [More Information Needed]
|
| 25 |
+
- **License:** [More Information Needed]
|
| 26 |
+
- **Finetuned from model [optional]:** [More Information Needed]
|
| 27 |
+
|
| 28 |
+
### Model Sources [optional]
|
| 29 |
+
|
| 30 |
+
<!-- Provide the basic links for the model. -->
|
| 31 |
+
|
| 32 |
+
- **Repository:** [More Information Needed]
|
| 33 |
+
- **Paper [optional]:** [More Information Needed]
|
| 34 |
+
- **Demo [optional]:** [More Information Needed]
|
| 35 |
+
|
| 36 |
+
## Uses
|
| 37 |
+
|
| 38 |
+
<!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
|
| 39 |
+
|
| 40 |
+
### Direct Use
|
| 41 |
+
|
| 42 |
+
<!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
|
| 43 |
+
|
| 44 |
+
[More Information Needed]
|
| 45 |
+
|
| 46 |
+
### Downstream Use [optional]
|
| 47 |
+
|
| 48 |
+
<!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
|
| 49 |
+
|
| 50 |
+
[More Information Needed]
|
| 51 |
+
|
| 52 |
+
### Out-of-Scope Use
|
| 53 |
+
|
| 54 |
+
<!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
|
| 55 |
+
|
| 56 |
+
[More Information Needed]
|
| 57 |
+
|
| 58 |
+
## Bias, Risks, and Limitations
|
| 59 |
+
|
| 60 |
+
<!-- This section is meant to convey both technical and sociotechnical limitations. -->
|
| 61 |
+
|
| 62 |
+
[More Information Needed]
|
| 63 |
+
|
| 64 |
+
### Recommendations
|
| 65 |
+
|
| 66 |
+
<!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
|
| 67 |
+
|
| 68 |
+
Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
|
| 69 |
+
|
| 70 |
+
## How to Get Started with the Model
|
| 71 |
+
|
| 72 |
+
Use the code below to get started with the model.
|
| 73 |
+
|
| 74 |
+
[More Information Needed]
|
| 75 |
+
|
| 76 |
+
## Training Details
|
| 77 |
+
|
| 78 |
+
### Training Data
|
| 79 |
+
|
| 80 |
+
<!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
|
| 81 |
+
|
| 82 |
+
[More Information Needed]
|
| 83 |
+
|
| 84 |
+
### Training Procedure
|
| 85 |
+
|
| 86 |
+
<!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
|
| 87 |
+
|
| 88 |
+
#### Preprocessing [optional]
|
| 89 |
+
|
| 90 |
+
[More Information Needed]
|
| 91 |
+
|
| 92 |
+
|
| 93 |
+
#### Training Hyperparameters
|
| 94 |
+
|
| 95 |
+
- **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
|
| 96 |
+
|
| 97 |
+
#### Speeds, Sizes, Times [optional]
|
| 98 |
+
|
| 99 |
+
<!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
|
| 100 |
+
|
| 101 |
+
[More Information Needed]
|
| 102 |
+
|
| 103 |
+
## Evaluation
|
| 104 |
+
|
| 105 |
+
<!-- This section describes the evaluation protocols and provides the results. -->
|
| 106 |
+
|
| 107 |
+
### Testing Data, Factors & Metrics
|
| 108 |
+
|
| 109 |
+
#### Testing Data
|
| 110 |
+
|
| 111 |
+
<!-- This should link to a Dataset Card if possible. -->
|
| 112 |
+
|
| 113 |
+
[More Information Needed]
|
| 114 |
+
|
| 115 |
+
#### Factors
|
| 116 |
+
|
| 117 |
+
<!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
|
| 118 |
+
|
| 119 |
+
[More Information Needed]
|
| 120 |
+
|
| 121 |
+
#### Metrics
|
| 122 |
+
|
| 123 |
+
<!-- These are the evaluation metrics being used, ideally with a description of why. -->
|
| 124 |
+
|
| 125 |
+
[More Information Needed]
|
| 126 |
+
|
| 127 |
+
### Results
|
| 128 |
+
|
| 129 |
+
[More Information Needed]
|
| 130 |
+
|
| 131 |
+
#### Summary
|
| 132 |
+
|
| 133 |
+
|
| 134 |
+
|
| 135 |
+
## Model Examination [optional]
|
| 136 |
+
|
| 137 |
+
<!-- Relevant interpretability work for the model goes here -->
|
| 138 |
+
|
| 139 |
+
[More Information Needed]
|
| 140 |
+
|
| 141 |
+
## Environmental Impact
|
| 142 |
+
|
| 143 |
+
<!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
|
| 144 |
+
|
| 145 |
+
Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
|
| 146 |
+
|
| 147 |
+
- **Hardware Type:** [More Information Needed]
|
| 148 |
+
- **Hours used:** [More Information Needed]
|
| 149 |
+
- **Cloud Provider:** [More Information Needed]
|
| 150 |
+
- **Compute Region:** [More Information Needed]
|
| 151 |
+
- **Carbon Emitted:** [More Information Needed]
|
| 152 |
+
|
| 153 |
+
## Technical Specifications [optional]
|
| 154 |
+
|
| 155 |
+
### Model Architecture and Objective
|
| 156 |
+
|
| 157 |
+
[More Information Needed]
|
| 158 |
+
|
| 159 |
+
### Compute Infrastructure
|
| 160 |
+
|
| 161 |
+
[More Information Needed]
|
| 162 |
+
|
| 163 |
+
#### Hardware
|
| 164 |
+
|
| 165 |
+
[More Information Needed]
|
| 166 |
+
|
| 167 |
+
#### Software
|
| 168 |
+
|
| 169 |
+
[More Information Needed]
|
| 170 |
+
|
| 171 |
+
## Citation [optional]
|
| 172 |
+
|
| 173 |
+
<!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
|
| 174 |
+
|
| 175 |
+
**BibTeX:**
|
| 176 |
+
|
| 177 |
+
[More Information Needed]
|
| 178 |
+
|
| 179 |
+
**APA:**
|
| 180 |
+
|
| 181 |
+
[More Information Needed]
|
| 182 |
+
|
| 183 |
+
## Glossary [optional]
|
| 184 |
+
|
| 185 |
+
<!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
|
| 186 |
+
|
| 187 |
+
[More Information Needed]
|
| 188 |
+
|
| 189 |
+
## More Information [optional]
|
| 190 |
+
|
| 191 |
+
[More Information Needed]
|
| 192 |
+
|
| 193 |
+
## Model Card Authors [optional]
|
| 194 |
+
|
| 195 |
+
[More Information Needed]
|
| 196 |
+
|
| 197 |
+
## Model Card Contact
|
| 198 |
+
|
| 199 |
+
[More Information Needed]
|
| 200 |
+
### Framework versions
|
| 201 |
+
|
| 202 |
+
- PEFT 0.15.2
|
v1-20250508-175113/checkpoint-500/adapter_config.json
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"alpha_pattern": {},
|
| 3 |
+
"auto_mapping": null,
|
| 4 |
+
"base_model_name_or_path": "/cpfs01/shared/llm_ddd/tanghuanze/ckpts/hf_hub/Qwen/Qwen2.5-VL-3B-Instruct",
|
| 5 |
+
"bias": "none",
|
| 6 |
+
"corda_config": null,
|
| 7 |
+
"eva_config": null,
|
| 8 |
+
"exclude_modules": null,
|
| 9 |
+
"fan_in_fan_out": false,
|
| 10 |
+
"inference_mode": true,
|
| 11 |
+
"init_lora_weights": true,
|
| 12 |
+
"layer_replication": null,
|
| 13 |
+
"layers_pattern": null,
|
| 14 |
+
"layers_to_transform": null,
|
| 15 |
+
"loftq_config": {},
|
| 16 |
+
"lora_alpha": 32,
|
| 17 |
+
"lora_bias": false,
|
| 18 |
+
"lora_dropout": 0.0,
|
| 19 |
+
"megatron_config": null,
|
| 20 |
+
"megatron_core": "megatron.core",
|
| 21 |
+
"modules_to_save": null,
|
| 22 |
+
"peft_type": "LORA",
|
| 23 |
+
"r": 64,
|
| 24 |
+
"rank_pattern": {},
|
| 25 |
+
"revision": null,
|
| 26 |
+
"target_modules": "^(model).*\\.(up_proj|o_proj|v_proj|k_proj|down_proj|q_proj|gate_proj)$",
|
| 27 |
+
"task_type": "CAUSAL_LM",
|
| 28 |
+
"trainable_token_indices": null,
|
| 29 |
+
"use_dora": false,
|
| 30 |
+
"use_rslora": false
|
| 31 |
+
}
|
v1-20250508-175113/checkpoint-500/adapter_model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4e28df83681e1e20d124b2be1931d9bf8458217cb1980149948a1b3c673e194c
|
| 3 |
+
size 239536776
|
v1-20250508-175113/checkpoint-500/additional_config.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"lora_dtype": null, "lorap_lr_ratio": null, "lorap_emb_lr": 1e-06}
|
v1-20250508-175113/checkpoint-500/args.json
ADDED
|
@@ -0,0 +1,371 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model": "/cpfs01/shared/llm_ddd/tanghuanze/ckpts/hf_hub/Qwen/Qwen2.5-VL-3B-Instruct",
|
| 3 |
+
"model_type": "qwen2_5_vl",
|
| 4 |
+
"model_revision": null,
|
| 5 |
+
"task_type": "causal_lm",
|
| 6 |
+
"torch_dtype": "bfloat16",
|
| 7 |
+
"attn_impl": null,
|
| 8 |
+
"num_labels": null,
|
| 9 |
+
"problem_type": null,
|
| 10 |
+
"rope_scaling": null,
|
| 11 |
+
"device_map": null,
|
| 12 |
+
"max_memory": {},
|
| 13 |
+
"local_repo_path": null,
|
| 14 |
+
"template": "qwen2_5_vl",
|
| 15 |
+
"system": null,
|
| 16 |
+
"max_length": 8192,
|
| 17 |
+
"truncation_strategy": "delete",
|
| 18 |
+
"max_pixels": null,
|
| 19 |
+
"agent_template": null,
|
| 20 |
+
"norm_bbox": null,
|
| 21 |
+
"response_prefix": null,
|
| 22 |
+
"padding_side": "right",
|
| 23 |
+
"loss_scale": "default",
|
| 24 |
+
"sequence_parallel_size": 1,
|
| 25 |
+
"use_chat_template": true,
|
| 26 |
+
"template_backend": "swift",
|
| 27 |
+
"dataset": [
|
| 28 |
+
"/cpfs01/shared/llm_ddd/zhangyulong/sa_work/msdata/wei682/amazon-qwen-file/updated_merged_data.json"
|
| 29 |
+
],
|
| 30 |
+
"val_dataset": [],
|
| 31 |
+
"split_dataset_ratio": 0.01,
|
| 32 |
+
"data_seed": 42,
|
| 33 |
+
"dataset_num_proc": 2,
|
| 34 |
+
"dataset_shuffle": true,
|
| 35 |
+
"val_dataset_shuffle": false,
|
| 36 |
+
"streaming": false,
|
| 37 |
+
"interleave_prob": null,
|
| 38 |
+
"stopping_strategy": "first_exhausted",
|
| 39 |
+
"shuffle_buffer_size": 1000,
|
| 40 |
+
"enable_cache": false,
|
| 41 |
+
"download_mode": "reuse_dataset_if_exists",
|
| 42 |
+
"columns": {},
|
| 43 |
+
"strict": false,
|
| 44 |
+
"remove_unused_columns": true,
|
| 45 |
+
"model_name": [
|
| 46 |
+
null,
|
| 47 |
+
null
|
| 48 |
+
],
|
| 49 |
+
"model_author": [
|
| 50 |
+
null,
|
| 51 |
+
null
|
| 52 |
+
],
|
| 53 |
+
"custom_dataset_info": [],
|
| 54 |
+
"quant_method": null,
|
| 55 |
+
"quant_bits": null,
|
| 56 |
+
"hqq_axis": null,
|
| 57 |
+
"bnb_4bit_compute_dtype": "bfloat16",
|
| 58 |
+
"bnb_4bit_quant_type": "nf4",
|
| 59 |
+
"bnb_4bit_use_double_quant": true,
|
| 60 |
+
"bnb_4bit_quant_storage": null,
|
| 61 |
+
"max_new_tokens": 64,
|
| 62 |
+
"temperature": 0.0,
|
| 63 |
+
"top_k": null,
|
| 64 |
+
"top_p": null,
|
| 65 |
+
"repetition_penalty": null,
|
| 66 |
+
"num_beams": 1,
|
| 67 |
+
"stream": false,
|
| 68 |
+
"stop_words": [],
|
| 69 |
+
"logprobs": false,
|
| 70 |
+
"top_logprobs": null,
|
| 71 |
+
"ckpt_dir": null,
|
| 72 |
+
"load_dataset_config": null,
|
| 73 |
+
"lora_modules": [],
|
| 74 |
+
"tuner_backend": "peft",
|
| 75 |
+
"train_type": "custom",
|
| 76 |
+
"adapters": [],
|
| 77 |
+
"external_plugins": [
|
| 78 |
+
"examples/train/multimodal/lora_llm_full_vit/custom_plugin.py"
|
| 79 |
+
],
|
| 80 |
+
"seed": 42,
|
| 81 |
+
"model_kwargs": {},
|
| 82 |
+
"load_args": false,
|
| 83 |
+
"load_data_args": false,
|
| 84 |
+
"use_hf": false,
|
| 85 |
+
"hub_token": null,
|
| 86 |
+
"custom_register_path": [],
|
| 87 |
+
"ignore_args_error": false,
|
| 88 |
+
"use_swift_lora": false,
|
| 89 |
+
"output_dir": "/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113",
|
| 90 |
+
"overwrite_output_dir": false,
|
| 91 |
+
"do_train": false,
|
| 92 |
+
"do_eval": false,
|
| 93 |
+
"do_predict": false,
|
| 94 |
+
"eval_strategy": "steps",
|
| 95 |
+
"prediction_loss_only": false,
|
| 96 |
+
"per_device_train_batch_size": 4,
|
| 97 |
+
"per_device_eval_batch_size": 4,
|
| 98 |
+
"per_gpu_train_batch_size": null,
|
| 99 |
+
"per_gpu_eval_batch_size": null,
|
| 100 |
+
"gradient_accumulation_steps": 4,
|
| 101 |
+
"eval_accumulation_steps": null,
|
| 102 |
+
"eval_delay": 0,
|
| 103 |
+
"torch_empty_cache_steps": null,
|
| 104 |
+
"learning_rate": 0.001,
|
| 105 |
+
"weight_decay": 0.1,
|
| 106 |
+
"adam_beta1": 0.9,
|
| 107 |
+
"adam_beta2": 0.95,
|
| 108 |
+
"adam_epsilon": 1e-08,
|
| 109 |
+
"max_grad_norm": 1.0,
|
| 110 |
+
"num_train_epochs": 1.0,
|
| 111 |
+
"max_steps": -1,
|
| 112 |
+
"lr_scheduler_type": "cosine",
|
| 113 |
+
"lr_scheduler_kwargs": null,
|
| 114 |
+
"warmup_ratio": 0.05,
|
| 115 |
+
"warmup_steps": 0,
|
| 116 |
+
"log_level": "passive",
|
| 117 |
+
"log_level_replica": "warning",
|
| 118 |
+
"log_on_each_node": true,
|
| 119 |
+
"logging_dir": "/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113/runs",
|
| 120 |
+
"logging_strategy": "steps",
|
| 121 |
+
"logging_first_step": true,
|
| 122 |
+
"logging_steps": 5,
|
| 123 |
+
"logging_nan_inf_filter": true,
|
| 124 |
+
"save_strategy": "steps",
|
| 125 |
+
"save_steps": 100.0,
|
| 126 |
+
"save_total_limit": 2,
|
| 127 |
+
"save_safetensors": true,
|
| 128 |
+
"save_on_each_node": false,
|
| 129 |
+
"save_only_model": true,
|
| 130 |
+
"restore_callback_states_from_checkpoint": false,
|
| 131 |
+
"no_cuda": false,
|
| 132 |
+
"use_cpu": false,
|
| 133 |
+
"use_mps_device": false,
|
| 134 |
+
"jit_mode_eval": false,
|
| 135 |
+
"use_ipex": false,
|
| 136 |
+
"bf16": true,
|
| 137 |
+
"fp16": false,
|
| 138 |
+
"fp16_opt_level": "O1",
|
| 139 |
+
"half_precision_backend": "auto",
|
| 140 |
+
"bf16_full_eval": false,
|
| 141 |
+
"fp16_full_eval": false,
|
| 142 |
+
"tf32": null,
|
| 143 |
+
"local_rank": 0,
|
| 144 |
+
"ddp_backend": null,
|
| 145 |
+
"tpu_num_cores": null,
|
| 146 |
+
"tpu_metrics_debug": false,
|
| 147 |
+
"debug": null,
|
| 148 |
+
"dataloader_drop_last": false,
|
| 149 |
+
"eval_steps": 100.0,
|
| 150 |
+
"dataloader_num_workers": 2,
|
| 151 |
+
"dataloader_prefetch_factor": null,
|
| 152 |
+
"past_index": -1,
|
| 153 |
+
"run_name": "/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113",
|
| 154 |
+
"disable_tqdm": null,
|
| 155 |
+
"label_names": null,
|
| 156 |
+
"load_best_model_at_end": false,
|
| 157 |
+
"metric_for_best_model": "loss",
|
| 158 |
+
"greater_is_better": false,
|
| 159 |
+
"ignore_data_skip": false,
|
| 160 |
+
"fsdp": "",
|
| 161 |
+
"fsdp_min_num_params": 0,
|
| 162 |
+
"fsdp_config": null,
|
| 163 |
+
"tp_size": 0,
|
| 164 |
+
"fsdp_transformer_layer_cls_to_wrap": null,
|
| 165 |
+
"accelerator_config": {
|
| 166 |
+
"dispatch_batches": false
|
| 167 |
+
},
|
| 168 |
+
"deepspeed": {
|
| 169 |
+
"fp16": {
|
| 170 |
+
"enabled": "auto",
|
| 171 |
+
"loss_scale": 0,
|
| 172 |
+
"loss_scale_window": 1000,
|
| 173 |
+
"initial_scale_power": 16,
|
| 174 |
+
"hysteresis": 2,
|
| 175 |
+
"min_loss_scale": 1
|
| 176 |
+
},
|
| 177 |
+
"bf16": {
|
| 178 |
+
"enabled": "auto"
|
| 179 |
+
},
|
| 180 |
+
"zero_optimization": {
|
| 181 |
+
"stage": 3,
|
| 182 |
+
"offload_optimizer": {
|
| 183 |
+
"device": "none",
|
| 184 |
+
"pin_memory": true
|
| 185 |
+
},
|
| 186 |
+
"offload_param": {
|
| 187 |
+
"device": "none",
|
| 188 |
+
"pin_memory": true
|
| 189 |
+
},
|
| 190 |
+
"overlap_comm": false,
|
| 191 |
+
"contiguous_gradients": true,
|
| 192 |
+
"sub_group_size": 1000000000.0,
|
| 193 |
+
"reduce_bucket_size": "auto",
|
| 194 |
+
"zero_quantized_weights": false,
|
| 195 |
+
"zero_quantized_gradients": false,
|
| 196 |
+
"stage3_prefetch_bucket_size": "auto",
|
| 197 |
+
"stage3_param_persistence_threshold": "auto",
|
| 198 |
+
"stage3_max_live_parameters": 1000000000.0,
|
| 199 |
+
"stage3_max_reuse_distance": 1000000000.0,
|
| 200 |
+
"stage3_gather_16bit_weights_on_model_save": true
|
| 201 |
+
},
|
| 202 |
+
"gradient_accumulation_steps": "auto",
|
| 203 |
+
"gradient_clipping": "auto",
|
| 204 |
+
"steps_per_print": 2000,
|
| 205 |
+
"train_batch_size": "auto",
|
| 206 |
+
"train_micro_batch_size_per_gpu": "auto",
|
| 207 |
+
"wall_clock_breakdown": false
|
| 208 |
+
},
|
| 209 |
+
"label_smoothing_factor": 0.0,
|
| 210 |
+
"optim": "adamw_torch",
|
| 211 |
+
"optim_args": null,
|
| 212 |
+
"adafactor": false,
|
| 213 |
+
"group_by_length": false,
|
| 214 |
+
"length_column_name": "length",
|
| 215 |
+
"report_to": [
|
| 216 |
+
"tensorboard"
|
| 217 |
+
],
|
| 218 |
+
"ddp_find_unused_parameters": null,
|
| 219 |
+
"ddp_bucket_cap_mb": null,
|
| 220 |
+
"ddp_broadcast_buffers": null,
|
| 221 |
+
"dataloader_pin_memory": true,
|
| 222 |
+
"dataloader_persistent_workers": false,
|
| 223 |
+
"skip_memory_metrics": true,
|
| 224 |
+
"use_legacy_prediction_loop": false,
|
| 225 |
+
"push_to_hub": false,
|
| 226 |
+
"resume_from_checkpoint": null,
|
| 227 |
+
"hub_model_id": null,
|
| 228 |
+
"hub_strategy": "every_save",
|
| 229 |
+
"hub_private_repo": null,
|
| 230 |
+
"hub_always_push": false,
|
| 231 |
+
"gradient_checkpointing": true,
|
| 232 |
+
"gradient_checkpointing_kwargs": null,
|
| 233 |
+
"include_inputs_for_metrics": false,
|
| 234 |
+
"include_for_metrics": [],
|
| 235 |
+
"eval_do_concat_batches": true,
|
| 236 |
+
"fp16_backend": "auto",
|
| 237 |
+
"push_to_hub_model_id": null,
|
| 238 |
+
"push_to_hub_organization": null,
|
| 239 |
+
"push_to_hub_token": null,
|
| 240 |
+
"mp_parameters": "",
|
| 241 |
+
"auto_find_batch_size": false,
|
| 242 |
+
"full_determinism": false,
|
| 243 |
+
"torchdynamo": null,
|
| 244 |
+
"ray_scope": "last",
|
| 245 |
+
"ddp_timeout": 1800,
|
| 246 |
+
"torch_compile": false,
|
| 247 |
+
"torch_compile_backend": null,
|
| 248 |
+
"torch_compile_mode": null,
|
| 249 |
+
"include_tokens_per_second": false,
|
| 250 |
+
"include_num_input_tokens_seen": false,
|
| 251 |
+
"neftune_noise_alpha": null,
|
| 252 |
+
"optim_target_modules": null,
|
| 253 |
+
"batch_eval_metrics": false,
|
| 254 |
+
"eval_on_start": false,
|
| 255 |
+
"use_liger_kernel": false,
|
| 256 |
+
"eval_use_gather_object": false,
|
| 257 |
+
"average_tokens_across_devices": false,
|
| 258 |
+
"sortish_sampler": false,
|
| 259 |
+
"predict_with_generate": false,
|
| 260 |
+
"generation_max_length": null,
|
| 261 |
+
"generation_num_beams": null,
|
| 262 |
+
"generation_config": null,
|
| 263 |
+
"check_model": true,
|
| 264 |
+
"acc_strategy": "token",
|
| 265 |
+
"train_dataloader_shuffle": true,
|
| 266 |
+
"metric_warmup_step": 0,
|
| 267 |
+
"fsdp_num": 1,
|
| 268 |
+
"acc_steps": 1,
|
| 269 |
+
"eval_use_evalscope": false,
|
| 270 |
+
"eval_datasets": [],
|
| 271 |
+
"eval_limit": null,
|
| 272 |
+
"eval_datasets_args": null,
|
| 273 |
+
"eval_generation_config": null,
|
| 274 |
+
"freeze_parameters": [
|
| 275 |
+
"visual",
|
| 276 |
+
"visual.merger"
|
| 277 |
+
],
|
| 278 |
+
"freeze_parameters_ratio": 0.0,
|
| 279 |
+
"trainable_parameters": [],
|
| 280 |
+
"freeze_llm": false,
|
| 281 |
+
"freeze_vit": true,
|
| 282 |
+
"freeze_aligner": true,
|
| 283 |
+
"target_modules": [
|
| 284 |
+
"all-linear"
|
| 285 |
+
],
|
| 286 |
+
"target_regex": null,
|
| 287 |
+
"modules_to_save": [],
|
| 288 |
+
"lora_rank": 64,
|
| 289 |
+
"lora_alpha": 32,
|
| 290 |
+
"lora_dropout": 0.05,
|
| 291 |
+
"lora_bias": "none",
|
| 292 |
+
"lora_dtype": null,
|
| 293 |
+
"lorap_lr_ratio": null,
|
| 294 |
+
"use_rslora": false,
|
| 295 |
+
"use_dora": false,
|
| 296 |
+
"lora_ga_batch_size": 2,
|
| 297 |
+
"lora_ga_iters": 2,
|
| 298 |
+
"lora_ga_max_length": 1024,
|
| 299 |
+
"lora_ga_direction": "ArB2r",
|
| 300 |
+
"lora_ga_scale": "stable",
|
| 301 |
+
"lora_ga_stable_gamma": 16,
|
| 302 |
+
"init_weights": true,
|
| 303 |
+
"fourier_n_frequency": 2000,
|
| 304 |
+
"fourier_scaling": 300.0,
|
| 305 |
+
"boft_block_size": 4,
|
| 306 |
+
"boft_block_num": 0,
|
| 307 |
+
"boft_n_butterfly_factor": 1,
|
| 308 |
+
"boft_dropout": 0.0,
|
| 309 |
+
"vera_rank": 256,
|
| 310 |
+
"vera_projection_prng_key": 0,
|
| 311 |
+
"vera_dropout": 0.0,
|
| 312 |
+
"vera_d_initial": 0.1,
|
| 313 |
+
"adapter_act": "gelu",
|
| 314 |
+
"adapter_length": 128,
|
| 315 |
+
"use_galore": false,
|
| 316 |
+
"galore_target_modules": null,
|
| 317 |
+
"galore_rank": 128,
|
| 318 |
+
"galore_update_proj_gap": 50,
|
| 319 |
+
"galore_scale": 1.0,
|
| 320 |
+
"galore_proj_type": "std",
|
| 321 |
+
"galore_optim_per_parameter": false,
|
| 322 |
+
"galore_with_embedding": false,
|
| 323 |
+
"galore_quantization": false,
|
| 324 |
+
"galore_proj_quant": false,
|
| 325 |
+
"galore_proj_bits": 4,
|
| 326 |
+
"galore_proj_group_size": 256,
|
| 327 |
+
"galore_cos_threshold": 0.4,
|
| 328 |
+
"galore_gamma_proj": 2,
|
| 329 |
+
"galore_queue_size": 5,
|
| 330 |
+
"adalora_target_r": 8,
|
| 331 |
+
"adalora_init_r": 12,
|
| 332 |
+
"adalora_tinit": 0,
|
| 333 |
+
"adalora_tfinal": 0,
|
| 334 |
+
"adalora_deltaT": 1,
|
| 335 |
+
"adalora_beta1": 0.85,
|
| 336 |
+
"adalora_beta2": 0.85,
|
| 337 |
+
"adalora_orth_reg_weight": 0.5,
|
| 338 |
+
"llamapro_num_new_blocks": 4,
|
| 339 |
+
"llamapro_num_groups": null,
|
| 340 |
+
"lisa_activated_layers": 0,
|
| 341 |
+
"lisa_step_interval": 20,
|
| 342 |
+
"reft_layer_key": null,
|
| 343 |
+
"reft_layers": null,
|
| 344 |
+
"reft_rank": 4,
|
| 345 |
+
"reft_intervention_type": "LoreftIntervention",
|
| 346 |
+
"reft_args": null,
|
| 347 |
+
"swanlab_token": null,
|
| 348 |
+
"swanlab_project": null,
|
| 349 |
+
"swanlab_workspace": null,
|
| 350 |
+
"swanlab_exp_name": null,
|
| 351 |
+
"swanlab_mode": "cloud",
|
| 352 |
+
"add_version": true,
|
| 353 |
+
"resume_only_model": false,
|
| 354 |
+
"create_checkpoint_symlink": false,
|
| 355 |
+
"packing": false,
|
| 356 |
+
"lazy_tokenize": true,
|
| 357 |
+
"loss_type": null,
|
| 358 |
+
"optimizer": "custom",
|
| 359 |
+
"metric": null,
|
| 360 |
+
"zero_hpz_partition_size": null,
|
| 361 |
+
"rank": 0,
|
| 362 |
+
"global_world_size": 4,
|
| 363 |
+
"local_world_size": 4,
|
| 364 |
+
"model_suffix": "Qwen2.5-VL-3B-Instruct",
|
| 365 |
+
"model_info": "ModelInfo(model_type='qwen2_5_vl', model_dir='/cpfs01/shared/llm_ddd/tanghuanze/ckpts/hf_hub/Qwen/Qwen2.5-VL-3B-Instruct', torch_dtype=torch.bfloat16, max_model_len=128000, quant_method=None, quant_bits=None, rope_scaling={'type': 'default', 'mrope_section': [16, 24, 24], 'rope_type': 'default'}, config=None, task_type='causal_lm', num_labels=None)",
|
| 366 |
+
"model_meta": "ModelMeta(model_type='qwen2_5_vl', model_groups=[ModelGroup(models=[Model(ms_model_id='Qwen/Qwen2.5-VL-3B-Instruct', hf_model_id='Qwen/Qwen2.5-VL-3B-Instruct', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-7B-Instruct', hf_model_id='Qwen/Qwen2.5-VL-7B-Instruct', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-32B-Instruct', hf_model_id='Qwen/Qwen2.5-VL-32B-Instruct', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-72B-Instruct', hf_model_id='Qwen/Qwen2.5-VL-72B-Instruct', model_path=None, ms_revision=None, hf_revision=None)], ignore_patterns=None, requires=None, tags=[]), ModelGroup(models=[Model(ms_model_id='Qwen/Qwen2.5-VL-3B-Instruct-AWQ', hf_model_id='Qwen/Qwen2.5-VL-3B-Instruct-AWQ', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-7B-Instruct-AWQ', hf_model_id='Qwen/Qwen2.5-VL-7B-Instruct-AWQ', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-32B-Instruct-AWQ', hf_model_id='Qwen/Qwen2.5-VL-32B-Instruct-AWQ', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-72B-Instruct-AWQ', hf_model_id='Qwen/Qwen2.5-VL-72B-Instruct-AWQ', model_path=None, ms_revision=None, hf_revision=None)], ignore_patterns=None, requires=None, tags=[])], template='qwen2_5_vl', get_function=<function get_model_tokenizer_qwen2_5_vl at 0x7fe3082e5bd0>, model_arch='qwen2_vl', architectures=['Qwen2_5_VLForConditionalGeneration'], additional_saved_files=[], torch_dtype=None, is_multimodal=True, is_reward=False, task_type=None, ignore_patterns=None, requires=['transformers>=4.49', 'qwen_vl_utils>=0.0.6', 'decord'], tags=[])",
|
| 367 |
+
"model_dir": "/cpfs01/shared/llm_ddd/tanghuanze/ckpts/hf_hub/Qwen/Qwen2.5-VL-3B-Instruct",
|
| 368 |
+
"hub": "<class 'swift.hub.hub.MSHub'>",
|
| 369 |
+
"evaluation_strategy": "steps",
|
| 370 |
+
"training_args": "Seq2SeqTrainingArguments(output_dir='/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113', overwrite_output_dir=False, do_train=False, do_eval=True, do_predict=False, eval_strategy=<IntervalStrategy.STEPS: 'steps'>, prediction_loss_only=False, per_device_train_batch_size=4, per_device_eval_batch_size=4, per_gpu_train_batch_size=None, per_gpu_eval_batch_size=None, gradient_accumulation_steps=4, eval_accumulation_steps=None, eval_delay=0, torch_empty_cache_steps=None, learning_rate=0.001, weight_decay=0.1, adam_beta1=0.9, adam_beta2=0.95, adam_epsilon=1e-08, max_grad_norm=1.0, num_train_epochs=1.0, max_steps=-1, lr_scheduler_type=<SchedulerType.COSINE: 'cosine'>, lr_scheduler_kwargs=None, warmup_ratio=0.05, warmup_steps=0, log_level='passive', log_level_replica='warning', log_on_each_node=True, logging_dir='/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113/runs', logging_strategy=<IntervalStrategy.STEPS: 'steps'>, logging_first_step=True, logging_steps=5, logging_nan_inf_filter=True, save_strategy=<SaveStrategy.STEPS: 'steps'>, save_steps=100, save_total_limit=2, save_safetensors=True, save_on_each_node=False, save_only_model=True, restore_callback_states_from_checkpoint=False, no_cuda=False, use_cpu=False, use_mps_device=False, seed=42, data_seed=42, jit_mode_eval=False, use_ipex=False, bf16=True, fp16=False, fp16_opt_level='O1', half_precision_backend='auto', bf16_full_eval=False, fp16_full_eval=False, tf32=None, local_rank=0, ddp_backend=None, tpu_num_cores=None, tpu_metrics_debug=False, debug=[], dataloader_drop_last=False, eval_steps=100, dataloader_num_workers=2, dataloader_prefetch_factor=10, past_index=-1, run_name='/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113', disable_tqdm=False, remove_unused_columns=False, label_names=None, load_best_model_at_end=False, metric_for_best_model='loss', greater_is_better=False, ignore_data_skip=False, fsdp=[], fsdp_min_num_params=0, fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False}, tp_size=0, fsdp_transformer_layer_cls_to_wrap=None, accelerator_config=AcceleratorConfig(split_batches=False, dispatch_batches=False, even_batches=True, use_seedable_sampler=True, non_blocking=False, gradient_accumulation_kwargs=None, use_configured_state=False), deepspeed={'fp16': {'enabled': 'auto', 'loss_scale': 0, 'loss_scale_window': 1000, 'initial_scale_power': 16, 'hysteresis': 2, 'min_loss_scale': 1}, 'bf16': {'enabled': 'auto'}, 'zero_optimization': {'stage': 3, 'offload_optimizer': {'device': 'none', 'pin_memory': True}, 'offload_param': {'device': 'none', 'pin_memory': True}, 'overlap_comm': False, 'contiguous_gradients': True, 'sub_group_size': 1000000000.0, 'reduce_bucket_size': 'auto', 'zero_quantized_weights': False, 'zero_quantized_gradients': False, 'stage3_prefetch_bucket_size': 'auto', 'stage3_param_persistence_threshold': 'auto', 'stage3_max_live_parameters': 1000000000.0, 'stage3_max_reuse_distance': 1000000000.0, 'stage3_gather_16bit_weights_on_model_save': True}, 'gradient_accumulation_steps': 'auto', 'gradient_clipping': 'auto', 'steps_per_print': 2000, 'train_batch_size': 'auto', 'train_micro_batch_size_per_gpu': 'auto', 'wall_clock_breakdown': False}, label_smoothing_factor=0.0, optim=<OptimizerNames.ADAMW_TORCH: 'adamw_torch'>, optim_args=None, adafactor=False, group_by_length=False, length_column_name='length', report_to=['tensorboard'], ddp_find_unused_parameters=None, ddp_bucket_cap_mb=None, ddp_broadcast_buffers=None, dataloader_pin_memory=True, dataloader_persistent_workers=False, skip_memory_metrics=True, use_legacy_prediction_loop=False, push_to_hub=False, resume_from_checkpoint=None, hub_model_id=None, hub_strategy=<HubStrategy.EVERY_SAVE: 'every_save'>, hub_token=None, hub_private_repo=None, hub_always_push=False, gradient_checkpointing=True, gradient_checkpointing_kwargs=None, include_inputs_for_metrics=False, include_for_metrics=[], eval_do_concat_batches=True, fp16_backend='auto', push_to_hub_model_id=None, push_to_hub_organization=None, push_to_hub_token=None, mp_parameters='', auto_find_batch_size=False, full_determinism=False, torchdynamo=None, ray_scope='last', ddp_timeout=1800, torch_compile=False, torch_compile_backend=None, torch_compile_mode=None, include_tokens_per_second=None, include_num_input_tokens_seen=None, neftune_noise_alpha=None, optim_target_modules=None, batch_eval_metrics=False, eval_on_start=False, use_liger_kernel=False, eval_use_gather_object=False, average_tokens_across_devices=None, sortish_sampler=False, predict_with_generate=False, generation_max_length=None, generation_num_beams=None, generation_config=None, check_model=True, acc_strategy='token', train_dataloader_shuffle=True, metric_warmup_step=0, fsdp_num=1, acc_steps=1, eval_use_evalscope=False, eval_datasets=[], eval_limit=None, eval_datasets_args=None, eval_generation_config=None, train_type='custom', optimizer='custom', local_repo_path=None, galore_config=None)"
|
| 371 |
+
}
|
v1-20250508-175113/checkpoint-500/trainer_state.json
ADDED
|
@@ -0,0 +1,1089 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_global_step": 500,
|
| 3 |
+
"best_metric": 1.65794051,
|
| 4 |
+
"best_model_checkpoint": "/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113/checkpoint-500",
|
| 5 |
+
"epoch": 0.8336807002917882,
|
| 6 |
+
"eval_steps": 100,
|
| 7 |
+
"global_step": 500,
|
| 8 |
+
"is_hyper_param_search": false,
|
| 9 |
+
"is_local_process_zero": true,
|
| 10 |
+
"is_world_process_zero": true,
|
| 11 |
+
"log_history": [
|
| 12 |
+
{
|
| 13 |
+
"epoch": 0.0016673614005835765,
|
| 14 |
+
"grad_norm": 64.40978700124327,
|
| 15 |
+
"learning_rate": 3.3333333333333333e-06,
|
| 16 |
+
"loss": 2.5885090827941895,
|
| 17 |
+
"memory(GiB)": 21.94,
|
| 18 |
+
"step": 1,
|
| 19 |
+
"token_acc": 0.5925233644859813,
|
| 20 |
+
"train_speed(iter/s)": 0.01568
|
| 21 |
+
},
|
| 22 |
+
{
|
| 23 |
+
"epoch": 0.008336807002917883,
|
| 24 |
+
"grad_norm": 42.49983691696958,
|
| 25 |
+
"learning_rate": 1.6666666666666667e-05,
|
| 26 |
+
"loss": 2.8374526500701904,
|
| 27 |
+
"memory(GiB)": 39.79,
|
| 28 |
+
"step": 5,
|
| 29 |
+
"token_acc": 0.5379928315412187,
|
| 30 |
+
"train_speed(iter/s)": 0.028676
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"epoch": 0.016673614005835766,
|
| 34 |
+
"grad_norm": 9.119435129423483,
|
| 35 |
+
"learning_rate": 3.3333333333333335e-05,
|
| 36 |
+
"loss": 2.485092544555664,
|
| 37 |
+
"memory(GiB)": 39.79,
|
| 38 |
+
"step": 10,
|
| 39 |
+
"token_acc": 0.503155996393147,
|
| 40 |
+
"train_speed(iter/s)": 0.032119
|
| 41 |
+
},
|
| 42 |
+
{
|
| 43 |
+
"epoch": 0.02501042100875365,
|
| 44 |
+
"grad_norm": 12.267838186475673,
|
| 45 |
+
"learning_rate": 5e-05,
|
| 46 |
+
"loss": 2.338067626953125,
|
| 47 |
+
"memory(GiB)": 39.79,
|
| 48 |
+
"step": 15,
|
| 49 |
+
"token_acc": 0.5031282586027112,
|
| 50 |
+
"train_speed(iter/s)": 0.033444
|
| 51 |
+
},
|
| 52 |
+
{
|
| 53 |
+
"epoch": 0.03334722801167153,
|
| 54 |
+
"grad_norm": 9.95333802907172,
|
| 55 |
+
"learning_rate": 6.666666666666667e-05,
|
| 56 |
+
"loss": 1.9811298370361328,
|
| 57 |
+
"memory(GiB)": 39.79,
|
| 58 |
+
"step": 20,
|
| 59 |
+
"token_acc": 0.5953382971835546,
|
| 60 |
+
"train_speed(iter/s)": 0.034146
|
| 61 |
+
},
|
| 62 |
+
{
|
| 63 |
+
"epoch": 0.041684035014589414,
|
| 64 |
+
"grad_norm": 9.427104390244967,
|
| 65 |
+
"learning_rate": 8.333333333333334e-05,
|
| 66 |
+
"loss": 2.093521499633789,
|
| 67 |
+
"memory(GiB)": 39.79,
|
| 68 |
+
"step": 25,
|
| 69 |
+
"token_acc": 0.5699672667757774,
|
| 70 |
+
"train_speed(iter/s)": 0.034594
|
| 71 |
+
},
|
| 72 |
+
{
|
| 73 |
+
"epoch": 0.0500208420175073,
|
| 74 |
+
"grad_norm": 7.827634062020569,
|
| 75 |
+
"learning_rate": 0.0001,
|
| 76 |
+
"loss": 2.170622634887695,
|
| 77 |
+
"memory(GiB)": 39.79,
|
| 78 |
+
"step": 30,
|
| 79 |
+
"token_acc": 0.537514886859865,
|
| 80 |
+
"train_speed(iter/s)": 0.034914
|
| 81 |
+
},
|
| 82 |
+
{
|
| 83 |
+
"epoch": 0.05835764902042518,
|
| 84 |
+
"grad_norm": 8.599374011882336,
|
| 85 |
+
"learning_rate": 9.998094856697883e-05,
|
| 86 |
+
"loss": 2.2315696716308593,
|
| 87 |
+
"memory(GiB)": 39.79,
|
| 88 |
+
"step": 35,
|
| 89 |
+
"token_acc": 0.53125,
|
| 90 |
+
"train_speed(iter/s)": 0.035133
|
| 91 |
+
},
|
| 92 |
+
{
|
| 93 |
+
"epoch": 0.06669445602334306,
|
| 94 |
+
"grad_norm": 3.6478108577238,
|
| 95 |
+
"learning_rate": 9.992380878619938e-05,
|
| 96 |
+
"loss": 2.2251216888427736,
|
| 97 |
+
"memory(GiB)": 39.79,
|
| 98 |
+
"step": 40,
|
| 99 |
+
"token_acc": 0.5307889672867223,
|
| 100 |
+
"train_speed(iter/s)": 0.035348
|
| 101 |
+
},
|
| 102 |
+
{
|
| 103 |
+
"epoch": 0.07503126302626094,
|
| 104 |
+
"grad_norm": 6.070802568263254,
|
| 105 |
+
"learning_rate": 9.982862420144985e-05,
|
| 106 |
+
"loss": 2.146166229248047,
|
| 107 |
+
"memory(GiB)": 39.79,
|
| 108 |
+
"step": 45,
|
| 109 |
+
"token_acc": 0.5032708242477104,
|
| 110 |
+
"train_speed(iter/s)": 0.035522
|
| 111 |
+
},
|
| 112 |
+
{
|
| 113 |
+
"epoch": 0.08336807002917883,
|
| 114 |
+
"grad_norm": 19.987724090220276,
|
| 115 |
+
"learning_rate": 9.96954673488399e-05,
|
| 116 |
+
"loss": 2.151926040649414,
|
| 117 |
+
"memory(GiB)": 39.79,
|
| 118 |
+
"step": 50,
|
| 119 |
+
"token_acc": 0.5182186234817814,
|
| 120 |
+
"train_speed(iter/s)": 0.035623
|
| 121 |
+
},
|
| 122 |
+
{
|
| 123 |
+
"epoch": 0.0917048770320967,
|
| 124 |
+
"grad_norm": 4.035690857782974,
|
| 125 |
+
"learning_rate": 9.95244397015239e-05,
|
| 126 |
+
"loss": 2.1567785263061525,
|
| 127 |
+
"memory(GiB)": 39.79,
|
| 128 |
+
"step": 55,
|
| 129 |
+
"token_acc": 0.5920617420066152,
|
| 130 |
+
"train_speed(iter/s)": 0.035682
|
| 131 |
+
},
|
| 132 |
+
{
|
| 133 |
+
"epoch": 0.1000416840350146,
|
| 134 |
+
"grad_norm": 5.595457490039321,
|
| 135 |
+
"learning_rate": 9.931567159237251e-05,
|
| 136 |
+
"loss": 2.2211952209472656,
|
| 137 |
+
"memory(GiB)": 39.79,
|
| 138 |
+
"step": 60,
|
| 139 |
+
"token_acc": 0.5970588235294118,
|
| 140 |
+
"train_speed(iter/s)": 0.035766
|
| 141 |
+
},
|
| 142 |
+
{
|
| 143 |
+
"epoch": 0.10837849103793247,
|
| 144 |
+
"grad_norm": 4.251189137101062,
|
| 145 |
+
"learning_rate": 9.906932211465173e-05,
|
| 146 |
+
"loss": 2.1470125198364256,
|
| 147 |
+
"memory(GiB)": 39.79,
|
| 148 |
+
"step": 65,
|
| 149 |
+
"token_acc": 0.5370370370370371,
|
| 150 |
+
"train_speed(iter/s)": 0.035799
|
| 151 |
+
},
|
| 152 |
+
{
|
| 153 |
+
"epoch": 0.11671529804085036,
|
| 154 |
+
"grad_norm": 3.884144144655882,
|
| 155 |
+
"learning_rate": 9.87855790007845e-05,
|
| 156 |
+
"loss": 2.022116279602051,
|
| 157 |
+
"memory(GiB)": 39.79,
|
| 158 |
+
"step": 70,
|
| 159 |
+
"token_acc": 0.5287525803597759,
|
| 160 |
+
"train_speed(iter/s)": 0.035828
|
| 161 |
+
},
|
| 162 |
+
{
|
| 163 |
+
"epoch": 0.12505210504376824,
|
| 164 |
+
"grad_norm": 2.210974142567793,
|
| 165 |
+
"learning_rate": 9.8464658479288e-05,
|
| 166 |
+
"loss": 2.1557781219482424,
|
| 167 |
+
"memory(GiB)": 39.79,
|
| 168 |
+
"step": 75,
|
| 169 |
+
"token_acc": 0.5915966386554622,
|
| 170 |
+
"train_speed(iter/s)": 0.035857
|
| 171 |
+
},
|
| 172 |
+
{
|
| 173 |
+
"epoch": 0.13338891204668613,
|
| 174 |
+
"grad_norm": 1.7772094345896288,
|
| 175 |
+
"learning_rate": 9.810680510999504e-05,
|
| 176 |
+
"loss": 2.187470626831055,
|
| 177 |
+
"memory(GiB)": 58.65,
|
| 178 |
+
"step": 80,
|
| 179 |
+
"token_acc": 0.6507300989166274,
|
| 180 |
+
"train_speed(iter/s)": 0.035894
|
| 181 |
+
},
|
| 182 |
+
{
|
| 183 |
+
"epoch": 0.14172571904960402,
|
| 184 |
+
"grad_norm": 1.7654534509312796,
|
| 185 |
+
"learning_rate": 9.771229159768547e-05,
|
| 186 |
+
"loss": 2.1243057250976562,
|
| 187 |
+
"memory(GiB)": 58.65,
|
| 188 |
+
"step": 85,
|
| 189 |
+
"token_acc": 0.6008230452674898,
|
| 190 |
+
"train_speed(iter/s)": 0.035946
|
| 191 |
+
},
|
| 192 |
+
{
|
| 193 |
+
"epoch": 0.15006252605252188,
|
| 194 |
+
"grad_norm": 3.6082285595666694,
|
| 195 |
+
"learning_rate": 9.728141858426952e-05,
|
| 196 |
+
"loss": 1.9325916290283203,
|
| 197 |
+
"memory(GiB)": 58.65,
|
| 198 |
+
"step": 90,
|
| 199 |
+
"token_acc": 0.6351851851851852,
|
| 200 |
+
"train_speed(iter/s)": 0.035958
|
| 201 |
+
},
|
| 202 |
+
{
|
| 203 |
+
"epoch": 0.15839933305543977,
|
| 204 |
+
"grad_norm": 5.966171320647249,
|
| 205 |
+
"learning_rate": 9.681451441968144e-05,
|
| 206 |
+
"loss": 2.167759323120117,
|
| 207 |
+
"memory(GiB)": 58.65,
|
| 208 |
+
"step": 95,
|
| 209 |
+
"token_acc": 0.5823655913978495,
|
| 210 |
+
"train_speed(iter/s)": 0.035985
|
| 211 |
+
},
|
| 212 |
+
{
|
| 213 |
+
"epoch": 0.16673614005835766,
|
| 214 |
+
"grad_norm": 2.313327380575682,
|
| 215 |
+
"learning_rate": 9.631193491165797e-05,
|
| 216 |
+
"loss": 2.035287094116211,
|
| 217 |
+
"memory(GiB)": 72.16,
|
| 218 |
+
"step": 100,
|
| 219 |
+
"token_acc": 0.7820299500831946,
|
| 220 |
+
"train_speed(iter/s)": 0.036004
|
| 221 |
+
},
|
| 222 |
+
{
|
| 223 |
+
"epoch": 0.16673614005835766,
|
| 224 |
+
"eval_loss": 1.9559379816055298,
|
| 225 |
+
"eval_runtime": 53.5583,
|
| 226 |
+
"eval_samples_per_second": 7.226,
|
| 227 |
+
"eval_steps_per_second": 0.467,
|
| 228 |
+
"eval_token_acc": 0.5837100353414975,
|
| 229 |
+
"step": 100
|
| 230 |
+
},
|
| 231 |
+
{
|
| 232 |
+
"epoch": 0.17507294706127552,
|
| 233 |
+
"grad_norm": 2.868885045077332,
|
| 234 |
+
"learning_rate": 9.577406305459251e-05,
|
| 235 |
+
"loss": 2.2079593658447267,
|
| 236 |
+
"memory(GiB)": 72.16,
|
| 237 |
+
"step": 105,
|
| 238 |
+
"token_acc": 0.5882740447957839,
|
| 239 |
+
"train_speed(iter/s)": 0.035314
|
| 240 |
+
},
|
| 241 |
+
{
|
| 242 |
+
"epoch": 0.1834097540641934,
|
| 243 |
+
"grad_norm": 4.276219061520432,
|
| 244 |
+
"learning_rate": 9.520130873767141e-05,
|
| 245 |
+
"loss": 1.9634353637695312,
|
| 246 |
+
"memory(GiB)": 72.16,
|
| 247 |
+
"step": 110,
|
| 248 |
+
"token_acc": 0.5720048406615571,
|
| 249 |
+
"train_speed(iter/s)": 0.035369
|
| 250 |
+
},
|
| 251 |
+
{
|
| 252 |
+
"epoch": 0.1917465610671113,
|
| 253 |
+
"grad_norm": 3.763743172552543,
|
| 254 |
+
"learning_rate": 9.459410843251494e-05,
|
| 255 |
+
"loss": 2.151229667663574,
|
| 256 |
+
"memory(GiB)": 72.16,
|
| 257 |
+
"step": 115,
|
| 258 |
+
"token_acc": 0.5875370919881305,
|
| 259 |
+
"train_speed(iter/s)": 0.035408
|
| 260 |
+
},
|
| 261 |
+
{
|
| 262 |
+
"epoch": 0.2000833680700292,
|
| 263 |
+
"grad_norm": 5.046897215873855,
|
| 264 |
+
"learning_rate": 9.395292486056087e-05,
|
| 265 |
+
"loss": 2.0171051025390625,
|
| 266 |
+
"memory(GiB)": 72.16,
|
| 267 |
+
"step": 120,
|
| 268 |
+
"token_acc": 0.5762273901808785,
|
| 269 |
+
"train_speed(iter/s)": 0.035439
|
| 270 |
+
},
|
| 271 |
+
{
|
| 272 |
+
"epoch": 0.20842017507294705,
|
| 273 |
+
"grad_norm": 7.217092259193579,
|
| 274 |
+
"learning_rate": 9.327824664044418e-05,
|
| 275 |
+
"loss": 1.8802574157714844,
|
| 276 |
+
"memory(GiB)": 72.79,
|
| 277 |
+
"step": 125,
|
| 278 |
+
"token_acc": 0.6241312204614957,
|
| 279 |
+
"train_speed(iter/s)": 0.035472
|
| 280 |
+
},
|
| 281 |
+
{
|
| 282 |
+
"epoch": 0.21675698207586494,
|
| 283 |
+
"grad_norm": 5.36767881783302,
|
| 284 |
+
"learning_rate": 9.257058791564174e-05,
|
| 285 |
+
"loss": 2.060834503173828,
|
| 286 |
+
"memory(GiB)": 72.79,
|
| 287 |
+
"step": 130,
|
| 288 |
+
"token_acc": 0.5179227941176471,
|
| 289 |
+
"train_speed(iter/s)": 0.035519
|
| 290 |
+
},
|
| 291 |
+
{
|
| 292 |
+
"epoch": 0.22509378907878283,
|
| 293 |
+
"grad_norm": 3.999365659434696,
|
| 294 |
+
"learning_rate": 9.183048796266547e-05,
|
| 295 |
+
"loss": 2.0895004272460938,
|
| 296 |
+
"memory(GiB)": 72.79,
|
| 297 |
+
"step": 135,
|
| 298 |
+
"token_acc": 0.5767557489123679,
|
| 299 |
+
"train_speed(iter/s)": 0.035567
|
| 300 |
+
},
|
| 301 |
+
{
|
| 302 |
+
"epoch": 0.23343059608170072,
|
| 303 |
+
"grad_norm": 6.057232659457883,
|
| 304 |
+
"learning_rate": 9.105851078010266e-05,
|
| 305 |
+
"loss": 2.1480243682861326,
|
| 306 |
+
"memory(GiB)": 72.79,
|
| 307 |
+
"step": 140,
|
| 308 |
+
"token_acc": 0.5369127516778524,
|
| 309 |
+
"train_speed(iter/s)": 0.035608
|
| 310 |
+
},
|
| 311 |
+
{
|
| 312 |
+
"epoch": 0.24176740308461858,
|
| 313 |
+
"grad_norm": 4.297950099035578,
|
| 314 |
+
"learning_rate": 9.025524465881683e-05,
|
| 315 |
+
"loss": 2.2559787750244142,
|
| 316 |
+
"memory(GiB)": 72.79,
|
| 317 |
+
"step": 145,
|
| 318 |
+
"token_acc": 0.5798551224560193,
|
| 319 |
+
"train_speed(iter/s)": 0.035642
|
| 320 |
+
},
|
| 321 |
+
{
|
| 322 |
+
"epoch": 0.25010421008753647,
|
| 323 |
+
"grad_norm": 4.244668680625598,
|
| 324 |
+
"learning_rate": 8.942130173363627e-05,
|
| 325 |
+
"loss": 1.9533634185791016,
|
| 326 |
+
"memory(GiB)": 72.79,
|
| 327 |
+
"step": 150,
|
| 328 |
+
"token_acc": 0.5807607497243661,
|
| 329 |
+
"train_speed(iter/s)": 0.035664
|
| 330 |
+
},
|
| 331 |
+
{
|
| 332 |
+
"epoch": 0.25844101709045436,
|
| 333 |
+
"grad_norm": 3.1871883558604828,
|
| 334 |
+
"learning_rate": 8.855731751687233e-05,
|
| 335 |
+
"loss": 1.998438835144043,
|
| 336 |
+
"memory(GiB)": 72.79,
|
| 337 |
+
"step": 155,
|
| 338 |
+
"token_acc": 0.5887660069848661,
|
| 339 |
+
"train_speed(iter/s)": 0.03569
|
| 340 |
+
},
|
| 341 |
+
{
|
| 342 |
+
"epoch": 0.26677782409337225,
|
| 343 |
+
"grad_norm": 5.6200981254634,
|
| 344 |
+
"learning_rate": 8.766395041402244e-05,
|
| 345 |
+
"loss": 2.0300174713134767,
|
| 346 |
+
"memory(GiB)": 72.79,
|
| 347 |
+
"step": 160,
|
| 348 |
+
"token_acc": 0.5340692805481538,
|
| 349 |
+
"train_speed(iter/s)": 0.035723
|
| 350 |
+
},
|
| 351 |
+
{
|
| 352 |
+
"epoch": 0.27511463109629014,
|
| 353 |
+
"grad_norm": 2.927029018387031,
|
| 354 |
+
"learning_rate": 8.674188122202756e-05,
|
| 355 |
+
"loss": 2.0101190567016602,
|
| 356 |
+
"memory(GiB)": 72.79,
|
| 357 |
+
"step": 165,
|
| 358 |
+
"token_acc": 0.5347068145800317,
|
| 359 |
+
"train_speed(iter/s)": 0.035744
|
| 360 |
+
},
|
| 361 |
+
{
|
| 362 |
+
"epoch": 0.28345143809920803,
|
| 363 |
+
"grad_norm": 2.4715904965940227,
|
| 364 |
+
"learning_rate": 8.579181261046577e-05,
|
| 365 |
+
"loss": 1.9930459976196289,
|
| 366 |
+
"memory(GiB)": 72.79,
|
| 367 |
+
"step": 170,
|
| 368 |
+
"token_acc": 0.535579436537903,
|
| 369 |
+
"train_speed(iter/s)": 0.035775
|
| 370 |
+
},
|
| 371 |
+
{
|
| 372 |
+
"epoch": 0.29178824510212586,
|
| 373 |
+
"grad_norm": 2.172373476373027,
|
| 374 |
+
"learning_rate": 8.48144685860778e-05,
|
| 375 |
+
"loss": 2.1063697814941404,
|
| 376 |
+
"memory(GiB)": 72.79,
|
| 377 |
+
"step": 175,
|
| 378 |
+
"token_acc": 0.5444211785821119,
|
| 379 |
+
"train_speed(iter/s)": 0.035788
|
| 380 |
+
},
|
| 381 |
+
{
|
| 382 |
+
"epoch": 0.30012505210504375,
|
| 383 |
+
"grad_norm": 15.509072899121204,
|
| 384 |
+
"learning_rate": 8.381059394103244e-05,
|
| 385 |
+
"loss": 1.990152931213379,
|
| 386 |
+
"memory(GiB)": 72.79,
|
| 387 |
+
"step": 180,
|
| 388 |
+
"token_acc": 0.5902404018658055,
|
| 389 |
+
"train_speed(iter/s)": 0.035808
|
| 390 |
+
},
|
| 391 |
+
{
|
| 392 |
+
"epoch": 0.30846185910796164,
|
| 393 |
+
"grad_norm": 3.2069182023630565,
|
| 394 |
+
"learning_rate": 8.278095368535214e-05,
|
| 395 |
+
"loss": 2.0244321823120117,
|
| 396 |
+
"memory(GiB)": 72.79,
|
| 397 |
+
"step": 185,
|
| 398 |
+
"token_acc": 0.6046979865771812,
|
| 399 |
+
"train_speed(iter/s)": 0.035814
|
| 400 |
+
},
|
| 401 |
+
{
|
| 402 |
+
"epoch": 0.31679866611087953,
|
| 403 |
+
"grad_norm": 2.681173674568493,
|
| 404 |
+
"learning_rate": 8.17263324639316e-05,
|
| 405 |
+
"loss": 1.9969100952148438,
|
| 406 |
+
"memory(GiB)": 75.54,
|
| 407 |
+
"step": 190,
|
| 408 |
+
"token_acc": 0.7210337578830222,
|
| 409 |
+
"train_speed(iter/s)": 0.035815
|
| 410 |
+
},
|
| 411 |
+
{
|
| 412 |
+
"epoch": 0.3251354731137974,
|
| 413 |
+
"grad_norm": 3.8461606303421405,
|
| 414 |
+
"learning_rate": 8.064753395859333e-05,
|
| 415 |
+
"loss": 1.9372346878051758,
|
| 416 |
+
"memory(GiB)": 75.54,
|
| 417 |
+
"step": 195,
|
| 418 |
+
"token_acc": 0.5546875,
|
| 419 |
+
"train_speed(iter/s)": 0.035833
|
| 420 |
+
},
|
| 421 |
+
{
|
| 422 |
+
"epoch": 0.3334722801167153,
|
| 423 |
+
"grad_norm": 3.6666870417160666,
|
| 424 |
+
"learning_rate": 7.954538027563601e-05,
|
| 425 |
+
"loss": 2.1287214279174806,
|
| 426 |
+
"memory(GiB)": 75.54,
|
| 427 |
+
"step": 200,
|
| 428 |
+
"token_acc": 0.5563304721030042,
|
| 429 |
+
"train_speed(iter/s)": 0.035849
|
| 430 |
+
},
|
| 431 |
+
{
|
| 432 |
+
"epoch": 0.3334722801167153,
|
| 433 |
+
"eval_loss": 1.8122097253799438,
|
| 434 |
+
"eval_runtime": 50.701,
|
| 435 |
+
"eval_samples_per_second": 7.633,
|
| 436 |
+
"eval_steps_per_second": 0.493,
|
| 437 |
+
"eval_token_acc": 0.5999835620941892,
|
| 438 |
+
"step": 200
|
| 439 |
+
},
|
| 440 |
+
{
|
| 441 |
+
"epoch": 0.3418090871196332,
|
| 442 |
+
"grad_norm": 3.5964229855404164,
|
| 443 |
+
"learning_rate": 7.842071131934246e-05,
|
| 444 |
+
"loss": 1.8944969177246094,
|
| 445 |
+
"memory(GiB)": 75.54,
|
| 446 |
+
"step": 205,
|
| 447 |
+
"token_acc": 0.6359223300970874,
|
| 448 |
+
"train_speed(iter/s)": 0.035526
|
| 449 |
+
},
|
| 450 |
+
{
|
| 451 |
+
"epoch": 0.35014589412255104,
|
| 452 |
+
"grad_norm": 2.580207777434119,
|
| 453 |
+
"learning_rate": 7.727438415192433e-05,
|
| 454 |
+
"loss": 1.9824462890625,
|
| 455 |
+
"memory(GiB)": 75.54,
|
| 456 |
+
"step": 210,
|
| 457 |
+
"token_acc": 0.5771408351026185,
|
| 458 |
+
"train_speed(iter/s)": 0.035546
|
| 459 |
+
},
|
| 460 |
+
{
|
| 461 |
+
"epoch": 0.3584827011254689,
|
| 462 |
+
"grad_norm": 2.8278774675330887,
|
| 463 |
+
"learning_rate": 7.610727234039167e-05,
|
| 464 |
+
"loss": 2.0116973876953126,
|
| 465 |
+
"memory(GiB)": 75.54,
|
| 466 |
+
"step": 215,
|
| 467 |
+
"token_acc": 0.6102067751869775,
|
| 468 |
+
"train_speed(iter/s)": 0.035564
|
| 469 |
+
},
|
| 470 |
+
{
|
| 471 |
+
"epoch": 0.3668195081283868,
|
| 472 |
+
"grad_norm": 5.855526666772205,
|
| 473 |
+
"learning_rate": 7.492026529084468e-05,
|
| 474 |
+
"loss": 1.9584331512451172,
|
| 475 |
+
"memory(GiB)": 75.54,
|
| 476 |
+
"step": 220,
|
| 477 |
+
"token_acc": 0.5683212493028444,
|
| 478 |
+
"train_speed(iter/s)": 0.035583
|
| 479 |
+
},
|
| 480 |
+
{
|
| 481 |
+
"epoch": 0.3751563151313047,
|
| 482 |
+
"grad_norm": 4.011785835207539,
|
| 483 |
+
"learning_rate": 7.371426757069537e-05,
|
| 484 |
+
"loss": 2.0375545501708983,
|
| 485 |
+
"memory(GiB)": 75.54,
|
| 486 |
+
"step": 225,
|
| 487 |
+
"token_acc": 0.5575163398692811,
|
| 488 |
+
"train_speed(iter/s)": 0.035599
|
| 489 |
+
},
|
| 490 |
+
{
|
| 491 |
+
"epoch": 0.3834931221342226,
|
| 492 |
+
"grad_norm": 2.3496604645389905,
|
| 493 |
+
"learning_rate": 7.249019821933529e-05,
|
| 494 |
+
"loss": 1.9074588775634767,
|
| 495 |
+
"memory(GiB)": 75.54,
|
| 496 |
+
"step": 230,
|
| 497 |
+
"token_acc": 0.571943887775551,
|
| 498 |
+
"train_speed(iter/s)": 0.035618
|
| 499 |
+
},
|
| 500 |
+
{
|
| 501 |
+
"epoch": 0.3918299291371405,
|
| 502 |
+
"grad_norm": 2.4061403829778873,
|
| 503 |
+
"learning_rate": 7.124899004777489e-05,
|
| 504 |
+
"loss": 2.0661600112915037,
|
| 505 |
+
"memory(GiB)": 75.54,
|
| 506 |
+
"step": 235,
|
| 507 |
+
"token_acc": 0.5840575367096195,
|
| 508 |
+
"train_speed(iter/s)": 0.035636
|
| 509 |
+
},
|
| 510 |
+
{
|
| 511 |
+
"epoch": 0.4001667361400584,
|
| 512 |
+
"grad_norm": 2.4741806087579885,
|
| 513 |
+
"learning_rate": 6.9991588927788e-05,
|
| 514 |
+
"loss": 2.0608776092529295,
|
| 515 |
+
"memory(GiB)": 75.54,
|
| 516 |
+
"step": 240,
|
| 517 |
+
"token_acc": 0.6465237166991553,
|
| 518 |
+
"train_speed(iter/s)": 0.035654
|
| 519 |
+
},
|
| 520 |
+
{
|
| 521 |
+
"epoch": 0.40850354314297627,
|
| 522 |
+
"grad_norm": 3.4537545471932733,
|
| 523 |
+
"learning_rate": 6.871895307110332e-05,
|
| 524 |
+
"loss": 1.938018798828125,
|
| 525 |
+
"memory(GiB)": 75.54,
|
| 526 |
+
"step": 245,
|
| 527 |
+
"token_acc": 0.6371398078975453,
|
| 528 |
+
"train_speed(iter/s)": 0.035664
|
| 529 |
+
},
|
| 530 |
+
{
|
| 531 |
+
"epoch": 0.4168403501458941,
|
| 532 |
+
"grad_norm": 16.72496162125171,
|
| 533 |
+
"learning_rate": 6.743205229919224e-05,
|
| 534 |
+
"loss": 1.7865478515625,
|
| 535 |
+
"memory(GiB)": 75.54,
|
| 536 |
+
"step": 250,
|
| 537 |
+
"token_acc": 0.5756791720569211,
|
| 538 |
+
"train_speed(iter/s)": 0.035675
|
| 539 |
+
},
|
| 540 |
+
{
|
| 541 |
+
"epoch": 0.425177157148812,
|
| 542 |
+
"grad_norm": 3.1932083384706664,
|
| 543 |
+
"learning_rate": 6.613186730420917e-05,
|
| 544 |
+
"loss": 2.0705270767211914,
|
| 545 |
+
"memory(GiB)": 75.54,
|
| 546 |
+
"step": 255,
|
| 547 |
+
"token_acc": 0.6203485633537447,
|
| 548 |
+
"train_speed(iter/s)": 0.035687
|
| 549 |
+
},
|
| 550 |
+
{
|
| 551 |
+
"epoch": 0.4335139641517299,
|
| 552 |
+
"grad_norm": 7.316201458214335,
|
| 553 |
+
"learning_rate": 6.4819388901648e-05,
|
| 554 |
+
"loss": 1.9050493240356445,
|
| 555 |
+
"memory(GiB)": 75.54,
|
| 556 |
+
"step": 260,
|
| 557 |
+
"token_acc": 0.5929742388758782,
|
| 558 |
+
"train_speed(iter/s)": 0.035706
|
| 559 |
+
},
|
| 560 |
+
{
|
| 561 |
+
"epoch": 0.44185077115464777,
|
| 562 |
+
"grad_norm": 2.7720764825330853,
|
| 563 |
+
"learning_rate": 6.349561727528388e-05,
|
| 564 |
+
"loss": 1.888047981262207,
|
| 565 |
+
"memory(GiB)": 75.54,
|
| 566 |
+
"step": 265,
|
| 567 |
+
"token_acc": 0.5771358328211432,
|
| 568 |
+
"train_speed(iter/s)": 0.035715
|
| 569 |
+
},
|
| 570 |
+
{
|
| 571 |
+
"epoch": 0.45018757815756566,
|
| 572 |
+
"grad_norm": 4.065193512944849,
|
| 573 |
+
"learning_rate": 6.216156121497578e-05,
|
| 574 |
+
"loss": 1.8664142608642578,
|
| 575 |
+
"memory(GiB)": 75.54,
|
| 576 |
+
"step": 270,
|
| 577 |
+
"token_acc": 0.6696696696696697,
|
| 578 |
+
"train_speed(iter/s)": 0.035727
|
| 579 |
+
},
|
| 580 |
+
{
|
| 581 |
+
"epoch": 0.45852438516048355,
|
| 582 |
+
"grad_norm": 5.3789773277344315,
|
| 583 |
+
"learning_rate": 6.0818237347910903e-05,
|
| 584 |
+
"loss": 1.9860942840576172,
|
| 585 |
+
"memory(GiB)": 75.54,
|
| 586 |
+
"step": 275,
|
| 587 |
+
"token_acc": 0.5549738219895288,
|
| 588 |
+
"train_speed(iter/s)": 0.035743
|
| 589 |
+
},
|
| 590 |
+
{
|
| 591 |
+
"epoch": 0.46686119216340144,
|
| 592 |
+
"grad_norm": 4.12258515072594,
|
| 593 |
+
"learning_rate": 5.946666936387637e-05,
|
| 594 |
+
"loss": 1.9592571258544922,
|
| 595 |
+
"memory(GiB)": 75.54,
|
| 596 |
+
"step": 280,
|
| 597 |
+
"token_acc": 0.5547355473554736,
|
| 598 |
+
"train_speed(iter/s)": 0.035764
|
| 599 |
+
},
|
| 600 |
+
{
|
| 601 |
+
"epoch": 0.4751979991663193,
|
| 602 |
+
"grad_norm": 20.985894221796727,
|
| 603 |
+
"learning_rate": 5.810788723514908e-05,
|
| 604 |
+
"loss": 2.0858516693115234,
|
| 605 |
+
"memory(GiB)": 75.54,
|
| 606 |
+
"step": 285,
|
| 607 |
+
"token_acc": 0.5434947049924357,
|
| 608 |
+
"train_speed(iter/s)": 0.035776
|
| 609 |
+
},
|
| 610 |
+
{
|
| 611 |
+
"epoch": 0.48353480616923716,
|
| 612 |
+
"grad_norm": 8.177745830458013,
|
| 613 |
+
"learning_rate": 5.674292643159764e-05,
|
| 614 |
+
"loss": 1.9558685302734375,
|
| 615 |
+
"memory(GiB)": 75.54,
|
| 616 |
+
"step": 290,
|
| 617 |
+
"token_acc": 0.5351539802440441,
|
| 618 |
+
"train_speed(iter/s)": 0.03579
|
| 619 |
+
},
|
| 620 |
+
{
|
| 621 |
+
"epoch": 0.49187161317215505,
|
| 622 |
+
"grad_norm": 4.0835586516577225,
|
| 623 |
+
"learning_rate": 5.537282713159507e-05,
|
| 624 |
+
"loss": 2.0921154022216797,
|
| 625 |
+
"memory(GiB)": 75.54,
|
| 626 |
+
"step": 295,
|
| 627 |
+
"token_acc": 0.5673590504451038,
|
| 628 |
+
"train_speed(iter/s)": 0.035795
|
| 629 |
+
},
|
| 630 |
+
{
|
| 631 |
+
"epoch": 0.5002084201750729,
|
| 632 |
+
"grad_norm": 12.193993471525227,
|
| 633 |
+
"learning_rate": 5.399863342934324e-05,
|
| 634 |
+
"loss": 1.867361068725586,
|
| 635 |
+
"memory(GiB)": 75.54,
|
| 636 |
+
"step": 300,
|
| 637 |
+
"token_acc": 0.5305164319248826,
|
| 638 |
+
"train_speed(iter/s)": 0.035811
|
| 639 |
+
},
|
| 640 |
+
{
|
| 641 |
+
"epoch": 0.5002084201750729,
|
| 642 |
+
"eval_loss": 1.742539644241333,
|
| 643 |
+
"eval_runtime": 50.716,
|
| 644 |
+
"eval_samples_per_second": 7.631,
|
| 645 |
+
"eval_steps_per_second": 0.493,
|
| 646 |
+
"eval_token_acc": 0.612147612394181,
|
| 647 |
+
"step": 300
|
| 648 |
+
},
|
| 649 |
+
{
|
| 650 |
+
"epoch": 0.5085452271779908,
|
| 651 |
+
"grad_norm": 3.6488789490883553,
|
| 652 |
+
"learning_rate": 5.262139253921319e-05,
|
| 653 |
+
"loss": 1.9004257202148438,
|
| 654 |
+
"memory(GiB)": 75.54,
|
| 655 |
+
"step": 305,
|
| 656 |
+
"token_acc": 0.6261458748505381,
|
| 657 |
+
"train_speed(iter/s)": 0.035585
|
| 658 |
+
},
|
| 659 |
+
{
|
| 660 |
+
"epoch": 0.5168820341809087,
|
| 661 |
+
"grad_norm": 2.862433077996505,
|
| 662 |
+
"learning_rate": 5.1242153997707823e-05,
|
| 663 |
+
"loss": 1.9045713424682618,
|
| 664 |
+
"memory(GiB)": 75.54,
|
| 665 |
+
"step": 310,
|
| 666 |
+
"token_acc": 0.6437185929648241,
|
| 667 |
+
"train_speed(iter/s)": 0.035598
|
| 668 |
+
},
|
| 669 |
+
{
|
| 670 |
+
"epoch": 0.5252188411838266,
|
| 671 |
+
"grad_norm": 1.8452496357053012,
|
| 672 |
+
"learning_rate": 4.9861968863654875e-05,
|
| 673 |
+
"loss": 1.8142269134521485,
|
| 674 |
+
"memory(GiB)": 75.54,
|
| 675 |
+
"step": 315,
|
| 676 |
+
"token_acc": 0.6176683562635771,
|
| 677 |
+
"train_speed(iter/s)": 0.035606
|
| 678 |
+
},
|
| 679 |
+
{
|
| 680 |
+
"epoch": 0.5335556481867445,
|
| 681 |
+
"grad_norm": 5.430011832866582,
|
| 682 |
+
"learning_rate": 4.84818889172399e-05,
|
| 683 |
+
"loss": 1.8209394454956054,
|
| 684 |
+
"memory(GiB)": 75.54,
|
| 685 |
+
"step": 320,
|
| 686 |
+
"token_acc": 0.6066109698510715,
|
| 687 |
+
"train_speed(iter/s)": 0.035613
|
| 688 |
+
},
|
| 689 |
+
{
|
| 690 |
+
"epoch": 0.5418924551896623,
|
| 691 |
+
"grad_norm": 2.600769211702097,
|
| 692 |
+
"learning_rate": 4.7102965858489377e-05,
|
| 693 |
+
"loss": 1.9349700927734375,
|
| 694 |
+
"memory(GiB)": 75.54,
|
| 695 |
+
"step": 325,
|
| 696 |
+
"token_acc": 0.6356394129979036,
|
| 697 |
+
"train_speed(iter/s)": 0.035624
|
| 698 |
+
},
|
| 699 |
+
{
|
| 700 |
+
"epoch": 0.5502292621925803,
|
| 701 |
+
"grad_norm": 3.3927364584396877,
|
| 702 |
+
"learning_rate": 4.572625050581516e-05,
|
| 703 |
+
"loss": 1.8674903869628907,
|
| 704 |
+
"memory(GiB)": 75.54,
|
| 705 |
+
"step": 330,
|
| 706 |
+
"token_acc": 0.5809617271835132,
|
| 707 |
+
"train_speed(iter/s)": 0.035636
|
| 708 |
+
},
|
| 709 |
+
{
|
| 710 |
+
"epoch": 0.5585660691954981,
|
| 711 |
+
"grad_norm": 4.347679564527899,
|
| 712 |
+
"learning_rate": 4.435279199523043e-05,
|
| 713 |
+
"loss": 1.9532249450683594,
|
| 714 |
+
"memory(GiB)": 75.54,
|
| 715 |
+
"step": 335,
|
| 716 |
+
"token_acc": 0.6082898709854515,
|
| 717 |
+
"train_speed(iter/s)": 0.03565
|
| 718 |
+
},
|
| 719 |
+
{
|
| 720 |
+
"epoch": 0.5669028761984161,
|
| 721 |
+
"grad_norm": 14.797523073893712,
|
| 722 |
+
"learning_rate": 4.298363698084809e-05,
|
| 723 |
+
"loss": 1.8467117309570313,
|
| 724 |
+
"memory(GiB)": 75.54,
|
| 725 |
+
"step": 340,
|
| 726 |
+
"token_acc": 0.5447761194029851,
|
| 727 |
+
"train_speed(iter/s)": 0.035659
|
| 728 |
+
},
|
| 729 |
+
{
|
| 730 |
+
"epoch": 0.5752396832013339,
|
| 731 |
+
"grad_norm": 9.174677465010568,
|
| 732 |
+
"learning_rate": 4.16198288372702e-05,
|
| 733 |
+
"loss": 1.8867809295654296,
|
| 734 |
+
"memory(GiB)": 75.54,
|
| 735 |
+
"step": 345,
|
| 736 |
+
"token_acc": 0.649881716796215,
|
| 737 |
+
"train_speed(iter/s)": 0.035671
|
| 738 |
+
},
|
| 739 |
+
{
|
| 740 |
+
"epoch": 0.5835764902042517,
|
| 741 |
+
"grad_norm": 11.517095197655033,
|
| 742 |
+
"learning_rate": 4.026240686447682e-05,
|
| 743 |
+
"loss": 1.9671783447265625,
|
| 744 |
+
"memory(GiB)": 75.54,
|
| 745 |
+
"step": 350,
|
| 746 |
+
"token_acc": 0.5656722200697404,
|
| 747 |
+
"train_speed(iter/s)": 0.035684
|
| 748 |
+
},
|
| 749 |
+
{
|
| 750 |
+
"epoch": 0.5919132972071697,
|
| 751 |
+
"grad_norm": 2.0947774902715133,
|
| 752 |
+
"learning_rate": 3.8912405495819786e-05,
|
| 753 |
+
"loss": 2.0355813980102537,
|
| 754 |
+
"memory(GiB)": 75.54,
|
| 755 |
+
"step": 355,
|
| 756 |
+
"token_acc": 0.5646427096241222,
|
| 757 |
+
"train_speed(iter/s)": 0.035694
|
| 758 |
+
},
|
| 759 |
+
{
|
| 760 |
+
"epoch": 0.6002501042100875,
|
| 761 |
+
"grad_norm": 2.2672434495907967,
|
| 762 |
+
"learning_rate": 3.757085350972523e-05,
|
| 763 |
+
"loss": 2.0212512969970704,
|
| 764 |
+
"memory(GiB)": 75.54,
|
| 765 |
+
"step": 360,
|
| 766 |
+
"token_acc": 0.603363412633306,
|
| 767 |
+
"train_speed(iter/s)": 0.035702
|
| 768 |
+
},
|
| 769 |
+
{
|
| 770 |
+
"epoch": 0.6085869112130055,
|
| 771 |
+
"grad_norm": 3.3476628567015885,
|
| 772 |
+
"learning_rate": 3.623877324570548e-05,
|
| 773 |
+
"loss": 1.888478660583496,
|
| 774 |
+
"memory(GiB)": 75.54,
|
| 775 |
+
"step": 365,
|
| 776 |
+
"token_acc": 0.5527710843373494,
|
| 777 |
+
"train_speed(iter/s)": 0.035708
|
| 778 |
+
},
|
| 779 |
+
{
|
| 780 |
+
"epoch": 0.6169237182159233,
|
| 781 |
+
"grad_norm": 6.163682031673968,
|
| 782 |
+
"learning_rate": 3.491717982527765e-05,
|
| 783 |
+
"loss": 2.1264991760253906,
|
| 784 |
+
"memory(GiB)": 75.54,
|
| 785 |
+
"step": 370,
|
| 786 |
+
"token_acc": 0.5708566853482786,
|
| 787 |
+
"train_speed(iter/s)": 0.035714
|
| 788 |
+
},
|
| 789 |
+
{
|
| 790 |
+
"epoch": 0.6252605252188412,
|
| 791 |
+
"grad_norm": 1.6777593794217385,
|
| 792 |
+
"learning_rate": 3.3607080378383005e-05,
|
| 793 |
+
"loss": 1.8802024841308593,
|
| 794 |
+
"memory(GiB)": 75.54,
|
| 795 |
+
"step": 375,
|
| 796 |
+
"token_acc": 0.5547480620155039,
|
| 797 |
+
"train_speed(iter/s)": 0.035723
|
| 798 |
+
},
|
| 799 |
+
{
|
| 800 |
+
"epoch": 0.6335973322217591,
|
| 801 |
+
"grad_norm": 2.8440899647216833,
|
| 802 |
+
"learning_rate": 3.230947327589602e-05,
|
| 803 |
+
"loss": 1.8647388458251952,
|
| 804 |
+
"memory(GiB)": 75.54,
|
| 805 |
+
"step": 380,
|
| 806 |
+
"token_acc": 0.5930644019815995,
|
| 807 |
+
"train_speed(iter/s)": 0.035732
|
| 808 |
+
},
|
| 809 |
+
{
|
| 810 |
+
"epoch": 0.6419341392246769,
|
| 811 |
+
"grad_norm": 37.911669332255286,
|
| 812 |
+
"learning_rate": 3.1025347368808775e-05,
|
| 813 |
+
"loss": 1.8130638122558593,
|
| 814 |
+
"memory(GiB)": 75.54,
|
| 815 |
+
"step": 385,
|
| 816 |
+
"token_acc": 0.5654246100519931,
|
| 817 |
+
"train_speed(iter/s)": 0.035741
|
| 818 |
+
},
|
| 819 |
+
{
|
| 820 |
+
"epoch": 0.6502709462275948,
|
| 821 |
+
"grad_norm": 4.022081032945518,
|
| 822 |
+
"learning_rate": 2.9755681234669663e-05,
|
| 823 |
+
"loss": 2.064923095703125,
|
| 824 |
+
"memory(GiB)": 75.54,
|
| 825 |
+
"step": 390,
|
| 826 |
+
"token_acc": 0.6108317214700193,
|
| 827 |
+
"train_speed(iter/s)": 0.035754
|
| 828 |
+
},
|
| 829 |
+
{
|
| 830 |
+
"epoch": 0.6586077532305127,
|
| 831 |
+
"grad_norm": 2.307367388031427,
|
| 832 |
+
"learning_rate": 2.85014424318512e-05,
|
| 833 |
+
"loss": 1.9649499893188476,
|
| 834 |
+
"memory(GiB)": 75.54,
|
| 835 |
+
"step": 395,
|
| 836 |
+
"token_acc": 0.5570539419087137,
|
| 837 |
+
"train_speed(iter/s)": 0.03576
|
| 838 |
+
},
|
| 839 |
+
{
|
| 840 |
+
"epoch": 0.6669445602334306,
|
| 841 |
+
"grad_norm": 4.204547110729611,
|
| 842 |
+
"learning_rate": 2.7263586762215197e-05,
|
| 843 |
+
"loss": 1.92861328125,
|
| 844 |
+
"memory(GiB)": 75.54,
|
| 845 |
+
"step": 400,
|
| 846 |
+
"token_acc": 0.5911730545876888,
|
| 847 |
+
"train_speed(iter/s)": 0.035772
|
| 848 |
+
},
|
| 849 |
+
{
|
| 850 |
+
"epoch": 0.6669445602334306,
|
| 851 |
+
"eval_loss": 1.7799798250198364,
|
| 852 |
+
"eval_runtime": 50.5835,
|
| 853 |
+
"eval_samples_per_second": 7.651,
|
| 854 |
+
"eval_steps_per_second": 0.494,
|
| 855 |
+
"eval_token_acc": 0.6100106846387771,
|
| 856 |
+
"step": 400
|
| 857 |
+
},
|
| 858 |
+
{
|
| 859 |
+
"epoch": 0.6752813672363485,
|
| 860 |
+
"grad_norm": 2.659254046452427,
|
| 861 |
+
"learning_rate": 2.6043057542736836e-05,
|
| 862 |
+
"loss": 1.9875520706176757,
|
| 863 |
+
"memory(GiB)": 75.54,
|
| 864 |
+
"step": 405,
|
| 865 |
+
"token_acc": 0.5904554527309018,
|
| 866 |
+
"train_speed(iter/s)": 0.03561
|
| 867 |
+
},
|
| 868 |
+
{
|
| 869 |
+
"epoch": 0.6836181742392664,
|
| 870 |
+
"grad_norm": 14.662841293469047,
|
| 871 |
+
"learning_rate": 2.4840784886643132e-05,
|
| 872 |
+
"loss": 1.7080364227294922,
|
| 873 |
+
"memory(GiB)": 75.54,
|
| 874 |
+
"step": 410,
|
| 875 |
+
"token_acc": 0.6345278725824801,
|
| 876 |
+
"train_speed(iter/s)": 0.035622
|
| 877 |
+
},
|
| 878 |
+
{
|
| 879 |
+
"epoch": 0.6919549812421842,
|
| 880 |
+
"grad_norm": 2.0901368954978015,
|
| 881 |
+
"learning_rate": 2.365768499461328e-05,
|
| 882 |
+
"loss": 1.9299034118652343,
|
| 883 |
+
"memory(GiB)": 75.54,
|
| 884 |
+
"step": 415,
|
| 885 |
+
"token_acc": 0.5383259911894274,
|
| 886 |
+
"train_speed(iter/s)": 0.035634
|
| 887 |
+
},
|
| 888 |
+
{
|
| 889 |
+
"epoch": 0.7002917882451021,
|
| 890 |
+
"grad_norm": 2.658074481592308,
|
| 891 |
+
"learning_rate": 2.249465945658135e-05,
|
| 892 |
+
"loss": 2.000414276123047,
|
| 893 |
+
"memory(GiB)": 75.54,
|
| 894 |
+
"step": 420,
|
| 895 |
+
"token_acc": 0.5644047135310849,
|
| 896 |
+
"train_speed(iter/s)": 0.035642
|
| 897 |
+
},
|
| 898 |
+
{
|
| 899 |
+
"epoch": 0.70862859524802,
|
| 900 |
+
"grad_norm": 3.8658249706668504,
|
| 901 |
+
"learning_rate": 2.1352594564672908e-05,
|
| 902 |
+
"loss": 1.9364568710327148,
|
| 903 |
+
"memory(GiB)": 75.54,
|
| 904 |
+
"step": 425,
|
| 905 |
+
"token_acc": 0.6048387096774194,
|
| 906 |
+
"train_speed(iter/s)": 0.03565
|
| 907 |
+
},
|
| 908 |
+
{
|
| 909 |
+
"epoch": 0.7169654022509379,
|
| 910 |
+
"grad_norm": 17.005290399571575,
|
| 911 |
+
"learning_rate": 2.0232360637799685e-05,
|
| 912 |
+
"loss": 1.9262733459472656,
|
| 913 |
+
"memory(GiB)": 75.54,
|
| 914 |
+
"step": 430,
|
| 915 |
+
"token_acc": 0.5591078066914498,
|
| 916 |
+
"train_speed(iter/s)": 0.035661
|
| 917 |
+
},
|
| 918 |
+
{
|
| 919 |
+
"epoch": 0.7253022092538558,
|
| 920 |
+
"grad_norm": 2.7462439370469673,
|
| 921 |
+
"learning_rate": 1.9134811358426757e-05,
|
| 922 |
+
"loss": 2.0130874633789064,
|
| 923 |
+
"memory(GiB)": 75.54,
|
| 924 |
+
"step": 435,
|
| 925 |
+
"token_acc": 0.5676373018798379,
|
| 926 |
+
"train_speed(iter/s)": 0.035672
|
| 927 |
+
},
|
| 928 |
+
{
|
| 929 |
+
"epoch": 0.7336390162567736,
|
| 930 |
+
"grad_norm": 3.9297703842696294,
|
| 931 |
+
"learning_rate": 1.806078312201745e-05,
|
| 932 |
+
"loss": 1.886693000793457,
|
| 933 |
+
"memory(GiB)": 75.54,
|
| 934 |
+
"step": 440,
|
| 935 |
+
"token_acc": 0.5958948043617703,
|
| 936 |
+
"train_speed(iter/s)": 0.035681
|
| 937 |
+
},
|
| 938 |
+
{
|
| 939 |
+
"epoch": 0.7419758232596916,
|
| 940 |
+
"grad_norm": 4.746458587201285,
|
| 941 |
+
"learning_rate": 1.7011094399652107e-05,
|
| 942 |
+
"loss": 1.8821983337402344,
|
| 943 |
+
"memory(GiB)": 75.54,
|
| 944 |
+
"step": 445,
|
| 945 |
+
"token_acc": 0.5943814687037949,
|
| 946 |
+
"train_speed(iter/s)": 0.035688
|
| 947 |
+
},
|
| 948 |
+
{
|
| 949 |
+
"epoch": 0.7503126302626094,
|
| 950 |
+
"grad_norm": 4.130883468895127,
|
| 951 |
+
"learning_rate": 1.59865451143062e-05,
|
| 952 |
+
"loss": 1.847772216796875,
|
| 953 |
+
"memory(GiB)": 75.54,
|
| 954 |
+
"step": 450,
|
| 955 |
+
"token_acc": 0.6304583182966438,
|
| 956 |
+
"train_speed(iter/s)": 0.035697
|
| 957 |
+
},
|
| 958 |
+
{
|
| 959 |
+
"epoch": 0.7586494372655272,
|
| 960 |
+
"grad_norm": 9.216193465389415,
|
| 961 |
+
"learning_rate": 1.4987916031263232e-05,
|
| 962 |
+
"loss": 1.8329124450683594,
|
| 963 |
+
"memory(GiB)": 75.54,
|
| 964 |
+
"step": 455,
|
| 965 |
+
"token_acc": 0.635618801207417,
|
| 966 |
+
"train_speed(iter/s)": 0.035704
|
| 967 |
+
},
|
| 968 |
+
{
|
| 969 |
+
"epoch": 0.7669862442684452,
|
| 970 |
+
"grad_norm": 5.297584387705496,
|
| 971 |
+
"learning_rate": 1.401596816312673e-05,
|
| 972 |
+
"loss": 1.8964439392089845,
|
| 973 |
+
"memory(GiB)": 75.54,
|
| 974 |
+
"step": 460,
|
| 975 |
+
"token_acc": 0.5654450261780105,
|
| 976 |
+
"train_speed(iter/s)": 0.035715
|
| 977 |
+
},
|
| 978 |
+
{
|
| 979 |
+
"epoch": 0.775323051271363,
|
| 980 |
+
"grad_norm": 2.687671502108267,
|
| 981 |
+
"learning_rate": 1.307144218988507e-05,
|
| 982 |
+
"loss": 1.9502939224243163,
|
| 983 |
+
"memory(GiB)": 75.54,
|
| 984 |
+
"step": 465,
|
| 985 |
+
"token_acc": 0.5579991375592928,
|
| 986 |
+
"train_speed(iter/s)": 0.035727
|
| 987 |
+
},
|
| 988 |
+
{
|
| 989 |
+
"epoch": 0.783659858274281,
|
| 990 |
+
"grad_norm": 3.767228791810687,
|
| 991 |
+
"learning_rate": 1.2155057894470928e-05,
|
| 992 |
+
"loss": 1.71820068359375,
|
| 993 |
+
"memory(GiB)": 75.54,
|
| 994 |
+
"step": 470,
|
| 995 |
+
"token_acc": 0.5902320748181503,
|
| 996 |
+
"train_speed(iter/s)": 0.035733
|
| 997 |
+
},
|
| 998 |
+
{
|
| 999 |
+
"epoch": 0.7919966652771988,
|
| 1000 |
+
"grad_norm": 4.582125037883684,
|
| 1001 |
+
"learning_rate": 1.126751361424529e-05,
|
| 1002 |
+
"loss": 1.8622875213623047,
|
| 1003 |
+
"memory(GiB)": 75.54,
|
| 1004 |
+
"step": 475,
|
| 1005 |
+
"token_acc": 0.5744859420898027,
|
| 1006 |
+
"train_speed(iter/s)": 0.035739
|
| 1007 |
+
},
|
| 1008 |
+
{
|
| 1009 |
+
"epoch": 0.8003334722801168,
|
| 1010 |
+
"grad_norm": 3.5101737578003767,
|
| 1011 |
+
"learning_rate": 1.0409485708824507e-05,
|
| 1012 |
+
"loss": 1.8682287216186524,
|
| 1013 |
+
"memory(GiB)": 75.54,
|
| 1014 |
+
"step": 480,
|
| 1015 |
+
"token_acc": 0.5714951094550536,
|
| 1016 |
+
"train_speed(iter/s)": 0.035748
|
| 1017 |
+
},
|
| 1018 |
+
{
|
| 1019 |
+
"epoch": 0.8086702792830346,
|
| 1020 |
+
"grad_norm": 2.414709851450202,
|
| 1021 |
+
"learning_rate": 9.581628044655394e-06,
|
| 1022 |
+
"loss": 1.8879878997802735,
|
| 1023 |
+
"memory(GiB)": 75.54,
|
| 1024 |
+
"step": 485,
|
| 1025 |
+
"token_acc": 0.656461583750368,
|
| 1026 |
+
"train_speed(iter/s)": 0.035756
|
| 1027 |
+
},
|
| 1028 |
+
{
|
| 1029 |
+
"epoch": 0.8170070862859525,
|
| 1030 |
+
"grad_norm": 3.810296788453775,
|
| 1031 |
+
"learning_rate": 8.78457149673152e-06,
|
| 1032 |
+
"loss": 1.8605712890625,
|
| 1033 |
+
"memory(GiB)": 75.54,
|
| 1034 |
+
"step": 490,
|
| 1035 |
+
"token_acc": 0.6178915862986365,
|
| 1036 |
+
"train_speed(iter/s)": 0.035763
|
| 1037 |
+
},
|
| 1038 |
+
{
|
| 1039 |
+
"epoch": 0.8253438932888704,
|
| 1040 |
+
"grad_norm": 5.265615022155901,
|
| 1041 |
+
"learning_rate": 8.018923467830403e-06,
|
| 1042 |
+
"loss": 1.8603355407714843,
|
| 1043 |
+
"memory(GiB)": 75.54,
|
| 1044 |
+
"step": 495,
|
| 1045 |
+
"token_acc": 0.5819091288036682,
|
| 1046 |
+
"train_speed(iter/s)": 0.035772
|
| 1047 |
+
},
|
| 1048 |
+
{
|
| 1049 |
+
"epoch": 0.8336807002917882,
|
| 1050 |
+
"grad_norm": 2.5025771365898435,
|
| 1051 |
+
"learning_rate": 7.28526742563762e-06,
|
| 1052 |
+
"loss": 1.69268741607666,
|
| 1053 |
+
"memory(GiB)": 75.54,
|
| 1054 |
+
"step": 500,
|
| 1055 |
+
"token_acc": 0.6207253886010363,
|
| 1056 |
+
"train_speed(iter/s)": 0.035779
|
| 1057 |
+
},
|
| 1058 |
+
{
|
| 1059 |
+
"epoch": 0.8336807002917882,
|
| 1060 |
+
"eval_loss": 1.6579405069351196,
|
| 1061 |
+
"eval_runtime": 50.7583,
|
| 1062 |
+
"eval_samples_per_second": 7.624,
|
| 1063 |
+
"eval_steps_per_second": 0.493,
|
| 1064 |
+
"eval_token_acc": 0.6231610092874168,
|
| 1065 |
+
"step": 500
|
| 1066 |
+
}
|
| 1067 |
+
],
|
| 1068 |
+
"logging_steps": 5,
|
| 1069 |
+
"max_steps": 599,
|
| 1070 |
+
"num_input_tokens_seen": 0,
|
| 1071 |
+
"num_train_epochs": 1,
|
| 1072 |
+
"save_steps": 100,
|
| 1073 |
+
"stateful_callbacks": {
|
| 1074 |
+
"TrainerControl": {
|
| 1075 |
+
"args": {
|
| 1076 |
+
"should_epoch_stop": false,
|
| 1077 |
+
"should_evaluate": false,
|
| 1078 |
+
"should_log": false,
|
| 1079 |
+
"should_save": true,
|
| 1080 |
+
"should_training_stop": false
|
| 1081 |
+
},
|
| 1082 |
+
"attributes": {}
|
| 1083 |
+
}
|
| 1084 |
+
},
|
| 1085 |
+
"total_flos": 797044869431296.0,
|
| 1086 |
+
"train_batch_size": 4,
|
| 1087 |
+
"trial_name": null,
|
| 1088 |
+
"trial_params": null
|
| 1089 |
+
}
|
v1-20250508-175113/checkpoint-500/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3cdbe863329b245f3e43ae627c8c9276a956c353361b2e3d0be81b45ee1195fb
|
| 3 |
+
size 8248
|
v1-20250508-175113/checkpoint-500/vit.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:25e78a407dc5a06df5c837e8aba44d35f2779575557b66e58fcbc702f253a265
|
| 3 |
+
size 1337416944
|
v1-20250508-175113/checkpoint-599/README.md
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
base_model: /cpfs01/shared/llm_ddd/tanghuanze/ckpts/hf_hub/Qwen/Qwen2.5-VL-3B-Instruct
|
| 3 |
+
library_name: peft
|
| 4 |
+
---
|
| 5 |
+
|
| 6 |
+
# Model Card for Model ID
|
| 7 |
+
|
| 8 |
+
<!-- Provide a quick summary of what the model is/does. -->
|
| 9 |
+
|
| 10 |
+
|
| 11 |
+
|
| 12 |
+
## Model Details
|
| 13 |
+
|
| 14 |
+
### Model Description
|
| 15 |
+
|
| 16 |
+
<!-- Provide a longer summary of what this model is. -->
|
| 17 |
+
|
| 18 |
+
|
| 19 |
+
|
| 20 |
+
- **Developed by:** [More Information Needed]
|
| 21 |
+
- **Funded by [optional]:** [More Information Needed]
|
| 22 |
+
- **Shared by [optional]:** [More Information Needed]
|
| 23 |
+
- **Model type:** [More Information Needed]
|
| 24 |
+
- **Language(s) (NLP):** [More Information Needed]
|
| 25 |
+
- **License:** [More Information Needed]
|
| 26 |
+
- **Finetuned from model [optional]:** [More Information Needed]
|
| 27 |
+
|
| 28 |
+
### Model Sources [optional]
|
| 29 |
+
|
| 30 |
+
<!-- Provide the basic links for the model. -->
|
| 31 |
+
|
| 32 |
+
- **Repository:** [More Information Needed]
|
| 33 |
+
- **Paper [optional]:** [More Information Needed]
|
| 34 |
+
- **Demo [optional]:** [More Information Needed]
|
| 35 |
+
|
| 36 |
+
## Uses
|
| 37 |
+
|
| 38 |
+
<!-- Address questions around how the model is intended to be used, including the foreseeable users of the model and those affected by the model. -->
|
| 39 |
+
|
| 40 |
+
### Direct Use
|
| 41 |
+
|
| 42 |
+
<!-- This section is for the model use without fine-tuning or plugging into a larger ecosystem/app. -->
|
| 43 |
+
|
| 44 |
+
[More Information Needed]
|
| 45 |
+
|
| 46 |
+
### Downstream Use [optional]
|
| 47 |
+
|
| 48 |
+
<!-- This section is for the model use when fine-tuned for a task, or when plugged into a larger ecosystem/app -->
|
| 49 |
+
|
| 50 |
+
[More Information Needed]
|
| 51 |
+
|
| 52 |
+
### Out-of-Scope Use
|
| 53 |
+
|
| 54 |
+
<!-- This section addresses misuse, malicious use, and uses that the model will not work well for. -->
|
| 55 |
+
|
| 56 |
+
[More Information Needed]
|
| 57 |
+
|
| 58 |
+
## Bias, Risks, and Limitations
|
| 59 |
+
|
| 60 |
+
<!-- This section is meant to convey both technical and sociotechnical limitations. -->
|
| 61 |
+
|
| 62 |
+
[More Information Needed]
|
| 63 |
+
|
| 64 |
+
### Recommendations
|
| 65 |
+
|
| 66 |
+
<!-- This section is meant to convey recommendations with respect to the bias, risk, and technical limitations. -->
|
| 67 |
+
|
| 68 |
+
Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
|
| 69 |
+
|
| 70 |
+
## How to Get Started with the Model
|
| 71 |
+
|
| 72 |
+
Use the code below to get started with the model.
|
| 73 |
+
|
| 74 |
+
[More Information Needed]
|
| 75 |
+
|
| 76 |
+
## Training Details
|
| 77 |
+
|
| 78 |
+
### Training Data
|
| 79 |
+
|
| 80 |
+
<!-- This should link to a Dataset Card, perhaps with a short stub of information on what the training data is all about as well as documentation related to data pre-processing or additional filtering. -->
|
| 81 |
+
|
| 82 |
+
[More Information Needed]
|
| 83 |
+
|
| 84 |
+
### Training Procedure
|
| 85 |
+
|
| 86 |
+
<!-- This relates heavily to the Technical Specifications. Content here should link to that section when it is relevant to the training procedure. -->
|
| 87 |
+
|
| 88 |
+
#### Preprocessing [optional]
|
| 89 |
+
|
| 90 |
+
[More Information Needed]
|
| 91 |
+
|
| 92 |
+
|
| 93 |
+
#### Training Hyperparameters
|
| 94 |
+
|
| 95 |
+
- **Training regime:** [More Information Needed] <!--fp32, fp16 mixed precision, bf16 mixed precision, bf16 non-mixed precision, fp16 non-mixed precision, fp8 mixed precision -->
|
| 96 |
+
|
| 97 |
+
#### Speeds, Sizes, Times [optional]
|
| 98 |
+
|
| 99 |
+
<!-- This section provides information about throughput, start/end time, checkpoint size if relevant, etc. -->
|
| 100 |
+
|
| 101 |
+
[More Information Needed]
|
| 102 |
+
|
| 103 |
+
## Evaluation
|
| 104 |
+
|
| 105 |
+
<!-- This section describes the evaluation protocols and provides the results. -->
|
| 106 |
+
|
| 107 |
+
### Testing Data, Factors & Metrics
|
| 108 |
+
|
| 109 |
+
#### Testing Data
|
| 110 |
+
|
| 111 |
+
<!-- This should link to a Dataset Card if possible. -->
|
| 112 |
+
|
| 113 |
+
[More Information Needed]
|
| 114 |
+
|
| 115 |
+
#### Factors
|
| 116 |
+
|
| 117 |
+
<!-- These are the things the evaluation is disaggregating by, e.g., subpopulations or domains. -->
|
| 118 |
+
|
| 119 |
+
[More Information Needed]
|
| 120 |
+
|
| 121 |
+
#### Metrics
|
| 122 |
+
|
| 123 |
+
<!-- These are the evaluation metrics being used, ideally with a description of why. -->
|
| 124 |
+
|
| 125 |
+
[More Information Needed]
|
| 126 |
+
|
| 127 |
+
### Results
|
| 128 |
+
|
| 129 |
+
[More Information Needed]
|
| 130 |
+
|
| 131 |
+
#### Summary
|
| 132 |
+
|
| 133 |
+
|
| 134 |
+
|
| 135 |
+
## Model Examination [optional]
|
| 136 |
+
|
| 137 |
+
<!-- Relevant interpretability work for the model goes here -->
|
| 138 |
+
|
| 139 |
+
[More Information Needed]
|
| 140 |
+
|
| 141 |
+
## Environmental Impact
|
| 142 |
+
|
| 143 |
+
<!-- Total emissions (in grams of CO2eq) and additional considerations, such as electricity usage, go here. Edit the suggested text below accordingly -->
|
| 144 |
+
|
| 145 |
+
Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
|
| 146 |
+
|
| 147 |
+
- **Hardware Type:** [More Information Needed]
|
| 148 |
+
- **Hours used:** [More Information Needed]
|
| 149 |
+
- **Cloud Provider:** [More Information Needed]
|
| 150 |
+
- **Compute Region:** [More Information Needed]
|
| 151 |
+
- **Carbon Emitted:** [More Information Needed]
|
| 152 |
+
|
| 153 |
+
## Technical Specifications [optional]
|
| 154 |
+
|
| 155 |
+
### Model Architecture and Objective
|
| 156 |
+
|
| 157 |
+
[More Information Needed]
|
| 158 |
+
|
| 159 |
+
### Compute Infrastructure
|
| 160 |
+
|
| 161 |
+
[More Information Needed]
|
| 162 |
+
|
| 163 |
+
#### Hardware
|
| 164 |
+
|
| 165 |
+
[More Information Needed]
|
| 166 |
+
|
| 167 |
+
#### Software
|
| 168 |
+
|
| 169 |
+
[More Information Needed]
|
| 170 |
+
|
| 171 |
+
## Citation [optional]
|
| 172 |
+
|
| 173 |
+
<!-- If there is a paper or blog post introducing the model, the APA and Bibtex information for that should go in this section. -->
|
| 174 |
+
|
| 175 |
+
**BibTeX:**
|
| 176 |
+
|
| 177 |
+
[More Information Needed]
|
| 178 |
+
|
| 179 |
+
**APA:**
|
| 180 |
+
|
| 181 |
+
[More Information Needed]
|
| 182 |
+
|
| 183 |
+
## Glossary [optional]
|
| 184 |
+
|
| 185 |
+
<!-- If relevant, include terms and calculations in this section that can help readers understand the model or model card. -->
|
| 186 |
+
|
| 187 |
+
[More Information Needed]
|
| 188 |
+
|
| 189 |
+
## More Information [optional]
|
| 190 |
+
|
| 191 |
+
[More Information Needed]
|
| 192 |
+
|
| 193 |
+
## Model Card Authors [optional]
|
| 194 |
+
|
| 195 |
+
[More Information Needed]
|
| 196 |
+
|
| 197 |
+
## Model Card Contact
|
| 198 |
+
|
| 199 |
+
[More Information Needed]
|
| 200 |
+
### Framework versions
|
| 201 |
+
|
| 202 |
+
- PEFT 0.15.2
|
v1-20250508-175113/checkpoint-599/adapter_config.json
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"alpha_pattern": {},
|
| 3 |
+
"auto_mapping": null,
|
| 4 |
+
"base_model_name_or_path": "/cpfs01/shared/llm_ddd/tanghuanze/ckpts/hf_hub/Qwen/Qwen2.5-VL-3B-Instruct",
|
| 5 |
+
"bias": "none",
|
| 6 |
+
"corda_config": null,
|
| 7 |
+
"eva_config": null,
|
| 8 |
+
"exclude_modules": null,
|
| 9 |
+
"fan_in_fan_out": false,
|
| 10 |
+
"inference_mode": true,
|
| 11 |
+
"init_lora_weights": true,
|
| 12 |
+
"layer_replication": null,
|
| 13 |
+
"layers_pattern": null,
|
| 14 |
+
"layers_to_transform": null,
|
| 15 |
+
"loftq_config": {},
|
| 16 |
+
"lora_alpha": 32,
|
| 17 |
+
"lora_bias": false,
|
| 18 |
+
"lora_dropout": 0.0,
|
| 19 |
+
"megatron_config": null,
|
| 20 |
+
"megatron_core": "megatron.core",
|
| 21 |
+
"modules_to_save": null,
|
| 22 |
+
"peft_type": "LORA",
|
| 23 |
+
"r": 64,
|
| 24 |
+
"rank_pattern": {},
|
| 25 |
+
"revision": null,
|
| 26 |
+
"target_modules": "^(model).*\\.(up_proj|o_proj|v_proj|k_proj|down_proj|q_proj|gate_proj)$",
|
| 27 |
+
"task_type": "CAUSAL_LM",
|
| 28 |
+
"trainable_token_indices": null,
|
| 29 |
+
"use_dora": false,
|
| 30 |
+
"use_rslora": false
|
| 31 |
+
}
|
v1-20250508-175113/checkpoint-599/adapter_model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:0d47492a5a400afd17f1880df3f626dc626e61626b2ec23c66afc5b4db3beb87
|
| 3 |
+
size 239536776
|
v1-20250508-175113/checkpoint-599/additional_config.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"lora_dtype": null, "lorap_lr_ratio": null, "lorap_emb_lr": 1e-06}
|
v1-20250508-175113/checkpoint-599/args.json
ADDED
|
@@ -0,0 +1,371 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model": "/cpfs01/shared/llm_ddd/tanghuanze/ckpts/hf_hub/Qwen/Qwen2.5-VL-3B-Instruct",
|
| 3 |
+
"model_type": "qwen2_5_vl",
|
| 4 |
+
"model_revision": null,
|
| 5 |
+
"task_type": "causal_lm",
|
| 6 |
+
"torch_dtype": "bfloat16",
|
| 7 |
+
"attn_impl": null,
|
| 8 |
+
"num_labels": null,
|
| 9 |
+
"problem_type": null,
|
| 10 |
+
"rope_scaling": null,
|
| 11 |
+
"device_map": null,
|
| 12 |
+
"max_memory": {},
|
| 13 |
+
"local_repo_path": null,
|
| 14 |
+
"template": "qwen2_5_vl",
|
| 15 |
+
"system": null,
|
| 16 |
+
"max_length": 8192,
|
| 17 |
+
"truncation_strategy": "delete",
|
| 18 |
+
"max_pixels": null,
|
| 19 |
+
"agent_template": null,
|
| 20 |
+
"norm_bbox": null,
|
| 21 |
+
"response_prefix": null,
|
| 22 |
+
"padding_side": "right",
|
| 23 |
+
"loss_scale": "default",
|
| 24 |
+
"sequence_parallel_size": 1,
|
| 25 |
+
"use_chat_template": true,
|
| 26 |
+
"template_backend": "swift",
|
| 27 |
+
"dataset": [
|
| 28 |
+
"/cpfs01/shared/llm_ddd/zhangyulong/sa_work/msdata/wei682/amazon-qwen-file/updated_merged_data.json"
|
| 29 |
+
],
|
| 30 |
+
"val_dataset": [],
|
| 31 |
+
"split_dataset_ratio": 0.01,
|
| 32 |
+
"data_seed": 42,
|
| 33 |
+
"dataset_num_proc": 2,
|
| 34 |
+
"dataset_shuffle": true,
|
| 35 |
+
"val_dataset_shuffle": false,
|
| 36 |
+
"streaming": false,
|
| 37 |
+
"interleave_prob": null,
|
| 38 |
+
"stopping_strategy": "first_exhausted",
|
| 39 |
+
"shuffle_buffer_size": 1000,
|
| 40 |
+
"enable_cache": false,
|
| 41 |
+
"download_mode": "reuse_dataset_if_exists",
|
| 42 |
+
"columns": {},
|
| 43 |
+
"strict": false,
|
| 44 |
+
"remove_unused_columns": true,
|
| 45 |
+
"model_name": [
|
| 46 |
+
null,
|
| 47 |
+
null
|
| 48 |
+
],
|
| 49 |
+
"model_author": [
|
| 50 |
+
null,
|
| 51 |
+
null
|
| 52 |
+
],
|
| 53 |
+
"custom_dataset_info": [],
|
| 54 |
+
"quant_method": null,
|
| 55 |
+
"quant_bits": null,
|
| 56 |
+
"hqq_axis": null,
|
| 57 |
+
"bnb_4bit_compute_dtype": "bfloat16",
|
| 58 |
+
"bnb_4bit_quant_type": "nf4",
|
| 59 |
+
"bnb_4bit_use_double_quant": true,
|
| 60 |
+
"bnb_4bit_quant_storage": null,
|
| 61 |
+
"max_new_tokens": 64,
|
| 62 |
+
"temperature": 0.0,
|
| 63 |
+
"top_k": null,
|
| 64 |
+
"top_p": null,
|
| 65 |
+
"repetition_penalty": null,
|
| 66 |
+
"num_beams": 1,
|
| 67 |
+
"stream": false,
|
| 68 |
+
"stop_words": [],
|
| 69 |
+
"logprobs": false,
|
| 70 |
+
"top_logprobs": null,
|
| 71 |
+
"ckpt_dir": null,
|
| 72 |
+
"load_dataset_config": null,
|
| 73 |
+
"lora_modules": [],
|
| 74 |
+
"tuner_backend": "peft",
|
| 75 |
+
"train_type": "custom",
|
| 76 |
+
"adapters": [],
|
| 77 |
+
"external_plugins": [
|
| 78 |
+
"examples/train/multimodal/lora_llm_full_vit/custom_plugin.py"
|
| 79 |
+
],
|
| 80 |
+
"seed": 42,
|
| 81 |
+
"model_kwargs": {},
|
| 82 |
+
"load_args": false,
|
| 83 |
+
"load_data_args": false,
|
| 84 |
+
"use_hf": false,
|
| 85 |
+
"hub_token": null,
|
| 86 |
+
"custom_register_path": [],
|
| 87 |
+
"ignore_args_error": false,
|
| 88 |
+
"use_swift_lora": false,
|
| 89 |
+
"output_dir": "/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113",
|
| 90 |
+
"overwrite_output_dir": false,
|
| 91 |
+
"do_train": false,
|
| 92 |
+
"do_eval": false,
|
| 93 |
+
"do_predict": false,
|
| 94 |
+
"eval_strategy": "steps",
|
| 95 |
+
"prediction_loss_only": false,
|
| 96 |
+
"per_device_train_batch_size": 4,
|
| 97 |
+
"per_device_eval_batch_size": 4,
|
| 98 |
+
"per_gpu_train_batch_size": null,
|
| 99 |
+
"per_gpu_eval_batch_size": null,
|
| 100 |
+
"gradient_accumulation_steps": 4,
|
| 101 |
+
"eval_accumulation_steps": null,
|
| 102 |
+
"eval_delay": 0,
|
| 103 |
+
"torch_empty_cache_steps": null,
|
| 104 |
+
"learning_rate": 0.001,
|
| 105 |
+
"weight_decay": 0.1,
|
| 106 |
+
"adam_beta1": 0.9,
|
| 107 |
+
"adam_beta2": 0.95,
|
| 108 |
+
"adam_epsilon": 1e-08,
|
| 109 |
+
"max_grad_norm": 1.0,
|
| 110 |
+
"num_train_epochs": 1.0,
|
| 111 |
+
"max_steps": -1,
|
| 112 |
+
"lr_scheduler_type": "cosine",
|
| 113 |
+
"lr_scheduler_kwargs": null,
|
| 114 |
+
"warmup_ratio": 0.05,
|
| 115 |
+
"warmup_steps": 0,
|
| 116 |
+
"log_level": "passive",
|
| 117 |
+
"log_level_replica": "warning",
|
| 118 |
+
"log_on_each_node": true,
|
| 119 |
+
"logging_dir": "/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113/runs",
|
| 120 |
+
"logging_strategy": "steps",
|
| 121 |
+
"logging_first_step": true,
|
| 122 |
+
"logging_steps": 5,
|
| 123 |
+
"logging_nan_inf_filter": true,
|
| 124 |
+
"save_strategy": "steps",
|
| 125 |
+
"save_steps": 100.0,
|
| 126 |
+
"save_total_limit": 2,
|
| 127 |
+
"save_safetensors": true,
|
| 128 |
+
"save_on_each_node": false,
|
| 129 |
+
"save_only_model": true,
|
| 130 |
+
"restore_callback_states_from_checkpoint": false,
|
| 131 |
+
"no_cuda": false,
|
| 132 |
+
"use_cpu": false,
|
| 133 |
+
"use_mps_device": false,
|
| 134 |
+
"jit_mode_eval": false,
|
| 135 |
+
"use_ipex": false,
|
| 136 |
+
"bf16": true,
|
| 137 |
+
"fp16": false,
|
| 138 |
+
"fp16_opt_level": "O1",
|
| 139 |
+
"half_precision_backend": "auto",
|
| 140 |
+
"bf16_full_eval": false,
|
| 141 |
+
"fp16_full_eval": false,
|
| 142 |
+
"tf32": null,
|
| 143 |
+
"local_rank": 0,
|
| 144 |
+
"ddp_backend": null,
|
| 145 |
+
"tpu_num_cores": null,
|
| 146 |
+
"tpu_metrics_debug": false,
|
| 147 |
+
"debug": null,
|
| 148 |
+
"dataloader_drop_last": false,
|
| 149 |
+
"eval_steps": 100.0,
|
| 150 |
+
"dataloader_num_workers": 2,
|
| 151 |
+
"dataloader_prefetch_factor": null,
|
| 152 |
+
"past_index": -1,
|
| 153 |
+
"run_name": "/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113",
|
| 154 |
+
"disable_tqdm": null,
|
| 155 |
+
"label_names": null,
|
| 156 |
+
"load_best_model_at_end": false,
|
| 157 |
+
"metric_for_best_model": "loss",
|
| 158 |
+
"greater_is_better": false,
|
| 159 |
+
"ignore_data_skip": false,
|
| 160 |
+
"fsdp": "",
|
| 161 |
+
"fsdp_min_num_params": 0,
|
| 162 |
+
"fsdp_config": null,
|
| 163 |
+
"tp_size": 0,
|
| 164 |
+
"fsdp_transformer_layer_cls_to_wrap": null,
|
| 165 |
+
"accelerator_config": {
|
| 166 |
+
"dispatch_batches": false
|
| 167 |
+
},
|
| 168 |
+
"deepspeed": {
|
| 169 |
+
"fp16": {
|
| 170 |
+
"enabled": "auto",
|
| 171 |
+
"loss_scale": 0,
|
| 172 |
+
"loss_scale_window": 1000,
|
| 173 |
+
"initial_scale_power": 16,
|
| 174 |
+
"hysteresis": 2,
|
| 175 |
+
"min_loss_scale": 1
|
| 176 |
+
},
|
| 177 |
+
"bf16": {
|
| 178 |
+
"enabled": "auto"
|
| 179 |
+
},
|
| 180 |
+
"zero_optimization": {
|
| 181 |
+
"stage": 3,
|
| 182 |
+
"offload_optimizer": {
|
| 183 |
+
"device": "none",
|
| 184 |
+
"pin_memory": true
|
| 185 |
+
},
|
| 186 |
+
"offload_param": {
|
| 187 |
+
"device": "none",
|
| 188 |
+
"pin_memory": true
|
| 189 |
+
},
|
| 190 |
+
"overlap_comm": false,
|
| 191 |
+
"contiguous_gradients": true,
|
| 192 |
+
"sub_group_size": 1000000000.0,
|
| 193 |
+
"reduce_bucket_size": "auto",
|
| 194 |
+
"zero_quantized_weights": false,
|
| 195 |
+
"zero_quantized_gradients": false,
|
| 196 |
+
"stage3_prefetch_bucket_size": "auto",
|
| 197 |
+
"stage3_param_persistence_threshold": "auto",
|
| 198 |
+
"stage3_max_live_parameters": 1000000000.0,
|
| 199 |
+
"stage3_max_reuse_distance": 1000000000.0,
|
| 200 |
+
"stage3_gather_16bit_weights_on_model_save": true
|
| 201 |
+
},
|
| 202 |
+
"gradient_accumulation_steps": "auto",
|
| 203 |
+
"gradient_clipping": "auto",
|
| 204 |
+
"steps_per_print": 2000,
|
| 205 |
+
"train_batch_size": "auto",
|
| 206 |
+
"train_micro_batch_size_per_gpu": "auto",
|
| 207 |
+
"wall_clock_breakdown": false
|
| 208 |
+
},
|
| 209 |
+
"label_smoothing_factor": 0.0,
|
| 210 |
+
"optim": "adamw_torch",
|
| 211 |
+
"optim_args": null,
|
| 212 |
+
"adafactor": false,
|
| 213 |
+
"group_by_length": false,
|
| 214 |
+
"length_column_name": "length",
|
| 215 |
+
"report_to": [
|
| 216 |
+
"tensorboard"
|
| 217 |
+
],
|
| 218 |
+
"ddp_find_unused_parameters": null,
|
| 219 |
+
"ddp_bucket_cap_mb": null,
|
| 220 |
+
"ddp_broadcast_buffers": null,
|
| 221 |
+
"dataloader_pin_memory": true,
|
| 222 |
+
"dataloader_persistent_workers": false,
|
| 223 |
+
"skip_memory_metrics": true,
|
| 224 |
+
"use_legacy_prediction_loop": false,
|
| 225 |
+
"push_to_hub": false,
|
| 226 |
+
"resume_from_checkpoint": null,
|
| 227 |
+
"hub_model_id": null,
|
| 228 |
+
"hub_strategy": "every_save",
|
| 229 |
+
"hub_private_repo": null,
|
| 230 |
+
"hub_always_push": false,
|
| 231 |
+
"gradient_checkpointing": true,
|
| 232 |
+
"gradient_checkpointing_kwargs": null,
|
| 233 |
+
"include_inputs_for_metrics": false,
|
| 234 |
+
"include_for_metrics": [],
|
| 235 |
+
"eval_do_concat_batches": true,
|
| 236 |
+
"fp16_backend": "auto",
|
| 237 |
+
"push_to_hub_model_id": null,
|
| 238 |
+
"push_to_hub_organization": null,
|
| 239 |
+
"push_to_hub_token": null,
|
| 240 |
+
"mp_parameters": "",
|
| 241 |
+
"auto_find_batch_size": false,
|
| 242 |
+
"full_determinism": false,
|
| 243 |
+
"torchdynamo": null,
|
| 244 |
+
"ray_scope": "last",
|
| 245 |
+
"ddp_timeout": 1800,
|
| 246 |
+
"torch_compile": false,
|
| 247 |
+
"torch_compile_backend": null,
|
| 248 |
+
"torch_compile_mode": null,
|
| 249 |
+
"include_tokens_per_second": false,
|
| 250 |
+
"include_num_input_tokens_seen": false,
|
| 251 |
+
"neftune_noise_alpha": null,
|
| 252 |
+
"optim_target_modules": null,
|
| 253 |
+
"batch_eval_metrics": false,
|
| 254 |
+
"eval_on_start": false,
|
| 255 |
+
"use_liger_kernel": false,
|
| 256 |
+
"eval_use_gather_object": false,
|
| 257 |
+
"average_tokens_across_devices": false,
|
| 258 |
+
"sortish_sampler": false,
|
| 259 |
+
"predict_with_generate": false,
|
| 260 |
+
"generation_max_length": null,
|
| 261 |
+
"generation_num_beams": null,
|
| 262 |
+
"generation_config": null,
|
| 263 |
+
"check_model": true,
|
| 264 |
+
"acc_strategy": "token",
|
| 265 |
+
"train_dataloader_shuffle": true,
|
| 266 |
+
"metric_warmup_step": 0,
|
| 267 |
+
"fsdp_num": 1,
|
| 268 |
+
"acc_steps": 1,
|
| 269 |
+
"eval_use_evalscope": false,
|
| 270 |
+
"eval_datasets": [],
|
| 271 |
+
"eval_limit": null,
|
| 272 |
+
"eval_datasets_args": null,
|
| 273 |
+
"eval_generation_config": null,
|
| 274 |
+
"freeze_parameters": [
|
| 275 |
+
"visual",
|
| 276 |
+
"visual.merger"
|
| 277 |
+
],
|
| 278 |
+
"freeze_parameters_ratio": 0.0,
|
| 279 |
+
"trainable_parameters": [],
|
| 280 |
+
"freeze_llm": false,
|
| 281 |
+
"freeze_vit": true,
|
| 282 |
+
"freeze_aligner": true,
|
| 283 |
+
"target_modules": [
|
| 284 |
+
"all-linear"
|
| 285 |
+
],
|
| 286 |
+
"target_regex": null,
|
| 287 |
+
"modules_to_save": [],
|
| 288 |
+
"lora_rank": 64,
|
| 289 |
+
"lora_alpha": 32,
|
| 290 |
+
"lora_dropout": 0.05,
|
| 291 |
+
"lora_bias": "none",
|
| 292 |
+
"lora_dtype": null,
|
| 293 |
+
"lorap_lr_ratio": null,
|
| 294 |
+
"use_rslora": false,
|
| 295 |
+
"use_dora": false,
|
| 296 |
+
"lora_ga_batch_size": 2,
|
| 297 |
+
"lora_ga_iters": 2,
|
| 298 |
+
"lora_ga_max_length": 1024,
|
| 299 |
+
"lora_ga_direction": "ArB2r",
|
| 300 |
+
"lora_ga_scale": "stable",
|
| 301 |
+
"lora_ga_stable_gamma": 16,
|
| 302 |
+
"init_weights": true,
|
| 303 |
+
"fourier_n_frequency": 2000,
|
| 304 |
+
"fourier_scaling": 300.0,
|
| 305 |
+
"boft_block_size": 4,
|
| 306 |
+
"boft_block_num": 0,
|
| 307 |
+
"boft_n_butterfly_factor": 1,
|
| 308 |
+
"boft_dropout": 0.0,
|
| 309 |
+
"vera_rank": 256,
|
| 310 |
+
"vera_projection_prng_key": 0,
|
| 311 |
+
"vera_dropout": 0.0,
|
| 312 |
+
"vera_d_initial": 0.1,
|
| 313 |
+
"adapter_act": "gelu",
|
| 314 |
+
"adapter_length": 128,
|
| 315 |
+
"use_galore": false,
|
| 316 |
+
"galore_target_modules": null,
|
| 317 |
+
"galore_rank": 128,
|
| 318 |
+
"galore_update_proj_gap": 50,
|
| 319 |
+
"galore_scale": 1.0,
|
| 320 |
+
"galore_proj_type": "std",
|
| 321 |
+
"galore_optim_per_parameter": false,
|
| 322 |
+
"galore_with_embedding": false,
|
| 323 |
+
"galore_quantization": false,
|
| 324 |
+
"galore_proj_quant": false,
|
| 325 |
+
"galore_proj_bits": 4,
|
| 326 |
+
"galore_proj_group_size": 256,
|
| 327 |
+
"galore_cos_threshold": 0.4,
|
| 328 |
+
"galore_gamma_proj": 2,
|
| 329 |
+
"galore_queue_size": 5,
|
| 330 |
+
"adalora_target_r": 8,
|
| 331 |
+
"adalora_init_r": 12,
|
| 332 |
+
"adalora_tinit": 0,
|
| 333 |
+
"adalora_tfinal": 0,
|
| 334 |
+
"adalora_deltaT": 1,
|
| 335 |
+
"adalora_beta1": 0.85,
|
| 336 |
+
"adalora_beta2": 0.85,
|
| 337 |
+
"adalora_orth_reg_weight": 0.5,
|
| 338 |
+
"llamapro_num_new_blocks": 4,
|
| 339 |
+
"llamapro_num_groups": null,
|
| 340 |
+
"lisa_activated_layers": 0,
|
| 341 |
+
"lisa_step_interval": 20,
|
| 342 |
+
"reft_layer_key": null,
|
| 343 |
+
"reft_layers": null,
|
| 344 |
+
"reft_rank": 4,
|
| 345 |
+
"reft_intervention_type": "LoreftIntervention",
|
| 346 |
+
"reft_args": null,
|
| 347 |
+
"swanlab_token": null,
|
| 348 |
+
"swanlab_project": null,
|
| 349 |
+
"swanlab_workspace": null,
|
| 350 |
+
"swanlab_exp_name": null,
|
| 351 |
+
"swanlab_mode": "cloud",
|
| 352 |
+
"add_version": true,
|
| 353 |
+
"resume_only_model": false,
|
| 354 |
+
"create_checkpoint_symlink": false,
|
| 355 |
+
"packing": false,
|
| 356 |
+
"lazy_tokenize": true,
|
| 357 |
+
"loss_type": null,
|
| 358 |
+
"optimizer": "custom",
|
| 359 |
+
"metric": null,
|
| 360 |
+
"zero_hpz_partition_size": null,
|
| 361 |
+
"rank": 0,
|
| 362 |
+
"global_world_size": 4,
|
| 363 |
+
"local_world_size": 4,
|
| 364 |
+
"model_suffix": "Qwen2.5-VL-3B-Instruct",
|
| 365 |
+
"model_info": "ModelInfo(model_type='qwen2_5_vl', model_dir='/cpfs01/shared/llm_ddd/tanghuanze/ckpts/hf_hub/Qwen/Qwen2.5-VL-3B-Instruct', torch_dtype=torch.bfloat16, max_model_len=128000, quant_method=None, quant_bits=None, rope_scaling={'type': 'default', 'mrope_section': [16, 24, 24], 'rope_type': 'default'}, config=None, task_type='causal_lm', num_labels=None)",
|
| 366 |
+
"model_meta": "ModelMeta(model_type='qwen2_5_vl', model_groups=[ModelGroup(models=[Model(ms_model_id='Qwen/Qwen2.5-VL-3B-Instruct', hf_model_id='Qwen/Qwen2.5-VL-3B-Instruct', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-7B-Instruct', hf_model_id='Qwen/Qwen2.5-VL-7B-Instruct', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-32B-Instruct', hf_model_id='Qwen/Qwen2.5-VL-32B-Instruct', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-72B-Instruct', hf_model_id='Qwen/Qwen2.5-VL-72B-Instruct', model_path=None, ms_revision=None, hf_revision=None)], ignore_patterns=None, requires=None, tags=[]), ModelGroup(models=[Model(ms_model_id='Qwen/Qwen2.5-VL-3B-Instruct-AWQ', hf_model_id='Qwen/Qwen2.5-VL-3B-Instruct-AWQ', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-7B-Instruct-AWQ', hf_model_id='Qwen/Qwen2.5-VL-7B-Instruct-AWQ', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-32B-Instruct-AWQ', hf_model_id='Qwen/Qwen2.5-VL-32B-Instruct-AWQ', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen2.5-VL-72B-Instruct-AWQ', hf_model_id='Qwen/Qwen2.5-VL-72B-Instruct-AWQ', model_path=None, ms_revision=None, hf_revision=None)], ignore_patterns=None, requires=None, tags=[])], template='qwen2_5_vl', get_function=<function get_model_tokenizer_qwen2_5_vl at 0x7fe3082e5bd0>, model_arch='qwen2_vl', architectures=['Qwen2_5_VLForConditionalGeneration'], additional_saved_files=[], torch_dtype=None, is_multimodal=True, is_reward=False, task_type=None, ignore_patterns=None, requires=['transformers>=4.49', 'qwen_vl_utils>=0.0.6', 'decord'], tags=[])",
|
| 367 |
+
"model_dir": "/cpfs01/shared/llm_ddd/tanghuanze/ckpts/hf_hub/Qwen/Qwen2.5-VL-3B-Instruct",
|
| 368 |
+
"hub": "<class 'swift.hub.hub.MSHub'>",
|
| 369 |
+
"evaluation_strategy": "steps",
|
| 370 |
+
"training_args": "Seq2SeqTrainingArguments(output_dir='/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113', overwrite_output_dir=False, do_train=False, do_eval=True, do_predict=False, eval_strategy=<IntervalStrategy.STEPS: 'steps'>, prediction_loss_only=False, per_device_train_batch_size=4, per_device_eval_batch_size=4, per_gpu_train_batch_size=None, per_gpu_eval_batch_size=None, gradient_accumulation_steps=4, eval_accumulation_steps=None, eval_delay=0, torch_empty_cache_steps=None, learning_rate=0.001, weight_decay=0.1, adam_beta1=0.9, adam_beta2=0.95, adam_epsilon=1e-08, max_grad_norm=1.0, num_train_epochs=1.0, max_steps=-1, lr_scheduler_type=<SchedulerType.COSINE: 'cosine'>, lr_scheduler_kwargs=None, warmup_ratio=0.05, warmup_steps=0, log_level='passive', log_level_replica='warning', log_on_each_node=True, logging_dir='/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113/runs', logging_strategy=<IntervalStrategy.STEPS: 'steps'>, logging_first_step=True, logging_steps=5, logging_nan_inf_filter=True, save_strategy=<SaveStrategy.STEPS: 'steps'>, save_steps=100, save_total_limit=2, save_safetensors=True, save_on_each_node=False, save_only_model=True, restore_callback_states_from_checkpoint=False, no_cuda=False, use_cpu=False, use_mps_device=False, seed=42, data_seed=42, jit_mode_eval=False, use_ipex=False, bf16=True, fp16=False, fp16_opt_level='O1', half_precision_backend='auto', bf16_full_eval=False, fp16_full_eval=False, tf32=None, local_rank=0, ddp_backend=None, tpu_num_cores=None, tpu_metrics_debug=False, debug=[], dataloader_drop_last=False, eval_steps=100, dataloader_num_workers=2, dataloader_prefetch_factor=10, past_index=-1, run_name='/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113', disable_tqdm=False, remove_unused_columns=False, label_names=None, load_best_model_at_end=False, metric_for_best_model='loss', greater_is_better=False, ignore_data_skip=False, fsdp=[], fsdp_min_num_params=0, fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False}, tp_size=0, fsdp_transformer_layer_cls_to_wrap=None, accelerator_config=AcceleratorConfig(split_batches=False, dispatch_batches=False, even_batches=True, use_seedable_sampler=True, non_blocking=False, gradient_accumulation_kwargs=None, use_configured_state=False), deepspeed={'fp16': {'enabled': 'auto', 'loss_scale': 0, 'loss_scale_window': 1000, 'initial_scale_power': 16, 'hysteresis': 2, 'min_loss_scale': 1}, 'bf16': {'enabled': 'auto'}, 'zero_optimization': {'stage': 3, 'offload_optimizer': {'device': 'none', 'pin_memory': True}, 'offload_param': {'device': 'none', 'pin_memory': True}, 'overlap_comm': False, 'contiguous_gradients': True, 'sub_group_size': 1000000000.0, 'reduce_bucket_size': 'auto', 'zero_quantized_weights': False, 'zero_quantized_gradients': False, 'stage3_prefetch_bucket_size': 'auto', 'stage3_param_persistence_threshold': 'auto', 'stage3_max_live_parameters': 1000000000.0, 'stage3_max_reuse_distance': 1000000000.0, 'stage3_gather_16bit_weights_on_model_save': True}, 'gradient_accumulation_steps': 'auto', 'gradient_clipping': 'auto', 'steps_per_print': 2000, 'train_batch_size': 'auto', 'train_micro_batch_size_per_gpu': 'auto', 'wall_clock_breakdown': False}, label_smoothing_factor=0.0, optim=<OptimizerNames.ADAMW_TORCH: 'adamw_torch'>, optim_args=None, adafactor=False, group_by_length=False, length_column_name='length', report_to=['tensorboard'], ddp_find_unused_parameters=None, ddp_bucket_cap_mb=None, ddp_broadcast_buffers=None, dataloader_pin_memory=True, dataloader_persistent_workers=False, skip_memory_metrics=True, use_legacy_prediction_loop=False, push_to_hub=False, resume_from_checkpoint=None, hub_model_id=None, hub_strategy=<HubStrategy.EVERY_SAVE: 'every_save'>, hub_token=None, hub_private_repo=None, hub_always_push=False, gradient_checkpointing=True, gradient_checkpointing_kwargs=None, include_inputs_for_metrics=False, include_for_metrics=[], eval_do_concat_batches=True, fp16_backend='auto', push_to_hub_model_id=None, push_to_hub_organization=None, push_to_hub_token=None, mp_parameters='', auto_find_batch_size=False, full_determinism=False, torchdynamo=None, ray_scope='last', ddp_timeout=1800, torch_compile=False, torch_compile_backend=None, torch_compile_mode=None, include_tokens_per_second=None, include_num_input_tokens_seen=None, neftune_noise_alpha=None, optim_target_modules=None, batch_eval_metrics=False, eval_on_start=False, use_liger_kernel=False, eval_use_gather_object=False, average_tokens_across_devices=None, sortish_sampler=False, predict_with_generate=False, generation_max_length=None, generation_num_beams=None, generation_config=None, check_model=True, acc_strategy='token', train_dataloader_shuffle=True, metric_warmup_step=0, fsdp_num=1, acc_steps=1, eval_use_evalscope=False, eval_datasets=[], eval_limit=None, eval_datasets_args=None, eval_generation_config=None, train_type='custom', optimizer='custom', local_repo_path=None, galore_config=None)"
|
| 371 |
+
}
|
v1-20250508-175113/checkpoint-599/trainer_state.json
ADDED
|
@@ -0,0 +1,1288 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_global_step": 599,
|
| 3 |
+
"best_metric": 1.63505709,
|
| 4 |
+
"best_model_checkpoint": "/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113/checkpoint-599",
|
| 5 |
+
"epoch": 0.9987494789495623,
|
| 6 |
+
"eval_steps": 100,
|
| 7 |
+
"global_step": 599,
|
| 8 |
+
"is_hyper_param_search": false,
|
| 9 |
+
"is_local_process_zero": true,
|
| 10 |
+
"is_world_process_zero": true,
|
| 11 |
+
"log_history": [
|
| 12 |
+
{
|
| 13 |
+
"epoch": 0.0016673614005835765,
|
| 14 |
+
"grad_norm": 64.40978700124327,
|
| 15 |
+
"learning_rate": 3.3333333333333333e-06,
|
| 16 |
+
"loss": 2.5885090827941895,
|
| 17 |
+
"memory(GiB)": 21.94,
|
| 18 |
+
"step": 1,
|
| 19 |
+
"token_acc": 0.5925233644859813,
|
| 20 |
+
"train_speed(iter/s)": 0.01568
|
| 21 |
+
},
|
| 22 |
+
{
|
| 23 |
+
"epoch": 0.008336807002917883,
|
| 24 |
+
"grad_norm": 42.49983691696958,
|
| 25 |
+
"learning_rate": 1.6666666666666667e-05,
|
| 26 |
+
"loss": 2.8374526500701904,
|
| 27 |
+
"memory(GiB)": 39.79,
|
| 28 |
+
"step": 5,
|
| 29 |
+
"token_acc": 0.5379928315412187,
|
| 30 |
+
"train_speed(iter/s)": 0.028676
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"epoch": 0.016673614005835766,
|
| 34 |
+
"grad_norm": 9.119435129423483,
|
| 35 |
+
"learning_rate": 3.3333333333333335e-05,
|
| 36 |
+
"loss": 2.485092544555664,
|
| 37 |
+
"memory(GiB)": 39.79,
|
| 38 |
+
"step": 10,
|
| 39 |
+
"token_acc": 0.503155996393147,
|
| 40 |
+
"train_speed(iter/s)": 0.032119
|
| 41 |
+
},
|
| 42 |
+
{
|
| 43 |
+
"epoch": 0.02501042100875365,
|
| 44 |
+
"grad_norm": 12.267838186475673,
|
| 45 |
+
"learning_rate": 5e-05,
|
| 46 |
+
"loss": 2.338067626953125,
|
| 47 |
+
"memory(GiB)": 39.79,
|
| 48 |
+
"step": 15,
|
| 49 |
+
"token_acc": 0.5031282586027112,
|
| 50 |
+
"train_speed(iter/s)": 0.033444
|
| 51 |
+
},
|
| 52 |
+
{
|
| 53 |
+
"epoch": 0.03334722801167153,
|
| 54 |
+
"grad_norm": 9.95333802907172,
|
| 55 |
+
"learning_rate": 6.666666666666667e-05,
|
| 56 |
+
"loss": 1.9811298370361328,
|
| 57 |
+
"memory(GiB)": 39.79,
|
| 58 |
+
"step": 20,
|
| 59 |
+
"token_acc": 0.5953382971835546,
|
| 60 |
+
"train_speed(iter/s)": 0.034146
|
| 61 |
+
},
|
| 62 |
+
{
|
| 63 |
+
"epoch": 0.041684035014589414,
|
| 64 |
+
"grad_norm": 9.427104390244967,
|
| 65 |
+
"learning_rate": 8.333333333333334e-05,
|
| 66 |
+
"loss": 2.093521499633789,
|
| 67 |
+
"memory(GiB)": 39.79,
|
| 68 |
+
"step": 25,
|
| 69 |
+
"token_acc": 0.5699672667757774,
|
| 70 |
+
"train_speed(iter/s)": 0.034594
|
| 71 |
+
},
|
| 72 |
+
{
|
| 73 |
+
"epoch": 0.0500208420175073,
|
| 74 |
+
"grad_norm": 7.827634062020569,
|
| 75 |
+
"learning_rate": 0.0001,
|
| 76 |
+
"loss": 2.170622634887695,
|
| 77 |
+
"memory(GiB)": 39.79,
|
| 78 |
+
"step": 30,
|
| 79 |
+
"token_acc": 0.537514886859865,
|
| 80 |
+
"train_speed(iter/s)": 0.034914
|
| 81 |
+
},
|
| 82 |
+
{
|
| 83 |
+
"epoch": 0.05835764902042518,
|
| 84 |
+
"grad_norm": 8.599374011882336,
|
| 85 |
+
"learning_rate": 9.998094856697883e-05,
|
| 86 |
+
"loss": 2.2315696716308593,
|
| 87 |
+
"memory(GiB)": 39.79,
|
| 88 |
+
"step": 35,
|
| 89 |
+
"token_acc": 0.53125,
|
| 90 |
+
"train_speed(iter/s)": 0.035133
|
| 91 |
+
},
|
| 92 |
+
{
|
| 93 |
+
"epoch": 0.06669445602334306,
|
| 94 |
+
"grad_norm": 3.6478108577238,
|
| 95 |
+
"learning_rate": 9.992380878619938e-05,
|
| 96 |
+
"loss": 2.2251216888427736,
|
| 97 |
+
"memory(GiB)": 39.79,
|
| 98 |
+
"step": 40,
|
| 99 |
+
"token_acc": 0.5307889672867223,
|
| 100 |
+
"train_speed(iter/s)": 0.035348
|
| 101 |
+
},
|
| 102 |
+
{
|
| 103 |
+
"epoch": 0.07503126302626094,
|
| 104 |
+
"grad_norm": 6.070802568263254,
|
| 105 |
+
"learning_rate": 9.982862420144985e-05,
|
| 106 |
+
"loss": 2.146166229248047,
|
| 107 |
+
"memory(GiB)": 39.79,
|
| 108 |
+
"step": 45,
|
| 109 |
+
"token_acc": 0.5032708242477104,
|
| 110 |
+
"train_speed(iter/s)": 0.035522
|
| 111 |
+
},
|
| 112 |
+
{
|
| 113 |
+
"epoch": 0.08336807002917883,
|
| 114 |
+
"grad_norm": 19.987724090220276,
|
| 115 |
+
"learning_rate": 9.96954673488399e-05,
|
| 116 |
+
"loss": 2.151926040649414,
|
| 117 |
+
"memory(GiB)": 39.79,
|
| 118 |
+
"step": 50,
|
| 119 |
+
"token_acc": 0.5182186234817814,
|
| 120 |
+
"train_speed(iter/s)": 0.035623
|
| 121 |
+
},
|
| 122 |
+
{
|
| 123 |
+
"epoch": 0.0917048770320967,
|
| 124 |
+
"grad_norm": 4.035690857782974,
|
| 125 |
+
"learning_rate": 9.95244397015239e-05,
|
| 126 |
+
"loss": 2.1567785263061525,
|
| 127 |
+
"memory(GiB)": 39.79,
|
| 128 |
+
"step": 55,
|
| 129 |
+
"token_acc": 0.5920617420066152,
|
| 130 |
+
"train_speed(iter/s)": 0.035682
|
| 131 |
+
},
|
| 132 |
+
{
|
| 133 |
+
"epoch": 0.1000416840350146,
|
| 134 |
+
"grad_norm": 5.595457490039321,
|
| 135 |
+
"learning_rate": 9.931567159237251e-05,
|
| 136 |
+
"loss": 2.2211952209472656,
|
| 137 |
+
"memory(GiB)": 39.79,
|
| 138 |
+
"step": 60,
|
| 139 |
+
"token_acc": 0.5970588235294118,
|
| 140 |
+
"train_speed(iter/s)": 0.035766
|
| 141 |
+
},
|
| 142 |
+
{
|
| 143 |
+
"epoch": 0.10837849103793247,
|
| 144 |
+
"grad_norm": 4.251189137101062,
|
| 145 |
+
"learning_rate": 9.906932211465173e-05,
|
| 146 |
+
"loss": 2.1470125198364256,
|
| 147 |
+
"memory(GiB)": 39.79,
|
| 148 |
+
"step": 65,
|
| 149 |
+
"token_acc": 0.5370370370370371,
|
| 150 |
+
"train_speed(iter/s)": 0.035799
|
| 151 |
+
},
|
| 152 |
+
{
|
| 153 |
+
"epoch": 0.11671529804085036,
|
| 154 |
+
"grad_norm": 3.884144144655882,
|
| 155 |
+
"learning_rate": 9.87855790007845e-05,
|
| 156 |
+
"loss": 2.022116279602051,
|
| 157 |
+
"memory(GiB)": 39.79,
|
| 158 |
+
"step": 70,
|
| 159 |
+
"token_acc": 0.5287525803597759,
|
| 160 |
+
"train_speed(iter/s)": 0.035828
|
| 161 |
+
},
|
| 162 |
+
{
|
| 163 |
+
"epoch": 0.12505210504376824,
|
| 164 |
+
"grad_norm": 2.210974142567793,
|
| 165 |
+
"learning_rate": 9.8464658479288e-05,
|
| 166 |
+
"loss": 2.1557781219482424,
|
| 167 |
+
"memory(GiB)": 39.79,
|
| 168 |
+
"step": 75,
|
| 169 |
+
"token_acc": 0.5915966386554622,
|
| 170 |
+
"train_speed(iter/s)": 0.035857
|
| 171 |
+
},
|
| 172 |
+
{
|
| 173 |
+
"epoch": 0.13338891204668613,
|
| 174 |
+
"grad_norm": 1.7772094345896288,
|
| 175 |
+
"learning_rate": 9.810680510999504e-05,
|
| 176 |
+
"loss": 2.187470626831055,
|
| 177 |
+
"memory(GiB)": 58.65,
|
| 178 |
+
"step": 80,
|
| 179 |
+
"token_acc": 0.6507300989166274,
|
| 180 |
+
"train_speed(iter/s)": 0.035894
|
| 181 |
+
},
|
| 182 |
+
{
|
| 183 |
+
"epoch": 0.14172571904960402,
|
| 184 |
+
"grad_norm": 1.7654534509312796,
|
| 185 |
+
"learning_rate": 9.771229159768547e-05,
|
| 186 |
+
"loss": 2.1243057250976562,
|
| 187 |
+
"memory(GiB)": 58.65,
|
| 188 |
+
"step": 85,
|
| 189 |
+
"token_acc": 0.6008230452674898,
|
| 190 |
+
"train_speed(iter/s)": 0.035946
|
| 191 |
+
},
|
| 192 |
+
{
|
| 193 |
+
"epoch": 0.15006252605252188,
|
| 194 |
+
"grad_norm": 3.6082285595666694,
|
| 195 |
+
"learning_rate": 9.728141858426952e-05,
|
| 196 |
+
"loss": 1.9325916290283203,
|
| 197 |
+
"memory(GiB)": 58.65,
|
| 198 |
+
"step": 90,
|
| 199 |
+
"token_acc": 0.6351851851851852,
|
| 200 |
+
"train_speed(iter/s)": 0.035958
|
| 201 |
+
},
|
| 202 |
+
{
|
| 203 |
+
"epoch": 0.15839933305543977,
|
| 204 |
+
"grad_norm": 5.966171320647249,
|
| 205 |
+
"learning_rate": 9.681451441968144e-05,
|
| 206 |
+
"loss": 2.167759323120117,
|
| 207 |
+
"memory(GiB)": 58.65,
|
| 208 |
+
"step": 95,
|
| 209 |
+
"token_acc": 0.5823655913978495,
|
| 210 |
+
"train_speed(iter/s)": 0.035985
|
| 211 |
+
},
|
| 212 |
+
{
|
| 213 |
+
"epoch": 0.16673614005835766,
|
| 214 |
+
"grad_norm": 2.313327380575682,
|
| 215 |
+
"learning_rate": 9.631193491165797e-05,
|
| 216 |
+
"loss": 2.035287094116211,
|
| 217 |
+
"memory(GiB)": 72.16,
|
| 218 |
+
"step": 100,
|
| 219 |
+
"token_acc": 0.7820299500831946,
|
| 220 |
+
"train_speed(iter/s)": 0.036004
|
| 221 |
+
},
|
| 222 |
+
{
|
| 223 |
+
"epoch": 0.16673614005835766,
|
| 224 |
+
"eval_loss": 1.9559379816055298,
|
| 225 |
+
"eval_runtime": 53.5583,
|
| 226 |
+
"eval_samples_per_second": 7.226,
|
| 227 |
+
"eval_steps_per_second": 0.467,
|
| 228 |
+
"eval_token_acc": 0.5837100353414975,
|
| 229 |
+
"step": 100
|
| 230 |
+
},
|
| 231 |
+
{
|
| 232 |
+
"epoch": 0.17507294706127552,
|
| 233 |
+
"grad_norm": 2.868885045077332,
|
| 234 |
+
"learning_rate": 9.577406305459251e-05,
|
| 235 |
+
"loss": 2.2079593658447267,
|
| 236 |
+
"memory(GiB)": 72.16,
|
| 237 |
+
"step": 105,
|
| 238 |
+
"token_acc": 0.5882740447957839,
|
| 239 |
+
"train_speed(iter/s)": 0.035314
|
| 240 |
+
},
|
| 241 |
+
{
|
| 242 |
+
"epoch": 0.1834097540641934,
|
| 243 |
+
"grad_norm": 4.276219061520432,
|
| 244 |
+
"learning_rate": 9.520130873767141e-05,
|
| 245 |
+
"loss": 1.9634353637695312,
|
| 246 |
+
"memory(GiB)": 72.16,
|
| 247 |
+
"step": 110,
|
| 248 |
+
"token_acc": 0.5720048406615571,
|
| 249 |
+
"train_speed(iter/s)": 0.035369
|
| 250 |
+
},
|
| 251 |
+
{
|
| 252 |
+
"epoch": 0.1917465610671113,
|
| 253 |
+
"grad_norm": 3.763743172552543,
|
| 254 |
+
"learning_rate": 9.459410843251494e-05,
|
| 255 |
+
"loss": 2.151229667663574,
|
| 256 |
+
"memory(GiB)": 72.16,
|
| 257 |
+
"step": 115,
|
| 258 |
+
"token_acc": 0.5875370919881305,
|
| 259 |
+
"train_speed(iter/s)": 0.035408
|
| 260 |
+
},
|
| 261 |
+
{
|
| 262 |
+
"epoch": 0.2000833680700292,
|
| 263 |
+
"grad_norm": 5.046897215873855,
|
| 264 |
+
"learning_rate": 9.395292486056087e-05,
|
| 265 |
+
"loss": 2.0171051025390625,
|
| 266 |
+
"memory(GiB)": 72.16,
|
| 267 |
+
"step": 120,
|
| 268 |
+
"token_acc": 0.5762273901808785,
|
| 269 |
+
"train_speed(iter/s)": 0.035439
|
| 270 |
+
},
|
| 271 |
+
{
|
| 272 |
+
"epoch": 0.20842017507294705,
|
| 273 |
+
"grad_norm": 7.217092259193579,
|
| 274 |
+
"learning_rate": 9.327824664044418e-05,
|
| 275 |
+
"loss": 1.8802574157714844,
|
| 276 |
+
"memory(GiB)": 72.79,
|
| 277 |
+
"step": 125,
|
| 278 |
+
"token_acc": 0.6241312204614957,
|
| 279 |
+
"train_speed(iter/s)": 0.035472
|
| 280 |
+
},
|
| 281 |
+
{
|
| 282 |
+
"epoch": 0.21675698207586494,
|
| 283 |
+
"grad_norm": 5.36767881783302,
|
| 284 |
+
"learning_rate": 9.257058791564174e-05,
|
| 285 |
+
"loss": 2.060834503173828,
|
| 286 |
+
"memory(GiB)": 72.79,
|
| 287 |
+
"step": 130,
|
| 288 |
+
"token_acc": 0.5179227941176471,
|
| 289 |
+
"train_speed(iter/s)": 0.035519
|
| 290 |
+
},
|
| 291 |
+
{
|
| 292 |
+
"epoch": 0.22509378907878283,
|
| 293 |
+
"grad_norm": 3.999365659434696,
|
| 294 |
+
"learning_rate": 9.183048796266547e-05,
|
| 295 |
+
"loss": 2.0895004272460938,
|
| 296 |
+
"memory(GiB)": 72.79,
|
| 297 |
+
"step": 135,
|
| 298 |
+
"token_acc": 0.5767557489123679,
|
| 299 |
+
"train_speed(iter/s)": 0.035567
|
| 300 |
+
},
|
| 301 |
+
{
|
| 302 |
+
"epoch": 0.23343059608170072,
|
| 303 |
+
"grad_norm": 6.057232659457883,
|
| 304 |
+
"learning_rate": 9.105851078010266e-05,
|
| 305 |
+
"loss": 2.1480243682861326,
|
| 306 |
+
"memory(GiB)": 72.79,
|
| 307 |
+
"step": 140,
|
| 308 |
+
"token_acc": 0.5369127516778524,
|
| 309 |
+
"train_speed(iter/s)": 0.035608
|
| 310 |
+
},
|
| 311 |
+
{
|
| 312 |
+
"epoch": 0.24176740308461858,
|
| 313 |
+
"grad_norm": 4.297950099035578,
|
| 314 |
+
"learning_rate": 9.025524465881683e-05,
|
| 315 |
+
"loss": 2.2559787750244142,
|
| 316 |
+
"memory(GiB)": 72.79,
|
| 317 |
+
"step": 145,
|
| 318 |
+
"token_acc": 0.5798551224560193,
|
| 319 |
+
"train_speed(iter/s)": 0.035642
|
| 320 |
+
},
|
| 321 |
+
{
|
| 322 |
+
"epoch": 0.25010421008753647,
|
| 323 |
+
"grad_norm": 4.244668680625598,
|
| 324 |
+
"learning_rate": 8.942130173363627e-05,
|
| 325 |
+
"loss": 1.9533634185791016,
|
| 326 |
+
"memory(GiB)": 72.79,
|
| 327 |
+
"step": 150,
|
| 328 |
+
"token_acc": 0.5807607497243661,
|
| 329 |
+
"train_speed(iter/s)": 0.035664
|
| 330 |
+
},
|
| 331 |
+
{
|
| 332 |
+
"epoch": 0.25844101709045436,
|
| 333 |
+
"grad_norm": 3.1871883558604828,
|
| 334 |
+
"learning_rate": 8.855731751687233e-05,
|
| 335 |
+
"loss": 1.998438835144043,
|
| 336 |
+
"memory(GiB)": 72.79,
|
| 337 |
+
"step": 155,
|
| 338 |
+
"token_acc": 0.5887660069848661,
|
| 339 |
+
"train_speed(iter/s)": 0.03569
|
| 340 |
+
},
|
| 341 |
+
{
|
| 342 |
+
"epoch": 0.26677782409337225,
|
| 343 |
+
"grad_norm": 5.6200981254634,
|
| 344 |
+
"learning_rate": 8.766395041402244e-05,
|
| 345 |
+
"loss": 2.0300174713134767,
|
| 346 |
+
"memory(GiB)": 72.79,
|
| 347 |
+
"step": 160,
|
| 348 |
+
"token_acc": 0.5340692805481538,
|
| 349 |
+
"train_speed(iter/s)": 0.035723
|
| 350 |
+
},
|
| 351 |
+
{
|
| 352 |
+
"epoch": 0.27511463109629014,
|
| 353 |
+
"grad_norm": 2.927029018387031,
|
| 354 |
+
"learning_rate": 8.674188122202756e-05,
|
| 355 |
+
"loss": 2.0101190567016602,
|
| 356 |
+
"memory(GiB)": 72.79,
|
| 357 |
+
"step": 165,
|
| 358 |
+
"token_acc": 0.5347068145800317,
|
| 359 |
+
"train_speed(iter/s)": 0.035744
|
| 360 |
+
},
|
| 361 |
+
{
|
| 362 |
+
"epoch": 0.28345143809920803,
|
| 363 |
+
"grad_norm": 2.4715904965940227,
|
| 364 |
+
"learning_rate": 8.579181261046577e-05,
|
| 365 |
+
"loss": 1.9930459976196289,
|
| 366 |
+
"memory(GiB)": 72.79,
|
| 367 |
+
"step": 170,
|
| 368 |
+
"token_acc": 0.535579436537903,
|
| 369 |
+
"train_speed(iter/s)": 0.035775
|
| 370 |
+
},
|
| 371 |
+
{
|
| 372 |
+
"epoch": 0.29178824510212586,
|
| 373 |
+
"grad_norm": 2.172373476373027,
|
| 374 |
+
"learning_rate": 8.48144685860778e-05,
|
| 375 |
+
"loss": 2.1063697814941404,
|
| 376 |
+
"memory(GiB)": 72.79,
|
| 377 |
+
"step": 175,
|
| 378 |
+
"token_acc": 0.5444211785821119,
|
| 379 |
+
"train_speed(iter/s)": 0.035788
|
| 380 |
+
},
|
| 381 |
+
{
|
| 382 |
+
"epoch": 0.30012505210504375,
|
| 383 |
+
"grad_norm": 15.509072899121204,
|
| 384 |
+
"learning_rate": 8.381059394103244e-05,
|
| 385 |
+
"loss": 1.990152931213379,
|
| 386 |
+
"memory(GiB)": 72.79,
|
| 387 |
+
"step": 180,
|
| 388 |
+
"token_acc": 0.5902404018658055,
|
| 389 |
+
"train_speed(iter/s)": 0.035808
|
| 390 |
+
},
|
| 391 |
+
{
|
| 392 |
+
"epoch": 0.30846185910796164,
|
| 393 |
+
"grad_norm": 3.2069182023630565,
|
| 394 |
+
"learning_rate": 8.278095368535214e-05,
|
| 395 |
+
"loss": 2.0244321823120117,
|
| 396 |
+
"memory(GiB)": 72.79,
|
| 397 |
+
"step": 185,
|
| 398 |
+
"token_acc": 0.6046979865771812,
|
| 399 |
+
"train_speed(iter/s)": 0.035814
|
| 400 |
+
},
|
| 401 |
+
{
|
| 402 |
+
"epoch": 0.31679866611087953,
|
| 403 |
+
"grad_norm": 2.681173674568493,
|
| 404 |
+
"learning_rate": 8.17263324639316e-05,
|
| 405 |
+
"loss": 1.9969100952148438,
|
| 406 |
+
"memory(GiB)": 75.54,
|
| 407 |
+
"step": 190,
|
| 408 |
+
"token_acc": 0.7210337578830222,
|
| 409 |
+
"train_speed(iter/s)": 0.035815
|
| 410 |
+
},
|
| 411 |
+
{
|
| 412 |
+
"epoch": 0.3251354731137974,
|
| 413 |
+
"grad_norm": 3.8461606303421405,
|
| 414 |
+
"learning_rate": 8.064753395859333e-05,
|
| 415 |
+
"loss": 1.9372346878051758,
|
| 416 |
+
"memory(GiB)": 75.54,
|
| 417 |
+
"step": 195,
|
| 418 |
+
"token_acc": 0.5546875,
|
| 419 |
+
"train_speed(iter/s)": 0.035833
|
| 420 |
+
},
|
| 421 |
+
{
|
| 422 |
+
"epoch": 0.3334722801167153,
|
| 423 |
+
"grad_norm": 3.6666870417160666,
|
| 424 |
+
"learning_rate": 7.954538027563601e-05,
|
| 425 |
+
"loss": 2.1287214279174806,
|
| 426 |
+
"memory(GiB)": 75.54,
|
| 427 |
+
"step": 200,
|
| 428 |
+
"token_acc": 0.5563304721030042,
|
| 429 |
+
"train_speed(iter/s)": 0.035849
|
| 430 |
+
},
|
| 431 |
+
{
|
| 432 |
+
"epoch": 0.3334722801167153,
|
| 433 |
+
"eval_loss": 1.8122097253799438,
|
| 434 |
+
"eval_runtime": 50.701,
|
| 435 |
+
"eval_samples_per_second": 7.633,
|
| 436 |
+
"eval_steps_per_second": 0.493,
|
| 437 |
+
"eval_token_acc": 0.5999835620941892,
|
| 438 |
+
"step": 200
|
| 439 |
+
},
|
| 440 |
+
{
|
| 441 |
+
"epoch": 0.3418090871196332,
|
| 442 |
+
"grad_norm": 3.5964229855404164,
|
| 443 |
+
"learning_rate": 7.842071131934246e-05,
|
| 444 |
+
"loss": 1.8944969177246094,
|
| 445 |
+
"memory(GiB)": 75.54,
|
| 446 |
+
"step": 205,
|
| 447 |
+
"token_acc": 0.6359223300970874,
|
| 448 |
+
"train_speed(iter/s)": 0.035526
|
| 449 |
+
},
|
| 450 |
+
{
|
| 451 |
+
"epoch": 0.35014589412255104,
|
| 452 |
+
"grad_norm": 2.580207777434119,
|
| 453 |
+
"learning_rate": 7.727438415192433e-05,
|
| 454 |
+
"loss": 1.9824462890625,
|
| 455 |
+
"memory(GiB)": 75.54,
|
| 456 |
+
"step": 210,
|
| 457 |
+
"token_acc": 0.5771408351026185,
|
| 458 |
+
"train_speed(iter/s)": 0.035546
|
| 459 |
+
},
|
| 460 |
+
{
|
| 461 |
+
"epoch": 0.3584827011254689,
|
| 462 |
+
"grad_norm": 2.8278774675330887,
|
| 463 |
+
"learning_rate": 7.610727234039167e-05,
|
| 464 |
+
"loss": 2.0116973876953126,
|
| 465 |
+
"memory(GiB)": 75.54,
|
| 466 |
+
"step": 215,
|
| 467 |
+
"token_acc": 0.6102067751869775,
|
| 468 |
+
"train_speed(iter/s)": 0.035564
|
| 469 |
+
},
|
| 470 |
+
{
|
| 471 |
+
"epoch": 0.3668195081283868,
|
| 472 |
+
"grad_norm": 5.855526666772205,
|
| 473 |
+
"learning_rate": 7.492026529084468e-05,
|
| 474 |
+
"loss": 1.9584331512451172,
|
| 475 |
+
"memory(GiB)": 75.54,
|
| 476 |
+
"step": 220,
|
| 477 |
+
"token_acc": 0.5683212493028444,
|
| 478 |
+
"train_speed(iter/s)": 0.035583
|
| 479 |
+
},
|
| 480 |
+
{
|
| 481 |
+
"epoch": 0.3751563151313047,
|
| 482 |
+
"grad_norm": 4.011785835207539,
|
| 483 |
+
"learning_rate": 7.371426757069537e-05,
|
| 484 |
+
"loss": 2.0375545501708983,
|
| 485 |
+
"memory(GiB)": 75.54,
|
| 486 |
+
"step": 225,
|
| 487 |
+
"token_acc": 0.5575163398692811,
|
| 488 |
+
"train_speed(iter/s)": 0.035599
|
| 489 |
+
},
|
| 490 |
+
{
|
| 491 |
+
"epoch": 0.3834931221342226,
|
| 492 |
+
"grad_norm": 2.3496604645389905,
|
| 493 |
+
"learning_rate": 7.249019821933529e-05,
|
| 494 |
+
"loss": 1.9074588775634767,
|
| 495 |
+
"memory(GiB)": 75.54,
|
| 496 |
+
"step": 230,
|
| 497 |
+
"token_acc": 0.571943887775551,
|
| 498 |
+
"train_speed(iter/s)": 0.035618
|
| 499 |
+
},
|
| 500 |
+
{
|
| 501 |
+
"epoch": 0.3918299291371405,
|
| 502 |
+
"grad_norm": 2.4061403829778873,
|
| 503 |
+
"learning_rate": 7.124899004777489e-05,
|
| 504 |
+
"loss": 2.0661600112915037,
|
| 505 |
+
"memory(GiB)": 75.54,
|
| 506 |
+
"step": 235,
|
| 507 |
+
"token_acc": 0.5840575367096195,
|
| 508 |
+
"train_speed(iter/s)": 0.035636
|
| 509 |
+
},
|
| 510 |
+
{
|
| 511 |
+
"epoch": 0.4001667361400584,
|
| 512 |
+
"grad_norm": 2.4741806087579885,
|
| 513 |
+
"learning_rate": 6.9991588927788e-05,
|
| 514 |
+
"loss": 2.0608776092529295,
|
| 515 |
+
"memory(GiB)": 75.54,
|
| 516 |
+
"step": 240,
|
| 517 |
+
"token_acc": 0.6465237166991553,
|
| 518 |
+
"train_speed(iter/s)": 0.035654
|
| 519 |
+
},
|
| 520 |
+
{
|
| 521 |
+
"epoch": 0.40850354314297627,
|
| 522 |
+
"grad_norm": 3.4537545471932733,
|
| 523 |
+
"learning_rate": 6.871895307110332e-05,
|
| 524 |
+
"loss": 1.938018798828125,
|
| 525 |
+
"memory(GiB)": 75.54,
|
| 526 |
+
"step": 245,
|
| 527 |
+
"token_acc": 0.6371398078975453,
|
| 528 |
+
"train_speed(iter/s)": 0.035664
|
| 529 |
+
},
|
| 530 |
+
{
|
| 531 |
+
"epoch": 0.4168403501458941,
|
| 532 |
+
"grad_norm": 16.72496162125171,
|
| 533 |
+
"learning_rate": 6.743205229919224e-05,
|
| 534 |
+
"loss": 1.7865478515625,
|
| 535 |
+
"memory(GiB)": 75.54,
|
| 536 |
+
"step": 250,
|
| 537 |
+
"token_acc": 0.5756791720569211,
|
| 538 |
+
"train_speed(iter/s)": 0.035675
|
| 539 |
+
},
|
| 540 |
+
{
|
| 541 |
+
"epoch": 0.425177157148812,
|
| 542 |
+
"grad_norm": 3.1932083384706664,
|
| 543 |
+
"learning_rate": 6.613186730420917e-05,
|
| 544 |
+
"loss": 2.0705270767211914,
|
| 545 |
+
"memory(GiB)": 75.54,
|
| 546 |
+
"step": 255,
|
| 547 |
+
"token_acc": 0.6203485633537447,
|
| 548 |
+
"train_speed(iter/s)": 0.035687
|
| 549 |
+
},
|
| 550 |
+
{
|
| 551 |
+
"epoch": 0.4335139641517299,
|
| 552 |
+
"grad_norm": 7.316201458214335,
|
| 553 |
+
"learning_rate": 6.4819388901648e-05,
|
| 554 |
+
"loss": 1.9050493240356445,
|
| 555 |
+
"memory(GiB)": 75.54,
|
| 556 |
+
"step": 260,
|
| 557 |
+
"token_acc": 0.5929742388758782,
|
| 558 |
+
"train_speed(iter/s)": 0.035706
|
| 559 |
+
},
|
| 560 |
+
{
|
| 561 |
+
"epoch": 0.44185077115464777,
|
| 562 |
+
"grad_norm": 2.7720764825330853,
|
| 563 |
+
"learning_rate": 6.349561727528388e-05,
|
| 564 |
+
"loss": 1.888047981262207,
|
| 565 |
+
"memory(GiB)": 75.54,
|
| 566 |
+
"step": 265,
|
| 567 |
+
"token_acc": 0.5771358328211432,
|
| 568 |
+
"train_speed(iter/s)": 0.035715
|
| 569 |
+
},
|
| 570 |
+
{
|
| 571 |
+
"epoch": 0.45018757815756566,
|
| 572 |
+
"grad_norm": 4.065193512944849,
|
| 573 |
+
"learning_rate": 6.216156121497578e-05,
|
| 574 |
+
"loss": 1.8664142608642578,
|
| 575 |
+
"memory(GiB)": 75.54,
|
| 576 |
+
"step": 270,
|
| 577 |
+
"token_acc": 0.6696696696696697,
|
| 578 |
+
"train_speed(iter/s)": 0.035727
|
| 579 |
+
},
|
| 580 |
+
{
|
| 581 |
+
"epoch": 0.45852438516048355,
|
| 582 |
+
"grad_norm": 5.3789773277344315,
|
| 583 |
+
"learning_rate": 6.0818237347910903e-05,
|
| 584 |
+
"loss": 1.9860942840576172,
|
| 585 |
+
"memory(GiB)": 75.54,
|
| 586 |
+
"step": 275,
|
| 587 |
+
"token_acc": 0.5549738219895288,
|
| 588 |
+
"train_speed(iter/s)": 0.035743
|
| 589 |
+
},
|
| 590 |
+
{
|
| 591 |
+
"epoch": 0.46686119216340144,
|
| 592 |
+
"grad_norm": 4.12258515072594,
|
| 593 |
+
"learning_rate": 5.946666936387637e-05,
|
| 594 |
+
"loss": 1.9592571258544922,
|
| 595 |
+
"memory(GiB)": 75.54,
|
| 596 |
+
"step": 280,
|
| 597 |
+
"token_acc": 0.5547355473554736,
|
| 598 |
+
"train_speed(iter/s)": 0.035764
|
| 599 |
+
},
|
| 600 |
+
{
|
| 601 |
+
"epoch": 0.4751979991663193,
|
| 602 |
+
"grad_norm": 20.985894221796727,
|
| 603 |
+
"learning_rate": 5.810788723514908e-05,
|
| 604 |
+
"loss": 2.0858516693115234,
|
| 605 |
+
"memory(GiB)": 75.54,
|
| 606 |
+
"step": 285,
|
| 607 |
+
"token_acc": 0.5434947049924357,
|
| 608 |
+
"train_speed(iter/s)": 0.035776
|
| 609 |
+
},
|
| 610 |
+
{
|
| 611 |
+
"epoch": 0.48353480616923716,
|
| 612 |
+
"grad_norm": 8.177745830458013,
|
| 613 |
+
"learning_rate": 5.674292643159764e-05,
|
| 614 |
+
"loss": 1.9558685302734375,
|
| 615 |
+
"memory(GiB)": 75.54,
|
| 616 |
+
"step": 290,
|
| 617 |
+
"token_acc": 0.5351539802440441,
|
| 618 |
+
"train_speed(iter/s)": 0.03579
|
| 619 |
+
},
|
| 620 |
+
{
|
| 621 |
+
"epoch": 0.49187161317215505,
|
| 622 |
+
"grad_norm": 4.0835586516577225,
|
| 623 |
+
"learning_rate": 5.537282713159507e-05,
|
| 624 |
+
"loss": 2.0921154022216797,
|
| 625 |
+
"memory(GiB)": 75.54,
|
| 626 |
+
"step": 295,
|
| 627 |
+
"token_acc": 0.5673590504451038,
|
| 628 |
+
"train_speed(iter/s)": 0.035795
|
| 629 |
+
},
|
| 630 |
+
{
|
| 631 |
+
"epoch": 0.5002084201750729,
|
| 632 |
+
"grad_norm": 12.193993471525227,
|
| 633 |
+
"learning_rate": 5.399863342934324e-05,
|
| 634 |
+
"loss": 1.867361068725586,
|
| 635 |
+
"memory(GiB)": 75.54,
|
| 636 |
+
"step": 300,
|
| 637 |
+
"token_acc": 0.5305164319248826,
|
| 638 |
+
"train_speed(iter/s)": 0.035811
|
| 639 |
+
},
|
| 640 |
+
{
|
| 641 |
+
"epoch": 0.5002084201750729,
|
| 642 |
+
"eval_loss": 1.742539644241333,
|
| 643 |
+
"eval_runtime": 50.716,
|
| 644 |
+
"eval_samples_per_second": 7.631,
|
| 645 |
+
"eval_steps_per_second": 0.493,
|
| 646 |
+
"eval_token_acc": 0.612147612394181,
|
| 647 |
+
"step": 300
|
| 648 |
+
},
|
| 649 |
+
{
|
| 650 |
+
"epoch": 0.5085452271779908,
|
| 651 |
+
"grad_norm": 3.6488789490883553,
|
| 652 |
+
"learning_rate": 5.262139253921319e-05,
|
| 653 |
+
"loss": 1.9004257202148438,
|
| 654 |
+
"memory(GiB)": 75.54,
|
| 655 |
+
"step": 305,
|
| 656 |
+
"token_acc": 0.6261458748505381,
|
| 657 |
+
"train_speed(iter/s)": 0.035585
|
| 658 |
+
},
|
| 659 |
+
{
|
| 660 |
+
"epoch": 0.5168820341809087,
|
| 661 |
+
"grad_norm": 2.862433077996505,
|
| 662 |
+
"learning_rate": 5.1242153997707823e-05,
|
| 663 |
+
"loss": 1.9045713424682618,
|
| 664 |
+
"memory(GiB)": 75.54,
|
| 665 |
+
"step": 310,
|
| 666 |
+
"token_acc": 0.6437185929648241,
|
| 667 |
+
"train_speed(iter/s)": 0.035598
|
| 668 |
+
},
|
| 669 |
+
{
|
| 670 |
+
"epoch": 0.5252188411838266,
|
| 671 |
+
"grad_norm": 1.8452496357053012,
|
| 672 |
+
"learning_rate": 4.9861968863654875e-05,
|
| 673 |
+
"loss": 1.8142269134521485,
|
| 674 |
+
"memory(GiB)": 75.54,
|
| 675 |
+
"step": 315,
|
| 676 |
+
"token_acc": 0.6176683562635771,
|
| 677 |
+
"train_speed(iter/s)": 0.035606
|
| 678 |
+
},
|
| 679 |
+
{
|
| 680 |
+
"epoch": 0.5335556481867445,
|
| 681 |
+
"grad_norm": 5.430011832866582,
|
| 682 |
+
"learning_rate": 4.84818889172399e-05,
|
| 683 |
+
"loss": 1.8209394454956054,
|
| 684 |
+
"memory(GiB)": 75.54,
|
| 685 |
+
"step": 320,
|
| 686 |
+
"token_acc": 0.6066109698510715,
|
| 687 |
+
"train_speed(iter/s)": 0.035613
|
| 688 |
+
},
|
| 689 |
+
{
|
| 690 |
+
"epoch": 0.5418924551896623,
|
| 691 |
+
"grad_norm": 2.600769211702097,
|
| 692 |
+
"learning_rate": 4.7102965858489377e-05,
|
| 693 |
+
"loss": 1.9349700927734375,
|
| 694 |
+
"memory(GiB)": 75.54,
|
| 695 |
+
"step": 325,
|
| 696 |
+
"token_acc": 0.6356394129979036,
|
| 697 |
+
"train_speed(iter/s)": 0.035624
|
| 698 |
+
},
|
| 699 |
+
{
|
| 700 |
+
"epoch": 0.5502292621925803,
|
| 701 |
+
"grad_norm": 3.3927364584396877,
|
| 702 |
+
"learning_rate": 4.572625050581516e-05,
|
| 703 |
+
"loss": 1.8674903869628907,
|
| 704 |
+
"memory(GiB)": 75.54,
|
| 705 |
+
"step": 330,
|
| 706 |
+
"token_acc": 0.5809617271835132,
|
| 707 |
+
"train_speed(iter/s)": 0.035636
|
| 708 |
+
},
|
| 709 |
+
{
|
| 710 |
+
"epoch": 0.5585660691954981,
|
| 711 |
+
"grad_norm": 4.347679564527899,
|
| 712 |
+
"learning_rate": 4.435279199523043e-05,
|
| 713 |
+
"loss": 1.9532249450683594,
|
| 714 |
+
"memory(GiB)": 75.54,
|
| 715 |
+
"step": 335,
|
| 716 |
+
"token_acc": 0.6082898709854515,
|
| 717 |
+
"train_speed(iter/s)": 0.03565
|
| 718 |
+
},
|
| 719 |
+
{
|
| 720 |
+
"epoch": 0.5669028761984161,
|
| 721 |
+
"grad_norm": 14.797523073893712,
|
| 722 |
+
"learning_rate": 4.298363698084809e-05,
|
| 723 |
+
"loss": 1.8467117309570313,
|
| 724 |
+
"memory(GiB)": 75.54,
|
| 725 |
+
"step": 340,
|
| 726 |
+
"token_acc": 0.5447761194029851,
|
| 727 |
+
"train_speed(iter/s)": 0.035659
|
| 728 |
+
},
|
| 729 |
+
{
|
| 730 |
+
"epoch": 0.5752396832013339,
|
| 731 |
+
"grad_norm": 9.174677465010568,
|
| 732 |
+
"learning_rate": 4.16198288372702e-05,
|
| 733 |
+
"loss": 1.8867809295654296,
|
| 734 |
+
"memory(GiB)": 75.54,
|
| 735 |
+
"step": 345,
|
| 736 |
+
"token_acc": 0.649881716796215,
|
| 737 |
+
"train_speed(iter/s)": 0.035671
|
| 738 |
+
},
|
| 739 |
+
{
|
| 740 |
+
"epoch": 0.5835764902042517,
|
| 741 |
+
"grad_norm": 11.517095197655033,
|
| 742 |
+
"learning_rate": 4.026240686447682e-05,
|
| 743 |
+
"loss": 1.9671783447265625,
|
| 744 |
+
"memory(GiB)": 75.54,
|
| 745 |
+
"step": 350,
|
| 746 |
+
"token_acc": 0.5656722200697404,
|
| 747 |
+
"train_speed(iter/s)": 0.035684
|
| 748 |
+
},
|
| 749 |
+
{
|
| 750 |
+
"epoch": 0.5919132972071697,
|
| 751 |
+
"grad_norm": 2.0947774902715133,
|
| 752 |
+
"learning_rate": 3.8912405495819786e-05,
|
| 753 |
+
"loss": 2.0355813980102537,
|
| 754 |
+
"memory(GiB)": 75.54,
|
| 755 |
+
"step": 355,
|
| 756 |
+
"token_acc": 0.5646427096241222,
|
| 757 |
+
"train_speed(iter/s)": 0.035694
|
| 758 |
+
},
|
| 759 |
+
{
|
| 760 |
+
"epoch": 0.6002501042100875,
|
| 761 |
+
"grad_norm": 2.2672434495907967,
|
| 762 |
+
"learning_rate": 3.757085350972523e-05,
|
| 763 |
+
"loss": 2.0212512969970704,
|
| 764 |
+
"memory(GiB)": 75.54,
|
| 765 |
+
"step": 360,
|
| 766 |
+
"token_acc": 0.603363412633306,
|
| 767 |
+
"train_speed(iter/s)": 0.035702
|
| 768 |
+
},
|
| 769 |
+
{
|
| 770 |
+
"epoch": 0.6085869112130055,
|
| 771 |
+
"grad_norm": 3.3476628567015885,
|
| 772 |
+
"learning_rate": 3.623877324570548e-05,
|
| 773 |
+
"loss": 1.888478660583496,
|
| 774 |
+
"memory(GiB)": 75.54,
|
| 775 |
+
"step": 365,
|
| 776 |
+
"token_acc": 0.5527710843373494,
|
| 777 |
+
"train_speed(iter/s)": 0.035708
|
| 778 |
+
},
|
| 779 |
+
{
|
| 780 |
+
"epoch": 0.6169237182159233,
|
| 781 |
+
"grad_norm": 6.163682031673968,
|
| 782 |
+
"learning_rate": 3.491717982527765e-05,
|
| 783 |
+
"loss": 2.1264991760253906,
|
| 784 |
+
"memory(GiB)": 75.54,
|
| 785 |
+
"step": 370,
|
| 786 |
+
"token_acc": 0.5708566853482786,
|
| 787 |
+
"train_speed(iter/s)": 0.035714
|
| 788 |
+
},
|
| 789 |
+
{
|
| 790 |
+
"epoch": 0.6252605252188412,
|
| 791 |
+
"grad_norm": 1.6777593794217385,
|
| 792 |
+
"learning_rate": 3.3607080378383005e-05,
|
| 793 |
+
"loss": 1.8802024841308593,
|
| 794 |
+
"memory(GiB)": 75.54,
|
| 795 |
+
"step": 375,
|
| 796 |
+
"token_acc": 0.5547480620155039,
|
| 797 |
+
"train_speed(iter/s)": 0.035723
|
| 798 |
+
},
|
| 799 |
+
{
|
| 800 |
+
"epoch": 0.6335973322217591,
|
| 801 |
+
"grad_norm": 2.8440899647216833,
|
| 802 |
+
"learning_rate": 3.230947327589602e-05,
|
| 803 |
+
"loss": 1.8647388458251952,
|
| 804 |
+
"memory(GiB)": 75.54,
|
| 805 |
+
"step": 380,
|
| 806 |
+
"token_acc": 0.5930644019815995,
|
| 807 |
+
"train_speed(iter/s)": 0.035732
|
| 808 |
+
},
|
| 809 |
+
{
|
| 810 |
+
"epoch": 0.6419341392246769,
|
| 811 |
+
"grad_norm": 37.911669332255286,
|
| 812 |
+
"learning_rate": 3.1025347368808775e-05,
|
| 813 |
+
"loss": 1.8130638122558593,
|
| 814 |
+
"memory(GiB)": 75.54,
|
| 815 |
+
"step": 385,
|
| 816 |
+
"token_acc": 0.5654246100519931,
|
| 817 |
+
"train_speed(iter/s)": 0.035741
|
| 818 |
+
},
|
| 819 |
+
{
|
| 820 |
+
"epoch": 0.6502709462275948,
|
| 821 |
+
"grad_norm": 4.022081032945518,
|
| 822 |
+
"learning_rate": 2.9755681234669663e-05,
|
| 823 |
+
"loss": 2.064923095703125,
|
| 824 |
+
"memory(GiB)": 75.54,
|
| 825 |
+
"step": 390,
|
| 826 |
+
"token_acc": 0.6108317214700193,
|
| 827 |
+
"train_speed(iter/s)": 0.035754
|
| 828 |
+
},
|
| 829 |
+
{
|
| 830 |
+
"epoch": 0.6586077532305127,
|
| 831 |
+
"grad_norm": 2.307367388031427,
|
| 832 |
+
"learning_rate": 2.85014424318512e-05,
|
| 833 |
+
"loss": 1.9649499893188476,
|
| 834 |
+
"memory(GiB)": 75.54,
|
| 835 |
+
"step": 395,
|
| 836 |
+
"token_acc": 0.5570539419087137,
|
| 837 |
+
"train_speed(iter/s)": 0.03576
|
| 838 |
+
},
|
| 839 |
+
{
|
| 840 |
+
"epoch": 0.6669445602334306,
|
| 841 |
+
"grad_norm": 4.204547110729611,
|
| 842 |
+
"learning_rate": 2.7263586762215197e-05,
|
| 843 |
+
"loss": 1.92861328125,
|
| 844 |
+
"memory(GiB)": 75.54,
|
| 845 |
+
"step": 400,
|
| 846 |
+
"token_acc": 0.5911730545876888,
|
| 847 |
+
"train_speed(iter/s)": 0.035772
|
| 848 |
+
},
|
| 849 |
+
{
|
| 850 |
+
"epoch": 0.6669445602334306,
|
| 851 |
+
"eval_loss": 1.7799798250198364,
|
| 852 |
+
"eval_runtime": 50.5835,
|
| 853 |
+
"eval_samples_per_second": 7.651,
|
| 854 |
+
"eval_steps_per_second": 0.494,
|
| 855 |
+
"eval_token_acc": 0.6100106846387771,
|
| 856 |
+
"step": 400
|
| 857 |
+
},
|
| 858 |
+
{
|
| 859 |
+
"epoch": 0.6752813672363485,
|
| 860 |
+
"grad_norm": 2.659254046452427,
|
| 861 |
+
"learning_rate": 2.6043057542736836e-05,
|
| 862 |
+
"loss": 1.9875520706176757,
|
| 863 |
+
"memory(GiB)": 75.54,
|
| 864 |
+
"step": 405,
|
| 865 |
+
"token_acc": 0.5904554527309018,
|
| 866 |
+
"train_speed(iter/s)": 0.03561
|
| 867 |
+
},
|
| 868 |
+
{
|
| 869 |
+
"epoch": 0.6836181742392664,
|
| 870 |
+
"grad_norm": 14.662841293469047,
|
| 871 |
+
"learning_rate": 2.4840784886643132e-05,
|
| 872 |
+
"loss": 1.7080364227294922,
|
| 873 |
+
"memory(GiB)": 75.54,
|
| 874 |
+
"step": 410,
|
| 875 |
+
"token_acc": 0.6345278725824801,
|
| 876 |
+
"train_speed(iter/s)": 0.035622
|
| 877 |
+
},
|
| 878 |
+
{
|
| 879 |
+
"epoch": 0.6919549812421842,
|
| 880 |
+
"grad_norm": 2.0901368954978015,
|
| 881 |
+
"learning_rate": 2.365768499461328e-05,
|
| 882 |
+
"loss": 1.9299034118652343,
|
| 883 |
+
"memory(GiB)": 75.54,
|
| 884 |
+
"step": 415,
|
| 885 |
+
"token_acc": 0.5383259911894274,
|
| 886 |
+
"train_speed(iter/s)": 0.035634
|
| 887 |
+
},
|
| 888 |
+
{
|
| 889 |
+
"epoch": 0.7002917882451021,
|
| 890 |
+
"grad_norm": 2.658074481592308,
|
| 891 |
+
"learning_rate": 2.249465945658135e-05,
|
| 892 |
+
"loss": 2.000414276123047,
|
| 893 |
+
"memory(GiB)": 75.54,
|
| 894 |
+
"step": 420,
|
| 895 |
+
"token_acc": 0.5644047135310849,
|
| 896 |
+
"train_speed(iter/s)": 0.035642
|
| 897 |
+
},
|
| 898 |
+
{
|
| 899 |
+
"epoch": 0.70862859524802,
|
| 900 |
+
"grad_norm": 3.8658249706668504,
|
| 901 |
+
"learning_rate": 2.1352594564672908e-05,
|
| 902 |
+
"loss": 1.9364568710327148,
|
| 903 |
+
"memory(GiB)": 75.54,
|
| 904 |
+
"step": 425,
|
| 905 |
+
"token_acc": 0.6048387096774194,
|
| 906 |
+
"train_speed(iter/s)": 0.03565
|
| 907 |
+
},
|
| 908 |
+
{
|
| 909 |
+
"epoch": 0.7169654022509379,
|
| 910 |
+
"grad_norm": 17.005290399571575,
|
| 911 |
+
"learning_rate": 2.0232360637799685e-05,
|
| 912 |
+
"loss": 1.9262733459472656,
|
| 913 |
+
"memory(GiB)": 75.54,
|
| 914 |
+
"step": 430,
|
| 915 |
+
"token_acc": 0.5591078066914498,
|
| 916 |
+
"train_speed(iter/s)": 0.035661
|
| 917 |
+
},
|
| 918 |
+
{
|
| 919 |
+
"epoch": 0.7253022092538558,
|
| 920 |
+
"grad_norm": 2.7462439370469673,
|
| 921 |
+
"learning_rate": 1.9134811358426757e-05,
|
| 922 |
+
"loss": 2.0130874633789064,
|
| 923 |
+
"memory(GiB)": 75.54,
|
| 924 |
+
"step": 435,
|
| 925 |
+
"token_acc": 0.5676373018798379,
|
| 926 |
+
"train_speed(iter/s)": 0.035672
|
| 927 |
+
},
|
| 928 |
+
{
|
| 929 |
+
"epoch": 0.7336390162567736,
|
| 930 |
+
"grad_norm": 3.9297703842696294,
|
| 931 |
+
"learning_rate": 1.806078312201745e-05,
|
| 932 |
+
"loss": 1.886693000793457,
|
| 933 |
+
"memory(GiB)": 75.54,
|
| 934 |
+
"step": 440,
|
| 935 |
+
"token_acc": 0.5958948043617703,
|
| 936 |
+
"train_speed(iter/s)": 0.035681
|
| 937 |
+
},
|
| 938 |
+
{
|
| 939 |
+
"epoch": 0.7419758232596916,
|
| 940 |
+
"grad_norm": 4.746458587201285,
|
| 941 |
+
"learning_rate": 1.7011094399652107e-05,
|
| 942 |
+
"loss": 1.8821983337402344,
|
| 943 |
+
"memory(GiB)": 75.54,
|
| 944 |
+
"step": 445,
|
| 945 |
+
"token_acc": 0.5943814687037949,
|
| 946 |
+
"train_speed(iter/s)": 0.035688
|
| 947 |
+
},
|
| 948 |
+
{
|
| 949 |
+
"epoch": 0.7503126302626094,
|
| 950 |
+
"grad_norm": 4.130883468895127,
|
| 951 |
+
"learning_rate": 1.59865451143062e-05,
|
| 952 |
+
"loss": 1.847772216796875,
|
| 953 |
+
"memory(GiB)": 75.54,
|
| 954 |
+
"step": 450,
|
| 955 |
+
"token_acc": 0.6304583182966438,
|
| 956 |
+
"train_speed(iter/s)": 0.035697
|
| 957 |
+
},
|
| 958 |
+
{
|
| 959 |
+
"epoch": 0.7586494372655272,
|
| 960 |
+
"grad_norm": 9.216193465389415,
|
| 961 |
+
"learning_rate": 1.4987916031263232e-05,
|
| 962 |
+
"loss": 1.8329124450683594,
|
| 963 |
+
"memory(GiB)": 75.54,
|
| 964 |
+
"step": 455,
|
| 965 |
+
"token_acc": 0.635618801207417,
|
| 966 |
+
"train_speed(iter/s)": 0.035704
|
| 967 |
+
},
|
| 968 |
+
{
|
| 969 |
+
"epoch": 0.7669862442684452,
|
| 970 |
+
"grad_norm": 5.297584387705496,
|
| 971 |
+
"learning_rate": 1.401596816312673e-05,
|
| 972 |
+
"loss": 1.8964439392089845,
|
| 973 |
+
"memory(GiB)": 75.54,
|
| 974 |
+
"step": 460,
|
| 975 |
+
"token_acc": 0.5654450261780105,
|
| 976 |
+
"train_speed(iter/s)": 0.035715
|
| 977 |
+
},
|
| 978 |
+
{
|
| 979 |
+
"epoch": 0.775323051271363,
|
| 980 |
+
"grad_norm": 2.687671502108267,
|
| 981 |
+
"learning_rate": 1.307144218988507e-05,
|
| 982 |
+
"loss": 1.9502939224243163,
|
| 983 |
+
"memory(GiB)": 75.54,
|
| 984 |
+
"step": 465,
|
| 985 |
+
"token_acc": 0.5579991375592928,
|
| 986 |
+
"train_speed(iter/s)": 0.035727
|
| 987 |
+
},
|
| 988 |
+
{
|
| 989 |
+
"epoch": 0.783659858274281,
|
| 990 |
+
"grad_norm": 3.767228791810687,
|
| 991 |
+
"learning_rate": 1.2155057894470928e-05,
|
| 992 |
+
"loss": 1.71820068359375,
|
| 993 |
+
"memory(GiB)": 75.54,
|
| 994 |
+
"step": 470,
|
| 995 |
+
"token_acc": 0.5902320748181503,
|
| 996 |
+
"train_speed(iter/s)": 0.035733
|
| 997 |
+
},
|
| 998 |
+
{
|
| 999 |
+
"epoch": 0.7919966652771988,
|
| 1000 |
+
"grad_norm": 4.582125037883684,
|
| 1001 |
+
"learning_rate": 1.126751361424529e-05,
|
| 1002 |
+
"loss": 1.8622875213623047,
|
| 1003 |
+
"memory(GiB)": 75.54,
|
| 1004 |
+
"step": 475,
|
| 1005 |
+
"token_acc": 0.5744859420898027,
|
| 1006 |
+
"train_speed(iter/s)": 0.035739
|
| 1007 |
+
},
|
| 1008 |
+
{
|
| 1009 |
+
"epoch": 0.8003334722801168,
|
| 1010 |
+
"grad_norm": 3.5101737578003767,
|
| 1011 |
+
"learning_rate": 1.0409485708824507e-05,
|
| 1012 |
+
"loss": 1.8682287216186524,
|
| 1013 |
+
"memory(GiB)": 75.54,
|
| 1014 |
+
"step": 480,
|
| 1015 |
+
"token_acc": 0.5714951094550536,
|
| 1016 |
+
"train_speed(iter/s)": 0.035748
|
| 1017 |
+
},
|
| 1018 |
+
{
|
| 1019 |
+
"epoch": 0.8086702792830346,
|
| 1020 |
+
"grad_norm": 2.414709851450202,
|
| 1021 |
+
"learning_rate": 9.581628044655394e-06,
|
| 1022 |
+
"loss": 1.8879878997802735,
|
| 1023 |
+
"memory(GiB)": 75.54,
|
| 1024 |
+
"step": 485,
|
| 1025 |
+
"token_acc": 0.656461583750368,
|
| 1026 |
+
"train_speed(iter/s)": 0.035756
|
| 1027 |
+
},
|
| 1028 |
+
{
|
| 1029 |
+
"epoch": 0.8170070862859525,
|
| 1030 |
+
"grad_norm": 3.810296788453775,
|
| 1031 |
+
"learning_rate": 8.78457149673152e-06,
|
| 1032 |
+
"loss": 1.8605712890625,
|
| 1033 |
+
"memory(GiB)": 75.54,
|
| 1034 |
+
"step": 490,
|
| 1035 |
+
"token_acc": 0.6178915862986365,
|
| 1036 |
+
"train_speed(iter/s)": 0.035763
|
| 1037 |
+
},
|
| 1038 |
+
{
|
| 1039 |
+
"epoch": 0.8253438932888704,
|
| 1040 |
+
"grad_norm": 5.265615022155901,
|
| 1041 |
+
"learning_rate": 8.018923467830403e-06,
|
| 1042 |
+
"loss": 1.8603355407714843,
|
| 1043 |
+
"memory(GiB)": 75.54,
|
| 1044 |
+
"step": 495,
|
| 1045 |
+
"token_acc": 0.5819091288036682,
|
| 1046 |
+
"train_speed(iter/s)": 0.035772
|
| 1047 |
+
},
|
| 1048 |
+
{
|
| 1049 |
+
"epoch": 0.8336807002917882,
|
| 1050 |
+
"grad_norm": 2.5025771365898435,
|
| 1051 |
+
"learning_rate": 7.28526742563762e-06,
|
| 1052 |
+
"loss": 1.69268741607666,
|
| 1053 |
+
"memory(GiB)": 75.54,
|
| 1054 |
+
"step": 500,
|
| 1055 |
+
"token_acc": 0.6207253886010363,
|
| 1056 |
+
"train_speed(iter/s)": 0.035779
|
| 1057 |
+
},
|
| 1058 |
+
{
|
| 1059 |
+
"epoch": 0.8336807002917882,
|
| 1060 |
+
"eval_loss": 1.6579405069351196,
|
| 1061 |
+
"eval_runtime": 50.7583,
|
| 1062 |
+
"eval_samples_per_second": 7.624,
|
| 1063 |
+
"eval_steps_per_second": 0.493,
|
| 1064 |
+
"eval_token_acc": 0.6231610092874168,
|
| 1065 |
+
"step": 500
|
| 1066 |
+
},
|
| 1067 |
+
{
|
| 1068 |
+
"epoch": 0.8420175072947061,
|
| 1069 |
+
"grad_norm": 7.055173813050995,
|
| 1070 |
+
"learning_rate": 6.584162458111148e-06,
|
| 1071 |
+
"loss": 1.8024673461914062,
|
| 1072 |
+
"memory(GiB)": 75.54,
|
| 1073 |
+
"step": 505,
|
| 1074 |
+
"token_acc": 0.655217965653897,
|
| 1075 |
+
"train_speed(iter/s)": 0.035646
|
| 1076 |
+
},
|
| 1077 |
+
{
|
| 1078 |
+
"epoch": 0.850354314297624,
|
| 1079 |
+
"grad_norm": 3.4632299338583885,
|
| 1080 |
+
"learning_rate": 5.916142847424127e-06,
|
| 1081 |
+
"loss": 1.6985977172851563,
|
| 1082 |
+
"memory(GiB)": 75.54,
|
| 1083 |
+
"step": 510,
|
| 1084 |
+
"token_acc": 0.6189883913764511,
|
| 1085 |
+
"train_speed(iter/s)": 0.035649
|
| 1086 |
+
},
|
| 1087 |
+
{
|
| 1088 |
+
"epoch": 0.8586911213005419,
|
| 1089 |
+
"grad_norm": 3.7217850034242863,
|
| 1090 |
+
"learning_rate": 5.281717662811381e-06,
|
| 1091 |
+
"loss": 1.8116317749023438,
|
| 1092 |
+
"memory(GiB)": 75.54,
|
| 1093 |
+
"step": 515,
|
| 1094 |
+
"token_acc": 0.7189467658843732,
|
| 1095 |
+
"train_speed(iter/s)": 0.035656
|
| 1096 |
+
},
|
| 1097 |
+
{
|
| 1098 |
+
"epoch": 0.8670279283034598,
|
| 1099 |
+
"grad_norm": 3.6951398715683865,
|
| 1100 |
+
"learning_rate": 4.681370372629368e-06,
|
| 1101 |
+
"loss": 1.657485008239746,
|
| 1102 |
+
"memory(GiB)": 75.54,
|
| 1103 |
+
"step": 520,
|
| 1104 |
+
"token_acc": 0.6326460481099656,
|
| 1105 |
+
"train_speed(iter/s)": 0.035663
|
| 1106 |
+
},
|
| 1107 |
+
{
|
| 1108 |
+
"epoch": 0.8753647353063777,
|
| 1109 |
+
"grad_norm": 1.8312455829592764,
|
| 1110 |
+
"learning_rate": 4.1155584759256015e-06,
|
| 1111 |
+
"loss": 1.7322940826416016,
|
| 1112 |
+
"memory(GiB)": 75.54,
|
| 1113 |
+
"step": 525,
|
| 1114 |
+
"token_acc": 0.5754838709677419,
|
| 1115 |
+
"train_speed(iter/s)": 0.035668
|
| 1116 |
+
},
|
| 1117 |
+
{
|
| 1118 |
+
"epoch": 0.8837015423092955,
|
| 1119 |
+
"grad_norm": 2.151076970470845,
|
| 1120 |
+
"learning_rate": 3.5847131537982137e-06,
|
| 1121 |
+
"loss": 1.8520671844482421,
|
| 1122 |
+
"memory(GiB)": 75.54,
|
| 1123 |
+
"step": 530,
|
| 1124 |
+
"token_acc": 0.6836757234371549,
|
| 1125 |
+
"train_speed(iter/s)": 0.035678
|
| 1126 |
+
},
|
| 1127 |
+
{
|
| 1128 |
+
"epoch": 0.8920383493122134,
|
| 1129 |
+
"grad_norm": 2.92427111277661,
|
| 1130 |
+
"learning_rate": 3.089238940811162e-06,
|
| 1131 |
+
"loss": 1.8487899780273438,
|
| 1132 |
+
"memory(GiB)": 75.54,
|
| 1133 |
+
"step": 535,
|
| 1134 |
+
"token_acc": 0.5323415265200517,
|
| 1135 |
+
"train_speed(iter/s)": 0.035687
|
| 1136 |
+
},
|
| 1137 |
+
{
|
| 1138 |
+
"epoch": 0.9003751563151313,
|
| 1139 |
+
"grad_norm": 2.7017957543139546,
|
| 1140 |
+
"learning_rate": 2.6295134167157343e-06,
|
| 1141 |
+
"loss": 1.7729415893554688,
|
| 1142 |
+
"memory(GiB)": 75.54,
|
| 1143 |
+
"step": 540,
|
| 1144 |
+
"token_acc": 0.5686113393590797,
|
| 1145 |
+
"train_speed(iter/s)": 0.035697
|
| 1146 |
+
},
|
| 1147 |
+
{
|
| 1148 |
+
"epoch": 0.9087119633180492,
|
| 1149 |
+
"grad_norm": 2.0280474276001033,
|
| 1150 |
+
"learning_rate": 2.2058869187131515e-06,
|
| 1151 |
+
"loss": 1.7561511993408203,
|
| 1152 |
+
"memory(GiB)": 75.54,
|
| 1153 |
+
"step": 545,
|
| 1154 |
+
"token_acc": 0.5832210242587601,
|
| 1155 |
+
"train_speed(iter/s)": 0.035701
|
| 1156 |
+
},
|
| 1157 |
+
{
|
| 1158 |
+
"epoch": 0.9170487703209671,
|
| 1159 |
+
"grad_norm": 3.339758576911185,
|
| 1160 |
+
"learning_rate": 1.8186822744775234e-06,
|
| 1161 |
+
"loss": 1.8700078964233398,
|
| 1162 |
+
"memory(GiB)": 75.54,
|
| 1163 |
+
"step": 550,
|
| 1164 |
+
"token_acc": 0.6577181208053692,
|
| 1165 |
+
"train_speed(iter/s)": 0.035708
|
| 1166 |
+
},
|
| 1167 |
+
{
|
| 1168 |
+
"epoch": 0.9253855773238849,
|
| 1169 |
+
"grad_norm": 5.51613595581924,
|
| 1170 |
+
"learning_rate": 1.4681945561426547e-06,
|
| 1171 |
+
"loss": 1.766021728515625,
|
| 1172 |
+
"memory(GiB)": 75.54,
|
| 1173 |
+
"step": 555,
|
| 1174 |
+
"token_acc": 0.5925414364640884,
|
| 1175 |
+
"train_speed(iter/s)": 0.035718
|
| 1176 |
+
},
|
| 1177 |
+
{
|
| 1178 |
+
"epoch": 0.9337223843268029,
|
| 1179 |
+
"grad_norm": 4.024065341173453,
|
| 1180 |
+
"learning_rate": 1.1546908554401659e-06,
|
| 1181 |
+
"loss": 1.8983577728271483,
|
| 1182 |
+
"memory(GiB)": 75.54,
|
| 1183 |
+
"step": 560,
|
| 1184 |
+
"token_acc": 0.5683696468820436,
|
| 1185 |
+
"train_speed(iter/s)": 0.035724
|
| 1186 |
+
},
|
| 1187 |
+
{
|
| 1188 |
+
"epoch": 0.9420591913297207,
|
| 1189 |
+
"grad_norm": 7.437280242056288,
|
| 1190 |
+
"learning_rate": 8.784100801602912e-07,
|
| 1191 |
+
"loss": 1.7441574096679688,
|
| 1192 |
+
"memory(GiB)": 75.54,
|
| 1193 |
+
"step": 565,
|
| 1194 |
+
"token_acc": 0.6148734985944289,
|
| 1195 |
+
"train_speed(iter/s)": 0.03573
|
| 1196 |
+
},
|
| 1197 |
+
{
|
| 1198 |
+
"epoch": 0.9503959983326385,
|
| 1199 |
+
"grad_norm": 6.1653456549827625,
|
| 1200 |
+
"learning_rate": 6.395627720904518e-07,
|
| 1201 |
+
"loss": 1.8380481719970703,
|
| 1202 |
+
"memory(GiB)": 75.54,
|
| 1203 |
+
"step": 570,
|
| 1204 |
+
"token_acc": 0.5575221238938053,
|
| 1205 |
+
"train_speed(iter/s)": 0.035736
|
| 1206 |
+
},
|
| 1207 |
+
{
|
| 1208 |
+
"epoch": 0.9587328053355565,
|
| 1209 |
+
"grad_norm": 3.928180437337867,
|
| 1210 |
+
"learning_rate": 4.383309465703145e-07,
|
| 1211 |
+
"loss": 1.8257612228393554,
|
| 1212 |
+
"memory(GiB)": 75.54,
|
| 1213 |
+
"step": 575,
|
| 1214 |
+
"token_acc": 0.5928525845564774,
|
| 1215 |
+
"train_speed(iter/s)": 0.035742
|
| 1216 |
+
},
|
| 1217 |
+
{
|
| 1218 |
+
"epoch": 0.9670696123384743,
|
| 1219 |
+
"grad_norm": 4.015367492611914,
|
| 1220 |
+
"learning_rate": 2.748679537857013e-07,
|
| 1221 |
+
"loss": 2.0145660400390626,
|
| 1222 |
+
"memory(GiB)": 75.54,
|
| 1223 |
+
"step": 580,
|
| 1224 |
+
"token_acc": 0.548,
|
| 1225 |
+
"train_speed(iter/s)": 0.035745
|
| 1226 |
+
},
|
| 1227 |
+
{
|
| 1228 |
+
"epoch": 0.9754064193413923,
|
| 1229 |
+
"grad_norm": 3.725103655434126,
|
| 1230 |
+
"learning_rate": 1.4929836190696323e-07,
|
| 1231 |
+
"loss": 1.9830612182617187,
|
| 1232 |
+
"memory(GiB)": 75.54,
|
| 1233 |
+
"step": 585,
|
| 1234 |
+
"token_acc": 0.5244872331519465,
|
| 1235 |
+
"train_speed(iter/s)": 0.035751
|
| 1236 |
+
},
|
| 1237 |
+
{
|
| 1238 |
+
"epoch": 0.9837432263443101,
|
| 1239 |
+
"grad_norm": 2.333001410662022,
|
| 1240 |
+
"learning_rate": 6.171786216085939e-08,
|
| 1241 |
+
"loss": 1.6920804977416992,
|
| 1242 |
+
"memory(GiB)": 75.54,
|
| 1243 |
+
"step": 590,
|
| 1244 |
+
"token_acc": 0.6742605915267785,
|
| 1245 |
+
"train_speed(iter/s)": 0.035754
|
| 1246 |
+
},
|
| 1247 |
+
{
|
| 1248 |
+
"epoch": 0.992080033347228,
|
| 1249 |
+
"grad_norm": 2.2401580217408386,
|
| 1250 |
+
"learning_rate": 1.2193195908388743e-08,
|
| 1251 |
+
"loss": 1.8399593353271484,
|
| 1252 |
+
"memory(GiB)": 75.54,
|
| 1253 |
+
"step": 595,
|
| 1254 |
+
"token_acc": 0.5763772954924875,
|
| 1255 |
+
"train_speed(iter/s)": 0.035759
|
| 1256 |
+
},
|
| 1257 |
+
{
|
| 1258 |
+
"epoch": 0.9987494789495623,
|
| 1259 |
+
"eval_loss": 1.6350570917129517,
|
| 1260 |
+
"eval_runtime": 50.6821,
|
| 1261 |
+
"eval_samples_per_second": 7.636,
|
| 1262 |
+
"eval_steps_per_second": 0.493,
|
| 1263 |
+
"eval_token_acc": 0.6257088846880907,
|
| 1264 |
+
"step": 599
|
| 1265 |
+
}
|
| 1266 |
+
],
|
| 1267 |
+
"logging_steps": 5,
|
| 1268 |
+
"max_steps": 599,
|
| 1269 |
+
"num_input_tokens_seen": 0,
|
| 1270 |
+
"num_train_epochs": 1,
|
| 1271 |
+
"save_steps": 100,
|
| 1272 |
+
"stateful_callbacks": {
|
| 1273 |
+
"TrainerControl": {
|
| 1274 |
+
"args": {
|
| 1275 |
+
"should_epoch_stop": false,
|
| 1276 |
+
"should_evaluate": false,
|
| 1277 |
+
"should_log": false,
|
| 1278 |
+
"should_save": true,
|
| 1279 |
+
"should_training_stop": true
|
| 1280 |
+
},
|
| 1281 |
+
"attributes": {}
|
| 1282 |
+
}
|
| 1283 |
+
},
|
| 1284 |
+
"total_flos": 954945089044480.0,
|
| 1285 |
+
"train_batch_size": 4,
|
| 1286 |
+
"trial_name": null,
|
| 1287 |
+
"trial_params": null
|
| 1288 |
+
}
|
v1-20250508-175113/checkpoint-599/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3cdbe863329b245f3e43ae627c8c9276a956c353361b2e3d0be81b45ee1195fb
|
| 3 |
+
size 8248
|
v1-20250508-175113/checkpoint-599/vit.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:414197ee1c895aa90d2c648448e6d02d19090b1941d0425b4009afb93446f26f
|
| 3 |
+
size 1337416944
|
v1-20250508-175113/images/eval_loss.png
ADDED
|
v1-20250508-175113/images/eval_runtime.png
ADDED
|
v1-20250508-175113/images/eval_samples_per_second.png
ADDED
|
v1-20250508-175113/images/eval_steps_per_second.png
ADDED
|
v1-20250508-175113/images/eval_token_acc.png
ADDED
|
v1-20250508-175113/images/train_epoch.png
ADDED
|
v1-20250508-175113/images/train_grad_norm.png
ADDED
|
v1-20250508-175113/images/train_learning_rate.png
ADDED
|
v1-20250508-175113/images/train_loss.png
ADDED
|
v1-20250508-175113/images/train_memory(GiB).png
ADDED
|
v1-20250508-175113/images/train_token_acc.png
ADDED
|
v1-20250508-175113/images/train_total_flos.png
ADDED
|
v1-20250508-175113/images/train_train_loss.png
ADDED
|
v1-20250508-175113/images/train_train_runtime.png
ADDED
|
v1-20250508-175113/images/train_train_samples_per_second.png
ADDED
|
v1-20250508-175113/images/train_train_speed(iter_s).png
ADDED
|
v1-20250508-175113/images/train_train_steps_per_second.png
ADDED
|
v1-20250508-175113/logging.jsonl
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"loss": 2.58850908, "token_acc": 0.59252336, "grad_norm": 64.409787, "learning_rate": 3.33e-06, "memory(GiB)": 21.94, "train_speed(iter/s)": 0.01568, "epoch": 0.00166736, "global_step/max_steps": "1/599", "percentage": "0.17%", "elapsed_time": "59s", "remaining_time": "9h 51m 17s"}
|
| 2 |
+
{"loss": 2.83745265, "token_acc": 0.53799283, "grad_norm": 42.49983692, "learning_rate": 1.667e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.028676, "epoch": 0.00833681, "global_step/max_steps": "5/599", "percentage": "0.83%", "elapsed_time": "2m 49s", "remaining_time": "5h 36m 26s"}
|
| 3 |
+
{"loss": 2.48509254, "token_acc": 0.503156, "grad_norm": 9.11943513, "learning_rate": 3.333e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.032119, "epoch": 0.01667361, "global_step/max_steps": "10/599", "percentage": "1.67%", "elapsed_time": "5m 6s", "remaining_time": "5h 1m 15s"}
|
| 4 |
+
{"loss": 2.33806763, "token_acc": 0.50312826, "grad_norm": 12.26783819, "learning_rate": 5e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.033444, "epoch": 0.02501042, "global_step/max_steps": "15/599", "percentage": "2.50%", "elapsed_time": "7m 24s", "remaining_time": "4h 48m 8s"}
|
| 5 |
+
{"loss": 1.98112984, "token_acc": 0.5953383, "grad_norm": 9.95333803, "learning_rate": 6.667e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.034146, "epoch": 0.03334723, "global_step/max_steps": "20/599", "percentage": "3.34%", "elapsed_time": "9m 41s", "remaining_time": "4h 40m 27s"}
|
| 6 |
+
{"loss": 2.0935215, "token_acc": 0.56996727, "grad_norm": 9.42710439, "learning_rate": 8.333e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.034594, "epoch": 0.04168404, "global_step/max_steps": "25/599", "percentage": "4.17%", "elapsed_time": "11m 58s", "remaining_time": "4h 34m 50s"}
|
| 7 |
+
{"loss": 2.17062263, "token_acc": 0.53751489, "grad_norm": 7.82763406, "learning_rate": 0.0001, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.034914, "epoch": 0.05002084, "global_step/max_steps": "30/599", "percentage": "5.01%", "elapsed_time": "14m 14s", "remaining_time": "4h 30m 13s"}
|
| 8 |
+
{"loss": 2.23156967, "token_acc": 0.53125, "grad_norm": 8.59937401, "learning_rate": 9.998e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035133, "epoch": 0.05835765, "global_step/max_steps": "35/599", "percentage": "5.84%", "elapsed_time": "16m 31s", "remaining_time": "4h 26m 21s"}
|
| 9 |
+
{"loss": 2.22512169, "token_acc": 0.53078897, "grad_norm": 3.64781086, "learning_rate": 9.992e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035348, "epoch": 0.06669446, "global_step/max_steps": "40/599", "percentage": "6.68%", "elapsed_time": "18m 47s", "remaining_time": "4h 22m 32s"}
|
| 10 |
+
{"loss": 2.14616623, "token_acc": 0.50327082, "grad_norm": 6.07080257, "learning_rate": 9.983e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035522, "epoch": 0.07503126, "global_step/max_steps": "45/599", "percentage": "7.51%", "elapsed_time": "21m 2s", "remaining_time": "4h 19m 1s"}
|
| 11 |
+
{"loss": 2.15192604, "token_acc": 0.51821862, "grad_norm": 19.98772409, "learning_rate": 9.97e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035623, "epoch": 0.08336807, "global_step/max_steps": "50/599", "percentage": "8.35%", "elapsed_time": "23m 19s", "remaining_time": "4h 16m 2s"}
|
| 12 |
+
{"loss": 2.15677853, "token_acc": 0.59206174, "grad_norm": 4.03569086, "learning_rate": 9.952e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035682, "epoch": 0.09170488, "global_step/max_steps": "55/599", "percentage": "9.18%", "elapsed_time": "25m 36s", "remaining_time": "4h 13m 21s"}
|
| 13 |
+
{"loss": 2.22119522, "token_acc": 0.59705882, "grad_norm": 5.59545749, "learning_rate": 9.932e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035766, "epoch": 0.10004168, "global_step/max_steps": "60/599", "percentage": "10.02%", "elapsed_time": "27m 53s", "remaining_time": "4h 10m 30s"}
|
| 14 |
+
{"loss": 2.14701252, "token_acc": 0.53703704, "grad_norm": 4.25118914, "learning_rate": 9.907e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035799, "epoch": 0.10837849, "global_step/max_steps": "65/599", "percentage": "10.85%", "elapsed_time": "30m 11s", "remaining_time": "4h 8m 0s"}
|
| 15 |
+
{"loss": 2.02211628, "token_acc": 0.52875258, "grad_norm": 3.88414414, "learning_rate": 9.879e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035828, "epoch": 0.1167153, "global_step/max_steps": "70/599", "percentage": "11.69%", "elapsed_time": "32m 29s", "remaining_time": "4h 5m 31s"}
|
| 16 |
+
{"loss": 2.15577812, "token_acc": 0.59159664, "grad_norm": 2.21097414, "learning_rate": 9.846e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035857, "epoch": 0.12505211, "global_step/max_steps": "75/599", "percentage": "12.52%", "elapsed_time": "34m 47s", "remaining_time": "4h 3m 2s"}
|
| 17 |
+
{"loss": 2.18747063, "token_acc": 0.6507301, "grad_norm": 1.77720943, "learning_rate": 9.811e-05, "memory(GiB)": 58.65, "train_speed(iter/s)": 0.035894, "epoch": 0.13338891, "global_step/max_steps": "80/599", "percentage": "13.36%", "elapsed_time": "37m 4s", "remaining_time": "4h 0m 30s"}
|
| 18 |
+
{"loss": 2.12430573, "token_acc": 0.60082305, "grad_norm": 1.76545345, "learning_rate": 9.771e-05, "memory(GiB)": 58.65, "train_speed(iter/s)": 0.035946, "epoch": 0.14172572, "global_step/max_steps": "85/599", "percentage": "14.19%", "elapsed_time": "39m 20s", "remaining_time": "3h 57m 52s"}
|
| 19 |
+
{"loss": 1.93259163, "token_acc": 0.63518519, "grad_norm": 3.60822856, "learning_rate": 9.728e-05, "memory(GiB)": 58.65, "train_speed(iter/s)": 0.035958, "epoch": 0.15006253, "global_step/max_steps": "90/599", "percentage": "15.03%", "elapsed_time": "41m 38s", "remaining_time": "3h 55m 30s"}
|
| 20 |
+
{"loss": 2.16775932, "token_acc": 0.58236559, "grad_norm": 5.96617132, "learning_rate": 9.681e-05, "memory(GiB)": 58.65, "train_speed(iter/s)": 0.035985, "epoch": 0.15839933, "global_step/max_steps": "95/599", "percentage": "15.86%", "elapsed_time": "43m 55s", "remaining_time": "3h 53m 2s"}
|
| 21 |
+
{"loss": 2.03528709, "token_acc": 0.78202995, "grad_norm": 2.31332738, "learning_rate": 9.631e-05, "memory(GiB)": 72.16, "train_speed(iter/s)": 0.036004, "epoch": 0.16673614, "global_step/max_steps": "100/599", "percentage": "16.69%", "elapsed_time": "46m 12s", "remaining_time": "3h 50m 37s"}
|
| 22 |
+
{"eval_loss": 1.95593798, "eval_token_acc": 0.58371004, "eval_runtime": 53.5583, "eval_samples_per_second": 7.226, "eval_steps_per_second": 0.467, "epoch": 0.16673614, "global_step/max_steps": "100/599", "percentage": "16.69%", "elapsed_time": "47m 6s", "remaining_time": "3h 55m 4s"}
|
| 23 |
+
{"loss": 2.20795937, "token_acc": 0.58827404, "grad_norm": 2.86888505, "learning_rate": 9.577e-05, "memory(GiB)": 72.16, "train_speed(iter/s)": 0.035314, "epoch": 0.17507295, "global_step/max_steps": "105/599", "percentage": "17.53%", "elapsed_time": "49m 28s", "remaining_time": "3h 52m 47s"}
|
| 24 |
+
{"loss": 1.96343536, "token_acc": 0.57200484, "grad_norm": 4.27621906, "learning_rate": 9.52e-05, "memory(GiB)": 72.16, "train_speed(iter/s)": 0.035369, "epoch": 0.18340975, "global_step/max_steps": "110/599", "percentage": "18.36%", "elapsed_time": "51m 45s", "remaining_time": "3h 50m 5s"}
|
| 25 |
+
{"loss": 2.15122967, "token_acc": 0.58753709, "grad_norm": 3.76374317, "learning_rate": 9.459e-05, "memory(GiB)": 72.16, "train_speed(iter/s)": 0.035408, "epoch": 0.19174656, "global_step/max_steps": "115/599", "percentage": "19.20%", "elapsed_time": "54m 3s", "remaining_time": "3h 47m 30s"}
|
| 26 |
+
{"loss": 2.0171051, "token_acc": 0.57622739, "grad_norm": 5.04689722, "learning_rate": 9.395e-05, "memory(GiB)": 72.16, "train_speed(iter/s)": 0.035439, "epoch": 0.20008337, "global_step/max_steps": "120/599", "percentage": "20.03%", "elapsed_time": "56m 21s", "remaining_time": "3h 44m 58s"}
|
| 27 |
+
{"loss": 1.88025742, "token_acc": 0.62413122, "grad_norm": 7.21709226, "learning_rate": 9.328e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035472, "epoch": 0.20842018, "global_step/max_steps": "125/599", "percentage": "20.87%", "elapsed_time": "58m 39s", "remaining_time": "3h 42m 25s"}
|
| 28 |
+
{"loss": 2.0608345, "token_acc": 0.51792279, "grad_norm": 5.36767882, "learning_rate": 9.257e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035519, "epoch": 0.21675698, "global_step/max_steps": "130/599", "percentage": "21.70%", "elapsed_time": "1h 0m 55s", "remaining_time": "3h 39m 48s"}
|
| 29 |
+
{"loss": 2.08950043, "token_acc": 0.57675575, "grad_norm": 3.99936566, "learning_rate": 9.183e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035567, "epoch": 0.22509379, "global_step/max_steps": "135/599", "percentage": "22.54%", "elapsed_time": "1h 3m 11s", "remaining_time": "3h 37m 10s"}
|
| 30 |
+
{"loss": 2.14802437, "token_acc": 0.53691275, "grad_norm": 6.05723266, "learning_rate": 9.106e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035608, "epoch": 0.2334306, "global_step/max_steps": "140/599", "percentage": "23.37%", "elapsed_time": "1h 5m 27s", "remaining_time": "3h 34m 35s"}
|
| 31 |
+
{"loss": 2.25597878, "token_acc": 0.57985512, "grad_norm": 4.2979501, "learning_rate": 9.026e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035642, "epoch": 0.2417674, "global_step/max_steps": "145/599", "percentage": "24.21%", "elapsed_time": "1h 7m 43s", "remaining_time": "3h 32m 3s"}
|
| 32 |
+
{"loss": 1.95336342, "token_acc": 0.58076075, "grad_norm": 4.24466868, "learning_rate": 8.942e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035664, "epoch": 0.25010421, "global_step/max_steps": "150/599", "percentage": "25.04%", "elapsed_time": "1h 10m 1s", "remaining_time": "3h 29m 36s"}
|
| 33 |
+
{"loss": 1.99843884, "token_acc": 0.58876601, "grad_norm": 3.18718836, "learning_rate": 8.856e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.03569, "epoch": 0.25844102, "global_step/max_steps": "155/599", "percentage": "25.88%", "elapsed_time": "1h 12m 18s", "remaining_time": "3h 27m 7s"}
|
| 34 |
+
{"loss": 2.03001747, "token_acc": 0.53406928, "grad_norm": 5.62009813, "learning_rate": 8.766e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035723, "epoch": 0.26677782, "global_step/max_steps": "160/599", "percentage": "26.71%", "elapsed_time": "1h 14m 34s", "remaining_time": "3h 24m 36s"}
|
| 35 |
+
{"loss": 2.01011906, "token_acc": 0.53470681, "grad_norm": 2.92702902, "learning_rate": 8.674e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035744, "epoch": 0.27511463, "global_step/max_steps": "165/599", "percentage": "27.55%", "elapsed_time": "1h 16m 51s", "remaining_time": "3h 22m 10s"}
|
| 36 |
+
{"loss": 1.993046, "token_acc": 0.53557944, "grad_norm": 2.4715905, "learning_rate": 8.579e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035775, "epoch": 0.28345144, "global_step/max_steps": "170/599", "percentage": "28.38%", "elapsed_time": "1h 19m 7s", "remaining_time": "3h 19m 40s"}
|
| 37 |
+
{"loss": 2.10636978, "token_acc": 0.54442118, "grad_norm": 2.17237348, "learning_rate": 8.481e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035788, "epoch": 0.29178825, "global_step/max_steps": "175/599", "percentage": "29.22%", "elapsed_time": "1h 21m 25s", "remaining_time": "3h 17m 16s"}
|
| 38 |
+
{"loss": 1.99015293, "token_acc": 0.5902404, "grad_norm": 15.5090729, "learning_rate": 8.381e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035808, "epoch": 0.30012505, "global_step/max_steps": "180/599", "percentage": "30.05%", "elapsed_time": "1h 23m 42s", "remaining_time": "3h 14m 51s"}
|
| 39 |
+
{"loss": 2.02443218, "token_acc": 0.60469799, "grad_norm": 3.2069182, "learning_rate": 8.278e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035814, "epoch": 0.30846186, "global_step/max_steps": "185/599", "percentage": "30.88%", "elapsed_time": "1h 26m 1s", "remaining_time": "3h 12m 29s"}
|
| 40 |
+
{"loss": 1.9969101, "token_acc": 0.72103376, "grad_norm": 2.68117367, "learning_rate": 8.173e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035815, "epoch": 0.31679867, "global_step/max_steps": "190/599", "percentage": "31.72%", "elapsed_time": "1h 28m 20s", "remaining_time": "3h 10m 10s"}
|
| 41 |
+
{"loss": 1.93723469, "token_acc": 0.5546875, "grad_norm": 3.84616063, "learning_rate": 8.065e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035833, "epoch": 0.32513547, "global_step/max_steps": "195/599", "percentage": "32.55%", "elapsed_time": "1h 30m 37s", "remaining_time": "3h 7m 45s"}
|
| 42 |
+
{"loss": 2.12872143, "token_acc": 0.55633047, "grad_norm": 3.66668704, "learning_rate": 7.955e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035849, "epoch": 0.33347228, "global_step/max_steps": "200/599", "percentage": "33.39%", "elapsed_time": "1h 32m 54s", "remaining_time": "3h 5m 21s"}
|
| 43 |
+
{"eval_loss": 1.81220973, "eval_token_acc": 0.59998356, "eval_runtime": 50.701, "eval_samples_per_second": 7.633, "eval_steps_per_second": 0.493, "epoch": 0.33347228, "global_step/max_steps": "200/599", "percentage": "33.39%", "elapsed_time": "1h 33m 45s", "remaining_time": "3h 7m 2s"}
|
| 44 |
+
{"loss": 1.89449692, "token_acc": 0.63592233, "grad_norm": 3.59642299, "learning_rate": 7.842e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035526, "epoch": 0.34180909, "global_step/max_steps": "205/599", "percentage": "34.22%", "elapsed_time": "1h 36m 5s", "remaining_time": "3h 4m 41s"}
|
| 45 |
+
{"loss": 1.98244629, "token_acc": 0.57714084, "grad_norm": 2.58020778, "learning_rate": 7.727e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035546, "epoch": 0.35014589, "global_step/max_steps": "210/599", "percentage": "35.06%", "elapsed_time": "1h 38m 23s", "remaining_time": "3h 2m 15s"}
|
| 46 |
+
{"loss": 2.01169739, "token_acc": 0.61020678, "grad_norm": 2.82787747, "learning_rate": 7.611e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035564, "epoch": 0.3584827, "global_step/max_steps": "215/599", "percentage": "35.89%", "elapsed_time": "1h 40m 41s", "remaining_time": "2h 59m 49s"}
|
| 47 |
+
{"loss": 1.95843315, "token_acc": 0.56832125, "grad_norm": 5.85552667, "learning_rate": 7.492e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035583, "epoch": 0.36681951, "global_step/max_steps": "220/599", "percentage": "36.73%", "elapsed_time": "1h 42m 58s", "remaining_time": "2h 57m 23s"}
|
| 48 |
+
{"loss": 2.03755455, "token_acc": 0.55751634, "grad_norm": 4.01178584, "learning_rate": 7.371e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035599, "epoch": 0.37515632, "global_step/max_steps": "225/599", "percentage": "37.56%", "elapsed_time": "1h 45m 15s", "remaining_time": "2h 54m 58s"}
|
| 49 |
+
{"loss": 1.90745888, "token_acc": 0.57194389, "grad_norm": 2.34966046, "learning_rate": 7.249e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035618, "epoch": 0.38349312, "global_step/max_steps": "230/599", "percentage": "38.40%", "elapsed_time": "1h 47m 32s", "remaining_time": "2h 52m 32s"}
|
| 50 |
+
{"loss": 2.06616001, "token_acc": 0.58405754, "grad_norm": 2.40614038, "learning_rate": 7.125e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035636, "epoch": 0.39182993, "global_step/max_steps": "235/599", "percentage": "39.23%", "elapsed_time": "1h 49m 50s", "remaining_time": "2h 50m 7s"}
|
| 51 |
+
{"loss": 2.06087761, "token_acc": 0.64652372, "grad_norm": 2.47418061, "learning_rate": 6.999e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035654, "epoch": 0.40016674, "global_step/max_steps": "240/599", "percentage": "40.07%", "elapsed_time": "1h 52m 6s", "remaining_time": "2h 47m 42s"}
|
| 52 |
+
{"loss": 1.9380188, "token_acc": 0.63713981, "grad_norm": 3.45375455, "learning_rate": 6.872e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035664, "epoch": 0.40850354, "global_step/max_steps": "245/599", "percentage": "40.90%", "elapsed_time": "1h 54m 25s", "remaining_time": "2h 45m 19s"}
|
| 53 |
+
{"loss": 1.78654785, "token_acc": 0.57567917, "grad_norm": 16.72496162, "learning_rate": 6.743e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035675, "epoch": 0.41684035, "global_step/max_steps": "250/599", "percentage": "41.74%", "elapsed_time": "1h 56m 43s", "remaining_time": "2h 42m 56s"}
|
| 54 |
+
{"loss": 2.07052708, "token_acc": 0.62034856, "grad_norm": 3.19320834, "learning_rate": 6.613e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035687, "epoch": 0.42517716, "global_step/max_steps": "255/599", "percentage": "42.57%", "elapsed_time": "1h 59m 1s", "remaining_time": "2h 40m 33s"}
|
| 55 |
+
{"loss": 1.90504932, "token_acc": 0.59297424, "grad_norm": 7.31620146, "learning_rate": 6.482e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035706, "epoch": 0.43351396, "global_step/max_steps": "260/599", "percentage": "43.41%", "elapsed_time": "2h 1m 17s", "remaining_time": "2h 38m 8s"}
|
| 56 |
+
{"loss": 1.88804798, "token_acc": 0.57713583, "grad_norm": 2.77207648, "learning_rate": 6.35e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035715, "epoch": 0.44185077, "global_step/max_steps": "265/599", "percentage": "44.24%", "elapsed_time": "2h 3m 35s", "remaining_time": "2h 35m 46s"}
|
| 57 |
+
{"loss": 1.86641426, "token_acc": 0.66966967, "grad_norm": 4.06519351, "learning_rate": 6.216e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035727, "epoch": 0.45018758, "global_step/max_steps": "270/599", "percentage": "45.08%", "elapsed_time": "2h 5m 52s", "remaining_time": "2h 33m 23s"}
|
| 58 |
+
{"loss": 1.98609428, "token_acc": 0.55497382, "grad_norm": 5.37897733, "learning_rate": 6.082e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035743, "epoch": 0.45852439, "global_step/max_steps": "275/599", "percentage": "45.91%", "elapsed_time": "2h 8m 9s", "remaining_time": "2h 30m 59s"}
|
| 59 |
+
{"loss": 1.95925713, "token_acc": 0.55473555, "grad_norm": 4.12258515, "learning_rate": 5.947e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035764, "epoch": 0.46686119, "global_step/max_steps": "280/599", "percentage": "46.74%", "elapsed_time": "2h 10m 24s", "remaining_time": "2h 28m 34s"}
|
| 60 |
+
{"loss": 2.08585167, "token_acc": 0.5434947, "grad_norm": 20.98589422, "learning_rate": 5.811e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035776, "epoch": 0.475198, "global_step/max_steps": "285/599", "percentage": "47.58%", "elapsed_time": "2h 12m 41s", "remaining_time": "2h 26m 11s"}
|
| 61 |
+
{"loss": 1.95586853, "token_acc": 0.53515398, "grad_norm": 8.17774583, "learning_rate": 5.674e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.03579, "epoch": 0.48353481, "global_step/max_steps": "290/599", "percentage": "48.41%", "elapsed_time": "2h 14m 58s", "remaining_time": "2h 23m 49s"}
|
| 62 |
+
{"loss": 2.0921154, "token_acc": 0.56735905, "grad_norm": 4.08355865, "learning_rate": 5.537e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035795, "epoch": 0.49187161, "global_step/max_steps": "295/599", "percentage": "49.25%", "elapsed_time": "2h 17m 16s", "remaining_time": "2h 21m 28s"}
|
| 63 |
+
{"loss": 1.86736107, "token_acc": 0.53051643, "grad_norm": 12.19399347, "learning_rate": 5.4e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035811, "epoch": 0.50020842, "global_step/max_steps": "300/599", "percentage": "50.08%", "elapsed_time": "2h 19m 32s", "remaining_time": "2h 19m 4s"}
|
| 64 |
+
{"eval_loss": 1.74253964, "eval_token_acc": 0.61214761, "eval_runtime": 50.716, "eval_samples_per_second": 7.631, "eval_steps_per_second": 0.493, "epoch": 0.50020842, "global_step/max_steps": "300/599", "percentage": "50.08%", "elapsed_time": "2h 20m 23s", "remaining_time": "2h 19m 55s"}
|
| 65 |
+
{"loss": 1.90042572, "token_acc": 0.62614587, "grad_norm": 3.64887895, "learning_rate": 5.262e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035585, "epoch": 0.50854523, "global_step/max_steps": "305/599", "percentage": "50.92%", "elapsed_time": "2h 22m 46s", "remaining_time": "2h 17m 37s"}
|
| 66 |
+
{"loss": 1.90457134, "token_acc": 0.64371859, "grad_norm": 2.86243308, "learning_rate": 5.124e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035598, "epoch": 0.51688203, "global_step/max_steps": "310/599", "percentage": "51.75%", "elapsed_time": "2h 25m 3s", "remaining_time": "2h 15m 14s"}
|
| 67 |
+
{"loss": 1.81422691, "token_acc": 0.61766836, "grad_norm": 1.84524964, "learning_rate": 4.986e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035606, "epoch": 0.52521884, "global_step/max_steps": "315/599", "percentage": "52.59%", "elapsed_time": "2h 27m 22s", "remaining_time": "2h 12m 52s"}
|
| 68 |
+
{"loss": 1.82093945, "token_acc": 0.60661097, "grad_norm": 5.43001183, "learning_rate": 4.848e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035613, "epoch": 0.53355565, "global_step/max_steps": "320/599", "percentage": "53.42%", "elapsed_time": "2h 29m 41s", "remaining_time": "2h 10m 30s"}
|
| 69 |
+
{"loss": 1.93497009, "token_acc": 0.63563941, "grad_norm": 2.60076921, "learning_rate": 4.71e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035624, "epoch": 0.54189246, "global_step/max_steps": "325/599", "percentage": "54.26%", "elapsed_time": "2h 31m 58s", "remaining_time": "2h 8m 7s"}
|
| 70 |
+
{"loss": 1.86749039, "token_acc": 0.58096173, "grad_norm": 3.39273646, "learning_rate": 4.573e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035636, "epoch": 0.55022926, "global_step/max_steps": "330/599", "percentage": "55.09%", "elapsed_time": "2h 34m 15s", "remaining_time": "2h 5m 45s"}
|
| 71 |
+
{"loss": 1.95322495, "token_acc": 0.60828987, "grad_norm": 4.34767956, "learning_rate": 4.435e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.03565, "epoch": 0.55856607, "global_step/max_steps": "335/599", "percentage": "55.93%", "elapsed_time": "2h 36m 32s", "remaining_time": "2h 3m 21s"}
|
| 72 |
+
{"loss": 1.84671173, "token_acc": 0.54477612, "grad_norm": 14.79752307, "learning_rate": 4.298e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035659, "epoch": 0.56690288, "global_step/max_steps": "340/599", "percentage": "56.76%", "elapsed_time": "2h 38m 50s", "remaining_time": "2h 0m 59s"}
|
| 73 |
+
{"loss": 1.88678093, "token_acc": 0.64988172, "grad_norm": 9.17467747, "learning_rate": 4.162e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035671, "epoch": 0.57523968, "global_step/max_steps": "345/599", "percentage": "57.60%", "elapsed_time": "2h 41m 7s", "remaining_time": "1h 58m 37s"}
|
| 74 |
+
{"loss": 1.96717834, "token_acc": 0.56567222, "grad_norm": 11.5170952, "learning_rate": 4.026e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035684, "epoch": 0.58357649, "global_step/max_steps": "350/599", "percentage": "58.43%", "elapsed_time": "2h 43m 23s", "remaining_time": "1h 56m 14s"}
|
| 75 |
+
{"loss": 2.0355814, "token_acc": 0.56464271, "grad_norm": 2.09477749, "learning_rate": 3.891e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035694, "epoch": 0.5919133, "global_step/max_steps": "355/599", "percentage": "59.27%", "elapsed_time": "2h 45m 41s", "remaining_time": "1h 53m 52s"}
|
| 76 |
+
{"loss": 2.0212513, "token_acc": 0.60336341, "grad_norm": 2.26724345, "learning_rate": 3.757e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035702, "epoch": 0.6002501, "global_step/max_steps": "360/599", "percentage": "60.10%", "elapsed_time": "2h 47m 59s", "remaining_time": "1h 51m 31s"}
|
| 77 |
+
{"loss": 1.88847866, "token_acc": 0.55277108, "grad_norm": 3.34766286, "learning_rate": 3.624e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035708, "epoch": 0.60858691, "global_step/max_steps": "365/599", "percentage": "60.93%", "elapsed_time": "2h 50m 17s", "remaining_time": "1h 49m 10s"}
|
| 78 |
+
{"loss": 2.12649918, "token_acc": 0.57085669, "grad_norm": 6.16368203, "learning_rate": 3.492e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035714, "epoch": 0.61692372, "global_step/max_steps": "370/599", "percentage": "61.77%", "elapsed_time": "2h 52m 35s", "remaining_time": "1h 46m 49s"}
|
| 79 |
+
{"loss": 1.88020248, "token_acc": 0.55474806, "grad_norm": 1.67775938, "learning_rate": 3.361e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035723, "epoch": 0.62526053, "global_step/max_steps": "375/599", "percentage": "62.60%", "elapsed_time": "2h 54m 53s", "remaining_time": "1h 44m 27s"}
|
| 80 |
+
{"loss": 1.86473885, "token_acc": 0.5930644, "grad_norm": 2.84408996, "learning_rate": 3.231e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035732, "epoch": 0.63359733, "global_step/max_steps": "380/599", "percentage": "63.44%", "elapsed_time": "2h 57m 10s", "remaining_time": "1h 42m 6s"}
|
| 81 |
+
{"loss": 1.81306381, "token_acc": 0.56542461, "grad_norm": 37.91166933, "learning_rate": 3.103e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035741, "epoch": 0.64193414, "global_step/max_steps": "385/599", "percentage": "64.27%", "elapsed_time": "2h 59m 27s", "remaining_time": "1h 39m 45s"}
|
| 82 |
+
{"loss": 2.0649231, "token_acc": 0.61083172, "grad_norm": 4.02208103, "learning_rate": 2.976e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035754, "epoch": 0.65027095, "global_step/max_steps": "390/599", "percentage": "65.11%", "elapsed_time": "3h 1m 43s", "remaining_time": "1h 37m 23s"}
|
| 83 |
+
{"loss": 1.96494999, "token_acc": 0.55705394, "grad_norm": 2.30736739, "learning_rate": 2.85e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.03576, "epoch": 0.65860775, "global_step/max_steps": "395/599", "percentage": "65.94%", "elapsed_time": "3h 4m 1s", "remaining_time": "1h 35m 2s"}
|
| 84 |
+
{"loss": 1.92861328, "token_acc": 0.59117305, "grad_norm": 4.20454711, "learning_rate": 2.726e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035772, "epoch": 0.66694456, "global_step/max_steps": "400/599", "percentage": "66.78%", "elapsed_time": "3h 6m 17s", "remaining_time": "1h 32m 40s"}
|
| 85 |
+
{"eval_loss": 1.77997983, "eval_token_acc": 0.61001068, "eval_runtime": 50.5835, "eval_samples_per_second": 7.651, "eval_steps_per_second": 0.494, "epoch": 0.66694456, "global_step/max_steps": "400/599", "percentage": "66.78%", "elapsed_time": "3h 7m 8s", "remaining_time": "1h 33m 5s"}
|
| 86 |
+
{"loss": 1.98755207, "token_acc": 0.59045545, "grad_norm": 2.65925405, "learning_rate": 2.604e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.03561, "epoch": 0.67528137, "global_step/max_steps": "405/599", "percentage": "67.61%", "elapsed_time": "3h 9m 28s", "remaining_time": "1h 30m 45s"}
|
| 87 |
+
{"loss": 1.70803642, "token_acc": 0.63452787, "grad_norm": 14.66284129, "learning_rate": 2.484e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035622, "epoch": 0.68361817, "global_step/max_steps": "410/599", "percentage": "68.45%", "elapsed_time": "3h 11m 45s", "remaining_time": "1h 28m 23s"}
|
| 88 |
+
{"loss": 1.92990341, "token_acc": 0.53832599, "grad_norm": 2.0901369, "learning_rate": 2.366e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035634, "epoch": 0.69195498, "global_step/max_steps": "415/599", "percentage": "69.28%", "elapsed_time": "3h 14m 1s", "remaining_time": "1h 26m 1s"}
|
| 89 |
+
{"loss": 2.00041428, "token_acc": 0.56440471, "grad_norm": 2.65807448, "learning_rate": 2.249e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035642, "epoch": 0.70029179, "global_step/max_steps": "420/599", "percentage": "70.12%", "elapsed_time": "3h 16m 19s", "remaining_time": "1h 23m 40s"}
|
| 90 |
+
{"loss": 1.93645687, "token_acc": 0.60483871, "grad_norm": 3.86582497, "learning_rate": 2.135e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.03565, "epoch": 0.7086286, "global_step/max_steps": "425/599", "percentage": "70.95%", "elapsed_time": "3h 18m 36s", "remaining_time": "1h 21m 18s"}
|
| 91 |
+
{"loss": 1.92627335, "token_acc": 0.55910781, "grad_norm": 17.0052904, "learning_rate": 2.023e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035661, "epoch": 0.7169654, "global_step/max_steps": "430/599", "percentage": "71.79%", "elapsed_time": "3h 20m 53s", "remaining_time": "1h 18m 57s"}
|
| 92 |
+
{"loss": 2.01308746, "token_acc": 0.5676373, "grad_norm": 2.74624394, "learning_rate": 1.913e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035672, "epoch": 0.72530221, "global_step/max_steps": "435/599", "percentage": "72.62%", "elapsed_time": "3h 23m 10s", "remaining_time": "1h 16m 35s"}
|
| 93 |
+
{"loss": 1.886693, "token_acc": 0.5958948, "grad_norm": 3.92977038, "learning_rate": 1.806e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035681, "epoch": 0.73363902, "global_step/max_steps": "440/599", "percentage": "73.46%", "elapsed_time": "3h 25m 27s", "remaining_time": "1h 14m 14s"}
|
| 94 |
+
{"loss": 1.88219833, "token_acc": 0.59438147, "grad_norm": 4.74645859, "learning_rate": 1.701e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035688, "epoch": 0.74197582, "global_step/max_steps": "445/599", "percentage": "74.29%", "elapsed_time": "3h 27m 44s", "remaining_time": "1h 11m 53s"}
|
| 95 |
+
{"loss": 1.84777222, "token_acc": 0.63045832, "grad_norm": 4.13088347, "learning_rate": 1.599e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035697, "epoch": 0.75031263, "global_step/max_steps": "450/599", "percentage": "75.13%", "elapsed_time": "3h 30m 1s", "remaining_time": "1h 9m 32s"}
|
| 96 |
+
{"loss": 1.83291245, "token_acc": 0.6356188, "grad_norm": 9.21619347, "learning_rate": 1.499e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035704, "epoch": 0.75864944, "global_step/max_steps": "455/599", "percentage": "75.96%", "elapsed_time": "3h 32m 19s", "remaining_time": "1h 7m 11s"}
|
| 97 |
+
{"loss": 1.89644394, "token_acc": 0.56544503, "grad_norm": 5.29758439, "learning_rate": 1.402e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035715, "epoch": 0.76698624, "global_step/max_steps": "460/599", "percentage": "76.79%", "elapsed_time": "3h 34m 35s", "remaining_time": "1h 4m 50s"}
|
| 98 |
+
{"loss": 1.95029392, "token_acc": 0.55799914, "grad_norm": 2.6876715, "learning_rate": 1.307e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035727, "epoch": 0.77532305, "global_step/max_steps": "465/599", "percentage": "77.63%", "elapsed_time": "3h 36m 50s", "remaining_time": "1h 2m 29s"}
|
| 99 |
+
{"loss": 1.71820068, "token_acc": 0.59023207, "grad_norm": 3.76722879, "learning_rate": 1.216e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035733, "epoch": 0.78365986, "global_step/max_steps": "470/599", "percentage": "78.46%", "elapsed_time": "3h 39m 8s", "remaining_time": "1h 0m 8s"}
|
| 100 |
+
{"loss": 1.86228752, "token_acc": 0.57448594, "grad_norm": 4.58212504, "learning_rate": 1.127e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035739, "epoch": 0.79199667, "global_step/max_steps": "475/599", "percentage": "79.30%", "elapsed_time": "3h 41m 26s", "remaining_time": "57m 48s"}
|
| 101 |
+
{"loss": 1.86822872, "token_acc": 0.57149511, "grad_norm": 3.51017376, "learning_rate": 1.041e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035748, "epoch": 0.80033347, "global_step/max_steps": "480/599", "percentage": "80.13%", "elapsed_time": "3h 43m 43s", "remaining_time": "55m 27s"}
|
| 102 |
+
{"loss": 1.8879879, "token_acc": 0.65646158, "grad_norm": 2.41470985, "learning_rate": 9.58e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035756, "epoch": 0.80867028, "global_step/max_steps": "485/599", "percentage": "80.97%", "elapsed_time": "3h 45m 59s", "remaining_time": "53m 7s"}
|
| 103 |
+
{"loss": 1.86057129, "token_acc": 0.61789159, "grad_norm": 3.81029679, "learning_rate": 8.78e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035763, "epoch": 0.81700709, "global_step/max_steps": "490/599", "percentage": "81.80%", "elapsed_time": "3h 48m 16s", "remaining_time": "50m 46s"}
|
| 104 |
+
{"loss": 1.86033554, "token_acc": 0.58190913, "grad_norm": 5.26561502, "learning_rate": 8.02e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035772, "epoch": 0.82534389, "global_step/max_steps": "495/599", "percentage": "82.64%", "elapsed_time": "3h 50m 33s", "remaining_time": "48m 26s"}
|
| 105 |
+
{"loss": 1.69268742, "token_acc": 0.62072539, "grad_norm": 2.50257714, "learning_rate": 7.29e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035779, "epoch": 0.8336807, "global_step/max_steps": "500/599", "percentage": "83.47%", "elapsed_time": "3h 52m 50s", "remaining_time": "46m 6s"}
|
| 106 |
+
{"eval_loss": 1.65794051, "eval_token_acc": 0.62316101, "eval_runtime": 50.7583, "eval_samples_per_second": 7.624, "eval_steps_per_second": 0.493, "epoch": 0.8336807, "global_step/max_steps": "500/599", "percentage": "83.47%", "elapsed_time": "3h 53m 41s", "remaining_time": "46m 16s"}
|
| 107 |
+
{"loss": 1.80246735, "token_acc": 0.65521797, "grad_norm": 7.05517381, "learning_rate": 6.58e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035646, "epoch": 0.84201751, "global_step/max_steps": "505/599", "percentage": "84.31%", "elapsed_time": "3h 56m 2s", "remaining_time": "43m 56s"}
|
| 108 |
+
{"loss": 1.69859772, "token_acc": 0.61898839, "grad_norm": 3.46322993, "learning_rate": 5.92e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035649, "epoch": 0.85035431, "global_step/max_steps": "510/599", "percentage": "85.14%", "elapsed_time": "3h 58m 21s", "remaining_time": "41m 35s"}
|
| 109 |
+
{"loss": 1.81163177, "token_acc": 0.71894677, "grad_norm": 3.721785, "learning_rate": 5.28e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035656, "epoch": 0.85869112, "global_step/max_steps": "515/599", "percentage": "85.98%", "elapsed_time": "4h 0m 38s", "remaining_time": "39m 15s"}
|
| 110 |
+
{"loss": 1.65748501, "token_acc": 0.63264605, "grad_norm": 3.69513987, "learning_rate": 4.68e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035663, "epoch": 0.86702793, "global_step/max_steps": "520/599", "percentage": "86.81%", "elapsed_time": "4h 2m 56s", "remaining_time": "36m 54s"}
|
| 111 |
+
{"loss": 1.73229408, "token_acc": 0.57548387, "grad_norm": 1.83124558, "learning_rate": 4.12e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035668, "epoch": 0.87536474, "global_step/max_steps": "525/599", "percentage": "87.65%", "elapsed_time": "4h 5m 14s", "remaining_time": "34m 34s"}
|
| 112 |
+
{"loss": 1.85206718, "token_acc": 0.68367572, "grad_norm": 2.15107697, "learning_rate": 3.58e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035678, "epoch": 0.88370154, "global_step/max_steps": "530/599", "percentage": "88.48%", "elapsed_time": "4h 7m 30s", "remaining_time": "32m 13s"}
|
| 113 |
+
{"loss": 1.84878998, "token_acc": 0.53234153, "grad_norm": 2.92427111, "learning_rate": 3.09e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035687, "epoch": 0.89203835, "global_step/max_steps": "535/599", "percentage": "89.32%", "elapsed_time": "4h 9m 47s", "remaining_time": "29m 52s"}
|
| 114 |
+
{"loss": 1.77294159, "token_acc": 0.56861134, "grad_norm": 2.70179575, "learning_rate": 2.63e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035697, "epoch": 0.90037516, "global_step/max_steps": "540/599", "percentage": "90.15%", "elapsed_time": "4h 12m 2s", "remaining_time": "27m 32s"}
|
| 115 |
+
{"loss": 1.7561512, "token_acc": 0.58322102, "grad_norm": 2.02804743, "learning_rate": 2.21e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035701, "epoch": 0.90871196, "global_step/max_steps": "545/599", "percentage": "90.98%", "elapsed_time": "4h 14m 21s", "remaining_time": "25m 12s"}
|
| 116 |
+
{"loss": 1.8700079, "token_acc": 0.65771812, "grad_norm": 3.33975858, "learning_rate": 1.82e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035708, "epoch": 0.91704877, "global_step/max_steps": "550/599", "percentage": "91.82%", "elapsed_time": "4h 16m 38s", "remaining_time": "22m 51s"}
|
| 117 |
+
{"loss": 1.76602173, "token_acc": 0.59254144, "grad_norm": 5.51613596, "learning_rate": 1.47e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035718, "epoch": 0.92538558, "global_step/max_steps": "555/599", "percentage": "92.65%", "elapsed_time": "4h 18m 54s", "remaining_time": "20m 31s"}
|
| 118 |
+
{"loss": 1.89835777, "token_acc": 0.56836965, "grad_norm": 4.02406534, "learning_rate": 1.15e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035724, "epoch": 0.93372238, "global_step/max_steps": "560/599", "percentage": "93.49%", "elapsed_time": "4h 21m 11s", "remaining_time": "18m 11s"}
|
| 119 |
+
{"loss": 1.74415741, "token_acc": 0.6148735, "grad_norm": 7.43728024, "learning_rate": 8.8e-07, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.03573, "epoch": 0.94205919, "global_step/max_steps": "565/599", "percentage": "94.32%", "elapsed_time": "4h 23m 28s", "remaining_time": "15m 51s"}
|
| 120 |
+
{"loss": 1.83804817, "token_acc": 0.55752212, "grad_norm": 6.16534565, "learning_rate": 6.4e-07, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035736, "epoch": 0.950396, "global_step/max_steps": "570/599", "percentage": "95.16%", "elapsed_time": "4h 25m 45s", "remaining_time": "13m 31s"}
|
| 121 |
+
{"loss": 1.82576122, "token_acc": 0.59285258, "grad_norm": 3.92818044, "learning_rate": 4.4e-07, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035742, "epoch": 0.95873281, "global_step/max_steps": "575/599", "percentage": "95.99%", "elapsed_time": "4h 28m 3s", "remaining_time": "11m 11s"}
|
| 122 |
+
{"loss": 2.01456604, "token_acc": 0.548, "grad_norm": 4.01536749, "learning_rate": 2.7e-07, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035745, "epoch": 0.96706961, "global_step/max_steps": "580/599", "percentage": "96.83%", "elapsed_time": "4h 30m 21s", "remaining_time": "8m 51s"}
|
| 123 |
+
{"loss": 1.98306122, "token_acc": 0.52448723, "grad_norm": 3.72510366, "learning_rate": 1.5e-07, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035751, "epoch": 0.97540642, "global_step/max_steps": "585/599", "percentage": "97.66%", "elapsed_time": "4h 32m 38s", "remaining_time": "6m 31s"}
|
| 124 |
+
{"loss": 1.6920805, "token_acc": 0.67426059, "grad_norm": 2.33300141, "learning_rate": 6e-08, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035754, "epoch": 0.98374323, "global_step/max_steps": "590/599", "percentage": "98.50%", "elapsed_time": "4h 34m 57s", "remaining_time": "4m 11s"}
|
| 125 |
+
{"loss": 1.83995934, "token_acc": 0.5763773, "grad_norm": 2.24015802, "learning_rate": 1e-08, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035759, "epoch": 0.99208003, "global_step/max_steps": "595/599", "percentage": "99.33%", "elapsed_time": "4h 37m 14s", "remaining_time": "1m 51s"}
|
| 126 |
+
{"eval_loss": 1.63505709, "eval_token_acc": 0.62570888, "eval_runtime": 50.6821, "eval_samples_per_second": 7.636, "eval_steps_per_second": 0.493, "epoch": 0.99874948, "global_step/max_steps": "599/599", "percentage": "100.00%", "elapsed_time": "4h 39m 56s", "remaining_time": "0s"}
|
| 127 |
+
{"train_runtime": 16799.6463, "train_samples_per_second": 2.284, "train_steps_per_second": 0.036, "total_flos": 954945089044480.0, "train_loss": 1.97028835, "epoch": 0.99874948, "global_step/max_steps": "599/599", "percentage": "100.00%", "elapsed_time": "4h 39m 59s", "remaining_time": "0s"}
|
| 128 |
+
{"model_parameter_info": "PeftModelForCausalLM: 3874.3572M Params (788.4186M Trainable [20.3497%]), 0.0024M Buffers.", "last_model_checkpoint": "/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113/checkpoint-599", "best_model_checkpoint": "/cpfs01/shared/llm_ddd/zhangyulong/sa_work/checkpoint/sft1e3/v1-20250508-175113/checkpoint-599", "best_metric": 1.63505709, "global_step": 599, "log_history": [{"loss": 2.5885090827941895, "token_acc": 0.5925233644859813, "grad_norm": 64.40978700124327, "learning_rate": 3.3333333333333333e-06, "memory(GiB)": 21.94, "train_speed(iter/s)": 0.01568, "epoch": 0.0016673614005835765, "step": 1}, {"loss": 2.8374526500701904, "token_acc": 0.5379928315412187, "grad_norm": 42.49983691696958, "learning_rate": 1.6666666666666667e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.028676, "epoch": 0.008336807002917883, "step": 5}, {"loss": 2.485092544555664, "token_acc": 0.503155996393147, "grad_norm": 9.119435129423483, "learning_rate": 3.3333333333333335e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.032119, "epoch": 0.016673614005835766, "step": 10}, {"loss": 2.338067626953125, "token_acc": 0.5031282586027112, "grad_norm": 12.267838186475673, "learning_rate": 5e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.033444, "epoch": 0.02501042100875365, "step": 15}, {"loss": 1.9811298370361328, "token_acc": 0.5953382971835546, "grad_norm": 9.95333802907172, "learning_rate": 6.666666666666667e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.034146, "epoch": 0.03334722801167153, "step": 20}, {"loss": 2.093521499633789, "token_acc": 0.5699672667757774, "grad_norm": 9.427104390244967, "learning_rate": 8.333333333333334e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.034594, "epoch": 0.041684035014589414, "step": 25}, {"loss": 2.170622634887695, "token_acc": 0.537514886859865, "grad_norm": 7.827634062020569, "learning_rate": 0.0001, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.034914, "epoch": 0.0500208420175073, "step": 30}, {"loss": 2.2315696716308593, "token_acc": 0.53125, "grad_norm": 8.599374011882336, "learning_rate": 9.998094856697883e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035133, "epoch": 0.05835764902042518, "step": 35}, {"loss": 2.2251216888427736, "token_acc": 0.5307889672867223, "grad_norm": 3.6478108577238, "learning_rate": 9.992380878619938e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035348, "epoch": 0.06669445602334306, "step": 40}, {"loss": 2.146166229248047, "token_acc": 0.5032708242477104, "grad_norm": 6.070802568263254, "learning_rate": 9.982862420144985e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035522, "epoch": 0.07503126302626094, "step": 45}, {"loss": 2.151926040649414, "token_acc": 0.5182186234817814, "grad_norm": 19.987724090220276, "learning_rate": 9.96954673488399e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035623, "epoch": 0.08336807002917883, "step": 50}, {"loss": 2.1567785263061525, "token_acc": 0.5920617420066152, "grad_norm": 4.035690857782974, "learning_rate": 9.95244397015239e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035682, "epoch": 0.0917048770320967, "step": 55}, {"loss": 2.2211952209472656, "token_acc": 0.5970588235294118, "grad_norm": 5.595457490039321, "learning_rate": 9.931567159237251e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035766, "epoch": 0.1000416840350146, "step": 60}, {"loss": 2.1470125198364256, "token_acc": 0.5370370370370371, "grad_norm": 4.251189137101062, "learning_rate": 9.906932211465173e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035799, "epoch": 0.10837849103793247, "step": 65}, {"loss": 2.022116279602051, "token_acc": 0.5287525803597759, "grad_norm": 3.884144144655882, "learning_rate": 9.87855790007845e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035828, "epoch": 0.11671529804085036, "step": 70}, {"loss": 2.1557781219482424, "token_acc": 0.5915966386554622, "grad_norm": 2.210974142567793, "learning_rate": 9.8464658479288e-05, "memory(GiB)": 39.79, "train_speed(iter/s)": 0.035857, "epoch": 0.12505210504376824, "step": 75}, {"loss": 2.187470626831055, "token_acc": 0.6507300989166274, "grad_norm": 1.7772094345896288, "learning_rate": 9.810680510999504e-05, "memory(GiB)": 58.65, "train_speed(iter/s)": 0.035894, "epoch": 0.13338891204668613, "step": 80}, {"loss": 2.1243057250976562, "token_acc": 0.6008230452674898, "grad_norm": 1.7654534509312796, "learning_rate": 9.771229159768547e-05, "memory(GiB)": 58.65, "train_speed(iter/s)": 0.035946, "epoch": 0.14172571904960402, "step": 85}, {"loss": 1.9325916290283203, "token_acc": 0.6351851851851852, "grad_norm": 3.6082285595666694, "learning_rate": 9.728141858426952e-05, "memory(GiB)": 58.65, "train_speed(iter/s)": 0.035958, "epoch": 0.15006252605252188, "step": 90}, {"loss": 2.167759323120117, "token_acc": 0.5823655913978495, "grad_norm": 5.966171320647249, "learning_rate": 9.681451441968144e-05, "memory(GiB)": 58.65, "train_speed(iter/s)": 0.035985, "epoch": 0.15839933305543977, "step": 95}, {"loss": 2.035287094116211, "token_acc": 0.7820299500831946, "grad_norm": 2.313327380575682, "learning_rate": 9.631193491165797e-05, "memory(GiB)": 72.16, "train_speed(iter/s)": 0.036004, "epoch": 0.16673614005835766, "step": 100}, {"eval_loss": 1.9559379816055298, "eval_token_acc": 0.5837100353414975, "eval_runtime": 53.5583, "eval_samples_per_second": 7.226, "eval_steps_per_second": 0.467, "epoch": 0.16673614005835766, "step": 100}, {"loss": 2.2079593658447267, "token_acc": 0.5882740447957839, "grad_norm": 2.868885045077332, "learning_rate": 9.577406305459251e-05, "memory(GiB)": 72.16, "train_speed(iter/s)": 0.035314, "epoch": 0.17507294706127552, "step": 105}, {"loss": 1.9634353637695312, "token_acc": 0.5720048406615571, "grad_norm": 4.276219061520432, "learning_rate": 9.520130873767141e-05, "memory(GiB)": 72.16, "train_speed(iter/s)": 0.035369, "epoch": 0.1834097540641934, "step": 110}, {"loss": 2.151229667663574, "token_acc": 0.5875370919881305, "grad_norm": 3.763743172552543, "learning_rate": 9.459410843251494e-05, "memory(GiB)": 72.16, "train_speed(iter/s)": 0.035408, "epoch": 0.1917465610671113, "step": 115}, {"loss": 2.0171051025390625, "token_acc": 0.5762273901808785, "grad_norm": 5.046897215873855, "learning_rate": 9.395292486056087e-05, "memory(GiB)": 72.16, "train_speed(iter/s)": 0.035439, "epoch": 0.2000833680700292, "step": 120}, {"loss": 1.8802574157714844, "token_acc": 0.6241312204614957, "grad_norm": 7.217092259193579, "learning_rate": 9.327824664044418e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035472, "epoch": 0.20842017507294705, "step": 125}, {"loss": 2.060834503173828, "token_acc": 0.5179227941176471, "grad_norm": 5.36767881783302, "learning_rate": 9.257058791564174e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035519, "epoch": 0.21675698207586494, "step": 130}, {"loss": 2.0895004272460938, "token_acc": 0.5767557489123679, "grad_norm": 3.999365659434696, "learning_rate": 9.183048796266547e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035567, "epoch": 0.22509378907878283, "step": 135}, {"loss": 2.1480243682861326, "token_acc": 0.5369127516778524, "grad_norm": 6.057232659457883, "learning_rate": 9.105851078010266e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035608, "epoch": 0.23343059608170072, "step": 140}, {"loss": 2.2559787750244142, "token_acc": 0.5798551224560193, "grad_norm": 4.297950099035578, "learning_rate": 9.025524465881683e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035642, "epoch": 0.24176740308461858, "step": 145}, {"loss": 1.9533634185791016, "token_acc": 0.5807607497243661, "grad_norm": 4.244668680625598, "learning_rate": 8.942130173363627e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035664, "epoch": 0.25010421008753647, "step": 150}, {"loss": 1.998438835144043, "token_acc": 0.5887660069848661, "grad_norm": 3.1871883558604828, "learning_rate": 8.855731751687233e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.03569, "epoch": 0.25844101709045436, "step": 155}, {"loss": 2.0300174713134767, "token_acc": 0.5340692805481538, "grad_norm": 5.6200981254634, "learning_rate": 8.766395041402244e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035723, "epoch": 0.26677782409337225, "step": 160}, {"loss": 2.0101190567016602, "token_acc": 0.5347068145800317, "grad_norm": 2.927029018387031, "learning_rate": 8.674188122202756e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035744, "epoch": 0.27511463109629014, "step": 165}, {"loss": 1.9930459976196289, "token_acc": 0.535579436537903, "grad_norm": 2.4715904965940227, "learning_rate": 8.579181261046577e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035775, "epoch": 0.28345143809920803, "step": 170}, {"loss": 2.1063697814941404, "token_acc": 0.5444211785821119, "grad_norm": 2.172373476373027, "learning_rate": 8.48144685860778e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035788, "epoch": 0.29178824510212586, "step": 175}, {"loss": 1.990152931213379, "token_acc": 0.5902404018658055, "grad_norm": 15.509072899121204, "learning_rate": 8.381059394103244e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035808, "epoch": 0.30012505210504375, "step": 180}, {"loss": 2.0244321823120117, "token_acc": 0.6046979865771812, "grad_norm": 3.2069182023630565, "learning_rate": 8.278095368535214e-05, "memory(GiB)": 72.79, "train_speed(iter/s)": 0.035814, "epoch": 0.30846185910796164, "step": 185}, {"loss": 1.9969100952148438, "token_acc": 0.7210337578830222, "grad_norm": 2.681173674568493, "learning_rate": 8.17263324639316e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035815, "epoch": 0.31679866611087953, "step": 190}, {"loss": 1.9372346878051758, "token_acc": 0.5546875, "grad_norm": 3.8461606303421405, "learning_rate": 8.064753395859333e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035833, "epoch": 0.3251354731137974, "step": 195}, {"loss": 2.1287214279174806, "token_acc": 0.5563304721030042, "grad_norm": 3.6666870417160666, "learning_rate": 7.954538027563601e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035849, "epoch": 0.3334722801167153, "step": 200}, {"eval_loss": 1.8122097253799438, "eval_token_acc": 0.5999835620941892, "eval_runtime": 50.701, "eval_samples_per_second": 7.633, "eval_steps_per_second": 0.493, "epoch": 0.3334722801167153, "step": 200}, {"loss": 1.8944969177246094, "token_acc": 0.6359223300970874, "grad_norm": 3.5964229855404164, "learning_rate": 7.842071131934246e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035526, "epoch": 0.3418090871196332, "step": 205}, {"loss": 1.9824462890625, "token_acc": 0.5771408351026185, "grad_norm": 2.580207777434119, "learning_rate": 7.727438415192433e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035546, "epoch": 0.35014589412255104, "step": 210}, {"loss": 2.0116973876953126, "token_acc": 0.6102067751869775, "grad_norm": 2.8278774675330887, "learning_rate": 7.610727234039167e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035564, "epoch": 0.3584827011254689, "step": 215}, {"loss": 1.9584331512451172, "token_acc": 0.5683212493028444, "grad_norm": 5.855526666772205, "learning_rate": 7.492026529084468e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035583, "epoch": 0.3668195081283868, "step": 220}, {"loss": 2.0375545501708983, "token_acc": 0.5575163398692811, "grad_norm": 4.011785835207539, "learning_rate": 7.371426757069537e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035599, "epoch": 0.3751563151313047, "step": 225}, {"loss": 1.9074588775634767, "token_acc": 0.571943887775551, "grad_norm": 2.3496604645389905, "learning_rate": 7.249019821933529e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035618, "epoch": 0.3834931221342226, "step": 230}, {"loss": 2.0661600112915037, "token_acc": 0.5840575367096195, "grad_norm": 2.4061403829778873, "learning_rate": 7.124899004777489e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035636, "epoch": 0.3918299291371405, "step": 235}, {"loss": 2.0608776092529295, "token_acc": 0.6465237166991553, "grad_norm": 2.4741806087579885, "learning_rate": 6.9991588927788e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035654, "epoch": 0.4001667361400584, "step": 240}, {"loss": 1.938018798828125, "token_acc": 0.6371398078975453, "grad_norm": 3.4537545471932733, "learning_rate": 6.871895307110332e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035664, "epoch": 0.40850354314297627, "step": 245}, {"loss": 1.7865478515625, "token_acc": 0.5756791720569211, "grad_norm": 16.72496162125171, "learning_rate": 6.743205229919224e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035675, "epoch": 0.4168403501458941, "step": 250}, {"loss": 2.0705270767211914, "token_acc": 0.6203485633537447, "grad_norm": 3.1932083384706664, "learning_rate": 6.613186730420917e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035687, "epoch": 0.425177157148812, "step": 255}, {"loss": 1.9050493240356445, "token_acc": 0.5929742388758782, "grad_norm": 7.316201458214335, "learning_rate": 6.4819388901648e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035706, "epoch": 0.4335139641517299, "step": 260}, {"loss": 1.888047981262207, "token_acc": 0.5771358328211432, "grad_norm": 2.7720764825330853, "learning_rate": 6.349561727528388e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035715, "epoch": 0.44185077115464777, "step": 265}, {"loss": 1.8664142608642578, "token_acc": 0.6696696696696697, "grad_norm": 4.065193512944849, "learning_rate": 6.216156121497578e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035727, "epoch": 0.45018757815756566, "step": 270}, {"loss": 1.9860942840576172, "token_acc": 0.5549738219895288, "grad_norm": 5.3789773277344315, "learning_rate": 6.0818237347910903e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035743, "epoch": 0.45852438516048355, "step": 275}, {"loss": 1.9592571258544922, "token_acc": 0.5547355473554736, "grad_norm": 4.12258515072594, "learning_rate": 5.946666936387637e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035764, "epoch": 0.46686119216340144, "step": 280}, {"loss": 2.0858516693115234, "token_acc": 0.5434947049924357, "grad_norm": 20.985894221796727, "learning_rate": 5.810788723514908e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035776, "epoch": 0.4751979991663193, "step": 285}, {"loss": 1.9558685302734375, "token_acc": 0.5351539802440441, "grad_norm": 8.177745830458013, "learning_rate": 5.674292643159764e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.03579, "epoch": 0.48353480616923716, "step": 290}, {"loss": 2.0921154022216797, "token_acc": 0.5673590504451038, "grad_norm": 4.0835586516577225, "learning_rate": 5.537282713159507e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035795, "epoch": 0.49187161317215505, "step": 295}, {"loss": 1.867361068725586, "token_acc": 0.5305164319248826, "grad_norm": 12.193993471525227, "learning_rate": 5.399863342934324e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035811, "epoch": 0.5002084201750729, "step": 300}, {"eval_loss": 1.742539644241333, "eval_token_acc": 0.612147612394181, "eval_runtime": 50.716, "eval_samples_per_second": 7.631, "eval_steps_per_second": 0.493, "epoch": 0.5002084201750729, "step": 300}, {"loss": 1.9004257202148438, "token_acc": 0.6261458748505381, "grad_norm": 3.6488789490883553, "learning_rate": 5.262139253921319e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035585, "epoch": 0.5085452271779908, "step": 305}, {"loss": 1.9045713424682618, "token_acc": 0.6437185929648241, "grad_norm": 2.862433077996505, "learning_rate": 5.1242153997707823e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035598, "epoch": 0.5168820341809087, "step": 310}, {"loss": 1.8142269134521485, "token_acc": 0.6176683562635771, "grad_norm": 1.8452496357053012, "learning_rate": 4.9861968863654875e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035606, "epoch": 0.5252188411838266, "step": 315}, {"loss": 1.8209394454956054, "token_acc": 0.6066109698510715, "grad_norm": 5.430011832866582, "learning_rate": 4.84818889172399e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035613, "epoch": 0.5335556481867445, "step": 320}, {"loss": 1.9349700927734375, "token_acc": 0.6356394129979036, "grad_norm": 2.600769211702097, "learning_rate": 4.7102965858489377e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035624, "epoch": 0.5418924551896623, "step": 325}, {"loss": 1.8674903869628907, "token_acc": 0.5809617271835132, "grad_norm": 3.3927364584396877, "learning_rate": 4.572625050581516e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035636, "epoch": 0.5502292621925803, "step": 330}, {"loss": 1.9532249450683594, "token_acc": 0.6082898709854515, "grad_norm": 4.347679564527899, "learning_rate": 4.435279199523043e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.03565, "epoch": 0.5585660691954981, "step": 335}, {"loss": 1.8467117309570313, "token_acc": 0.5447761194029851, "grad_norm": 14.797523073893712, "learning_rate": 4.298363698084809e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035659, "epoch": 0.5669028761984161, "step": 340}, {"loss": 1.8867809295654296, "token_acc": 0.649881716796215, "grad_norm": 9.174677465010568, "learning_rate": 4.16198288372702e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035671, "epoch": 0.5752396832013339, "step": 345}, {"loss": 1.9671783447265625, "token_acc": 0.5656722200697404, "grad_norm": 11.517095197655033, "learning_rate": 4.026240686447682e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035684, "epoch": 0.5835764902042517, "step": 350}, {"loss": 2.0355813980102537, "token_acc": 0.5646427096241222, "grad_norm": 2.0947774902715133, "learning_rate": 3.8912405495819786e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035694, "epoch": 0.5919132972071697, "step": 355}, {"loss": 2.0212512969970704, "token_acc": 0.603363412633306, "grad_norm": 2.2672434495907967, "learning_rate": 3.757085350972523e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035702, "epoch": 0.6002501042100875, "step": 360}, {"loss": 1.888478660583496, "token_acc": 0.5527710843373494, "grad_norm": 3.3476628567015885, "learning_rate": 3.623877324570548e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035708, "epoch": 0.6085869112130055, "step": 365}, {"loss": 2.1264991760253906, "token_acc": 0.5708566853482786, "grad_norm": 6.163682031673968, "learning_rate": 3.491717982527765e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035714, "epoch": 0.6169237182159233, "step": 370}, {"loss": 1.8802024841308593, "token_acc": 0.5547480620155039, "grad_norm": 1.6777593794217385, "learning_rate": 3.3607080378383005e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035723, "epoch": 0.6252605252188412, "step": 375}, {"loss": 1.8647388458251952, "token_acc": 0.5930644019815995, "grad_norm": 2.8440899647216833, "learning_rate": 3.230947327589602e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035732, "epoch": 0.6335973322217591, "step": 380}, {"loss": 1.8130638122558593, "token_acc": 0.5654246100519931, "grad_norm": 37.911669332255286, "learning_rate": 3.1025347368808775e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035741, "epoch": 0.6419341392246769, "step": 385}, {"loss": 2.064923095703125, "token_acc": 0.6108317214700193, "grad_norm": 4.022081032945518, "learning_rate": 2.9755681234669663e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035754, "epoch": 0.6502709462275948, "step": 390}, {"loss": 1.9649499893188476, "token_acc": 0.5570539419087137, "grad_norm": 2.307367388031427, "learning_rate": 2.85014424318512e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.03576, "epoch": 0.6586077532305127, "step": 395}, {"loss": 1.92861328125, "token_acc": 0.5911730545876888, "grad_norm": 4.204547110729611, "learning_rate": 2.7263586762215197e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035772, "epoch": 0.6669445602334306, "step": 400}, {"eval_loss": 1.7799798250198364, "eval_token_acc": 0.6100106846387771, "eval_runtime": 50.5835, "eval_samples_per_second": 7.651, "eval_steps_per_second": 0.494, "epoch": 0.6669445602334306, "step": 400}, {"loss": 1.9875520706176757, "token_acc": 0.5904554527309018, "grad_norm": 2.659254046452427, "learning_rate": 2.6043057542736836e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.03561, "epoch": 0.6752813672363485, "step": 405}, {"loss": 1.7080364227294922, "token_acc": 0.6345278725824801, "grad_norm": 14.662841293469047, "learning_rate": 2.4840784886643132e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035622, "epoch": 0.6836181742392664, "step": 410}, {"loss": 1.9299034118652343, "token_acc": 0.5383259911894274, "grad_norm": 2.0901368954978015, "learning_rate": 2.365768499461328e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035634, "epoch": 0.6919549812421842, "step": 415}, {"loss": 2.000414276123047, "token_acc": 0.5644047135310849, "grad_norm": 2.658074481592308, "learning_rate": 2.249465945658135e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035642, "epoch": 0.7002917882451021, "step": 420}, {"loss": 1.9364568710327148, "token_acc": 0.6048387096774194, "grad_norm": 3.8658249706668504, "learning_rate": 2.1352594564672908e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.03565, "epoch": 0.70862859524802, "step": 425}, {"loss": 1.9262733459472656, "token_acc": 0.5591078066914498, "grad_norm": 17.005290399571575, "learning_rate": 2.0232360637799685e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035661, "epoch": 0.7169654022509379, "step": 430}, {"loss": 2.0130874633789064, "token_acc": 0.5676373018798379, "grad_norm": 2.7462439370469673, "learning_rate": 1.9134811358426757e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035672, "epoch": 0.7253022092538558, "step": 435}, {"loss": 1.886693000793457, "token_acc": 0.5958948043617703, "grad_norm": 3.9297703842696294, "learning_rate": 1.806078312201745e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035681, "epoch": 0.7336390162567736, "step": 440}, {"loss": 1.8821983337402344, "token_acc": 0.5943814687037949, "grad_norm": 4.746458587201285, "learning_rate": 1.7011094399652107e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035688, "epoch": 0.7419758232596916, "step": 445}, {"loss": 1.847772216796875, "token_acc": 0.6304583182966438, "grad_norm": 4.130883468895127, "learning_rate": 1.59865451143062e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035697, "epoch": 0.7503126302626094, "step": 450}, {"loss": 1.8329124450683594, "token_acc": 0.635618801207417, "grad_norm": 9.216193465389415, "learning_rate": 1.4987916031263232e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035704, "epoch": 0.7586494372655272, "step": 455}, {"loss": 1.8964439392089845, "token_acc": 0.5654450261780105, "grad_norm": 5.297584387705496, "learning_rate": 1.401596816312673e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035715, "epoch": 0.7669862442684452, "step": 460}, {"loss": 1.9502939224243163, "token_acc": 0.5579991375592928, "grad_norm": 2.687671502108267, "learning_rate": 1.307144218988507e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035727, "epoch": 0.775323051271363, "step": 465}, {"loss": 1.71820068359375, "token_acc": 0.5902320748181503, "grad_norm": 3.767228791810687, "learning_rate": 1.2155057894470928e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035733, "epoch": 0.783659858274281, "step": 470}, {"loss": 1.8622875213623047, "token_acc": 0.5744859420898027, "grad_norm": 4.582125037883684, "learning_rate": 1.126751361424529e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035739, "epoch": 0.7919966652771988, "step": 475}, {"loss": 1.8682287216186524, "token_acc": 0.5714951094550536, "grad_norm": 3.5101737578003767, "learning_rate": 1.0409485708824507e-05, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035748, "epoch": 0.8003334722801168, "step": 480}, {"loss": 1.8879878997802735, "token_acc": 0.656461583750368, "grad_norm": 2.414709851450202, "learning_rate": 9.581628044655394e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035756, "epoch": 0.8086702792830346, "step": 485}, {"loss": 1.8605712890625, "token_acc": 0.6178915862986365, "grad_norm": 3.810296788453775, "learning_rate": 8.78457149673152e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035763, "epoch": 0.8170070862859525, "step": 490}, {"loss": 1.8603355407714843, "token_acc": 0.5819091288036682, "grad_norm": 5.265615022155901, "learning_rate": 8.018923467830403e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035772, "epoch": 0.8253438932888704, "step": 495}, {"loss": 1.69268741607666, "token_acc": 0.6207253886010363, "grad_norm": 2.5025771365898435, "learning_rate": 7.28526742563762e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035779, "epoch": 0.8336807002917882, "step": 500}, {"eval_loss": 1.6579405069351196, "eval_token_acc": 0.6231610092874168, "eval_runtime": 50.7583, "eval_samples_per_second": 7.624, "eval_steps_per_second": 0.493, "epoch": 0.8336807002917882, "step": 500}, {"loss": 1.8024673461914062, "token_acc": 0.655217965653897, "grad_norm": 7.055173813050995, "learning_rate": 6.584162458111148e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035646, "epoch": 0.8420175072947061, "step": 505}, {"loss": 1.6985977172851563, "token_acc": 0.6189883913764511, "grad_norm": 3.4632299338583885, "learning_rate": 5.916142847424127e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035649, "epoch": 0.850354314297624, "step": 510}, {"loss": 1.8116317749023438, "token_acc": 0.7189467658843732, "grad_norm": 3.7217850034242863, "learning_rate": 5.281717662811381e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035656, "epoch": 0.8586911213005419, "step": 515}, {"loss": 1.657485008239746, "token_acc": 0.6326460481099656, "grad_norm": 3.6951398715683865, "learning_rate": 4.681370372629368e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035663, "epoch": 0.8670279283034598, "step": 520}, {"loss": 1.7322940826416016, "token_acc": 0.5754838709677419, "grad_norm": 1.8312455829592764, "learning_rate": 4.1155584759256015e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035668, "epoch": 0.8753647353063777, "step": 525}, {"loss": 1.8520671844482421, "token_acc": 0.6836757234371549, "grad_norm": 2.151076970470845, "learning_rate": 3.5847131537982137e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035678, "epoch": 0.8837015423092955, "step": 530}, {"loss": 1.8487899780273438, "token_acc": 0.5323415265200517, "grad_norm": 2.92427111277661, "learning_rate": 3.089238940811162e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035687, "epoch": 0.8920383493122134, "step": 535}, {"loss": 1.7729415893554688, "token_acc": 0.5686113393590797, "grad_norm": 2.7017957543139546, "learning_rate": 2.6295134167157343e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035697, "epoch": 0.9003751563151313, "step": 540}, {"loss": 1.7561511993408203, "token_acc": 0.5832210242587601, "grad_norm": 2.0280474276001033, "learning_rate": 2.2058869187131515e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035701, "epoch": 0.9087119633180492, "step": 545}, {"loss": 1.8700078964233398, "token_acc": 0.6577181208053692, "grad_norm": 3.339758576911185, "learning_rate": 1.8186822744775234e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035708, "epoch": 0.9170487703209671, "step": 550}, {"loss": 1.766021728515625, "token_acc": 0.5925414364640884, "grad_norm": 5.51613595581924, "learning_rate": 1.4681945561426547e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035718, "epoch": 0.9253855773238849, "step": 555}, {"loss": 1.8983577728271483, "token_acc": 0.5683696468820436, "grad_norm": 4.024065341173453, "learning_rate": 1.1546908554401659e-06, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035724, "epoch": 0.9337223843268029, "step": 560}, {"loss": 1.7441574096679688, "token_acc": 0.6148734985944289, "grad_norm": 7.437280242056288, "learning_rate": 8.784100801602912e-07, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.03573, "epoch": 0.9420591913297207, "step": 565}, {"loss": 1.8380481719970703, "token_acc": 0.5575221238938053, "grad_norm": 6.1653456549827625, "learning_rate": 6.395627720904518e-07, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035736, "epoch": 0.9503959983326385, "step": 570}, {"loss": 1.8257612228393554, "token_acc": 0.5928525845564774, "grad_norm": 3.928180437337867, "learning_rate": 4.383309465703145e-07, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035742, "epoch": 0.9587328053355565, "step": 575}, {"loss": 2.0145660400390626, "token_acc": 0.548, "grad_norm": 4.015367492611914, "learning_rate": 2.748679537857013e-07, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035745, "epoch": 0.9670696123384743, "step": 580}, {"loss": 1.9830612182617187, "token_acc": 0.5244872331519465, "grad_norm": 3.725103655434126, "learning_rate": 1.4929836190696323e-07, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035751, "epoch": 0.9754064193413923, "step": 585}, {"loss": 1.6920804977416992, "token_acc": 0.6742605915267785, "grad_norm": 2.333001410662022, "learning_rate": 6.171786216085939e-08, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035754, "epoch": 0.9837432263443101, "step": 590}, {"loss": 1.8399593353271484, "token_acc": 0.5763772954924875, "grad_norm": 2.2401580217408386, "learning_rate": 1.2193195908388743e-08, "memory(GiB)": 75.54, "train_speed(iter/s)": 0.035759, "epoch": 0.992080033347228, "step": 595}, {"eval_loss": 1.6350570917129517, "eval_token_acc": 0.6257088846880907, "eval_runtime": 50.6821, "eval_samples_per_second": 7.636, "eval_steps_per_second": 0.493, "epoch": 0.9987494789495623, "step": 599}, {"train_runtime": 16799.6463, "train_samples_per_second": 2.284, "train_steps_per_second": 0.036, "total_flos": 954945089044480.0, "train_loss": 1.970288353889733, "epoch": 0.9987494789495623, "step": 599}], "memory": 75.537109375}
|
v1-20250508-175113/runs/events.out.tfevents.1746697894.dsw-2138-644b4b995d-5j9mw.1228873.0
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f5d4cb97524ff4ec71a0c3b8ead1473972696d8400d6555008d67cbffd5059dc
|
| 3 |
+
size 56004
|
v1-20250508-175113/val_dataset.jsonl
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|