diff --git a/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/args.json b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/args.json new file mode 100644 index 0000000000000000000000000000000000000000..d71a2c036c84e0c237b2873a8290f8f60f1ab85c --- /dev/null +++ b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/args.json @@ -0,0 +1,382 @@ +{ + "output_dir": "/NHNHOME/data/sanghyeok/qwen-cua/runs/ocu_exact_rt_sig_ac_clean_gcon_cmp/v0-20260613-023235", + "per_device_train_batch_size": 1, + "num_train_epochs": 1.0, + "max_steps": -1, + "learning_rate": 1e-05, + "lr_scheduler_type": "cosine", + "lr_scheduler_kwargs": null, + "warmup_steps": 200.0, + "optim": "adamw_torch_fused", + "optim_args": null, + "weight_decay": 0.0, + "adam_beta1": 0.9, + "adam_beta2": 0.95, + "adam_epsilon": 1e-08, + "optim_target_modules": null, + "gradient_accumulation_steps": 16, + "average_tokens_across_devices": true, + "max_grad_norm": 1.0, + "label_smoothing_factor": 0.0, + "bf16": true, + "fp16": false, + "bf16_full_eval": false, + "fp16_full_eval": false, + "tf32": true, + "gradient_checkpointing": true, + "gradient_checkpointing_kwargs": null, + "torch_compile": false, + "torch_compile_backend": null, + "torch_compile_mode": null, + "use_liger_kernel": false, + "liger_kernel_config": null, + "use_cache": false, + "neftune_noise_alpha": null, + "torch_empty_cache_steps": null, + "auto_find_batch_size": false, + "logging_strategy": "steps", + "logging_steps": 1, + "logging_first_step": true, + "log_on_each_node": true, + "logging_nan_inf_filter": true, + "include_num_input_tokens_seen": false, + "log_level": "passive", + "log_level_replica": "warning", + "disable_tqdm": null, + "report_to": [ + "wandb" + ], + "run_name": "ocu_exact_rt_sig_ac_clean_gcon_cmp", + "project": "huggingface", + "trackio_space_id": null, + "trackio_bucket_id": null, + "trackio_static_space_id": null, + "eval_strategy": "steps", + "eval_steps": 100.0, + "eval_delay": 0, + "per_device_eval_batch_size": 1, + "prediction_loss_only": false, + "eval_on_start": false, + "eval_do_concat_batches": true, + "eval_use_gather_object": false, + "eval_accumulation_steps": null, + "include_for_metrics": [], + "batch_eval_metrics": false, + "save_only_model": false, + "save_strategy": "steps", + "save_steps": 100.0, + "save_on_each_node": false, + "save_total_limit": 3, + "enable_jit_checkpoint": false, + "push_to_hub": false, + "hub_token": null, + "hub_private_repo": null, + "hub_model_id": null, + "hub_strategy": "every_save", + "hub_always_push": false, + "hub_revision": null, + "load_best_model_at_end": false, + "metric_for_best_model": "loss", + "greater_is_better": false, + "ignore_data_skip": false, + "restore_callback_states_from_checkpoint": false, + "full_determinism": false, + "seed": 42, + "data_seed": 42, + "use_cpu": false, + "accelerator_config": { + "dispatch_batches": false + }, + "parallelism_config": null, + "dataloader_drop_last": false, + "dataloader_num_workers": 8, + "dataloader_pin_memory": true, + "dataloader_persistent_workers": true, + "dataloader_prefetch_factor": 4, + "remove_unused_columns": true, + "label_names": null, + "train_sampling_strategy": "random", + "length_column_name": "length", + "ddp_find_unused_parameters": null, + "ddp_bucket_cap_mb": null, + "ddp_broadcast_buffers": null, + "ddp_static_graph": null, + "ddp_backend": null, + "ddp_timeout": 18000000, + "fsdp": [], + "fsdp_config": null, + "deepspeed": { + "bf16": { + "enabled": true + }, + "fp16": { + "enabled": false + }, + "zero_optimization": { + "stage": 2, + "overlap_comm": true, + "contiguous_gradients": true, + "reduce_bucket_size": 500000000.0, + "allgather_bucket_size": 500000000.0, + "reduce_scatter": true, + "round_robin_gradients": true + }, + "gradient_clipping": "auto", + "gradient_accumulation_steps": "auto", + "train_batch_size": "auto", + "train_micro_batch_size_per_gpu": "auto", + "steps_per_print": 100, + "wall_clock_breakdown": false, + "_comment_bf16_safety": "DeepSpeed 0.19 BF16_Optimizer keeps fp32 master grad partitions (fp32_groups_flat_partition + accumulate_hp_grads_and_remove_lp hook), so gradient accumulation across many micro-batches is done in fp32. bf16 is only used for the param master copy and forward/backward activations. This is safe at grad_accum=16+; no need to force fp32 collectives." + }, + "debug": null, + "skip_memory_metrics": true, + "do_train": false, + "do_eval": false, + "do_predict": false, + "resume_from_checkpoint": null, + "warmup_ratio": null, + "logging_dir": "/NHNHOME/data/sanghyeok/qwen-cua/runs/ocu_exact_rt_sig_ac_clean_gcon_cmp/v0-20260613-023235/runs", + "local_rank": 0, + "sortish_sampler": false, + "predict_with_generate": false, + "generation_max_length": null, + "generation_num_beams": null, + "generation_config": null, + "tuner_backend": "peft", + "vit_gradient_checkpointing": false, + "router_aux_loss_coef": 0.0, + "enable_dft_loss": false, + "enable_channel_loss": false, + "safe_serialization": true, + "max_shard_size": "5GB", + "check_model": true, + "acc_strategy": "token", + "train_dataloader_shuffle": true, + "group_by_length": false, + "max_epochs": null, + "aligner_lr": null, + "vit_lr": null, + "use_logits_to_keep": null, + "ds3_gather_for_generation": true, + "resume_only_model": false, + "optimizer": null, + "loss_type": null, + "eval_metric": null, + "callbacks": [], + "early_stop_interval": null, + "eval_use_evalscope": false, + "eval_dataset": [], + "eval_dataset_args": null, + "eval_limit": null, + "eval_generation_config": null, + "extra_eval_args": null, + "tuner_type": "full", + "use_galore": false, + "galore_target_modules": null, + "galore_rank": 128, + "galore_update_proj_gap": 50, + "galore_scale": 1.0, + "galore_proj_type": "std", + "galore_optim_per_parameter": false, + "galore_with_embedding": false, + "galore_quantization": false, + "galore_proj_quant": false, + "galore_proj_bits": 4, + "galore_proj_group_size": 256, + "galore_cos_threshold": 0.4, + "galore_gamma_proj": 2, + "galore_queue_size": 5, + "lisa_activated_layers": 0, + "lisa_step_interval": 20, + "use_flash_ckpt": false, + "use_ray": false, + "ray_exp_name": null, + "device_groups": null, + "model": "/NHNHOME/data/sanghyeok/qwen-cua/hf_cache/hub/models--Qwen--Qwen3.5-4B/snapshots/851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", + "model_type": "qwen3_5", + "model_revision": null, + "task_type": "causal_lm", + "torch_dtype": "bfloat16", + "attn_impl": "flash_attention_2", + "experts_impl": null, + "new_special_tokens": [], + "num_labels": null, + "problem_type": null, + "rope_scaling": null, + "device_map": null, + "max_memory": {}, + "max_model_len": null, + "local_repo_path": null, + "init_strategy": null, + "template": "qwen3_5", + "system": null, + "max_length": 12288, + "truncation_strategy": "delete", + "max_pixels": 16777216, + "agent_template": null, + "norm_bbox": null, + "use_chat_template": true, + "padding_side": "right", + "padding_free": false, + "loss_scale": "default", + "sequence_parallel_size": 1, + "template_backend": "swift", + "response_prefix": null, + "enable_thinking": null, + "add_non_thinking_prefix": true, + "dataset": [ + "/NHNHOME/data/sanghyeok/qwen-cua/data/AgentNet/l1/ubuntu/train_wm.jsonl#35000", + "/NHNHOME/data/sanghyeok/qwen-cua/data/AgentNet/l2/ubuntu/train_wm.jsonl#35000", + "/NHNHOME/data/sanghyeok/qwen-cua/data/AgentNet/l3/ubuntu/train_wm.jsonl#35000", + "/NHNHOME/data/sanghyeok/qwen-cua/data/AgentNet/l1/win_mac/train_wm.jsonl#100000", + "/NHNHOME/data/sanghyeok/qwen-cua/data/AgentNet/l2/win_mac/train_wm.jsonl#100000", + "/NHNHOME/data/sanghyeok/qwen-cua/data/AgentNet/l3/win_mac/train_wm.jsonl#100000", + "/NHNHOME/data/sanghyeok/qwen-cua/data/GUI-360/l1/train_wm.jsonl#50000", + "/NHNHOME/data/sanghyeok/qwen-cua/data/GUI-360/l2/train_wm.jsonl#50000", + "/NHNHOME/data/sanghyeok/qwen-cua/data/GUI-360/l3/train_wm.jsonl#50000" + ], + "val_dataset": [ + "/NHNHOME/data/sanghyeok/qwen-cua/data/AgentNet/l2/ubuntu/val_wm.jsonl" + ], + "cached_dataset": [], + "cached_val_dataset": [], + "split_dataset_ratio": 0.0, + "dataset_num_proc": 4, + "load_from_cache_file": false, + "dataset_shuffle": true, + "val_dataset_shuffle": false, + "streaming": false, + "interleave_prob": null, + "stopping_strategy": "first_exhausted", + "shuffle_buffer_size": 1000, + "download_mode": "reuse_dataset_if_exists", + "columns": {}, + "strict": false, + "disable_auto_column_mapping": false, + "model_name": null, + "model_author": null, + "custom_dataset_info": [], + "quant_method": null, + "quant_bits": null, + "hqq_axis": null, + "bnb_4bit_compute_dtype": "bfloat16", + "bnb_4bit_quant_type": "nf4", + "bnb_4bit_use_double_quant": true, + "bnb_4bit_quant_storage": null, + "max_new_tokens": 64, + "temperature": 0.0, + "top_k": null, + "top_p": null, + "repetition_penalty": null, + "num_beams": 1, + "stream": false, + "stop_words": [], + "logprobs": false, + "top_logprobs": null, + "structured_outputs_regex": null, + "adapters": [], + "external_plugins": [], + "custom_register_path": [], + "model_kwargs": {}, + "enable_npu_model_patch": true, + "load_args": false, + "load_data_args": false, + "packing": false, + "packing_length": null, + "packing_num_proc": 1, + "lazy_tokenize": true, + "use_hf": false, + "ignore_args_error": false, + "use_swift_lora": false, + "freeze_parameters": [ + "model.visual", + "model.visual.merger" + ], + "freeze_parameters_regex": null, + "freeze_parameters_ratio": 0.0, + "trainable_parameters": [], + "trainable_parameters_regex": null, + "freeze_llm": false, + "freeze_vit": true, + "freeze_aligner": true, + "target_modules": [ + "all-linear" + ], + "target_regex": null, + "target_parameters": null, + "modules_to_save": [], + "lora_rank": 8, + "lora_alpha": 32, + "lora_dropout": 0.05, + "lora_bias": "none", + "lora_dtype": null, + "lorap_lr_ratio": null, + "use_rslora": false, + "use_dora": false, + "lora_ga_batch_size": 2, + "lora_ga_iters": 2, + "lora_ga_max_length": 1024, + "lora_ga_direction": "ArB2r", + "lora_ga_scale": "stable", + "lora_ga_stable_gamma": 16, + "init_weights": true, + "fourier_n_frequency": 2000, + "fourier_scaling": 300.0, + "boft_block_size": 4, + "boft_block_num": 0, + "boft_n_butterfly_factor": 1, + "boft_dropout": 0.0, + "vera_rank": 256, + "vera_projection_prng_key": 0, + "vera_dropout": 0.0, + "vera_d_initial": 0.1, + "adapter_act": "gelu", + "adapter_length": 128, + "adalora_target_r": 8, + "adalora_init_r": 12, + "adalora_tinit": 0, + "adalora_tfinal": 0, + "adalora_deltaT": 1, + "adalora_beta1": 0.85, + "adalora_beta2": 0.85, + "adalora_orth_reg_weight": 0.5, + "llamapro_num_new_blocks": 4, + "llamapro_num_groups": null, + "reft_layer_key": null, + "reft_layers": null, + "reft_rank": 4, + "reft_intervention_type": "LoreftIntervention", + "reft_args": null, + "swanlab_token": null, + "swanlab_project": "ms-swift", + "swanlab_workspace": null, + "swanlab_exp_name": null, + "swanlab_notification_method": null, + "swanlab_webhook_url": null, + "swanlab_secret": null, + "swanlab_sender_email": null, + "swanlab_receiver_email": null, + "swanlab_smtp_server": null, + "swanlab_smtp_port": null, + "swanlab_email_language": "zh", + "swanlab_mode": "cloud", + "add_version": true, + "create_checkpoint_symlink": false, + "zero_hpz_partition_size": null, + "deepspeed_autotp_size": null, + "swift_version": "4.2.1", + "ckpt_dir": null, + "rank": 0, + "global_world_size": 8, + "local_world_size": 8, + "model_suffix": "Qwen3.5-4B", + "model_info": "ModelInfo(model_type='qwen3_5', model_dir='/NHNHOME/data/sanghyeok/qwen-cua/hf_cache/hub/models--Qwen--Qwen3.5-4B/snapshots/851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a', torch_dtype=torch.bfloat16, max_model_len=262144, quant_method=None, quant_bits=None, rope_scaling=None, is_moe_model=False, is_multimodal=True, config=None, task_type='causal_lm', num_labels=None)", + "model_meta": "ModelMeta(model_type='qwen3_5', model_groups=[ModelGroup(models=[Model(ms_model_id='Qwen/Qwen3.5-0.8B', hf_model_id='Qwen/Qwen3.5-0.8B', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen3.5-2B', hf_model_id='Qwen/Qwen3.5-2B', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen3.5-4B', hf_model_id='Qwen/Qwen3.5-4B', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen3.5-9B', hf_model_id='Qwen/Qwen3.5-9B', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen3.5-27B', hf_model_id='Qwen/Qwen3.5-27B', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen3.5-27B-FP8', hf_model_id='Qwen/Qwen3.5-27B-FP8', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen3.5-0.8B-Base', hf_model_id='Qwen/Qwen3.5-0.8B-Base', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen3.5-2B-Base', hf_model_id='Qwen/Qwen3.5-2B-Base', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen3.5-4B-Base', hf_model_id='Qwen/Qwen3.5-4B-Base', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen3.5-9B-Base', hf_model_id='Qwen/Qwen3.5-9B-Base', model_path=None, ms_revision=None, hf_revision=None)], template='qwen3_5', ignore_patterns=None, requires=None, tags=[]), ModelGroup(models=[Model(ms_model_id='Qwen/Qwen3.6-27B', hf_model_id='Qwen/Qwen3.6-27B', model_path=None, ms_revision=None, hf_revision=None), Model(ms_model_id='Qwen/Qwen3.6-27B-FP8', hf_model_id='Qwen/Qwen3.6-27B-FP8', model_path=None, ms_revision=None, hf_revision=None)], template='qwen3_5', ignore_patterns=None, requires=None, tags=[])], loader=, template='qwen3_5', model_arch=MultiModelKeys(arch_name='qwen2_vl', embedding=None, module_list=None, lm_head=None, q_proj=None, k_proj=None, v_proj=None, o_proj=None, attention=None, mlp=None, down_proj=None, qkv_proj=None, qk_proj=None, qa_proj=None, qb_proj=None, kv_proj=None, kva_proj=None, kvb_proj=None, language_model=['model.language_model', 'lm_head'], aligner=['model.visual.merger'], vision_tower=['model.visual'], generator=[]), mcore_model_type=None, architectures=['Qwen3_5ForConditionalGeneration'], additional_saved_files=[], torch_dtype=None, is_multimodal=True, is_reward=False, task_type=None, ignore_patterns=None, requires=['transformers>=5.0.0.dev', 'qwen_vl_utils>=0.0.14', 'decord'], tags=[])", + "model_dir": "/NHNHOME/data/sanghyeok/qwen-cua/hf_cache/hub/models--Qwen--Qwen3.5-4B/snapshots/851bf6e806efd8d0a36b00ddf55e13ccb7b8cd0a", + "template_meta": "QwenTemplateMeta(template_type='qwen3_5', prefix=[], prompt=['<|im_start|>user\\n{{QUERY}}<|im_end|>\\n<|im_start|>assistant\\n'], chat_sep=['<|im_end|>\\n'], suffix=['<|im_end|>\\n'], template_cls=, system_prefix=['<|im_start|>system\\n{{SYSTEM}}<|im_end|>\\n'], default_system=None, auto_add_bos=False, stop_words=['<|endoftext|>'], agent_template='qwen3_5', is_thinking=True, thinking_prefix='\\n', non_thinking_prefix='\\n\\n\\n\\n', history_thinking_prefix='')", + "_val_dataset_exists": true, + "hub": "", + "evaluation_strategy": "steps", + "training_args": "Seq2SeqTrainingArguments(output_dir='/NHNHOME/data/sanghyeok/qwen-cua/runs/ocu_exact_rt_sig_ac_clean_gcon_cmp/v0-20260613-023235', per_device_train_batch_size=1, num_train_epochs=1.0, max_steps=-1, learning_rate=1e-05, lr_scheduler_type=, lr_scheduler_kwargs=None, warmup_steps=200.0, optim=, optim_args=None, weight_decay=0.0, adam_beta1=0.9, adam_beta2=0.95, adam_epsilon=1e-08, optim_target_modules=None, gradient_accumulation_steps=16, average_tokens_across_devices=None, max_grad_norm=1.0, label_smoothing_factor=0.0, bf16=True, fp16=False, bf16_full_eval=False, fp16_full_eval=False, tf32=True, gradient_checkpointing=True, gradient_checkpointing_kwargs=None, torch_compile=False, torch_compile_backend=None, torch_compile_mode=None, use_liger_kernel=False, liger_kernel_config=None, use_cache=False, neftune_noise_alpha=None, torch_empty_cache_steps=None, auto_find_batch_size=False, logging_strategy=, logging_steps=1, logging_first_step=True, log_on_each_node=True, logging_nan_inf_filter=True, include_num_input_tokens_seen=None, log_level='passive', log_level_replica='warning', disable_tqdm=False, report_to=['wandb'], run_name='ocu_exact_rt_sig_ac_clean_gcon_cmp', project='huggingface', trackio_space_id=None, trackio_bucket_id=None, trackio_static_space_id=None, eval_strategy=, eval_steps=100, eval_delay=0, per_device_eval_batch_size=1, prediction_loss_only=False, eval_on_start=False, eval_do_concat_batches=True, eval_use_gather_object=False, eval_accumulation_steps=None, include_for_metrics=[], batch_eval_metrics=False, save_only_model=False, save_strategy=, save_steps=100, save_on_each_node=False, save_total_limit=3, enable_jit_checkpoint=False, push_to_hub=False, hub_token=None, hub_private_repo=None, hub_model_id=None, hub_strategy=, hub_always_push=False, hub_revision=None, load_best_model_at_end=False, metric_for_best_model='loss', greater_is_better=False, ignore_data_skip=False, restore_callback_states_from_checkpoint=False, full_determinism=False, seed=42, data_seed=42, use_cpu=False, accelerator_config=AcceleratorConfig(split_batches=False, dispatch_batches=False, even_batches=True, use_seedable_sampler=True, non_blocking=False, gradient_accumulation_kwargs=None, use_configured_state=False), parallelism_config=None, dataloader_drop_last=False, dataloader_num_workers=8, dataloader_pin_memory=True, dataloader_persistent_workers=True, dataloader_prefetch_factor=4, remove_unused_columns=False, label_names=None, train_sampling_strategy='random', length_column_name='length', ddp_find_unused_parameters=None, ddp_bucket_cap_mb=None, ddp_broadcast_buffers=None, ddp_static_graph=None, ddp_backend=None, ddp_timeout=18000000, fsdp=[], fsdp_config={'min_num_params': 0, 'xla': False, 'xla_fsdp_v2': False, 'xla_fsdp_grad_ckpt': False}, deepspeed={'bf16': {'enabled': True}, 'fp16': {'enabled': False}, 'zero_optimization': {'stage': 2, 'overlap_comm': True, 'contiguous_gradients': True, 'reduce_bucket_size': 500000000.0, 'allgather_bucket_size': 500000000.0, 'reduce_scatter': True, 'round_robin_gradients': True}, 'gradient_clipping': 'auto', 'gradient_accumulation_steps': 'auto', 'train_batch_size': 'auto', 'train_micro_batch_size_per_gpu': 'auto', 'steps_per_print': 100, 'wall_clock_breakdown': False, '_comment_bf16_safety': 'DeepSpeed 0.19 BF16_Optimizer keeps fp32 master grad partitions (fp32_groups_flat_partition + accumulate_hp_grads_and_remove_lp hook), so gradient accumulation across many micro-batches is done in fp32. bf16 is only used for the param master copy and forward/backward activations. This is safe at grad_accum=16+; no need to force fp32 collectives.'}, debug=[], skip_memory_metrics=True, do_train=False, do_eval=True, do_predict=False, resume_from_checkpoint=None, warmup_ratio=None, logging_dir='/NHNHOME/data/sanghyeok/qwen-cua/runs/ocu_exact_rt_sig_ac_clean_gcon_cmp/v0-20260613-023235/runs', local_rank=0, sortish_sampler=False, predict_with_generate=False, generation_max_length=None, generation_num_beams=None, generation_config=None, tuner_backend='peft', vit_gradient_checkpointing=False, router_aux_loss_coef=0.0, enable_dft_loss=False, enable_channel_loss=False, safe_serialization=True, max_shard_size='5GB', check_model=True, acc_strategy='token', train_dataloader_shuffle=True, group_by_length=False, max_epochs=None, aligner_lr=None, vit_lr=None, use_logits_to_keep=None, ds3_gather_for_generation=True, resume_only_model=False, optimizer=None, loss_type=None, eval_metric=None, callbacks=[], early_stop_interval=None, eval_use_evalscope=False, eval_dataset=[], eval_dataset_args=None, eval_limit=None, eval_generation_config=None, extra_eval_args=None, tuner_type='full', use_galore=False, galore_target_modules=None, galore_rank=128, galore_update_proj_gap=50, galore_scale=1.0, galore_proj_type='std', galore_optim_per_parameter=False, galore_with_embedding=False, galore_quantization=False, galore_proj_quant=False, galore_proj_bits=4, galore_proj_group_size=256, galore_cos_threshold=0.4, galore_gamma_proj=2, galore_queue_size=5, lisa_activated_layers=0, lisa_step_interval=20, use_flash_ckpt=False)" +} \ No newline at end of file diff --git a/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/chat_template.jinja b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..a585dec894e63da457d9440ec6aa7caa16d20860 --- /dev/null +++ b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/chat_template.jinja @@ -0,0 +1,154 @@ +{%- set image_count = namespace(value=0) %} +{%- set video_count = namespace(value=0) %} +{%- macro render_content(content, do_vision_count, is_system_content=false) %} + {%- if content is string %} + {{- content }} + {%- elif content is iterable and content is not mapping %} + {%- for item in content %} + {%- if 'image' in item or 'image_url' in item or item.type == 'image' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain images.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set image_count.value = image_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Picture ' ~ image_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|image_pad|><|vision_end|>' }} + {%- elif 'video' in item or item.type == 'video' %} + {%- if is_system_content %} + {{- raise_exception('System message cannot contain videos.') }} + {%- endif %} + {%- if do_vision_count %} + {%- set video_count.value = video_count.value + 1 %} + {%- endif %} + {%- if add_vision_id %} + {{- 'Video ' ~ video_count.value ~ ': ' }} + {%- endif %} + {{- '<|vision_start|><|video_pad|><|vision_end|>' }} + {%- elif 'text' in item %} + {{- item.text }} + {%- else %} + {{- raise_exception('Unexpected item type in content.') }} + {%- endif %} + {%- endfor %} + {%- elif content is none or content is undefined %} + {{- '' }} + {%- else %} + {{- raise_exception('Unexpected content type.') }} + {%- endif %} +{%- endmacro %} +{%- if not messages %} + {{- raise_exception('No messages provided.') }} +{%- endif %} +{%- if tools and tools is iterable and tools is not mapping %} + {{- '<|im_start|>system\n' }} + {{- "# Tools\n\nYou have access to the following functions:\n\n" }} + {%- for tool in tools %} + {{- "\n" }} + {{- tool | tojson }} + {%- endfor %} + {{- "\n" }} + {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n\n\n\nvalue_1\n\n\nThis is the value for the second parameter\nthat can span\nmultiple lines\n\n\n\n\n\nReminder:\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n' }} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {%- if content %} + {{- '\n\n' + content }} + {%- endif %} + {%- endif %} + {{- '<|im_end|>\n' }} +{%- else %} + {%- if messages[0].role == 'system' %} + {%- set content = render_content(messages[0].content, false, true)|trim %} + {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }} + {%- endif %} +{%- endif %} +{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %} +{%- for message in messages[::-1] %} + {%- set index = (messages|length - 1) - loop.index0 %} + {%- if ns.multi_step_tool and message.role == "user" %} + {%- set content = render_content(message.content, false)|trim %} + {%- if not(content.startswith('') and content.endswith('')) %} + {%- set ns.multi_step_tool = false %} + {%- set ns.last_query_index = index %} + {%- endif %} + {%- endif %} +{%- endfor %} +{%- if ns.multi_step_tool %} + {{- raise_exception('No user query found in messages.') }} +{%- endif %} +{%- for message in messages %} + {%- set content = render_content(message.content, true)|trim %} + {%- if message.role == "system" %} + {%- if not loop.first %} + {{- raise_exception('System message must be at the beginning.') }} + {%- endif %} + {%- elif message.role == "user" %} + {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }} + {%- elif message.role == "assistant" %} + {%- set reasoning_content = '' %} + {%- if message.reasoning_content is string %} + {%- set reasoning_content = message.reasoning_content %} + {%- else %} + {%- if '' in content %} + {%- set reasoning_content = content.split('')[0].rstrip('\n').split('')[-1].lstrip('\n') %} + {%- set content = content.split('')[-1].lstrip('\n') %} + {%- endif %} + {%- endif %} + {%- set reasoning_content = reasoning_content|trim %} + {%- if loop.index0 > ns.last_query_index %} + {{- '<|im_start|>' + message.role + '\n\n' + reasoning_content + '\n\n\n' + content }} + {%- else %} + {{- '<|im_start|>' + message.role + '\n' + content }} + {%- endif %} + {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %} + {%- for tool_call in message.tool_calls %} + {%- if tool_call.function is defined %} + {%- set tool_call = tool_call.function %} + {%- endif %} + {%- if loop.first %} + {%- if content|trim %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n\n' }} + {%- endif %} + {%- else %} + {{- '\n\n\n' }} + {%- endif %} + {%- if tool_call.arguments is defined %} + {%- for args_name, args_value in tool_call.arguments|items %} + {{- '\n' }} + {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %} + {{- args_value }} + {{- '\n\n' }} + {%- endfor %} + {%- endif %} + {{- '\n' }} + {%- endfor %} + {%- endif %} + {{- '<|im_end|>\n' }} + {%- elif message.role == "tool" %} + {%- if loop.previtem and loop.previtem.role != "tool" %} + {{- '<|im_start|>user' }} + {%- endif %} + {{- '\n\n' }} + {{- content }} + {{- '\n' }} + {%- if not loop.last and loop.nextitem.role != "tool" %} + {{- '<|im_end|>\n' }} + {%- elif loop.last %} + {{- '<|im_end|>\n' }} + {%- endif %} + {%- else %} + {{- raise_exception('Unexpected message role.') }} + {%- endif %} +{%- endfor %} +{%- if add_generation_prompt %} + {{- '<|im_start|>assistant\n' }} + {%- if enable_thinking is defined and enable_thinking is false %} + {{- '\n\n\n\n' }} + {%- else %} + {{- '\n' }} + {%- endif %} +{%- endif %} \ No newline at end of file diff --git a/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/config.json b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/config.json new file mode 100644 index 0000000000000000000000000000000000000000..314297e9ae3a6f7ca90b01d29ee0a46cfbf53c2a --- /dev/null +++ b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/config.json @@ -0,0 +1,119 @@ +{ + "architectures": [ + "Qwen35VLLatentForConditionalGeneration" + ], + "dtype": "bfloat16", + "eos_token_id": 248046, + "hidden_size": 2560, + "image_token_id": 248056, + "latent_end_id": 248078, + "latent_pad_id": 248079, + "latent_start_id": 248077, + "model_type": "qwen3_5", + "pad_token_id": 248044, + "text_config": { + "attention_bias": false, + "attention_dropout": 0.0, + "attn_output_gate": true, + "bos_token_id": null, + "dtype": "bfloat16", + "eos_token_id": 248044, + "full_attention_interval": 4, + "head_dim": 256, + "hidden_act": "silu", + "hidden_size": 2560, + "initializer_range": 0.02, + "intermediate_size": 9216, + "latent_end_id": 248078, + "latent_pad_id": 248079, + "latent_start_id": 248077, + "layer_types": [ + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention", + "linear_attention", + "linear_attention", + "linear_attention", + "full_attention" + ], + "linear_conv_kernel_dim": 4, + "linear_key_head_dim": 128, + "linear_num_key_heads": 16, + "linear_num_value_heads": 32, + "linear_value_head_dim": 128, + "mamba_ssm_dtype": "float32", + "max_position_embeddings": 262144, + "mlp_only_layers": [], + "model_type": "qwen3_5_text", + "mtp_num_hidden_layers": 1, + "mtp_use_dedicated_embeddings": false, + "num_attention_heads": 16, + "num_hidden_layers": 32, + "num_key_value_heads": 4, + "pad_token_id": 248044, + "partial_rotary_factor": 0.25, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "mrope_interleaved": true, + "mrope_section": [ + 11, + 11, + 10 + ], + "partial_rotary_factor": 0.25, + "rope_theta": 10000000, + "rope_type": "default" + }, + "tie_word_embeddings": true, + "use_cache": false, + "vocab_size": 248080 + }, + "tie_word_embeddings": true, + "transformers_version": "5.8.1", + "use_cache": false, + "video_token_id": 248057, + "vision_config": { + "deepstack_visual_indexes": [], + "depth": 24, + "dtype": "bfloat16", + "hidden_act": "gelu_pytorch_tanh", + "hidden_size": 1024, + "in_channels": 3, + "initializer_range": 0.02, + "intermediate_size": 4096, + "model_type": "qwen3_5_vision", + "num_heads": 16, + "num_position_embeddings": 2304, + "out_hidden_size": 2560, + "patch_size": 16, + "spatial_merge_size": 2, + "temporal_patch_size": 2 + }, + "vision_end_token_id": 248054, + "vision_start_token_id": 248053 +} diff --git a/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/generation_config.json b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/generation_config.json new file mode 100644 index 0000000000000000000000000000000000000000..1f1ab970fed81cb788c2945bed33833d247b9091 --- /dev/null +++ b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/generation_config.json @@ -0,0 +1,9 @@ +{ + "_from_model_config": true, + "eos_token_id": [ + 248044, + 248046 + ], + "transformers_version": "5.8.1", + "use_cache": true +} diff --git a/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/latest b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/latest new file mode 100644 index 0000000000000000000000000000000000000000..f0b47ce15fff9a01b2a416a473b2148085048a50 --- /dev/null +++ b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/latest @@ -0,0 +1 @@ +global_step500 \ No newline at end of file diff --git a/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/model-00001-of-00003.safetensors b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/model-00001-of-00003.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..1b92feec2ced675b35c19ba120061e67e9738266 --- /dev/null +++ b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/model-00001-of-00003.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:28886d6e1ef67875feb52aaa041379a821cc6624a35951b06a5337397a459565 +size 4989860336 diff --git a/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/model-00002-of-00003.safetensors b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/model-00002-of-00003.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..fe787be3f6573fd033a4ef8f86bc07f3e0a09697 --- /dev/null +++ b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/model-00002-of-00003.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:31990c5b6810a68820674a5d036908c5c7a900fe7599430964e5e4a2c815c65b +size 4993498424 diff --git a/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/model-00003-of-00003.safetensors b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/model-00003-of-00003.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..4ab8023a740649342f1bc7e224320898de73054c --- /dev/null +++ b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/model-00003-of-00003.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9132529e66cac3079a828b70ff720888e735b62dcc8a839f7c42730ef896d42c +size 398327248 diff --git a/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/model.safetensors.index.json b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/model.safetensors.index.json new file mode 100644 index 0000000000000000000000000000000000000000..6a37ec4c02c77592002094503fd812d32a3c3c13 --- /dev/null +++ b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/model.safetensors.index.json @@ -0,0 +1,747 @@ +{ + "metadata": { + "total_parameters": 4555712896, + "total_size": 10381595392 + }, + "weight_map": { + "alignment_projector.action_proj.bias": "model-00001-of-00003.safetensors", + "alignment_projector.action_proj.weight": "model-00001-of-00003.safetensors", + "alignment_projector.ln_action.bias": "model-00001-of-00003.safetensors", + "alignment_projector.ln_action.weight": "model-00001-of-00003.safetensors", + "alignment_projector.ln_state.bias": "model-00001-of-00003.safetensors", + "alignment_projector.ln_state.weight": "model-00001-of-00003.safetensors", + "alignment_projector.mlp.1.bias": "model-00001-of-00003.safetensors", + "alignment_projector.mlp.1.weight": "model-00001-of-00003.safetensors", + "alignment_projector.mlp.3.bias": "model-00001-of-00003.safetensors", + "alignment_projector.mlp.3.weight": "model-00001-of-00003.safetensors", + "alignment_projector.norm.bias": "model-00001-of-00003.safetensors", + "alignment_projector.norm.weight": "model-00001-of-00003.safetensors", + "alignment_projector.state_proj.bias": "model-00001-of-00003.safetensors", + "alignment_projector.state_proj.weight": "model-00001-of-00003.safetensors", + "latent_init_embedding": "model-00001-of-00003.safetensors", + "lm_head.weight": "model-00001-of-00003.safetensors", + "model.language_model.embed_tokens.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.0.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.1.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.10.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.10.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.11.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.11.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.11.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.11.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.11.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.11.self_attn.k_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.11.self_attn.k_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.11.self_attn.o_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.11.self_attn.q_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.11.self_attn.q_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.11.self_attn.v_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.12.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.13.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.14.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.self_attn.k_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.self_attn.k_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.self_attn.o_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.self_attn.q_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.self_attn.q_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.15.self_attn.v_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.16.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.17.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.18.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.self_attn.k_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.self_attn.k_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.self_attn.o_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.self_attn.q_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.self_attn.q_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.19.self_attn.v_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.2.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.2.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.20.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.20.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.21.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.22.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.self_attn.k_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.self_attn.k_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.self_attn.o_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.self_attn.q_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.self_attn.q_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.23.self_attn.v_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.24.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.25.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.26.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.self_attn.k_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.self_attn.k_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.self_attn.o_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.self_attn.q_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.self_attn.q_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.27.self_attn.v_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.28.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.29.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.3.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.self_attn.k_norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.self_attn.k_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.self_attn.o_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.self_attn.q_norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.self_attn.q_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.3.self_attn.v_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.30.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.A_log": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.conv1d.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.dt_bias": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.in_proj_a.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.in_proj_b.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.in_proj_qkv.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.in_proj_z.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.linear_attn.out_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.30.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.31.input_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.31.mlp.down_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.31.mlp.gate_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.31.mlp.up_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.31.post_attention_layernorm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.31.self_attn.k_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.31.self_attn.k_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.31.self_attn.o_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.31.self_attn.q_norm.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.31.self_attn.q_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.31.self_attn.v_proj.weight": "model-00002-of-00003.safetensors", + "model.language_model.layers.4.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.4.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.5.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.6.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.self_attn.k_norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.self_attn.k_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.self_attn.o_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.self_attn.q_norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.self_attn.q_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.7.self_attn.v_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.8.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.input_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.A_log": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.conv1d.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.dt_bias": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.in_proj_a.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.in_proj_b.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.in_proj_qkv.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.in_proj_z.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.norm.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.linear_attn.out_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.mlp.down_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.mlp.gate_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.mlp.up_proj.weight": "model-00001-of-00003.safetensors", + "model.language_model.layers.9.post_attention_layernorm.weight": "model-00001-of-00003.safetensors", + "model.language_model.norm.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.0.attn.proj.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.0.attn.proj.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.0.attn.qkv.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.0.attn.qkv.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.0.mlp.linear_fc1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.0.mlp.linear_fc1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.0.mlp.linear_fc2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.0.mlp.linear_fc2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.0.norm1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.0.norm1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.0.norm2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.0.norm2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.1.attn.proj.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.1.attn.proj.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.1.attn.qkv.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.1.attn.qkv.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.1.mlp.linear_fc1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.1.mlp.linear_fc1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.1.mlp.linear_fc2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.1.mlp.linear_fc2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.1.norm1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.1.norm1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.1.norm2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.1.norm2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.10.attn.proj.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.10.attn.proj.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.10.attn.qkv.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.10.attn.qkv.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.10.mlp.linear_fc1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.10.mlp.linear_fc1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.10.mlp.linear_fc2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.10.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.10.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.10.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.10.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.10.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.11.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.12.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.13.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.14.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.15.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.16.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.17.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.18.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.19.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.2.attn.proj.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.2.attn.proj.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.2.attn.qkv.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.2.attn.qkv.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.2.mlp.linear_fc1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.2.mlp.linear_fc1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.2.mlp.linear_fc2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.2.mlp.linear_fc2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.2.norm1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.2.norm1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.2.norm2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.2.norm2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.20.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.20.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.21.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.22.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.attn.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.attn.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.attn.qkv.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.attn.qkv.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.mlp.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.mlp.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.mlp.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.mlp.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.norm1.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.norm1.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.norm2.bias": "model-00003-of-00003.safetensors", + "model.visual.blocks.23.norm2.weight": "model-00003-of-00003.safetensors", + "model.visual.blocks.3.attn.proj.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.3.attn.proj.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.3.attn.qkv.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.3.attn.qkv.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.3.mlp.linear_fc1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.3.mlp.linear_fc1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.3.mlp.linear_fc2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.3.mlp.linear_fc2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.3.norm1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.3.norm1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.3.norm2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.3.norm2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.4.attn.proj.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.4.attn.proj.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.4.attn.qkv.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.4.attn.qkv.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.4.mlp.linear_fc1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.4.mlp.linear_fc1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.4.mlp.linear_fc2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.4.mlp.linear_fc2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.4.norm1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.4.norm1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.4.norm2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.4.norm2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.5.attn.proj.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.5.attn.proj.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.5.attn.qkv.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.5.attn.qkv.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.5.mlp.linear_fc1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.5.mlp.linear_fc1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.5.mlp.linear_fc2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.5.mlp.linear_fc2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.5.norm1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.5.norm1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.5.norm2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.5.norm2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.6.attn.proj.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.6.attn.proj.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.6.attn.qkv.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.6.attn.qkv.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.6.mlp.linear_fc1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.6.mlp.linear_fc1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.6.mlp.linear_fc2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.6.mlp.linear_fc2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.6.norm1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.6.norm1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.6.norm2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.6.norm2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.7.attn.proj.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.7.attn.proj.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.7.attn.qkv.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.7.attn.qkv.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.7.mlp.linear_fc1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.7.mlp.linear_fc1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.7.mlp.linear_fc2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.7.mlp.linear_fc2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.7.norm1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.7.norm1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.7.norm2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.7.norm2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.8.attn.proj.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.8.attn.proj.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.8.attn.qkv.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.8.attn.qkv.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.8.mlp.linear_fc1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.8.mlp.linear_fc1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.8.mlp.linear_fc2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.8.mlp.linear_fc2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.8.norm1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.8.norm1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.8.norm2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.8.norm2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.9.attn.proj.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.9.attn.proj.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.9.attn.qkv.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.9.attn.qkv.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.9.mlp.linear_fc1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.9.mlp.linear_fc1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.9.mlp.linear_fc2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.9.mlp.linear_fc2.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.9.norm1.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.9.norm1.weight": "model-00002-of-00003.safetensors", + "model.visual.blocks.9.norm2.bias": "model-00002-of-00003.safetensors", + "model.visual.blocks.9.norm2.weight": "model-00002-of-00003.safetensors", + "model.visual.merger.linear_fc1.bias": "model-00003-of-00003.safetensors", + "model.visual.merger.linear_fc1.weight": "model-00003-of-00003.safetensors", + "model.visual.merger.linear_fc2.bias": "model-00003-of-00003.safetensors", + "model.visual.merger.linear_fc2.weight": "model-00003-of-00003.safetensors", + "model.visual.merger.norm.bias": "model-00003-of-00003.safetensors", + "model.visual.merger.norm.weight": "model-00003-of-00003.safetensors", + "model.visual.patch_embed.proj.bias": "model-00003-of-00003.safetensors", + "model.visual.patch_embed.proj.weight": "model-00003-of-00003.safetensors", + "model.visual.pos_embed.weight": "model-00003-of-00003.safetensors" + } +} diff --git a/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/preprocessor_config.json b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/preprocessor_config.json new file mode 100644 index 0000000000000000000000000000000000000000..2ea84a437d448ff71b08df68fdd949d5cc4ebb64 --- /dev/null +++ b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/preprocessor_config.json @@ -0,0 +1,21 @@ +{ + "size": { + "longest_edge": 16777216, + "shortest_edge": 65536 + }, + "patch_size": 16, + "temporal_patch_size": 2, + "merge_size": 2, + "image_mean": [ + 0.5, + 0.5, + 0.5 + ], + "image_std": [ + 0.5, + 0.5, + 0.5 + ], + "processor_class": "Qwen3VLProcessor", + "image_processor_type": "Qwen2VLImageProcessorFast" +} \ No newline at end of file diff --git a/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/processor_config.json b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/processor_config.json new file mode 100644 index 0000000000000000000000000000000000000000..33818c7f9e991ad735fd240209f4fa73e6c28c50 --- /dev/null +++ b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/processor_config.json @@ -0,0 +1,60 @@ +{ + "image_processor": { + "do_convert_rgb": true, + "do_normalize": true, + "do_rescale": true, + "do_resize": true, + "image_mean": [ + 0.5, + 0.5, + 0.5 + ], + "image_processor_type": "Qwen2VLImageProcessor", + "image_std": [ + 0.5, + 0.5, + 0.5 + ], + "merge_size": 2, + "patch_size": 16, + "resample": 3, + "rescale_factor": 0.00392156862745098, + "size": { + "longest_edge": 16777216, + "shortest_edge": 65536 + }, + "temporal_patch_size": 2 + }, + "processor_class": "Qwen3VLProcessor", + "video_processor": { + "do_convert_rgb": true, + "do_normalize": true, + "do_rescale": true, + "do_resize": true, + "do_sample_frames": true, + "fps": 2, + "image_mean": [ + 0.5, + 0.5, + 0.5 + ], + "image_std": [ + 0.5, + 0.5, + 0.5 + ], + "max_frames": 768, + "merge_size": 2, + "min_frames": 4, + "patch_size": 16, + "resample": 3, + "rescale_factor": 0.00392156862745098, + "return_metadata": false, + "size": { + "longest_edge": 25165824, + "shortest_edge": 4096 + }, + "temporal_patch_size": 2, + "video_processor_type": "Qwen3VLVideoProcessor" + } +} diff --git a/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/tokenizer.json b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..9e8ac8634b7c90a05a65182d43516ddf7371c521 --- /dev/null +++ b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4012eb888c8edaca05866241a3a8319ca02d45c367eef3da25b8ba4ef002f5b3 +size 19989885 diff --git a/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/tokenizer_config.json b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..1d134cd298be1e3be25db393d93a1cefe80e3214 --- /dev/null +++ b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/tokenizer_config.json @@ -0,0 +1,33 @@ +{ + "add_prefix_space": false, + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "backend": "tokenizers", + "bos_token": null, + "clean_up_tokenization_spaces": false, + "eos_token": "<|im_end|>", + "errors": "replace", + "image_token": "<|image_pad|>", + "is_local": true, + "local_files_only": false, + "model_max_length": 262144, + "model_specific_special_tokens": { + "audio_bos_token": "<|audio_start|>", + "audio_eos_token": "<|audio_end|>", + "audio_token": "<|audio_pad|>", + "image_token": "<|image_pad|>", + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" + }, + "pad_token": "<|endoftext|>", + "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+", + "processor_class": "Qwen3VLProcessor", + "split_special_tokens": false, + "tokenizer_class": "Qwen2Tokenizer", + "unk_token": null, + "video_token": "<|video_pad|>", + "vision_bos_token": "<|vision_start|>", + "vision_eos_token": "<|vision_end|>" +} diff --git a/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/trainer_state.json b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..d9c5d1f0f0f3adc9a2837619abfd68b4fd18f179 --- /dev/null +++ b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/trainer_state.json @@ -0,0 +1,7104 @@ +{ + "best_global_step": 500, + "best_metric": 0.71396095, + "best_model_checkpoint": "/NHNHOME/data/sanghyeok/qwen-cua/runs/ocu_exact_rt_sig_ac_clean_gcon_cmp/v0-20260613-023235/checkpoint-500", + "epoch": 0.11531531531531532, + "eval_steps": 100, + "global_step": 500, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.00023063063063063062, + "grad_norm": 439.0276794433594, + "learning_rate": 5.0000000000000004e-08, + "loss": 26.146907806396484, + "loss_alignment": 0.99169921875, + "loss_alignment_w": 0.495849609375, + "loss_sft": 1.1391561850905418, + "loss_total_v6": 1.6341817677021027, + "step": 1 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.00046126126126126124, + "grad_norm": 500.2991638183594, + "learning_rate": 1.0000000000000001e-07, + "loss": 26.98201560974121, + "loss_alignment": 0.9990234375, + "loss_alignment_w": 0.49951171875, + "loss_sft": 1.1877798289060593, + "loss_total_v6": 1.686376005411148, + "step": 2 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0006918918918918919, + "grad_norm": 534.3580932617188, + "learning_rate": 1.5000000000000002e-07, + "loss": 28.312280654907227, + "loss_alignment": 0.9970703125, + "loss_alignment_w": 0.49853515625, + "loss_sft": 1.270722895860672, + "loss_total_v6": 1.7695174664258957, + "step": 3 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0009225225225225225, + "grad_norm": 507.1813049316406, + "learning_rate": 2.0000000000000002e-07, + "loss": 27.52889633178711, + "loss_alignment": 1.0, + "loss_alignment_w": 0.5, + "loss_sft": 1.2208765000104904, + "loss_total_v6": 1.7205560207366943, + "step": 4 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0011531531531531532, + "grad_norm": 472.163818359375, + "learning_rate": 2.5000000000000004e-07, + "loss": 26.532962799072266, + "loss_alignment": 0.998046875, + "loss_alignment_w": 0.4990234375, + "loss_sft": 1.1594393029808998, + "loss_total_v6": 1.658310130238533, + "step": 5 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0013837837837837837, + "grad_norm": 516.5152587890625, + "learning_rate": 3.0000000000000004e-07, + "loss": 27.257396697998047, + "loss_alignment": 0.998046875, + "loss_alignment_w": 0.4990234375, + "loss_sft": 1.2058456540107727, + "loss_total_v6": 1.7035873383283615, + "step": 6 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0016144144144144145, + "grad_norm": 500.1063232421875, + "learning_rate": 3.5000000000000004e-07, + "loss": 27.427608489990234, + "loss_alignment": 0.990234375, + "loss_alignment_w": 0.4951171875, + "loss_sft": 1.2192762345075607, + "loss_total_v6": 1.7142255455255508, + "step": 7 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.001845045045045045, + "grad_norm": 548.2033081054688, + "learning_rate": 4.0000000000000003e-07, + "loss": 27.40910530090332, + "loss_alignment": 0.99755859375, + "loss_alignment_w": 0.498779296875, + "loss_sft": 1.2146102488040924, + "loss_total_v6": 1.7130690813064575, + "step": 8 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0020756756756756755, + "grad_norm": 437.5999450683594, + "learning_rate": 4.5000000000000003e-07, + "loss": 26.050594329833984, + "loss_alignment": 0.998046875, + "loss_alignment_w": 0.4990234375, + "loss_sft": 1.1290166974067688, + "loss_total_v6": 1.6281621903181076, + "step": 9 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0023063063063063064, + "grad_norm": 419.8359069824219, + "learning_rate": 5.000000000000001e-07, + "loss": 25.78015899658203, + "loss_alignment": 1.0, + "loss_alignment_w": 0.5, + "loss_sft": 1.1117788255214691, + "loss_total_v6": 1.6112600564956665, + "step": 10 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.002536936936936937, + "grad_norm": 365.5787048339844, + "learning_rate": 5.5e-07, + "loss": 25.50562286376953, + "loss_alignment": 0.99072265625, + "loss_alignment_w": 0.495361328125, + "loss_sft": 1.0993351265788078, + "loss_total_v6": 1.5941013991832733, + "step": 11 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0027675675675675675, + "grad_norm": 314.37579345703125, + "learning_rate": 6.000000000000001e-07, + "loss": 24.77506446838379, + "loss_alignment": 0.99853515625, + "loss_alignment_w": 0.499267578125, + "loss_sft": 1.0497995540499687, + "loss_total_v6": 1.5484415590763092, + "step": 12 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0029981981981981984, + "grad_norm": 291.4332580566406, + "learning_rate": 6.5e-07, + "loss": 23.8707332611084, + "loss_alignment": 0.9912109375, + "loss_alignment_w": 0.49560546875, + "loss_sft": 0.9965900033712387, + "loss_total_v6": 1.491920828819275, + "step": 13 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.003228828828828829, + "grad_norm": 191.8142547607422, + "learning_rate": 7.000000000000001e-07, + "loss": 22.40311050415039, + "loss_alignment": 0.9990234375, + "loss_alignment_w": 0.49951171875, + "loss_sft": 0.9021322727203369, + "loss_total_v6": 1.4001943916082382, + "step": 14 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0034594594594594594, + "grad_norm": 159.45960998535156, + "learning_rate": 7.5e-07, + "loss": 22.563995361328125, + "loss_alignment": 0.99658203125, + "loss_alignment_w": 0.498291015625, + "loss_sft": 0.9119739681482315, + "loss_total_v6": 1.4102496802806854, + "step": 15 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.00369009009009009, + "grad_norm": 152.0430908203125, + "learning_rate": 8.000000000000001e-07, + "loss": 21.12885284423828, + "loss_alignment": 0.9970703125, + "loss_alignment_w": 0.49853515625, + "loss_sft": 0.8228878527879715, + "loss_total_v6": 1.3205533027648926, + "step": 16 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.003920720720720721, + "grad_norm": 132.17184448242188, + "learning_rate": 8.500000000000001e-07, + "loss": 21.82652473449707, + "loss_alignment": 0.990234375, + "loss_alignment_w": 0.4951171875, + "loss_sft": 0.8704443722963333, + "loss_total_v6": 1.3641577810049057, + "step": 17 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.004151351351351351, + "grad_norm": 133.6917266845703, + "learning_rate": 9.000000000000001e-07, + "loss": 20.7487850189209, + "loss_alignment": 0.9970703125, + "loss_alignment_w": 0.49853515625, + "loss_sft": 0.7991642877459526, + "loss_total_v6": 1.296799123287201, + "step": 18 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.004381981981981982, + "grad_norm": 112.31056213378906, + "learning_rate": 9.500000000000001e-07, + "loss": 20.871936798095703, + "loss_alignment": 0.998046875, + "loss_alignment_w": 0.4990234375, + "loss_sft": 0.8064186871051788, + "loss_total_v6": 1.3044960796833038, + "step": 19 + }, + { + "action_cond_aligned": 0.9765625, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.004612612612612613, + "grad_norm": 147.16363525390625, + "learning_rate": 1.0000000000000002e-06, + "loss": 20.460861206054688, + "loss_alignment": 0.9697265625, + "loss_alignment_w": 0.48486328125, + "loss_sft": 0.79487144947052, + "loss_total_v6": 1.278803899884224, + "step": 20 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.004843243243243243, + "grad_norm": 116.33827209472656, + "learning_rate": 1.0500000000000001e-06, + "loss": 19.67446517944336, + "loss_alignment": 0.98486328125, + "loss_alignment_w": 0.492431640625, + "loss_sft": 0.7394044697284698, + "loss_total_v6": 1.2296540588140488, + "step": 21 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.005073873873873874, + "grad_norm": 96.91616821289062, + "learning_rate": 1.1e-06, + "loss": 19.485607147216797, + "loss_alignment": 0.974609375, + "loss_alignment_w": 0.4873046875, + "loss_sft": 0.7328041419386864, + "loss_total_v6": 1.2178505212068558, + "step": 22 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.005304504504504505, + "grad_norm": 84.47233581542969, + "learning_rate": 1.1500000000000002e-06, + "loss": 19.21877098083496, + "loss_alignment": 0.9833984375, + "loss_alignment_w": 0.49169921875, + "loss_sft": 0.7122510671615601, + "loss_total_v6": 1.201173186302185, + "step": 23 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.005535135135135135, + "grad_norm": 85.31909942626953, + "learning_rate": 1.2000000000000002e-06, + "loss": 18.99209976196289, + "loss_alignment": 0.99072265625, + "loss_alignment_w": 0.495361328125, + "loss_sft": 0.6947730183601379, + "loss_total_v6": 1.1870063245296478, + "step": 24 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.005765765765765766, + "grad_norm": 91.31905364990234, + "learning_rate": 1.25e-06, + "loss": 18.692764282226562, + "loss_alignment": 0.98291015625, + "loss_alignment_w": 0.491455078125, + "loss_sft": 0.678139753639698, + "loss_total_v6": 1.1682978421449661, + "step": 25 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.005996396396396397, + "grad_norm": 81.24137878417969, + "learning_rate": 1.3e-06, + "loss": 18.92623519897461, + "loss_alignment": 0.97607421875, + "loss_alignment_w": 0.488037109375, + "loss_sft": 0.6972634568810463, + "loss_total_v6": 1.182889699935913, + "step": 26 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.006227027027027027, + "grad_norm": 107.83047485351562, + "learning_rate": 1.3500000000000002e-06, + "loss": 18.555086135864258, + "loss_alignment": 0.97314453125, + "loss_alignment_w": 0.486572265625, + "loss_sft": 0.6738530471920967, + "loss_total_v6": 1.1596928983926773, + "step": 27 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.006457657657657658, + "grad_norm": 92.99987030029297, + "learning_rate": 1.4000000000000001e-06, + "loss": 18.18427276611328, + "loss_alignment": 0.96630859375, + "loss_alignment_w": 0.483154296875, + "loss_sft": 0.6532101556658745, + "loss_total_v6": 1.1365170627832413, + "step": 28 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.006688288288288288, + "grad_norm": 71.58002471923828, + "learning_rate": 1.45e-06, + "loss": 18.10936164855957, + "loss_alignment": 0.96044921875, + "loss_alignment_w": 0.480224609375, + "loss_sft": 0.650633879005909, + "loss_total_v6": 1.1318350434303284, + "step": 29 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.006918918918918919, + "grad_norm": 63.12832260131836, + "learning_rate": 1.5e-06, + "loss": 17.84610939025879, + "loss_alignment": 0.94580078125, + "loss_alignment_w": 0.472900390625, + "loss_sft": 0.6412454918026924, + "loss_total_v6": 1.1153818517923355, + "step": 30 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.00714954954954955, + "grad_norm": 58.475467681884766, + "learning_rate": 1.5500000000000002e-06, + "loss": 17.696842193603516, + "loss_alignment": 0.94921875, + "loss_alignment_w": 0.474609375, + "loss_sft": 0.6289864778518677, + "loss_total_v6": 1.1060525625944138, + "step": 31 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.00738018018018018, + "grad_norm": 53.76437759399414, + "learning_rate": 1.6000000000000001e-06, + "loss": 17.36236572265625, + "loss_alignment": 0.94091796875, + "loss_alignment_w": 0.470458984375, + "loss_sft": 0.6130409464240074, + "loss_total_v6": 1.0851478427648544, + "step": 32 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.007610810810810811, + "grad_norm": 53.21007537841797, + "learning_rate": 1.6500000000000003e-06, + "loss": 17.258296966552734, + "loss_alignment": 0.94384765625, + "loss_alignment_w": 0.471923828125, + "loss_sft": 0.6057430803775787, + "loss_total_v6": 1.0786435157060623, + "step": 33 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.007841441441441442, + "grad_norm": 56.078094482421875, + "learning_rate": 1.7000000000000002e-06, + "loss": 17.20782470703125, + "loss_alignment": 0.939453125, + "loss_alignment_w": 0.4697265625, + "loss_sft": 0.6055335029959679, + "loss_total_v6": 1.075488954782486, + "step": 34 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.008072072072072072, + "grad_norm": 49.554718017578125, + "learning_rate": 1.75e-06, + "loss": 16.908308029174805, + "loss_alignment": 0.9296875, + "loss_alignment_w": 0.46484375, + "loss_sft": 0.5928562059998512, + "loss_total_v6": 1.0567691922187805, + "step": 35 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.008302702702702702, + "grad_norm": 55.66875457763672, + "learning_rate": 1.8000000000000001e-06, + "loss": 17.240007400512695, + "loss_alignment": 0.9267578125, + "loss_alignment_w": 0.46337890625, + "loss_sft": 0.6156473979353905, + "loss_total_v6": 1.077500432729721, + "step": 36 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.008533333333333334, + "grad_norm": 51.596351623535156, + "learning_rate": 1.85e-06, + "loss": 17.088653564453125, + "loss_alignment": 0.91064453125, + "loss_alignment_w": 0.455322265625, + "loss_sft": 0.6127490475773811, + "loss_total_v6": 1.0680407881736755, + "step": 37 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.008763963963963964, + "grad_norm": 51.967376708984375, + "learning_rate": 1.9000000000000002e-06, + "loss": 16.826202392578125, + "loss_alignment": 0.89990234375, + "loss_alignment_w": 0.449951171875, + "loss_sft": 0.6014423742890358, + "loss_total_v6": 1.0516376942396164, + "step": 38 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.008994594594594594, + "grad_norm": 48.50193405151367, + "learning_rate": 1.9500000000000004e-06, + "loss": 16.247900009155273, + "loss_alignment": 0.8935546875, + "loss_alignment_w": 0.44677734375, + "loss_sft": 0.5678618922829628, + "loss_total_v6": 1.015493743121624, + "step": 39 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.009225225225225226, + "grad_norm": 44.81880569458008, + "learning_rate": 2.0000000000000003e-06, + "loss": 16.415573120117188, + "loss_alignment": 0.88671875, + "loss_alignment_w": 0.443359375, + "loss_sft": 0.5835599601268768, + "loss_total_v6": 1.0259732753038406, + "step": 40 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.009455855855855856, + "grad_norm": 42.74640655517578, + "learning_rate": 2.05e-06, + "loss": 16.188968658447266, + "loss_alignment": 0.875, + "loss_alignment_w": 0.4375, + "loss_sft": 0.575088694691658, + "loss_total_v6": 1.0118104964494705, + "step": 41 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.009686486486486486, + "grad_norm": 42.12061309814453, + "learning_rate": 2.1000000000000002e-06, + "loss": 16.455902099609375, + "loss_alignment": 0.8603515625, + "loss_alignment_w": 0.43017578125, + "loss_sft": 0.5992183610796928, + "loss_total_v6": 1.0284938737750053, + "step": 42 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.009917117117117118, + "grad_norm": 50.96394348144531, + "learning_rate": 2.15e-06, + "loss": 16.0130558013916, + "loss_alignment": 0.85400390625, + "loss_alignment_w": 0.427001953125, + "loss_sft": 0.5734478533267975, + "loss_total_v6": 1.0008160173892975, + "step": 43 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.010147747747747748, + "grad_norm": 48.8274040222168, + "learning_rate": 2.2e-06, + "loss": 15.956930160522461, + "loss_alignment": 0.8505859375, + "loss_alignment_w": 0.42529296875, + "loss_sft": 0.5732358172535896, + "loss_total_v6": 0.997308112680912, + "step": 44 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.010378378378378378, + "grad_norm": 47.16791534423828, + "learning_rate": 2.25e-06, + "loss": 15.690572738647461, + "loss_alignment": 0.8369140625, + "loss_alignment_w": 0.41845703125, + "loss_sft": 0.562920905649662, + "loss_total_v6": 0.9806607589125633, + "step": 45 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.01060900900900901, + "grad_norm": 55.0712776184082, + "learning_rate": 2.3000000000000004e-06, + "loss": 15.520903587341309, + "loss_alignment": 0.8310546875, + "loss_alignment_w": 0.41552734375, + "loss_sft": 0.5540255829691887, + "loss_total_v6": 0.9700564816594124, + "step": 46 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.01083963963963964, + "grad_norm": 41.40166091918945, + "learning_rate": 2.35e-06, + "loss": 15.894697189331055, + "loss_alignment": 0.82861328125, + "loss_alignment_w": 0.414306640625, + "loss_sft": 0.5793102607131004, + "loss_total_v6": 0.9934185519814491, + "step": 47 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.01107027027027027, + "grad_norm": 43.96187973022461, + "learning_rate": 2.4000000000000003e-06, + "loss": 15.351356506347656, + "loss_alignment": 0.81494140625, + "loss_alignment_w": 0.407470703125, + "loss_sft": 0.5529656559228897, + "loss_total_v6": 0.9594597890973091, + "step": 48 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.011300900900900902, + "grad_norm": 42.595890045166016, + "learning_rate": 2.4500000000000003e-06, + "loss": 15.166361808776855, + "loss_alignment": 0.8115234375, + "loss_alignment_w": 0.40576171875, + "loss_sft": 0.5419680662453175, + "loss_total_v6": 0.9478976279497147, + "step": 49 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.011531531531531532, + "grad_norm": 44.41075134277344, + "learning_rate": 2.5e-06, + "loss": 14.95013427734375, + "loss_alignment": 0.7802734375, + "loss_alignment_w": 0.39013671875, + "loss_sft": 0.5427208133041859, + "loss_total_v6": 0.934383399784565, + "step": 50 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.011762162162162162, + "grad_norm": 48.160213470458984, + "learning_rate": 2.55e-06, + "loss": 14.87997055053711, + "loss_alignment": 0.76318359375, + "loss_alignment_w": 0.381591796875, + "loss_sft": 0.5490930341184139, + "loss_total_v6": 0.9299981966614723, + "step": 51 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.011992792792792794, + "grad_norm": 41.55876922607422, + "learning_rate": 2.6e-06, + "loss": 15.20147705078125, + "loss_alignment": 0.767578125, + "loss_alignment_w": 0.3837890625, + "loss_sft": 0.5661049000918865, + "loss_total_v6": 0.9500923231244087, + "step": 52 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.012223423423423424, + "grad_norm": 43.1703987121582, + "learning_rate": 2.6500000000000005e-06, + "loss": 15.258073806762695, + "loss_alignment": 0.76318359375, + "loss_alignment_w": 0.381591796875, + "loss_sft": 0.5720684006810188, + "loss_total_v6": 0.953629620373249, + "step": 53 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.012454054054054054, + "grad_norm": 49.836978912353516, + "learning_rate": 2.7000000000000004e-06, + "loss": 14.69503402709961, + "loss_alignment": 0.7451171875, + "loss_alignment_w": 0.37255859375, + "loss_sft": 0.5460946299135685, + "loss_total_v6": 0.9184396043419838, + "step": 54 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.012684684684684684, + "grad_norm": 44.47955322265625, + "learning_rate": 2.7500000000000004e-06, + "loss": 14.992738723754883, + "loss_alignment": 0.7431640625, + "loss_alignment_w": 0.37158203125, + "loss_sft": 0.5651436746120453, + "loss_total_v6": 0.937046155333519, + "step": 55 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.012915315315315316, + "grad_norm": 48.49378967285156, + "learning_rate": 2.8000000000000003e-06, + "loss": 14.763578414916992, + "loss_alignment": 0.73583984375, + "loss_alignment_w": 0.367919921875, + "loss_sft": 0.5535982549190521, + "loss_total_v6": 0.9227236285805702, + "step": 56 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.013145945945945946, + "grad_norm": 49.66253662109375, + "learning_rate": 2.85e-06, + "loss": 14.56563663482666, + "loss_alignment": 0.7333984375, + "loss_alignment_w": 0.36669921875, + "loss_sft": 0.541760977357626, + "loss_total_v6": 0.9103522822260857, + "step": 57 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.013376576576576576, + "grad_norm": 54.137325286865234, + "learning_rate": 2.9e-06, + "loss": 14.35837173461914, + "loss_alignment": 0.73193359375, + "loss_alignment_w": 0.365966796875, + "loss_sft": 0.531950231641531, + "loss_total_v6": 0.8973982334136963, + "step": 58 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.013607207207207208, + "grad_norm": 143.80364990234375, + "learning_rate": 2.95e-06, + "loss": 14.914738655090332, + "loss_alignment": 0.71875, + "loss_alignment_w": 0.359375, + "loss_sft": 0.5725367590785027, + "loss_total_v6": 0.9321711510419846, + "step": 59 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.013837837837837838, + "grad_norm": 47.25136947631836, + "learning_rate": 3e-06, + "loss": 14.622476577758789, + "loss_alignment": 0.71142578125, + "loss_alignment_w": 0.355712890625, + "loss_sft": 0.5577341131865978, + "loss_total_v6": 0.9139047861099243, + "step": 60 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.014068468468468468, + "grad_norm": 49.77809524536133, + "learning_rate": 3.05e-06, + "loss": 14.11844539642334, + "loss_alignment": 0.712890625, + "loss_alignment_w": 0.3564453125, + "loss_sft": 0.5258659459650517, + "loss_total_v6": 0.8824028447270393, + "step": 61 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0142990990990991, + "grad_norm": 41.52340316772461, + "learning_rate": 3.1000000000000004e-06, + "loss": 14.74616527557373, + "loss_alignment": 0.708984375, + "loss_alignment_w": 0.3544921875, + "loss_sft": 0.5668990090489388, + "loss_total_v6": 0.921635314822197, + "step": 62 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.01452972972972973, + "grad_norm": 44.59492874145508, + "learning_rate": 3.1500000000000003e-06, + "loss": 13.965381622314453, + "loss_alignment": 0.697265625, + "loss_alignment_w": 0.3486328125, + "loss_sft": 0.5241120010614395, + "loss_total_v6": 0.8728363737463951, + "step": 63 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.01476036036036036, + "grad_norm": 42.699703216552734, + "learning_rate": 3.2000000000000003e-06, + "loss": 13.913717269897461, + "loss_alignment": 0.69921875, + "loss_alignment_w": 0.349609375, + "loss_sft": 0.5196775309741497, + "loss_total_v6": 0.8696073442697525, + "step": 64 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.014990990990990992, + "grad_norm": 38.93959045410156, + "learning_rate": 3.2500000000000002e-06, + "loss": 14.431289672851562, + "loss_alignment": 0.69482421875, + "loss_alignment_w": 0.347412109375, + "loss_sft": 0.55478760227561, + "loss_total_v6": 0.9019556120038033, + "step": 65 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.015221621621621622, + "grad_norm": 42.72892761230469, + "learning_rate": 3.3000000000000006e-06, + "loss": 14.124234199523926, + "loss_alignment": 0.68505859375, + "loss_alignment_w": 0.342529296875, + "loss_sft": 0.539838582277298, + "loss_total_v6": 0.8827646225690842, + "step": 66 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.015452252252252252, + "grad_norm": 39.99228286743164, + "learning_rate": 3.3500000000000005e-06, + "loss": 13.90633773803711, + "loss_alignment": 0.67724609375, + "loss_alignment_w": 0.338623046875, + "loss_sft": 0.5312401689589024, + "loss_total_v6": 0.8691460639238358, + "step": 67 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.015682882882882884, + "grad_norm": 65.71363067626953, + "learning_rate": 3.4000000000000005e-06, + "loss": 14.427297592163086, + "loss_alignment": 0.673828125, + "loss_alignment_w": 0.3369140625, + "loss_sft": 0.565020926296711, + "loss_total_v6": 0.9017061218619347, + "step": 68 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.015913513513513514, + "grad_norm": 44.44363784790039, + "learning_rate": 3.45e-06, + "loss": 13.518455505371094, + "loss_alignment": 0.66015625, + "loss_alignment_w": 0.330078125, + "loss_sft": 0.5146117098629475, + "loss_total_v6": 0.8449034467339516, + "step": 69 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.016144144144144144, + "grad_norm": 54.572872161865234, + "learning_rate": 3.5e-06, + "loss": 13.862147331237793, + "loss_alignment": 0.662109375, + "loss_alignment_w": 0.3310546875, + "loss_sft": 0.5353753082454205, + "loss_total_v6": 0.8663842082023621, + "step": 70 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.016374774774774774, + "grad_norm": 39.83063888549805, + "learning_rate": 3.5500000000000003e-06, + "loss": 13.652727127075195, + "loss_alignment": 0.6494140625, + "loss_alignment_w": 0.32470703125, + "loss_sft": 0.5283747911453247, + "loss_total_v6": 0.8532954826951027, + "step": 71 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.016605405405405404, + "grad_norm": 39.1588020324707, + "learning_rate": 3.6000000000000003e-06, + "loss": 13.264286041259766, + "loss_alignment": 0.64306640625, + "loss_alignment_w": 0.321533203125, + "loss_sft": 0.5074999630451202, + "loss_total_v6": 0.8290178999304771, + "step": 72 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.016836036036036037, + "grad_norm": 39.07815933227539, + "learning_rate": 3.65e-06, + "loss": 13.098067283630371, + "loss_alignment": 0.63623046875, + "loss_alignment_w": 0.318115234375, + "loss_sft": 0.5004987083375454, + "loss_total_v6": 0.8186292052268982, + "step": 73 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.017066666666666667, + "grad_norm": 39.64657974243164, + "learning_rate": 3.7e-06, + "loss": 13.649795532226562, + "loss_alignment": 0.63818359375, + "loss_alignment_w": 0.319091796875, + "loss_sft": 0.534356165677309, + "loss_total_v6": 0.8531122654676437, + "step": 74 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.017297297297297298, + "grad_norm": 40.814876556396484, + "learning_rate": 3.7500000000000005e-06, + "loss": 13.793352127075195, + "loss_alignment": 0.62646484375, + "loss_alignment_w": 0.313232421875, + "loss_sft": 0.5481502413749695, + "loss_total_v6": 0.8620845302939415, + "step": 75 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.017527927927927928, + "grad_norm": 44.19852828979492, + "learning_rate": 3.8000000000000005e-06, + "loss": 13.07166862487793, + "loss_alignment": 0.63134765625, + "loss_alignment_w": 0.315673828125, + "loss_sft": 0.5021752044558525, + "loss_total_v6": 0.81697928160429, + "step": 76 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.017758558558558558, + "grad_norm": 37.5131950378418, + "learning_rate": 3.85e-06, + "loss": 13.6424560546875, + "loss_alignment": 0.63232421875, + "loss_alignment_w": 0.316162109375, + "loss_sft": 0.5367965288460255, + "loss_total_v6": 0.8526534661650658, + "step": 77 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.017989189189189188, + "grad_norm": 45.17282485961914, + "learning_rate": 3.900000000000001e-06, + "loss": 13.174717903137207, + "loss_alignment": 0.6201171875, + "loss_alignment_w": 0.31005859375, + "loss_sft": 0.5134375765919685, + "loss_total_v6": 0.8234198689460754, + "step": 78 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.01821981981981982, + "grad_norm": 36.88737869262695, + "learning_rate": 3.95e-06, + "loss": 13.664063453674316, + "loss_alignment": 0.61279296875, + "loss_alignment_w": 0.306396484375, + "loss_sft": 0.547241285443306, + "loss_total_v6": 0.8540039286017418, + "step": 79 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.01845045045045045, + "grad_norm": 43.357059478759766, + "learning_rate": 4.000000000000001e-06, + "loss": 13.295951843261719, + "loss_alignment": 0.6142578125, + "loss_alignment_w": 0.30712890625, + "loss_sft": 0.5241732709109783, + "loss_total_v6": 0.8309969827532768, + "step": 80 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.01868108108108108, + "grad_norm": 40.89968490600586, + "learning_rate": 4.05e-06, + "loss": 13.297460556030273, + "loss_alignment": 0.60205078125, + "loss_alignment_w": 0.301025390625, + "loss_sft": 0.5292266607284546, + "loss_total_v6": 0.8310913071036339, + "step": 81 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.01891171171171171, + "grad_norm": 35.835838317871094, + "learning_rate": 4.1e-06, + "loss": 13.24339485168457, + "loss_alignment": 0.6123046875, + "loss_alignment_w": 0.30615234375, + "loss_sft": 0.5218649581074715, + "loss_total_v6": 0.8277121484279633, + "step": 82 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.01914234234234234, + "grad_norm": 36.579681396484375, + "learning_rate": 4.15e-06, + "loss": 13.207765579223633, + "loss_alignment": 0.6123046875, + "loss_alignment_w": 0.30615234375, + "loss_sft": 0.5198365561664104, + "loss_total_v6": 0.825485348701477, + "step": 83 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.019372972972972972, + "grad_norm": 40.91501235961914, + "learning_rate": 4.2000000000000004e-06, + "loss": 13.323108673095703, + "loss_alignment": 0.60400390625, + "loss_alignment_w": 0.302001953125, + "loss_sft": 0.5312722288072109, + "loss_total_v6": 0.8326943442225456, + "step": 84 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.019603603603603605, + "grad_norm": 42.157230377197266, + "learning_rate": 4.25e-06, + "loss": 12.847209930419922, + "loss_alignment": 0.595703125, + "loss_alignment_w": 0.2978515625, + "loss_sft": 0.504839688539505, + "loss_total_v6": 0.8029506355524063, + "step": 85 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.019834234234234235, + "grad_norm": 33.32569885253906, + "learning_rate": 4.3e-06, + "loss": 13.206663131713867, + "loss_alignment": 0.59814453125, + "loss_alignment_w": 0.299072265625, + "loss_sft": 0.5257948227226734, + "loss_total_v6": 0.8254164159297943, + "step": 86 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.020064864864864865, + "grad_norm": 37.24618148803711, + "learning_rate": 4.350000000000001e-06, + "loss": 12.897462844848633, + "loss_alignment": 0.58447265625, + "loss_alignment_w": 0.292236328125, + "loss_sft": 0.5138703398406506, + "loss_total_v6": 0.8060914054512978, + "step": 87 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.020295495495495496, + "grad_norm": 40.878604888916016, + "learning_rate": 4.4e-06, + "loss": 13.272189140319824, + "loss_alignment": 0.583984375, + "loss_alignment_w": 0.2919921875, + "loss_sft": 0.5375959165394306, + "loss_total_v6": 0.8295117989182472, + "step": 88 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.020526126126126126, + "grad_norm": 50.75047302246094, + "learning_rate": 4.450000000000001e-06, + "loss": 13.007837295532227, + "loss_alignment": 0.57470703125, + "loss_alignment_w": 0.287353515625, + "loss_sft": 0.525697335600853, + "loss_total_v6": 0.8129898235201836, + "step": 89 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.020756756756756756, + "grad_norm": 40.63207244873047, + "learning_rate": 4.5e-06, + "loss": 12.732479095458984, + "loss_alignment": 0.57373046875, + "loss_alignment_w": 0.286865234375, + "loss_sft": 0.5092961899936199, + "loss_total_v6": 0.7957799732685089, + "step": 90 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.020987387387387386, + "grad_norm": 40.568904876708984, + "learning_rate": 4.5500000000000005e-06, + "loss": 12.692139625549316, + "loss_alignment": 0.57373046875, + "loss_alignment_w": 0.286865234375, + "loss_sft": 0.5061340965330601, + "loss_total_v6": 0.7932587340474129, + "step": 91 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.02121801801801802, + "grad_norm": 41.535316467285156, + "learning_rate": 4.600000000000001e-06, + "loss": 12.72914981842041, + "loss_alignment": 0.5595703125, + "loss_alignment_w": 0.27978515625, + "loss_sft": 0.515481498092413, + "loss_total_v6": 0.7955718412995338, + "step": 92 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.02144864864864865, + "grad_norm": 40.7745475769043, + "learning_rate": 4.65e-06, + "loss": 12.506476402282715, + "loss_alignment": 0.56787109375, + "loss_alignment_w": 0.283935546875, + "loss_sft": 0.4983143284916878, + "loss_total_v6": 0.7816547602415085, + "step": 93 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.02167927927927928, + "grad_norm": 37.23093795776367, + "learning_rate": 4.7e-06, + "loss": 12.715709686279297, + "loss_alignment": 0.56201171875, + "loss_alignment_w": 0.281005859375, + "loss_sft": 0.5136191733181477, + "loss_total_v6": 0.7947318330407143, + "step": 94 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.02190990990990991, + "grad_norm": 37.72247314453125, + "learning_rate": 4.75e-06, + "loss": 12.836080551147461, + "loss_alignment": 0.56640625, + "loss_alignment_w": 0.283203125, + "loss_sft": 0.520028468221426, + "loss_total_v6": 0.8022550344467163, + "step": 95 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.02214054054054054, + "grad_norm": 40.37207794189453, + "learning_rate": 4.800000000000001e-06, + "loss": 12.288612365722656, + "loss_alignment": 0.55322265625, + "loss_alignment_w": 0.276611328125, + "loss_sft": 0.4917015917599201, + "loss_total_v6": 0.7680382430553436, + "step": 96 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.02237117117117117, + "grad_norm": 39.364845275878906, + "learning_rate": 4.85e-06, + "loss": 12.279783248901367, + "loss_alignment": 0.54638671875, + "loss_alignment_w": 0.273193359375, + "loss_sft": 0.49365221336483955, + "loss_total_v6": 0.7674864679574966, + "step": 97 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.022601801801801803, + "grad_norm": 38.37849426269531, + "learning_rate": 4.9000000000000005e-06, + "loss": 12.229836463928223, + "loss_alignment": 0.560546875, + "loss_alignment_w": 0.2802734375, + "loss_sft": 0.4845643602311611, + "loss_total_v6": 0.7643647789955139, + "step": 98 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.022832432432432433, + "grad_norm": 36.85342025756836, + "learning_rate": 4.95e-06, + "loss": 12.803227424621582, + "loss_alignment": 0.5498046875, + "loss_alignment_w": 0.27490234375, + "loss_sft": 0.5249789580702782, + "loss_total_v6": 0.8002017512917519, + "step": 99 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.023063063063063063, + "grad_norm": 36.0045166015625, + "learning_rate": 5e-06, + "loss": 12.904866218566895, + "loss_alignment": 0.54443359375, + "loss_alignment_w": 0.272216796875, + "loss_sft": 0.5344746857881546, + "loss_total_v6": 0.8065541312098503, + "step": 100 + }, + { + "epoch": 0.023063063063063063, + "eval_loss": 0.7603949308395386, + "eval_loss_alignment": 0.5316736230022832, + "eval_loss_alignment_w": 0.2658368115011416, + "eval_loss_sft": 0.4945581131465903, + "eval_loss_total_v6": 0.760394924987941, + "eval_loss_wm_recon": 0.0, + "eval_loss_wm_recon_w": 0.0, + "eval_runtime": 211.5521, + "eval_samples_per_second": 8.249, + "eval_steps_per_second": 1.035, + "step": 100 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.023293693693693693, + "grad_norm": 36.68770217895508, + "learning_rate": 5.050000000000001e-06, + "loss": 12.502948760986328, + "loss_alignment": 0.54296875, + "loss_alignment_w": 0.271484375, + "loss_sft": 0.5100720003247261, + "loss_total_v6": 0.7814343273639679, + "step": 101 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.023524324324324324, + "grad_norm": 34.5877571105957, + "learning_rate": 5.1e-06, + "loss": 12.602604866027832, + "loss_alignment": 0.5439453125, + "loss_alignment_w": 0.27197265625, + "loss_sft": 0.5160869397222996, + "loss_total_v6": 0.7876628413796425, + "step": 102 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.023754954954954954, + "grad_norm": 37.871517181396484, + "learning_rate": 5.150000000000001e-06, + "loss": 11.892693519592285, + "loss_alignment": 0.54248046875, + "loss_alignment_w": 0.271240234375, + "loss_sft": 0.4720226228237152, + "loss_total_v6": 0.7432933524250984, + "step": 103 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.023985585585585587, + "grad_norm": 38.78010559082031, + "learning_rate": 5.2e-06, + "loss": 12.868436813354492, + "loss_alignment": 0.54052734375, + "loss_alignment_w": 0.270263671875, + "loss_sft": 0.5340746864676476, + "loss_total_v6": 0.8042773082852364, + "step": 104 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.024216216216216217, + "grad_norm": 38.05021286010742, + "learning_rate": 5.2500000000000006e-06, + "loss": 12.377391815185547, + "loss_alignment": 0.5302734375, + "loss_alignment_w": 0.26513671875, + "loss_sft": 0.5085723623633385, + "loss_total_v6": 0.7735870108008385, + "step": 105 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.024446846846846847, + "grad_norm": 36.38343048095703, + "learning_rate": 5.300000000000001e-06, + "loss": 11.81425666809082, + "loss_alignment": 0.5283203125, + "loss_alignment_w": 0.26416015625, + "loss_sft": 0.4738341085612774, + "loss_total_v6": 0.7383910268545151, + "step": 106 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.024677477477477477, + "grad_norm": 35.74465560913086, + "learning_rate": 5.3500000000000004e-06, + "loss": 12.362968444824219, + "loss_alignment": 0.53369140625, + "loss_alignment_w": 0.266845703125, + "loss_sft": 0.5052904970943928, + "loss_total_v6": 0.7726855054497719, + "step": 107 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.024908108108108108, + "grad_norm": 33.50358963012695, + "learning_rate": 5.400000000000001e-06, + "loss": 12.476561546325684, + "loss_alignment": 0.5322265625, + "loss_alignment_w": 0.26611328125, + "loss_sft": 0.5130919553339481, + "loss_total_v6": 0.7797850593924522, + "step": 108 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.025138738738738738, + "grad_norm": 39.65915298461914, + "learning_rate": 5.450000000000001e-06, + "loss": 12.096356391906738, + "loss_alignment": 0.5263671875, + "loss_alignment_w": 0.26318359375, + "loss_sft": 0.49323543161153793, + "loss_total_v6": 0.7560222744941711, + "step": 109 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.025369369369369368, + "grad_norm": 35.459716796875, + "learning_rate": 5.500000000000001e-06, + "loss": 12.067989349365234, + "loss_alignment": 0.524658203125, + "loss_alignment_w": 0.2623291015625, + "loss_sft": 0.4919813238084316, + "loss_total_v6": 0.7542493790388107, + "step": 110 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0256, + "grad_norm": 36.72193145751953, + "learning_rate": 5.550000000000001e-06, + "loss": 12.380197525024414, + "loss_alignment": 0.5244140625, + "loss_alignment_w": 0.26220703125, + "loss_sft": 0.511448472738266, + "loss_total_v6": 0.7737623304128647, + "step": 111 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.02583063063063063, + "grad_norm": 39.09083557128906, + "learning_rate": 5.600000000000001e-06, + "loss": 11.901430130004883, + "loss_alignment": 0.5166015625, + "loss_alignment_w": 0.25830078125, + "loss_sft": 0.48535552248358727, + "loss_total_v6": 0.7438393905758858, + "step": 112 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.02606126126126126, + "grad_norm": 35.048606872558594, + "learning_rate": 5.65e-06, + "loss": 12.05987548828125, + "loss_alignment": 0.5205078125, + "loss_alignment_w": 0.26025390625, + "loss_sft": 0.49371713027358055, + "loss_total_v6": 0.7537421733140945, + "step": 113 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.02629189189189189, + "grad_norm": 37.565067291259766, + "learning_rate": 5.7e-06, + "loss": 11.644591331481934, + "loss_alignment": 0.521484375, + "loss_alignment_w": 0.2607421875, + "loss_sft": 0.46687688678503036, + "loss_total_v6": 0.7277869433164597, + "step": 114 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.02652252252252252, + "grad_norm": 33.95093536376953, + "learning_rate": 5.75e-06, + "loss": 12.189088821411133, + "loss_alignment": 0.5146484375, + "loss_alignment_w": 0.25732421875, + "loss_sft": 0.5042039565742016, + "loss_total_v6": 0.7618180811405182, + "step": 115 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.02675315315315315, + "grad_norm": 35.17734909057617, + "learning_rate": 5.8e-06, + "loss": 11.92866325378418, + "loss_alignment": 0.51708984375, + "loss_alignment_w": 0.258544921875, + "loss_sft": 0.48763736337423325, + "loss_total_v6": 0.74554143846035, + "step": 116 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.026983783783783785, + "grad_norm": 35.09998321533203, + "learning_rate": 5.85e-06, + "loss": 11.819519996643066, + "loss_alignment": 0.51708984375, + "loss_alignment_w": 0.258544921875, + "loss_sft": 0.48003773391246796, + "loss_total_v6": 0.7387200146913528, + "step": 117 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.027214414414414415, + "grad_norm": 33.770320892333984, + "learning_rate": 5.9e-06, + "loss": 11.795919418334961, + "loss_alignment": 0.509765625, + "loss_alignment_w": 0.2548828125, + "loss_sft": 0.48152292519807816, + "loss_total_v6": 0.7372449561953545, + "step": 118 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.027445045045045045, + "grad_norm": 35.33518981933594, + "learning_rate": 5.950000000000001e-06, + "loss": 11.388572692871094, + "loss_alignment": 0.510986328125, + "loss_alignment_w": 0.2554931640625, + "loss_sft": 0.45624687522649765, + "loss_total_v6": 0.7117858231067657, + "step": 119 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.027675675675675675, + "grad_norm": 33.83852005004883, + "learning_rate": 6e-06, + "loss": 11.63688850402832, + "loss_alignment": 0.498779296875, + "loss_alignment_w": 0.2493896484375, + "loss_sft": 0.47761065885424614, + "loss_total_v6": 0.7273054718971252, + "step": 120 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.027906306306306305, + "grad_norm": 34.03733444213867, + "learning_rate": 6.0500000000000005e-06, + "loss": 11.487631797790527, + "loss_alignment": 0.509521484375, + "loss_alignment_w": 0.2547607421875, + "loss_sft": 0.46312466636300087, + "loss_total_v6": 0.7179769799113274, + "step": 121 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.028136936936936936, + "grad_norm": 35.97840118408203, + "learning_rate": 6.1e-06, + "loss": 12.252918243408203, + "loss_alignment": 0.504150390625, + "loss_alignment_w": 0.2520751953125, + "loss_sft": 0.5137475207448006, + "loss_total_v6": 0.7658074200153351, + "step": 122 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.02836756756756757, + "grad_norm": 36.2728385925293, + "learning_rate": 6.15e-06, + "loss": 11.812028884887695, + "loss_alignment": 0.496337890625, + "loss_alignment_w": 0.2481689453125, + "loss_sft": 0.49028119072318077, + "loss_total_v6": 0.7382517755031586, + "step": 123 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0285981981981982, + "grad_norm": 32.81572341918945, + "learning_rate": 6.200000000000001e-06, + "loss": 11.900299072265625, + "loss_alignment": 0.493896484375, + "loss_alignment_w": 0.2469482421875, + "loss_sft": 0.49695784226059914, + "loss_total_v6": 0.7437687516212463, + "step": 124 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.02882882882882883, + "grad_norm": 35.772552490234375, + "learning_rate": 6.25e-06, + "loss": 11.7321195602417, + "loss_alignment": 0.49853515625, + "loss_alignment_w": 0.249267578125, + "loss_sft": 0.4839593768119812, + "loss_total_v6": 0.7332574725151062, + "step": 125 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.02905945945945946, + "grad_norm": 30.98957633972168, + "learning_rate": 6.300000000000001e-06, + "loss": 11.462678909301758, + "loss_alignment": 0.50830078125, + "loss_alignment_w": 0.254150390625, + "loss_sft": 0.4621907062828541, + "loss_total_v6": 0.7164174169301987, + "step": 126 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.02929009009009009, + "grad_norm": 32.6396484375, + "learning_rate": 6.35e-06, + "loss": 11.754329681396484, + "loss_alignment": 0.498291015625, + "loss_alignment_w": 0.2491455078125, + "loss_sft": 0.4860188625752926, + "loss_total_v6": 0.7346455678343773, + "step": 127 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.02952072072072072, + "grad_norm": 31.852096557617188, + "learning_rate": 6.4000000000000006e-06, + "loss": 11.584247589111328, + "loss_alignment": 0.4970703125, + "loss_alignment_w": 0.24853515625, + "loss_sft": 0.4757702462375164, + "loss_total_v6": 0.7240155190229416, + "step": 128 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.029751351351351353, + "grad_norm": 32.783714294433594, + "learning_rate": 6.450000000000001e-06, + "loss": 11.665525436401367, + "loss_alignment": 0.490478515625, + "loss_alignment_w": 0.2452392578125, + "loss_sft": 0.48405444622039795, + "loss_total_v6": 0.7290953397750854, + "step": 129 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.029981981981981983, + "grad_norm": 30.54566764831543, + "learning_rate": 6.5000000000000004e-06, + "loss": 11.98303508758545, + "loss_alignment": 0.49267578125, + "loss_alignment_w": 0.246337890625, + "loss_sft": 0.5026780776679516, + "loss_total_v6": 0.7489396929740906, + "step": 130 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.030212612612612613, + "grad_norm": 41.61016082763672, + "learning_rate": 6.550000000000001e-06, + "loss": 11.350701332092285, + "loss_alignment": 0.4892578125, + "loss_alignment_w": 0.24462890625, + "loss_sft": 0.4652171954512596, + "loss_total_v6": 0.7094188630580902, + "step": 131 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.030443243243243243, + "grad_norm": 31.93428611755371, + "learning_rate": 6.600000000000001e-06, + "loss": 11.879814147949219, + "loss_alignment": 0.49853515625, + "loss_alignment_w": 0.249267578125, + "loss_sft": 0.4936327338218689, + "loss_total_v6": 0.742488332092762, + "step": 132 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.030673873873873873, + "grad_norm": 34.60322189331055, + "learning_rate": 6.650000000000001e-06, + "loss": 11.332307815551758, + "loss_alignment": 0.486083984375, + "loss_alignment_w": 0.2430419921875, + "loss_sft": 0.4650440663099289, + "loss_total_v6": 0.7082691863179207, + "step": 133 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.030904504504504503, + "grad_norm": 32.404293060302734, + "learning_rate": 6.700000000000001e-06, + "loss": 11.678876876831055, + "loss_alignment": 0.48876953125, + "loss_alignment_w": 0.244384765625, + "loss_sft": 0.485545065253973, + "loss_total_v6": 0.7299298197031021, + "step": 134 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.031135135135135134, + "grad_norm": 32.375484466552734, + "learning_rate": 6.750000000000001e-06, + "loss": 11.51457691192627, + "loss_alignment": 0.49169921875, + "loss_alignment_w": 0.245849609375, + "loss_sft": 0.47378091141581535, + "loss_total_v6": 0.7196610569953918, + "step": 135 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03136576576576577, + "grad_norm": 34.645286560058594, + "learning_rate": 6.800000000000001e-06, + "loss": 11.535043716430664, + "loss_alignment": 0.483642578125, + "loss_alignment_w": 0.2418212890625, + "loss_sft": 0.47914944216609, + "loss_total_v6": 0.7209402024745941, + "step": 136 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.031596396396396394, + "grad_norm": 30.149227142333984, + "learning_rate": 6.850000000000001e-06, + "loss": 11.960624694824219, + "loss_alignment": 0.48974609375, + "loss_alignment_w": 0.244873046875, + "loss_sft": 0.5028491243720055, + "loss_total_v6": 0.7475390210747719, + "step": 137 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03182702702702703, + "grad_norm": 31.145723342895508, + "learning_rate": 6.9e-06, + "loss": 11.0299072265625, + "loss_alignment": 0.487548828125, + "loss_alignment_w": 0.2437744140625, + "loss_sft": 0.44496918469667435, + "loss_total_v6": 0.6893692016601562, + "step": 138 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03205765765765766, + "grad_norm": 31.852170944213867, + "learning_rate": 6.95e-06, + "loss": 11.626943588256836, + "loss_alignment": 0.488525390625, + "loss_alignment_w": 0.2442626953125, + "loss_sft": 0.48266541212797165, + "loss_total_v6": 0.7266839742660522, + "step": 139 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03228828828828829, + "grad_norm": 51.08174133300781, + "learning_rate": 7e-06, + "loss": 11.608294486999512, + "loss_alignment": 0.488037109375, + "loss_alignment_w": 0.2440185546875, + "loss_sft": 0.48185084015130997, + "loss_total_v6": 0.7255184352397919, + "step": 140 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03251891891891892, + "grad_norm": 34.58945083618164, + "learning_rate": 7.05e-06, + "loss": 11.699002265930176, + "loss_alignment": 0.4833984375, + "loss_alignment_w": 0.24169921875, + "loss_sft": 0.4895189329981804, + "loss_total_v6": 0.7311876714229584, + "step": 141 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03274954954954955, + "grad_norm": 38.467018127441406, + "learning_rate": 7.100000000000001e-06, + "loss": 11.408781051635742, + "loss_alignment": 0.4853515625, + "loss_alignment_w": 0.24267578125, + "loss_sft": 0.4702662564814091, + "loss_total_v6": 0.7130488529801369, + "step": 142 + }, + { + "action_cond_aligned": 0.9765625, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03298018018018018, + "grad_norm": 32.846343994140625, + "learning_rate": 7.15e-06, + "loss": 11.273344039916992, + "loss_alignment": 0.47509765625, + "loss_alignment_w": 0.237548828125, + "loss_sft": 0.4669131003320217, + "loss_total_v6": 0.7045839875936508, + "step": 143 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03321081081081081, + "grad_norm": 38.2901725769043, + "learning_rate": 7.2000000000000005e-06, + "loss": 11.750795364379883, + "loss_alignment": 0.48828125, + "loss_alignment_w": 0.244140625, + "loss_sft": 0.4904976524412632, + "loss_total_v6": 0.7344246879220009, + "step": 144 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03344144144144144, + "grad_norm": 32.86590576171875, + "learning_rate": 7.25e-06, + "loss": 11.431882858276367, + "loss_alignment": 0.482177734375, + "loss_alignment_w": 0.2410888671875, + "loss_sft": 0.47392264008522034, + "loss_total_v6": 0.7144926935434341, + "step": 145 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.033672072072072075, + "grad_norm": 29.40862464904785, + "learning_rate": 7.3e-06, + "loss": 11.947632789611816, + "loss_alignment": 0.4873046875, + "loss_alignment_w": 0.24365234375, + "loss_sft": 0.5028916262090206, + "loss_total_v6": 0.7467270866036415, + "step": 146 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0339027027027027, + "grad_norm": 31.30272674560547, + "learning_rate": 7.350000000000001e-06, + "loss": 11.27760124206543, + "loss_alignment": 0.494384765625, + "loss_alignment_w": 0.2471923828125, + "loss_sft": 0.45785602182149887, + "loss_total_v6": 0.704850047826767, + "step": 147 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.034133333333333335, + "grad_norm": 33.112213134765625, + "learning_rate": 7.4e-06, + "loss": 11.63190746307373, + "loss_alignment": 0.479736328125, + "loss_alignment_w": 0.2398681640625, + "loss_sft": 0.48727861419320107, + "loss_total_v6": 0.726994201540947, + "step": 148 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03436396396396396, + "grad_norm": 33.42546463012695, + "learning_rate": 7.450000000000001e-06, + "loss": 11.333279609680176, + "loss_alignment": 0.477783203125, + "loss_alignment_w": 0.2388916015625, + "loss_sft": 0.4691179320216179, + "loss_total_v6": 0.7083299607038498, + "step": 149 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.034594594594594595, + "grad_norm": 30.737577438354492, + "learning_rate": 7.500000000000001e-06, + "loss": 11.238577842712402, + "loss_alignment": 0.48193359375, + "loss_alignment_w": 0.240966796875, + "loss_sft": 0.46106286346912384, + "loss_total_v6": 0.7024111449718475, + "step": 150 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03482522522522523, + "grad_norm": 133.33059692382812, + "learning_rate": 7.5500000000000006e-06, + "loss": 12.227761268615723, + "loss_alignment": 0.481201171875, + "loss_alignment_w": 0.2406005859375, + "loss_sft": 0.5238786414265633, + "loss_total_v6": 0.7642350494861603, + "step": 151 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.035055855855855855, + "grad_norm": 42.39982604980469, + "learning_rate": 7.600000000000001e-06, + "loss": 11.003484725952148, + "loss_alignment": 0.4833984375, + "loss_alignment_w": 0.24169921875, + "loss_sft": 0.4461558982729912, + "loss_total_v6": 0.6877177953720093, + "step": 152 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03528648648648649, + "grad_norm": 145.70811462402344, + "learning_rate": 7.650000000000001e-06, + "loss": 11.646322250366211, + "loss_alignment": 0.4794921875, + "loss_alignment_w": 0.23974609375, + "loss_sft": 0.4882711172103882, + "loss_total_v6": 0.7278951480984688, + "step": 153 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.035517117117117115, + "grad_norm": 35.15251159667969, + "learning_rate": 7.7e-06, + "loss": 11.495044708251953, + "loss_alignment": 0.487060546875, + "loss_alignment_w": 0.2435302734375, + "loss_sft": 0.47515418007969856, + "loss_total_v6": 0.7184402868151665, + "step": 154 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03574774774774775, + "grad_norm": 29.29365348815918, + "learning_rate": 7.75e-06, + "loss": 11.912647247314453, + "loss_alignment": 0.492919921875, + "loss_alignment_w": 0.2464599609375, + "loss_sft": 0.4979431964457035, + "loss_total_v6": 0.7445404678583145, + "step": 155 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.035978378378378376, + "grad_norm": 30.243995666503906, + "learning_rate": 7.800000000000002e-06, + "loss": 11.600605010986328, + "loss_alignment": 0.484130859375, + "loss_alignment_w": 0.2420654296875, + "loss_sft": 0.482606228441, + "loss_total_v6": 0.7250378578901291, + "step": 156 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03620900900900901, + "grad_norm": 40.444183349609375, + "learning_rate": 7.850000000000001e-06, + "loss": 11.138669967651367, + "loss_alignment": 0.4853515625, + "loss_alignment_w": 0.24267578125, + "loss_sft": 0.45378104597330093, + "loss_total_v6": 0.6961668729782104, + "step": 157 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03643963963963964, + "grad_norm": 30.37200927734375, + "learning_rate": 7.9e-06, + "loss": 11.368667602539062, + "loss_alignment": 0.488037109375, + "loss_alignment_w": 0.2440185546875, + "loss_sft": 0.4664315991103649, + "loss_total_v6": 0.7105417177081108, + "step": 158 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03667027027027027, + "grad_norm": 32.30575942993164, + "learning_rate": 7.950000000000002e-06, + "loss": 11.416982650756836, + "loss_alignment": 0.48388671875, + "loss_alignment_w": 0.241943359375, + "loss_sft": 0.47129761055111885, + "loss_total_v6": 0.7135613784193993, + "step": 159 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0369009009009009, + "grad_norm": 28.06721305847168, + "learning_rate": 8.000000000000001e-06, + "loss": 10.358413696289062, + "loss_alignment": 0.480712890625, + "loss_alignment_w": 0.2403564453125, + "loss_sft": 0.40750209614634514, + "loss_total_v6": 0.6474008038640022, + "step": 160 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03713153153153153, + "grad_norm": 29.02115821838379, + "learning_rate": 8.050000000000001e-06, + "loss": 11.356025695800781, + "loss_alignment": 0.48046875, + "loss_alignment_w": 0.240234375, + "loss_sft": 0.46907468885183334, + "loss_total_v6": 0.709751583635807, + "step": 161 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03736216216216216, + "grad_norm": 30.30998992919922, + "learning_rate": 8.1e-06, + "loss": 11.658432006835938, + "loss_alignment": 0.492919921875, + "loss_alignment_w": 0.2464599609375, + "loss_sft": 0.48173435777425766, + "loss_total_v6": 0.7286520376801491, + "step": 162 + }, + { + "action_cond_aligned": 0.9765625, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03759279279279279, + "grad_norm": 31.25248908996582, + "learning_rate": 8.15e-06, + "loss": 11.242467880249023, + "loss_alignment": 0.477294921875, + "loss_alignment_w": 0.2386474609375, + "loss_sft": 0.4635184481739998, + "loss_total_v6": 0.7026542350649834, + "step": 163 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03782342342342342, + "grad_norm": 32.52606201171875, + "learning_rate": 8.2e-06, + "loss": 11.442350387573242, + "loss_alignment": 0.48095703125, + "loss_alignment_w": 0.240478515625, + "loss_sft": 0.47434795647859573, + "loss_total_v6": 0.7151468992233276, + "step": 164 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03805405405405406, + "grad_norm": 33.31549072265625, + "learning_rate": 8.25e-06, + "loss": 11.126302719116211, + "loss_alignment": 0.48583984375, + "loss_alignment_w": 0.242919921875, + "loss_sft": 0.4524587467312813, + "loss_total_v6": 0.695393942296505, + "step": 165 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03828468468468468, + "grad_norm": 30.15898323059082, + "learning_rate": 8.3e-06, + "loss": 11.807147026062012, + "loss_alignment": 0.4912109375, + "loss_alignment_w": 0.24560546875, + "loss_sft": 0.49244802817702293, + "loss_total_v6": 0.7379466965794563, + "step": 166 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03851531531531532, + "grad_norm": 28.19438362121582, + "learning_rate": 8.35e-06, + "loss": 11.425878524780273, + "loss_alignment": 0.484375, + "loss_alignment_w": 0.2421875, + "loss_sft": 0.4717315286397934, + "loss_total_v6": 0.7141174003481865, + "step": 167 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.038745945945945943, + "grad_norm": 27.054704666137695, + "learning_rate": 8.400000000000001e-06, + "loss": 11.794561386108398, + "loss_alignment": 0.49267578125, + "loss_alignment_w": 0.246337890625, + "loss_sft": 0.4914630576968193, + "loss_total_v6": 0.7371600717306137, + "step": 168 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03897657657657658, + "grad_norm": 30.420682907104492, + "learning_rate": 8.45e-06, + "loss": 11.920896530151367, + "loss_alignment": 0.476318359375, + "loss_alignment_w": 0.2381591796875, + "loss_sft": 0.5066221505403519, + "loss_total_v6": 0.7450560182332993, + "step": 169 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03920720720720721, + "grad_norm": 34.55197525024414, + "learning_rate": 8.5e-06, + "loss": 11.988027572631836, + "loss_alignment": 0.488037109375, + "loss_alignment_w": 0.2440185546875, + "loss_sft": 0.5058740526437759, + "loss_total_v6": 0.7492517307400703, + "step": 170 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03943783783783784, + "grad_norm": 30.191083908081055, + "learning_rate": 8.550000000000001e-06, + "loss": 11.615690231323242, + "loss_alignment": 0.48291015625, + "loss_alignment_w": 0.241455078125, + "loss_sft": 0.48486124724149704, + "loss_total_v6": 0.7259806171059608, + "step": 171 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.03966846846846847, + "grad_norm": 44.080665588378906, + "learning_rate": 8.6e-06, + "loss": 11.60816478729248, + "loss_alignment": 0.483154296875, + "loss_alignment_w": 0.2415771484375, + "loss_sft": 0.48379581421613693, + "loss_total_v6": 0.7255102917551994, + "step": 172 + }, + { + "action_cond_aligned": 0.9765625, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0398990990990991, + "grad_norm": 32.76341247558594, + "learning_rate": 8.65e-06, + "loss": 10.936525344848633, + "loss_alignment": 0.465087890625, + "loss_alignment_w": 0.2325439453125, + "loss_sft": 0.45146186649799347, + "loss_total_v6": 0.6835328042507172, + "step": 173 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04012972972972973, + "grad_norm": 33.15988540649414, + "learning_rate": 8.700000000000001e-06, + "loss": 11.159172058105469, + "loss_alignment": 0.482177734375, + "loss_alignment_w": 0.2410888671875, + "loss_sft": 0.4561151750385761, + "loss_total_v6": 0.6974482089281082, + "step": 174 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04036036036036036, + "grad_norm": 31.26700210571289, + "learning_rate": 8.750000000000001e-06, + "loss": 11.683128356933594, + "loss_alignment": 0.4892578125, + "loss_alignment_w": 0.24462890625, + "loss_sft": 0.4853987395763397, + "loss_total_v6": 0.7301954925060272, + "step": 175 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04059099099099099, + "grad_norm": 28.46639633178711, + "learning_rate": 8.8e-06, + "loss": 11.497734069824219, + "loss_alignment": 0.488525390625, + "loss_alignment_w": 0.2442626953125, + "loss_sft": 0.47437621280550957, + "loss_total_v6": 0.7186083868145943, + "step": 176 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.040821621621621625, + "grad_norm": 27.947410583496094, + "learning_rate": 8.85e-06, + "loss": 11.242040634155273, + "loss_alignment": 0.486328125, + "loss_alignment_w": 0.2431640625, + "loss_sft": 0.4594787172973156, + "loss_total_v6": 0.7026275172829628, + "step": 177 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04105225225225225, + "grad_norm": 38.73674774169922, + "learning_rate": 8.900000000000001e-06, + "loss": 11.371100425720215, + "loss_alignment": 0.49072265625, + "loss_alignment_w": 0.245361328125, + "loss_sft": 0.46583597734570503, + "loss_total_v6": 0.7106937766075134, + "step": 178 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.041282882882882885, + "grad_norm": 28.507984161376953, + "learning_rate": 8.95e-06, + "loss": 11.638439178466797, + "loss_alignment": 0.483642578125, + "loss_alignment_w": 0.2418212890625, + "loss_sft": 0.4851081557571888, + "loss_total_v6": 0.7274024710059166, + "step": 179 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04151351351351351, + "grad_norm": 29.600019454956055, + "learning_rate": 9e-06, + "loss": 11.631031036376953, + "loss_alignment": 0.494384765625, + "loss_alignment_w": 0.2471923828125, + "loss_sft": 0.4801437593996525, + "loss_total_v6": 0.7269394248723984, + "step": 180 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.041744144144144145, + "grad_norm": 32.25768280029297, + "learning_rate": 9.050000000000001e-06, + "loss": 11.346786499023438, + "loss_alignment": 0.4794921875, + "loss_alignment_w": 0.23974609375, + "loss_sft": 0.46959590539336205, + "loss_total_v6": 0.7091741636395454, + "step": 181 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04197477477477477, + "grad_norm": 29.78536033630371, + "learning_rate": 9.100000000000001e-06, + "loss": 11.559186935424805, + "loss_alignment": 0.486083984375, + "loss_alignment_w": 0.2430419921875, + "loss_sft": 0.4795445501804352, + "loss_total_v6": 0.7224492207169533, + "step": 182 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.042205405405405405, + "grad_norm": 25.27552604675293, + "learning_rate": 9.15e-06, + "loss": 11.353900909423828, + "loss_alignment": 0.472900390625, + "loss_alignment_w": 0.2364501953125, + "loss_sft": 0.47371796891093254, + "loss_total_v6": 0.7096188366413116, + "step": 183 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04243603603603604, + "grad_norm": 25.628002166748047, + "learning_rate": 9.200000000000002e-06, + "loss": 11.49113655090332, + "loss_alignment": 0.485595703125, + "loss_alignment_w": 0.2427978515625, + "loss_sft": 0.4756118543446064, + "loss_total_v6": 0.7181960716843605, + "step": 184 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.042666666666666665, + "grad_norm": 27.595272064208984, + "learning_rate": 9.250000000000001e-06, + "loss": 11.143836975097656, + "loss_alignment": 0.48291015625, + "loss_alignment_w": 0.241455078125, + "loss_sft": 0.454897403717041, + "loss_total_v6": 0.6964897811412811, + "step": 185 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0428972972972973, + "grad_norm": 29.77658462524414, + "learning_rate": 9.3e-06, + "loss": 11.16983413696289, + "loss_alignment": 0.478271484375, + "loss_alignment_w": 0.2391357421875, + "loss_sft": 0.4590399041771889, + "loss_total_v6": 0.6981146037578583, + "step": 186 + }, + { + "action_cond_aligned": 0.9765625, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.043127927927927925, + "grad_norm": 35.990234375, + "learning_rate": 9.350000000000002e-06, + "loss": 11.228672981262207, + "loss_alignment": 0.478271484375, + "loss_alignment_w": 0.2391357421875, + "loss_sft": 0.46323613449931145, + "loss_total_v6": 0.7017920315265656, + "step": 187 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04335855855855856, + "grad_norm": 27.21523666381836, + "learning_rate": 9.4e-06, + "loss": 11.243139266967773, + "loss_alignment": 0.48486328125, + "loss_alignment_w": 0.242431640625, + "loss_sft": 0.46050866320729256, + "loss_total_v6": 0.7026961669325829, + "step": 188 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04358918918918919, + "grad_norm": 29.033153533935547, + "learning_rate": 9.450000000000001e-06, + "loss": 11.232537269592285, + "loss_alignment": 0.474609375, + "loss_alignment_w": 0.2373046875, + "loss_sft": 0.4645763114094734, + "loss_total_v6": 0.7020335868000984, + "step": 189 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04381981981981982, + "grad_norm": 27.855985641479492, + "learning_rate": 9.5e-06, + "loss": 11.404376983642578, + "loss_alignment": 0.474609375, + "loss_alignment_w": 0.2373046875, + "loss_sft": 0.47614022344350815, + "loss_total_v6": 0.7127735242247581, + "step": 190 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04405045045045045, + "grad_norm": 28.248476028442383, + "learning_rate": 9.55e-06, + "loss": 11.351865768432617, + "loss_alignment": 0.48828125, + "loss_alignment_w": 0.244140625, + "loss_sft": 0.4654577746987343, + "loss_total_v6": 0.7094916179776192, + "step": 191 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04428108108108108, + "grad_norm": 26.08788299560547, + "learning_rate": 9.600000000000001e-06, + "loss": 11.035432815551758, + "loss_alignment": 0.46875, + "loss_alignment_w": 0.234375, + "loss_sft": 0.45584307610988617, + "loss_total_v6": 0.6897145509719849, + "step": 192 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04451171171171171, + "grad_norm": 27.452720642089844, + "learning_rate": 9.65e-06, + "loss": 11.101813316345215, + "loss_alignment": 0.482177734375, + "loss_alignment_w": 0.2410888671875, + "loss_sft": 0.4523624815046787, + "loss_total_v6": 0.6938633397221565, + "step": 193 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04474234234234234, + "grad_norm": 34.1194953918457, + "learning_rate": 9.7e-06, + "loss": 11.296146392822266, + "loss_alignment": 0.476318359375, + "loss_alignment_w": 0.2381591796875, + "loss_sft": 0.4678651764988899, + "loss_total_v6": 0.7060091122984886, + "step": 194 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04497297297297297, + "grad_norm": 31.091306686401367, + "learning_rate": 9.75e-06, + "loss": 11.614121437072754, + "loss_alignment": 0.4716796875, + "loss_alignment_w": 0.23583984375, + "loss_sft": 0.4895086772739887, + "loss_total_v6": 0.7258825600147247, + "step": 195 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.045203603603603607, + "grad_norm": 30.47385597229004, + "learning_rate": 9.800000000000001e-06, + "loss": 11.521017074584961, + "loss_alignment": 0.478271484375, + "loss_alignment_w": 0.2391357421875, + "loss_sft": 0.4809277579188347, + "loss_total_v6": 0.7200635299086571, + "step": 196 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04543423423423423, + "grad_norm": 30.350555419921875, + "learning_rate": 9.85e-06, + "loss": 11.37847900390625, + "loss_alignment": 0.492919921875, + "loss_alignment_w": 0.2464599609375, + "loss_sft": 0.46583936363458633, + "loss_total_v6": 0.7111549153923988, + "step": 197 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04566486486486487, + "grad_norm": 29.04558753967285, + "learning_rate": 9.9e-06, + "loss": 11.232992172241211, + "loss_alignment": 0.476806640625, + "loss_alignment_w": 0.2384033203125, + "loss_sft": 0.4642080254852772, + "loss_total_v6": 0.7020620331168175, + "step": 198 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04589549549549549, + "grad_norm": 30.018320083618164, + "learning_rate": 9.950000000000001e-06, + "loss": 11.418879508972168, + "loss_alignment": 0.482177734375, + "loss_alignment_w": 0.2410888671875, + "loss_sft": 0.4721790961921215, + "loss_total_v6": 0.7136799544095993, + "step": 199 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04612612612612613, + "grad_norm": 27.505163192749023, + "learning_rate": 1e-05, + "loss": 12.028244018554688, + "loss_alignment": 0.47314453125, + "loss_alignment_w": 0.236572265625, + "loss_sft": 0.5147199705243111, + "loss_total_v6": 0.751765251159668, + "step": 200 + }, + { + "epoch": 0.04612612612612613, + "eval_loss": 0.7332190275192261, + "eval_loss_alignment": 0.467300763413242, + "eval_loss_alignment_w": 0.233650381706621, + "eval_loss_sft": 0.4995686298653007, + "eval_loss_total_v6": 0.7332190115634165, + "eval_loss_wm_recon": 0.0, + "eval_loss_wm_recon_w": 0.0, + "eval_runtime": 210.3881, + "eval_samples_per_second": 8.294, + "eval_steps_per_second": 1.041, + "step": 200 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04635675675675675, + "grad_norm": 26.609739303588867, + "learning_rate": 9.999998557623385e-06, + "loss": 11.591358184814453, + "loss_alignment": 0.481689453125, + "loss_alignment_w": 0.2408447265625, + "loss_sft": 0.48395080864429474, + "loss_total_v6": 0.7244598492980003, + "step": 201 + }, + { + "action_cond_aligned": 0.9765625, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04658738738738739, + "grad_norm": 27.139442443847656, + "learning_rate": 9.999994230494368e-06, + "loss": 11.43282699584961, + "loss_alignment": 0.471923828125, + "loss_alignment_w": 0.2359619140625, + "loss_sft": 0.4785134941339493, + "loss_total_v6": 0.7145517021417618, + "step": 202 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04681801801801802, + "grad_norm": 25.55487060546875, + "learning_rate": 9.999987018615447e-06, + "loss": 11.808528900146484, + "loss_alignment": 0.485595703125, + "loss_alignment_w": 0.2427978515625, + "loss_sft": 0.4952657222747803, + "loss_total_v6": 0.7380330488085747, + "step": 203 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04704864864864865, + "grad_norm": 25.824718475341797, + "learning_rate": 9.999976921990784e-06, + "loss": 11.070340156555176, + "loss_alignment": 0.489013671875, + "loss_alignment_w": 0.2445068359375, + "loss_sft": 0.4470384605228901, + "loss_total_v6": 0.6918962821364403, + "step": 204 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04727927927927928, + "grad_norm": 26.750333786010742, + "learning_rate": 9.999963940626203e-06, + "loss": 12.185787200927734, + "loss_alignment": 0.488525390625, + "loss_alignment_w": 0.2442626953125, + "loss_sft": 0.5171811766922474, + "loss_total_v6": 0.7616117298603058, + "step": 205 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04750990990990991, + "grad_norm": 27.215356826782227, + "learning_rate": 9.999948074529194e-06, + "loss": 11.26993179321289, + "loss_alignment": 0.48193359375, + "loss_alignment_w": 0.240966796875, + "loss_sft": 0.46352604776620865, + "loss_total_v6": 0.7043707743287086, + "step": 206 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04774054054054054, + "grad_norm": 27.858125686645508, + "learning_rate": 9.999929323708913e-06, + "loss": 11.215011596679688, + "loss_alignment": 0.474853515625, + "loss_alignment_w": 0.2374267578125, + "loss_sft": 0.46316054463386536, + "loss_total_v6": 0.7009382322430611, + "step": 207 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.047971171171171174, + "grad_norm": 26.925840377807617, + "learning_rate": 9.999907688176173e-06, + "loss": 11.664392471313477, + "loss_alignment": 0.470458984375, + "loss_alignment_w": 0.2352294921875, + "loss_sft": 0.49353567138314247, + "loss_total_v6": 0.7290245741605759, + "step": 208 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0482018018018018, + "grad_norm": 29.078609466552734, + "learning_rate": 9.999883167943461e-06, + "loss": 11.895009994506836, + "loss_alignment": 0.47900390625, + "loss_alignment_w": 0.239501953125, + "loss_sft": 0.503692027181387, + "loss_total_v6": 0.743438147008419, + "step": 209 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.048432432432432435, + "grad_norm": 31.282981872558594, + "learning_rate": 9.999855763024923e-06, + "loss": 10.957780838012695, + "loss_alignment": 0.48486328125, + "loss_alignment_w": 0.242431640625, + "loss_sft": 0.44291793182492256, + "loss_total_v6": 0.6848613247275352, + "step": 210 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04866306306306306, + "grad_norm": 29.630348205566406, + "learning_rate": 9.99982547343637e-06, + "loss": 11.374975204467773, + "loss_alignment": 0.476318359375, + "loss_alignment_w": 0.2381591796875, + "loss_sft": 0.47274621576070786, + "loss_total_v6": 0.7109359204769135, + "step": 211 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.048893693693693695, + "grad_norm": 28.66404151916504, + "learning_rate": 9.999792299195278e-06, + "loss": 11.906631469726562, + "loss_alignment": 0.479736328125, + "loss_alignment_w": 0.2398681640625, + "loss_sft": 0.5040827319025993, + "loss_total_v6": 0.7441645190119743, + "step": 212 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04912432432432432, + "grad_norm": 27.231538772583008, + "learning_rate": 9.999756240320786e-06, + "loss": 11.032928466796875, + "loss_alignment": 0.4775390625, + "loss_alignment_w": 0.23876953125, + "loss_sft": 0.4503154903650284, + "loss_total_v6": 0.6895580515265465, + "step": 213 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.049354954954954955, + "grad_norm": 27.36345863342285, + "learning_rate": 9.9997172968337e-06, + "loss": 11.203835487365723, + "loss_alignment": 0.47900390625, + "loss_alignment_w": 0.239501953125, + "loss_sft": 0.46067672222852707, + "loss_total_v6": 0.7002396956086159, + "step": 214 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.04958558558558559, + "grad_norm": 27.78192138671875, + "learning_rate": 9.999675468756485e-06, + "loss": 11.40892219543457, + "loss_alignment": 0.480224609375, + "loss_alignment_w": 0.2401123046875, + "loss_sft": 0.47288430109620094, + "loss_total_v6": 0.71305762976408, + "step": 215 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.049816216216216215, + "grad_norm": 24.95342445373535, + "learning_rate": 9.999630756113278e-06, + "loss": 11.767892837524414, + "loss_alignment": 0.48681640625, + "loss_alignment_w": 0.243408203125, + "loss_sft": 0.49206985905766487, + "loss_total_v6": 0.7354933395981789, + "step": 216 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05004684684684685, + "grad_norm": 27.6690731048584, + "learning_rate": 9.999583158929873e-06, + "loss": 10.98437213897705, + "loss_alignment": 0.4833984375, + "loss_alignment_w": 0.24169921875, + "loss_sft": 0.44471724331378937, + "loss_total_v6": 0.6865232735872269, + "step": 217 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.050277477477477475, + "grad_norm": 27.07967758178711, + "learning_rate": 9.999532677233733e-06, + "loss": 11.298315048217773, + "loss_alignment": 0.478759765625, + "loss_alignment_w": 0.2393798828125, + "loss_sft": 0.4667648524045944, + "loss_total_v6": 0.7061447277665138, + "step": 218 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05050810810810811, + "grad_norm": 28.920127868652344, + "learning_rate": 9.999479311053982e-06, + "loss": 11.27223014831543, + "loss_alignment": 0.486572265625, + "loss_alignment_w": 0.2432861328125, + "loss_sft": 0.46081627160310745, + "loss_total_v6": 0.7045144066214561, + "step": 219 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.050738738738738735, + "grad_norm": 27.202350616455078, + "learning_rate": 9.999423060421412e-06, + "loss": 11.516592025756836, + "loss_alignment": 0.47705078125, + "loss_alignment_w": 0.238525390625, + "loss_sft": 0.48124635964632034, + "loss_total_v6": 0.7197869941592216, + "step": 220 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05096936936936937, + "grad_norm": 27.233062744140625, + "learning_rate": 9.999363925368472e-06, + "loss": 10.901599884033203, + "loss_alignment": 0.4814453125, + "loss_alignment_w": 0.24072265625, + "loss_sft": 0.4406272992491722, + "loss_total_v6": 0.6813499554991722, + "step": 221 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0512, + "grad_norm": 29.04269790649414, + "learning_rate": 9.999301905929286e-06, + "loss": 11.448999404907227, + "loss_alignment": 0.475830078125, + "loss_alignment_w": 0.2379150390625, + "loss_sft": 0.47734224423766136, + "loss_total_v6": 0.7155624479055405, + "step": 222 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05143063063063063, + "grad_norm": 27.673601150512695, + "learning_rate": 9.999237002139633e-06, + "loss": 11.85905647277832, + "loss_alignment": 0.470458984375, + "loss_alignment_w": 0.2352294921875, + "loss_sft": 0.5056716352701187, + "loss_total_v6": 0.741191029548645, + "step": 223 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05166126126126126, + "grad_norm": 26.25150489807129, + "learning_rate": 9.999169214036958e-06, + "loss": 11.293829917907715, + "loss_alignment": 0.4814453125, + "loss_alignment_w": 0.24072265625, + "loss_sft": 0.4656299538910389, + "loss_total_v6": 0.7058643326163292, + "step": 224 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05189189189189189, + "grad_norm": 25.452255249023438, + "learning_rate": 9.999098541660375e-06, + "loss": 11.324202537536621, + "loss_alignment": 0.474609375, + "loss_alignment_w": 0.2373046875, + "loss_sft": 0.4703053720295429, + "loss_total_v6": 0.7077626511454582, + "step": 225 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05212252252252252, + "grad_norm": 27.15583610534668, + "learning_rate": 9.999024985050653e-06, + "loss": 11.486753463745117, + "loss_alignment": 0.473876953125, + "loss_alignment_w": 0.2369384765625, + "loss_sft": 0.4809836111962795, + "loss_total_v6": 0.7179221138358116, + "step": 226 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.052353153153153156, + "grad_norm": 30.19793701171875, + "learning_rate": 9.998948544250237e-06, + "loss": 11.549861907958984, + "loss_alignment": 0.478271484375, + "loss_alignment_w": 0.2391357421875, + "loss_sft": 0.4830205626785755, + "loss_total_v6": 0.7218663766980171, + "step": 227 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05258378378378378, + "grad_norm": 25.87264060974121, + "learning_rate": 9.998869219303227e-06, + "loss": 11.455377578735352, + "loss_alignment": 0.4775390625, + "loss_alignment_w": 0.23876953125, + "loss_sft": 0.47748148813843727, + "loss_total_v6": 0.7159611061215401, + "step": 228 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.052814414414414416, + "grad_norm": 29.141202926635742, + "learning_rate": 9.998787010255388e-06, + "loss": 11.162013053894043, + "loss_alignment": 0.48193359375, + "loss_alignment_w": 0.240966796875, + "loss_sft": 0.4562928043305874, + "loss_total_v6": 0.6976258084177971, + "step": 229 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05304504504504504, + "grad_norm": 27.606157302856445, + "learning_rate": 9.998701917154152e-06, + "loss": 10.48721981048584, + "loss_alignment": 0.478515625, + "loss_alignment_w": 0.2392578125, + "loss_sft": 0.4162239283323288, + "loss_total_v6": 0.6554512158036232, + "step": 230 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05327567567567568, + "grad_norm": 35.75673294067383, + "learning_rate": 9.998613940048613e-06, + "loss": 11.355308532714844, + "loss_alignment": 0.48193359375, + "loss_alignment_w": 0.240966796875, + "loss_sft": 0.46892303973436356, + "loss_total_v6": 0.7097067534923553, + "step": 231 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0535063063063063, + "grad_norm": 26.841440200805664, + "learning_rate": 9.99852307898953e-06, + "loss": 11.133645057678223, + "loss_alignment": 0.482177734375, + "loss_alignment_w": 0.2410888671875, + "loss_sft": 0.4547792263329029, + "loss_total_v6": 0.6958528310060501, + "step": 232 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05373693693693694, + "grad_norm": 26.802459716796875, + "learning_rate": 9.998429334029323e-06, + "loss": 11.569849014282227, + "loss_alignment": 0.472900390625, + "loss_alignment_w": 0.2364501953125, + "loss_sft": 0.48608554527163506, + "loss_total_v6": 0.7231156006455421, + "step": 233 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05396756756756757, + "grad_norm": 27.97140121459961, + "learning_rate": 9.998332705222083e-06, + "loss": 11.270265579223633, + "loss_alignment": 0.4775390625, + "loss_alignment_w": 0.23876953125, + "loss_sft": 0.4656373858451843, + "loss_total_v6": 0.7043916285037994, + "step": 234 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0541981981981982, + "grad_norm": 30.06894874572754, + "learning_rate": 9.998233192623556e-06, + "loss": 11.586647033691406, + "loss_alignment": 0.482421875, + "loss_alignment_w": 0.2412109375, + "loss_sft": 0.4824203886091709, + "loss_total_v6": 0.7241654247045517, + "step": 235 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05442882882882883, + "grad_norm": 27.803905487060547, + "learning_rate": 9.998130796291156e-06, + "loss": 11.388806343078613, + "loss_alignment": 0.483154296875, + "loss_alignment_w": 0.2415771484375, + "loss_sft": 0.4702385365962982, + "loss_total_v6": 0.7118004187941551, + "step": 236 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05465945945945946, + "grad_norm": 27.515819549560547, + "learning_rate": 9.998025516283964e-06, + "loss": 11.616597175598145, + "loss_alignment": 0.477783203125, + "loss_alignment_w": 0.2388916015625, + "loss_sft": 0.48729831352829933, + "loss_total_v6": 0.726037323474884, + "step": 237 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05489009009009009, + "grad_norm": 31.365758895874023, + "learning_rate": 9.997917352662718e-06, + "loss": 11.232580184936523, + "loss_alignment": 0.470458984375, + "loss_alignment_w": 0.2352294921875, + "loss_sft": 0.4669898711144924, + "loss_total_v6": 0.7020362466573715, + "step": 238 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05512072072072072, + "grad_norm": 27.221965789794922, + "learning_rate": 9.997806305489826e-06, + "loss": 11.391910552978516, + "loss_alignment": 0.476806640625, + "loss_alignment_w": 0.2384033203125, + "loss_sft": 0.4738199934363365, + "loss_total_v6": 0.711994431912899, + "step": 239 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05535135135135135, + "grad_norm": 28.87374496459961, + "learning_rate": 9.997692374829352e-06, + "loss": 11.407242774963379, + "loss_alignment": 0.481201171875, + "loss_alignment_w": 0.2406005859375, + "loss_sft": 0.4728556051850319, + "loss_total_v6": 0.7129526734352112, + "step": 240 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.055581981981981984, + "grad_norm": 25.56194496154785, + "learning_rate": 9.997575560747033e-06, + "loss": 11.545186996459961, + "loss_alignment": 0.481689453125, + "loss_alignment_w": 0.2408447265625, + "loss_sft": 0.48051585629582405, + "loss_total_v6": 0.7215742096304893, + "step": 241 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05581261261261261, + "grad_norm": 28.05449676513672, + "learning_rate": 9.997455863310263e-06, + "loss": 11.452713966369629, + "loss_alignment": 0.47802734375, + "loss_alignment_w": 0.239013671875, + "loss_sft": 0.4767351858317852, + "loss_total_v6": 0.7157946228981018, + "step": 242 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.056043243243243245, + "grad_norm": 26.37656021118164, + "learning_rate": 9.997333282588103e-06, + "loss": 11.68358039855957, + "loss_alignment": 0.489501953125, + "loss_alignment_w": 0.2447509765625, + "loss_sft": 0.48524389415979385, + "loss_total_v6": 0.730223760008812, + "step": 243 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05627387387387387, + "grad_norm": 27.562246322631836, + "learning_rate": 9.997207818651273e-06, + "loss": 10.854132652282715, + "loss_alignment": 0.472412109375, + "loss_alignment_w": 0.2362060546875, + "loss_sft": 0.4424366429448128, + "loss_total_v6": 0.6783833056688309, + "step": 244 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.056504504504504505, + "grad_norm": 29.82301902770996, + "learning_rate": 9.997079471572163e-06, + "loss": 11.16312026977539, + "loss_alignment": 0.47412109375, + "loss_alignment_w": 0.237060546875, + "loss_sft": 0.4598105102777481, + "loss_total_v6": 0.6976950317621231, + "step": 245 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05673513513513514, + "grad_norm": 27.683420181274414, + "learning_rate": 9.996948241424819e-06, + "loss": 11.051734924316406, + "loss_alignment": 0.478515625, + "loss_alignment_w": 0.2392578125, + "loss_sft": 0.451536625623703, + "loss_total_v6": 0.690733402967453, + "step": 246 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.056965765765765765, + "grad_norm": 28.251258850097656, + "learning_rate": 9.99681412828496e-06, + "loss": 11.429145812988281, + "loss_alignment": 0.47802734375, + "loss_alignment_w": 0.239013671875, + "loss_sft": 0.47520115971565247, + "loss_total_v6": 0.7143216729164124, + "step": 247 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0571963963963964, + "grad_norm": 25.78322410583496, + "learning_rate": 9.996677132229957e-06, + "loss": 11.652090072631836, + "loss_alignment": 0.481689453125, + "loss_alignment_w": 0.2408447265625, + "loss_sft": 0.48699889704585075, + "loss_total_v6": 0.7282555922865868, + "step": 248 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.057427027027027025, + "grad_norm": 25.888643264770508, + "learning_rate": 9.996537253338852e-06, + "loss": 10.796924591064453, + "loss_alignment": 0.487060546875, + "loss_alignment_w": 0.2435302734375, + "loss_sft": 0.4313385412096977, + "loss_total_v6": 0.6748077794909477, + "step": 249 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05765765765765766, + "grad_norm": 26.317501068115234, + "learning_rate": 9.99639449169235e-06, + "loss": 11.169139862060547, + "loss_alignment": 0.47900390625, + "loss_alignment_w": 0.239501953125, + "loss_sft": 0.4585082419216633, + "loss_total_v6": 0.698071226477623, + "step": 250 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.057888288288288285, + "grad_norm": 26.609331130981445, + "learning_rate": 9.996248847372813e-06, + "loss": 10.822470664978027, + "loss_alignment": 0.478759765625, + "loss_alignment_w": 0.2393798828125, + "loss_sft": 0.4365819878876209, + "loss_total_v6": 0.6764043793082237, + "step": 251 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05811891891891892, + "grad_norm": 26.543106079101562, + "learning_rate": 9.996100320464274e-06, + "loss": 10.933443069458008, + "loss_alignment": 0.476806640625, + "loss_alignment_w": 0.2384033203125, + "loss_sft": 0.4450284615159035, + "loss_total_v6": 0.6833402216434479, + "step": 252 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05834954954954955, + "grad_norm": 34.36637878417969, + "learning_rate": 9.995948911052427e-06, + "loss": 11.750804901123047, + "loss_alignment": 0.485595703125, + "loss_alignment_w": 0.2427978515625, + "loss_sft": 0.49133751541376114, + "loss_total_v6": 0.7344252914190292, + "step": 253 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05858018018018018, + "grad_norm": 28.136125564575195, + "learning_rate": 9.995794619224626e-06, + "loss": 11.889636993408203, + "loss_alignment": 0.474609375, + "loss_alignment_w": 0.2373046875, + "loss_sft": 0.5060264728963375, + "loss_total_v6": 0.7431022673845291, + "step": 254 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05881081081081081, + "grad_norm": 27.95440673828125, + "learning_rate": 9.995637445069889e-06, + "loss": 11.263141632080078, + "loss_alignment": 0.4755859375, + "loss_alignment_w": 0.23779296875, + "loss_sft": 0.4655735269188881, + "loss_total_v6": 0.7039463371038437, + "step": 255 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05904144144144144, + "grad_norm": 30.204607009887695, + "learning_rate": 9.995477388678898e-06, + "loss": 11.705924034118652, + "loss_alignment": 0.4873046875, + "loss_alignment_w": 0.24365234375, + "loss_sft": 0.4884103797376156, + "loss_total_v6": 0.731620229780674, + "step": 256 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05927207207207207, + "grad_norm": 29.34585189819336, + "learning_rate": 9.995314450143999e-06, + "loss": 11.041016578674316, + "loss_alignment": 0.478271484375, + "loss_alignment_w": 0.2391357421875, + "loss_sft": 0.45089730620384216, + "loss_total_v6": 0.6900635585188866, + "step": 257 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.059502702702702706, + "grad_norm": 29.012117385864258, + "learning_rate": 9.995148629559196e-06, + "loss": 11.591495513916016, + "loss_alignment": 0.47705078125, + "loss_alignment_w": 0.238525390625, + "loss_sft": 0.48583633452653885, + "loss_total_v6": 0.7244685143232346, + "step": 258 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.05973333333333333, + "grad_norm": 26.302356719970703, + "learning_rate": 9.994979927020163e-06, + "loss": 11.520719528198242, + "loss_alignment": 0.471435546875, + "loss_alignment_w": 0.2357177734375, + "loss_sft": 0.48396099358797073, + "loss_total_v6": 0.7200449556112289, + "step": 259 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.059963963963963966, + "grad_norm": 28.707950592041016, + "learning_rate": 9.994808342624234e-06, + "loss": 11.087664604187012, + "loss_alignment": 0.462158203125, + "loss_alignment_w": 0.2310791015625, + "loss_sft": 0.46196099370718, + "loss_total_v6": 0.69297906011343, + "step": 260 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06019459459459459, + "grad_norm": 25.830787658691406, + "learning_rate": 9.9946338764704e-06, + "loss": 11.12302017211914, + "loss_alignment": 0.479248046875, + "loss_alignment_w": 0.2396240234375, + "loss_sft": 0.4554884433746338, + "loss_total_v6": 0.6951887607574463, + "step": 261 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.060425225225225226, + "grad_norm": 27.2542724609375, + "learning_rate": 9.994456528659322e-06, + "loss": 11.253007888793945, + "loss_alignment": 0.4677734375, + "loss_alignment_w": 0.23388671875, + "loss_sft": 0.4692279323935509, + "loss_total_v6": 0.7033130005002022, + "step": 262 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06065585585585585, + "grad_norm": 29.192886352539062, + "learning_rate": 9.994276299293321e-06, + "loss": 11.074997901916504, + "loss_alignment": 0.475341796875, + "loss_alignment_w": 0.2376708984375, + "loss_sft": 0.4543943591415882, + "loss_total_v6": 0.6921873465180397, + "step": 263 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06088648648648649, + "grad_norm": 26.80851936340332, + "learning_rate": 9.994093188476383e-06, + "loss": 11.838074684143066, + "loss_alignment": 0.490966796875, + "loss_alignment_w": 0.2454833984375, + "loss_sft": 0.49494557455182076, + "loss_total_v6": 0.7398796454071999, + "step": 264 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06111711711711712, + "grad_norm": 28.07490348815918, + "learning_rate": 9.993907196314148e-06, + "loss": 11.411355972290039, + "loss_alignment": 0.47021484375, + "loss_alignment_w": 0.235107421875, + "loss_sft": 0.47785820811986923, + "loss_total_v6": 0.7132097482681274, + "step": 265 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06134774774774775, + "grad_norm": 27.885679244995117, + "learning_rate": 9.99371832291393e-06, + "loss": 11.488171577453613, + "loss_alignment": 0.48046875, + "loss_alignment_w": 0.240234375, + "loss_sft": 0.4770592041313648, + "loss_total_v6": 0.7180107459425926, + "step": 266 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06157837837837838, + "grad_norm": 23.356637954711914, + "learning_rate": 9.993526568384694e-06, + "loss": 11.155753135681152, + "loss_alignment": 0.476806640625, + "loss_alignment_w": 0.2384033203125, + "loss_sft": 0.4586786590516567, + "loss_total_v6": 0.6972345635294914, + "step": 267 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06180900900900901, + "grad_norm": 23.480131149291992, + "learning_rate": 9.99333193283708e-06, + "loss": 11.213006973266602, + "loss_alignment": 0.476806640625, + "loss_alignment_w": 0.2384033203125, + "loss_sft": 0.4623028300702572, + "loss_total_v6": 0.7008129507303238, + "step": 268 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06203963963963964, + "grad_norm": 26.763906478881836, + "learning_rate": 9.993134416383376e-06, + "loss": 10.984155654907227, + "loss_alignment": 0.474853515625, + "loss_alignment_w": 0.2374267578125, + "loss_sft": 0.44934238120913506, + "loss_total_v6": 0.6865097507834435, + "step": 269 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06227027027027027, + "grad_norm": 27.472946166992188, + "learning_rate": 9.992934019137544e-06, + "loss": 10.612754821777344, + "loss_alignment": 0.468994140625, + "loss_alignment_w": 0.2344970703125, + "loss_sft": 0.4288763552904129, + "loss_total_v6": 0.6632971614599228, + "step": 270 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06250090090090091, + "grad_norm": 29.511455535888672, + "learning_rate": 9.992730741215202e-06, + "loss": 11.636491775512695, + "loss_alignment": 0.478759765625, + "loss_alignment_w": 0.2393798828125, + "loss_sft": 0.48764149844646454, + "loss_total_v6": 0.7272807732224464, + "step": 271 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06273153153153153, + "grad_norm": 29.413599014282227, + "learning_rate": 9.99252458273363e-06, + "loss": 10.934126853942871, + "loss_alignment": 0.480712890625, + "loss_alignment_w": 0.2403564453125, + "loss_sft": 0.44314859062433243, + "loss_total_v6": 0.6833829432725906, + "step": 272 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06296216216216216, + "grad_norm": 28.205230712890625, + "learning_rate": 9.992315543811773e-06, + "loss": 11.02242374420166, + "loss_alignment": 0.48095703125, + "loss_alignment_w": 0.240478515625, + "loss_sft": 0.44849928468465805, + "loss_total_v6": 0.6889015063643456, + "step": 273 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06319279279279279, + "grad_norm": 25.988523483276367, + "learning_rate": 9.992103624570234e-06, + "loss": 11.323556900024414, + "loss_alignment": 0.4814453125, + "loss_alignment_w": 0.24072265625, + "loss_sft": 0.4666334055364132, + "loss_total_v6": 0.7077222764492035, + "step": 274 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06342342342342343, + "grad_norm": 26.359764099121094, + "learning_rate": 9.991888825131283e-06, + "loss": 11.35845947265625, + "loss_alignment": 0.477783203125, + "loss_alignment_w": 0.2388916015625, + "loss_sft": 0.47139356657862663, + "loss_total_v6": 0.709903709590435, + "step": 275 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06365405405405405, + "grad_norm": 25.42605209350586, + "learning_rate": 9.991671145618847e-06, + "loss": 11.420595169067383, + "loss_alignment": 0.478271484375, + "loss_alignment_w": 0.2391357421875, + "loss_sft": 0.47494135797023773, + "loss_total_v6": 0.713787168264389, + "step": 276 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06388468468468468, + "grad_norm": 31.570829391479492, + "learning_rate": 9.991450586158515e-06, + "loss": 11.147132873535156, + "loss_alignment": 0.481201171875, + "loss_alignment_w": 0.2406005859375, + "loss_sft": 0.45650720223784447, + "loss_total_v6": 0.6966958194971085, + "step": 277 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06411531531531532, + "grad_norm": 27.17643165588379, + "learning_rate": 9.991227146877542e-06, + "loss": 11.183236122131348, + "loss_alignment": 0.46826171875, + "loss_alignment_w": 0.234130859375, + "loss_sft": 0.46454673632979393, + "loss_total_v6": 0.6989522501826286, + "step": 278 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06434594594594595, + "grad_norm": 25.899641036987305, + "learning_rate": 9.991000827904839e-06, + "loss": 11.463985443115234, + "loss_alignment": 0.47216796875, + "loss_alignment_w": 0.236083984375, + "loss_sft": 0.48043037205934525, + "loss_total_v6": 0.7164991050958633, + "step": 279 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06457657657657657, + "grad_norm": 25.007986068725586, + "learning_rate": 9.99077162937098e-06, + "loss": 11.334939956665039, + "loss_alignment": 0.4775390625, + "loss_alignment_w": 0.23876953125, + "loss_sft": 0.4697252996265888, + "loss_total_v6": 0.7084337547421455, + "step": 280 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0648072072072072, + "grad_norm": 22.644859313964844, + "learning_rate": 9.990539551408207e-06, + "loss": 11.15483283996582, + "loss_alignment": 0.475341796875, + "loss_alignment_w": 0.2376708984375, + "loss_sft": 0.4594145491719246, + "loss_total_v6": 0.6971770003437996, + "step": 281 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06503783783783784, + "grad_norm": 27.784229278564453, + "learning_rate": 9.990304594150411e-06, + "loss": 11.314251899719238, + "loss_alignment": 0.473876953125, + "loss_alignment_w": 0.2369384765625, + "loss_sft": 0.4703090898692608, + "loss_total_v6": 0.707140751183033, + "step": 282 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06526846846846847, + "grad_norm": 23.214994430541992, + "learning_rate": 9.990066757733152e-06, + "loss": 11.279536247253418, + "loss_alignment": 0.470703125, + "loss_alignment_w": 0.2353515625, + "loss_sft": 0.46914640441536903, + "loss_total_v6": 0.704971008002758, + "step": 283 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0654990990990991, + "grad_norm": 24.606393814086914, + "learning_rate": 9.989826042293653e-06, + "loss": 11.392138481140137, + "loss_alignment": 0.4765625, + "loss_alignment_w": 0.23828125, + "loss_sft": 0.47301026806235313, + "loss_total_v6": 0.7120086923241615, + "step": 284 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06572972972972974, + "grad_norm": 24.229402542114258, + "learning_rate": 9.989582447970792e-06, + "loss": 11.512886047363281, + "loss_alignment": 0.474609375, + "loss_alignment_w": 0.2373046875, + "loss_sft": 0.4819760397076607, + "loss_total_v6": 0.7195553779602051, + "step": 285 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06596036036036036, + "grad_norm": 25.056901931762695, + "learning_rate": 9.989335974905112e-06, + "loss": 11.496524810791016, + "loss_alignment": 0.48291015625, + "loss_alignment_w": 0.241455078125, + "loss_sft": 0.4768030717968941, + "loss_total_v6": 0.7185327857732773, + "step": 286 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06619099099099099, + "grad_norm": 25.460979461669922, + "learning_rate": 9.989086623238815e-06, + "loss": 10.97523307800293, + "loss_alignment": 0.480712890625, + "loss_alignment_w": 0.2403564453125, + "loss_sft": 0.44478684663772583, + "loss_total_v6": 0.6859520226716995, + "step": 287 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06642162162162162, + "grad_norm": 24.90329933166504, + "learning_rate": 9.988834393115768e-06, + "loss": 11.153392791748047, + "loss_alignment": 0.472900390625, + "loss_alignment_w": 0.2364501953125, + "loss_sft": 0.46101831644773483, + "loss_total_v6": 0.6970870345830917, + "step": 288 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06665225225225226, + "grad_norm": 23.27044677734375, + "learning_rate": 9.98857928468149e-06, + "loss": 10.606639862060547, + "loss_alignment": 0.4755859375, + "loss_alignment_w": 0.23779296875, + "loss_sft": 0.4249389208853245, + "loss_total_v6": 0.6629149988293648, + "step": 289 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06688288288288288, + "grad_norm": 22.40992546081543, + "learning_rate": 9.988321298083167e-06, + "loss": 11.145687103271484, + "loss_alignment": 0.47998046875, + "loss_alignment_w": 0.239990234375, + "loss_sft": 0.4563557803630829, + "loss_total_v6": 0.6966054141521454, + "step": 290 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06711351351351351, + "grad_norm": 23.94482421875, + "learning_rate": 9.98806043346965e-06, + "loss": 11.150928497314453, + "loss_alignment": 0.471923828125, + "loss_alignment_w": 0.2359619140625, + "loss_sft": 0.4612000659108162, + "loss_total_v6": 0.6969330906867981, + "step": 291 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06734414414414415, + "grad_norm": 25.551071166992188, + "learning_rate": 9.987796690991438e-06, + "loss": 11.572139739990234, + "loss_alignment": 0.47607421875, + "loss_alignment_w": 0.238037109375, + "loss_sft": 0.4850538298487663, + "loss_total_v6": 0.7232587784528732, + "step": 292 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06757477477477478, + "grad_norm": 25.478410720825195, + "learning_rate": 9.987530070800703e-06, + "loss": 11.232378959655762, + "loss_alignment": 0.47021484375, + "loss_alignment_w": 0.235107421875, + "loss_sft": 0.466611061245203, + "loss_total_v6": 0.7020236477255821, + "step": 293 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0678054054054054, + "grad_norm": 23.91053009033203, + "learning_rate": 9.987260573051268e-06, + "loss": 10.89527702331543, + "loss_alignment": 0.475830078125, + "loss_alignment_w": 0.2379150390625, + "loss_sft": 0.44320761412382126, + "loss_total_v6": 0.6809547990560532, + "step": 294 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06803603603603603, + "grad_norm": 26.719926834106445, + "learning_rate": 9.986988197898621e-06, + "loss": 11.345191955566406, + "loss_alignment": 0.47412109375, + "loss_alignment_w": 0.237060546875, + "loss_sft": 0.4722733311355114, + "loss_total_v6": 0.709074467420578, + "step": 295 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06826666666666667, + "grad_norm": 25.598859786987305, + "learning_rate": 9.98671294549991e-06, + "loss": 11.003934860229492, + "loss_alignment": 0.46728515625, + "loss_alignment_w": 0.233642578125, + "loss_sft": 0.45417963713407516, + "loss_total_v6": 0.6877459362149239, + "step": 296 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0684972972972973, + "grad_norm": 24.08054542541504, + "learning_rate": 9.986434816013941e-06, + "loss": 11.355661392211914, + "loss_alignment": 0.474365234375, + "loss_alignment_w": 0.2371826171875, + "loss_sft": 0.4725157134234905, + "loss_total_v6": 0.7097288370132446, + "step": 297 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06872792792792792, + "grad_norm": 26.83513641357422, + "learning_rate": 9.98615380960118e-06, + "loss": 11.276330947875977, + "loss_alignment": 0.46923828125, + "loss_alignment_w": 0.234619140625, + "loss_sft": 0.47028883919119835, + "loss_total_v6": 0.7047706469893456, + "step": 298 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06895855855855856, + "grad_norm": 27.924551010131836, + "learning_rate": 9.985869926423757e-06, + "loss": 11.02277660369873, + "loss_alignment": 0.471923828125, + "loss_alignment_w": 0.2359619140625, + "loss_sft": 0.45303791761398315, + "loss_total_v6": 0.6889235451817513, + "step": 299 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06918918918918919, + "grad_norm": 25.309782028198242, + "learning_rate": 9.985583166645455e-06, + "loss": 11.677335739135742, + "loss_alignment": 0.47412109375, + "loss_alignment_w": 0.237060546875, + "loss_sft": 0.4929559677839279, + "loss_total_v6": 0.7298334538936615, + "step": 300 + }, + { + "epoch": 0.06918918918918919, + "eval_loss": 0.7310951948165894, + "eval_loss_alignment": 0.4600657284531963, + "eval_loss_alignment_w": 0.23003286422659816, + "eval_loss_sft": 0.5010623624335685, + "eval_loss_total_v6": 0.7310952265410935, + "eval_loss_wm_recon": 0.0, + "eval_loss_wm_recon_w": 0.0, + "eval_runtime": 209.9647, + "eval_samples_per_second": 8.311, + "eval_steps_per_second": 1.043, + "step": 300 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06941981981981982, + "grad_norm": 23.8580379486084, + "learning_rate": 9.985293530431722e-06, + "loss": 10.627340316772461, + "loss_alignment": 0.472412109375, + "loss_alignment_w": 0.2362060546875, + "loss_sft": 0.4281553290784359, + "loss_total_v6": 0.6642087921500206, + "step": 301 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06965045045045046, + "grad_norm": 24.241756439208984, + "learning_rate": 9.985001017949664e-06, + "loss": 10.83974838256836, + "loss_alignment": 0.471435546875, + "loss_alignment_w": 0.2357177734375, + "loss_sft": 0.44144609197974205, + "loss_total_v6": 0.6774843111634254, + "step": 302 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.06988108108108108, + "grad_norm": 26.51390266418457, + "learning_rate": 9.984705629368046e-06, + "loss": 11.197379112243652, + "loss_alignment": 0.472900390625, + "loss_alignment_w": 0.2364501953125, + "loss_sft": 0.46358437091112137, + "loss_total_v6": 0.6998362168669701, + "step": 303 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07011171171171171, + "grad_norm": 30.686939239501953, + "learning_rate": 9.984407364857292e-06, + "loss": 10.83891487121582, + "loss_alignment": 0.4677734375, + "loss_alignment_w": 0.23388671875, + "loss_sft": 0.44336235150694847, + "loss_total_v6": 0.67743219435215, + "step": 304 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07034234234234234, + "grad_norm": 26.79740333557129, + "learning_rate": 9.984106224589488e-06, + "loss": 11.532442092895508, + "loss_alignment": 0.476318359375, + "loss_alignment_w": 0.2381591796875, + "loss_sft": 0.48280157521367073, + "loss_total_v6": 0.720777653157711, + "step": 305 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07057297297297298, + "grad_norm": 27.83164405822754, + "learning_rate": 9.983802208738373e-06, + "loss": 11.20138931274414, + "loss_alignment": 0.472412109375, + "loss_alignment_w": 0.2362060546875, + "loss_sft": 0.4636518470942974, + "loss_total_v6": 0.7000868022441864, + "step": 306 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0708036036036036, + "grad_norm": 63.708900451660156, + "learning_rate": 9.983495317479353e-06, + "loss": 10.748090744018555, + "loss_alignment": 0.477783203125, + "loss_alignment_w": 0.2388916015625, + "loss_sft": 0.4329250827431679, + "loss_total_v6": 0.6717556342482567, + "step": 307 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07103423423423423, + "grad_norm": 26.875289916992188, + "learning_rate": 9.983185550989488e-06, + "loss": 11.434309959411621, + "loss_alignment": 0.46533203125, + "loss_alignment_w": 0.232666015625, + "loss_sft": 0.4818410351872444, + "loss_total_v6": 0.7146443873643875, + "step": 308 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07126486486486487, + "grad_norm": 32.60738754272461, + "learning_rate": 9.982872909447496e-06, + "loss": 10.952850341796875, + "loss_alignment": 0.4697265625, + "loss_alignment_w": 0.23486328125, + "loss_sft": 0.4497204050421715, + "loss_total_v6": 0.6845531389117241, + "step": 309 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0714954954954955, + "grad_norm": 24.358631134033203, + "learning_rate": 9.982557393033758e-06, + "loss": 11.318194389343262, + "loss_alignment": 0.483154296875, + "loss_alignment_w": 0.2415771484375, + "loss_sft": 0.46579474955797195, + "loss_total_v6": 0.7073871418833733, + "step": 310 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07172612612612612, + "grad_norm": 22.65622901916504, + "learning_rate": 9.982239001930311e-06, + "loss": 11.56165599822998, + "loss_alignment": 0.486572265625, + "loss_alignment_w": 0.2432861328125, + "loss_sft": 0.4797446019947529, + "loss_total_v6": 0.7226034849882126, + "step": 311 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07195675675675675, + "grad_norm": 23.091514587402344, + "learning_rate": 9.981917736320852e-06, + "loss": 11.022138595581055, + "loss_alignment": 0.468017578125, + "loss_alignment_w": 0.2340087890625, + "loss_sft": 0.4545087106525898, + "loss_total_v6": 0.6888836771249771, + "step": 312 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07218738738738739, + "grad_norm": 24.828413009643555, + "learning_rate": 9.981593596390733e-06, + "loss": 10.889232635498047, + "loss_alignment": 0.483154296875, + "loss_alignment_w": 0.2415771484375, + "loss_sft": 0.4396102577447891, + "loss_total_v6": 0.6805770546197891, + "step": 313 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07241801801801802, + "grad_norm": 24.954437255859375, + "learning_rate": 9.98126658232697e-06, + "loss": 11.442669868469238, + "loss_alignment": 0.46728515625, + "loss_alignment_w": 0.233642578125, + "loss_sft": 0.4812039025127888, + "loss_total_v6": 0.7151668965816498, + "step": 314 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07264864864864864, + "grad_norm": 24.41240882873535, + "learning_rate": 9.980936694318231e-06, + "loss": 10.970152854919434, + "loss_alignment": 0.4638671875, + "loss_alignment_w": 0.23193359375, + "loss_sft": 0.4538687877357006, + "loss_total_v6": 0.6856345310807228, + "step": 315 + }, + { + "action_cond_aligned": 0.9765625, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07287927927927929, + "grad_norm": 24.66547203063965, + "learning_rate": 9.980603932554844e-06, + "loss": 11.209514617919922, + "loss_alignment": 0.4638671875, + "loss_alignment_w": 0.23193359375, + "loss_sft": 0.4688289426267147, + "loss_total_v6": 0.7005946561694145, + "step": 316 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07310990990990991, + "grad_norm": 40.1005973815918, + "learning_rate": 9.9802682972288e-06, + "loss": 11.070292472839355, + "loss_alignment": 0.4736328125, + "loss_alignment_w": 0.23681640625, + "loss_sft": 0.455244705080986, + "loss_total_v6": 0.6918932795524597, + "step": 317 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07334054054054054, + "grad_norm": 24.93593406677246, + "learning_rate": 9.979929788533742e-06, + "loss": 11.129278182983398, + "loss_alignment": 0.480712890625, + "loss_alignment_w": 0.2403564453125, + "loss_sft": 0.4550403766334057, + "loss_total_v6": 0.695579931139946, + "step": 318 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07357117117117117, + "grad_norm": 24.48813819885254, + "learning_rate": 9.979588406664972e-06, + "loss": 10.4683256149292, + "loss_alignment": 0.46875, + "loss_alignment_w": 0.234375, + "loss_sft": 0.41930027306079865, + "loss_total_v6": 0.6542703658342361, + "step": 319 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0738018018018018, + "grad_norm": 24.166595458984375, + "learning_rate": 9.979244151819454e-06, + "loss": 11.378369331359863, + "loss_alignment": 0.480712890625, + "loss_alignment_w": 0.2403564453125, + "loss_sft": 0.47051696851849556, + "loss_total_v6": 0.7111480757594109, + "step": 320 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07403243243243243, + "grad_norm": 25.6979923248291, + "learning_rate": 9.978897024195801e-06, + "loss": 11.367805480957031, + "loss_alignment": 0.470703125, + "loss_alignment_w": 0.2353515625, + "loss_sft": 0.47528891637921333, + "loss_total_v6": 0.710487850010395, + "step": 321 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07426306306306306, + "grad_norm": 24.469079971313477, + "learning_rate": 9.978547023994292e-06, + "loss": 11.171829223632812, + "loss_alignment": 0.4677734375, + "loss_alignment_w": 0.23388671875, + "loss_sft": 0.4652376212179661, + "loss_total_v6": 0.6982393190264702, + "step": 322 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0744936936936937, + "grad_norm": 24.80580711364746, + "learning_rate": 9.978194151416857e-06, + "loss": 10.842912673950195, + "loss_alignment": 0.468994140625, + "loss_alignment_w": 0.2344970703125, + "loss_sft": 0.4427119828760624, + "loss_total_v6": 0.6776820719242096, + "step": 323 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07472432432432433, + "grad_norm": 27.03061866760254, + "learning_rate": 9.977838406667091e-06, + "loss": 11.06953239440918, + "loss_alignment": 0.46875, + "loss_alignment_w": 0.234375, + "loss_sft": 0.4572266414761543, + "loss_total_v6": 0.6918457671999931, + "step": 324 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07495495495495495, + "grad_norm": 24.431278228759766, + "learning_rate": 9.977479789950238e-06, + "loss": 11.063437461853027, + "loss_alignment": 0.47314453125, + "loss_alignment_w": 0.236572265625, + "loss_sft": 0.4548773542046547, + "loss_total_v6": 0.6914648562669754, + "step": 325 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07518558558558558, + "grad_norm": 24.36257553100586, + "learning_rate": 9.977118301473199e-06, + "loss": 11.04661750793457, + "loss_alignment": 0.47900390625, + "loss_alignment_w": 0.239501953125, + "loss_sft": 0.4509727209806442, + "loss_total_v6": 0.6904136165976524, + "step": 326 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07541621621621622, + "grad_norm": 24.105876922607422, + "learning_rate": 9.976753941444541e-06, + "loss": 10.442359924316406, + "loss_alignment": 0.479248046875, + "loss_alignment_w": 0.2396240234375, + "loss_sft": 0.4133590795099735, + "loss_total_v6": 0.6526474580168724, + "step": 327 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07564684684684685, + "grad_norm": 25.406007766723633, + "learning_rate": 9.976386710074479e-06, + "loss": 11.315403938293457, + "loss_alignment": 0.481201171875, + "loss_alignment_w": 0.2406005859375, + "loss_sft": 0.46716149523854256, + "loss_total_v6": 0.7072127684950829, + "step": 328 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07587747747747747, + "grad_norm": 24.866226196289062, + "learning_rate": 9.976016607574886e-06, + "loss": 10.748985290527344, + "loss_alignment": 0.47216796875, + "loss_alignment_w": 0.236083984375, + "loss_sft": 0.4352239966392517, + "loss_total_v6": 0.6718115285038948, + "step": 329 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07610810810810811, + "grad_norm": 22.840526580810547, + "learning_rate": 9.975643634159294e-06, + "loss": 10.77231502532959, + "loss_alignment": 0.464111328125, + "loss_alignment_w": 0.2320556640625, + "loss_sft": 0.44145815819501877, + "loss_total_v6": 0.673269659280777, + "step": 330 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07633873873873874, + "grad_norm": 23.417579650878906, + "learning_rate": 9.975267790042892e-06, + "loss": 10.975736618041992, + "loss_alignment": 0.46533203125, + "loss_alignment_w": 0.232666015625, + "loss_sft": 0.4530428573489189, + "loss_total_v6": 0.6859835386276245, + "step": 331 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07656936936936937, + "grad_norm": 22.53042221069336, + "learning_rate": 9.97488907544252e-06, + "loss": 10.891094207763672, + "loss_alignment": 0.470947265625, + "loss_alignment_w": 0.2354736328125, + "loss_sft": 0.445219736546278, + "loss_total_v6": 0.6806933879852295, + "step": 332 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0768, + "grad_norm": 24.009929656982422, + "learning_rate": 9.974507490576682e-06, + "loss": 10.634273529052734, + "loss_alignment": 0.476318359375, + "loss_alignment_w": 0.2381591796875, + "loss_sft": 0.4269406944513321, + "loss_total_v6": 0.6646421179175377, + "step": 333 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07703063063063063, + "grad_norm": 22.457494735717773, + "learning_rate": 9.974123035665531e-06, + "loss": 10.797586441040039, + "loss_alignment": 0.47314453125, + "loss_alignment_w": 0.236572265625, + "loss_sft": 0.43806326016783714, + "loss_total_v6": 0.6748491525650024, + "step": 334 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07726126126126126, + "grad_norm": 56.6547737121582, + "learning_rate": 9.973735710930878e-06, + "loss": 11.001415252685547, + "loss_alignment": 0.460205078125, + "loss_alignment_w": 0.2301025390625, + "loss_sft": 0.45768430083990097, + "loss_total_v6": 0.6875884905457497, + "step": 335 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07749189189189189, + "grad_norm": 26.911483764648438, + "learning_rate": 9.973345516596192e-06, + "loss": 11.123373031616211, + "loss_alignment": 0.469970703125, + "loss_alignment_w": 0.2349853515625, + "loss_sft": 0.4604238271713257, + "loss_total_v6": 0.695210836827755, + "step": 336 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07772252252252253, + "grad_norm": 28.79522132873535, + "learning_rate": 9.972952452886594e-06, + "loss": 11.34846019744873, + "loss_alignment": 0.4697265625, + "loss_alignment_w": 0.23486328125, + "loss_sft": 0.47418662533164024, + "loss_total_v6": 0.7092787474393845, + "step": 337 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07795315315315315, + "grad_norm": 27.409423828125, + "learning_rate": 9.972556520028863e-06, + "loss": 11.20928955078125, + "loss_alignment": 0.470947265625, + "loss_alignment_w": 0.2354736328125, + "loss_sft": 0.4652442894876003, + "loss_total_v6": 0.7005805969238281, + "step": 338 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07818378378378378, + "grad_norm": 26.526700973510742, + "learning_rate": 9.972157718251433e-06, + "loss": 11.071608543395996, + "loss_alignment": 0.473388671875, + "loss_alignment_w": 0.2366943359375, + "loss_sft": 0.4555405452847481, + "loss_total_v6": 0.6919755190610886, + "step": 339 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07841441441441442, + "grad_norm": 24.09821128845215, + "learning_rate": 9.971756047784393e-06, + "loss": 10.828761100769043, + "loss_alignment": 0.47265625, + "loss_alignment_w": 0.236328125, + "loss_sft": 0.4407135583460331, + "loss_total_v6": 0.6767975762486458, + "step": 340 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07864504504504505, + "grad_norm": 23.540786743164062, + "learning_rate": 9.971351508859488e-06, + "loss": 10.812807083129883, + "loss_alignment": 0.473388671875, + "loss_alignment_w": 0.2366943359375, + "loss_sft": 0.43933502957224846, + "loss_total_v6": 0.6758004799485207, + "step": 341 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07887567567567567, + "grad_norm": 31.989015579223633, + "learning_rate": 9.970944101710115e-06, + "loss": 10.717317581176758, + "loss_alignment": 0.474609375, + "loss_alignment_w": 0.2373046875, + "loss_sft": 0.4323139563202858, + "loss_total_v6": 0.6698322966694832, + "step": 342 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0791063063063063, + "grad_norm": 24.011783599853516, + "learning_rate": 9.970533826571329e-06, + "loss": 10.97474479675293, + "loss_alignment": 0.46923828125, + "loss_alignment_w": 0.234619140625, + "loss_sft": 0.451775386929512, + "loss_total_v6": 0.6859215199947357, + "step": 343 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07933693693693694, + "grad_norm": 21.678264617919922, + "learning_rate": 9.970120683679837e-06, + "loss": 11.028223037719727, + "loss_alignment": 0.47314453125, + "loss_alignment_w": 0.236572265625, + "loss_sft": 0.4529053010046482, + "loss_total_v6": 0.6892639696598053, + "step": 344 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.07956756756756757, + "grad_norm": 25.6877498626709, + "learning_rate": 9.969704673274006e-06, + "loss": 11.212571144104004, + "loss_alignment": 0.475341796875, + "loss_alignment_w": 0.2376708984375, + "loss_sft": 0.46297749131917953, + "loss_total_v6": 0.7007856965065002, + "step": 345 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0797981981981982, + "grad_norm": 28.886327743530273, + "learning_rate": 9.969285795593851e-06, + "loss": 11.452106475830078, + "loss_alignment": 0.475830078125, + "loss_alignment_w": 0.2379150390625, + "loss_sft": 0.4779179282486439, + "loss_total_v6": 0.7157566696405411, + "step": 346 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08002882882882884, + "grad_norm": 25.05923080444336, + "learning_rate": 9.96886405088104e-06, + "loss": 11.433916091918945, + "loss_alignment": 0.46728515625, + "loss_alignment_w": 0.233642578125, + "loss_sft": 0.4807788133621216, + "loss_total_v6": 0.7146197557449341, + "step": 347 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08025945945945946, + "grad_norm": 25.03101921081543, + "learning_rate": 9.968439439378905e-06, + "loss": 11.498796463012695, + "loss_alignment": 0.47900390625, + "loss_alignment_w": 0.239501953125, + "loss_sft": 0.47932542115449905, + "loss_total_v6": 0.7186747938394547, + "step": 348 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08049009009009009, + "grad_norm": 24.129573822021484, + "learning_rate": 9.968011961332424e-06, + "loss": 11.097419738769531, + "loss_alignment": 0.4697265625, + "loss_alignment_w": 0.23486328125, + "loss_sft": 0.4586948864161968, + "loss_total_v6": 0.6935886964201927, + "step": 349 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08072072072072072, + "grad_norm": 23.218379974365234, + "learning_rate": 9.967581616988229e-06, + "loss": 11.301342010498047, + "loss_alignment": 0.462890625, + "loss_alignment_w": 0.2314453125, + "loss_sft": 0.4750259146094322, + "loss_total_v6": 0.7063338831067085, + "step": 350 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08095135135135136, + "grad_norm": 22.20384407043457, + "learning_rate": 9.967148406594608e-06, + "loss": 11.354997634887695, + "loss_alignment": 0.472900390625, + "loss_alignment_w": 0.2364501953125, + "loss_sft": 0.47319139912724495, + "loss_total_v6": 0.7096873819828033, + "step": 351 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08118198198198198, + "grad_norm": 23.051889419555664, + "learning_rate": 9.966712330401503e-06, + "loss": 10.890063285827637, + "loss_alignment": 0.47509765625, + "loss_alignment_w": 0.237548828125, + "loss_sft": 0.4434158280491829, + "loss_total_v6": 0.6806289702653885, + "step": 352 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08141261261261261, + "grad_norm": 22.29570960998535, + "learning_rate": 9.966273388660508e-06, + "loss": 11.435673713684082, + "loss_alignment": 0.47265625, + "loss_alignment_w": 0.236328125, + "loss_sft": 0.4782794415950775, + "loss_total_v6": 0.7147296145558357, + "step": 353 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08164324324324325, + "grad_norm": 24.159191131591797, + "learning_rate": 9.965831581624872e-06, + "loss": 10.609786987304688, + "loss_alignment": 0.46240234375, + "loss_alignment_w": 0.231201171875, + "loss_sft": 0.4317273795604706, + "loss_total_v6": 0.6631116792559624, + "step": 354 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08187387387387388, + "grad_norm": 22.40285873413086, + "learning_rate": 9.965386909549492e-06, + "loss": 11.039311408996582, + "loss_alignment": 0.468505859375, + "loss_alignment_w": 0.2342529296875, + "loss_sft": 0.45564302802085876, + "loss_total_v6": 0.68995700776577, + "step": 355 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0821045045045045, + "grad_norm": 24.469518661499023, + "learning_rate": 9.964939372690927e-06, + "loss": 11.439233779907227, + "loss_alignment": 0.4775390625, + "loss_alignment_w": 0.23876953125, + "loss_sft": 0.47656402364373207, + "loss_total_v6": 0.7149520814418793, + "step": 356 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08233513513513513, + "grad_norm": 22.939502716064453, + "learning_rate": 9.96448897130738e-06, + "loss": 10.833956718444824, + "loss_alignment": 0.473876953125, + "loss_alignment_w": 0.2369384765625, + "loss_sft": 0.4398480989038944, + "loss_total_v6": 0.6771222576498985, + "step": 357 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08256576576576577, + "grad_norm": 22.33869171142578, + "learning_rate": 9.96403570565871e-06, + "loss": 10.93720817565918, + "loss_alignment": 0.471435546875, + "loss_alignment_w": 0.2357177734375, + "loss_sft": 0.44787298887968063, + "loss_total_v6": 0.6835754811763763, + "step": 358 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0827963963963964, + "grad_norm": 24.324840545654297, + "learning_rate": 9.963579576006433e-06, + "loss": 10.493080139160156, + "loss_alignment": 0.452880859375, + "loss_alignment_w": 0.2264404296875, + "loss_sft": 0.42879723757505417, + "loss_total_v6": 0.6558174937963486, + "step": 359 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08302702702702702, + "grad_norm": 23.83697509765625, + "learning_rate": 9.96312058261371e-06, + "loss": 10.711538314819336, + "loss_alignment": 0.46826171875, + "loss_alignment_w": 0.234130859375, + "loss_sft": 0.4358743838965893, + "loss_total_v6": 0.6694711893796921, + "step": 360 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08325765765765766, + "grad_norm": 23.62445068359375, + "learning_rate": 9.962658725745358e-06, + "loss": 11.202364921569824, + "loss_alignment": 0.468505859375, + "loss_alignment_w": 0.2342529296875, + "loss_sft": 0.4659406691789627, + "loss_total_v6": 0.7001478001475334, + "step": 361 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08348828828828829, + "grad_norm": 24.545530319213867, + "learning_rate": 9.962194005667844e-06, + "loss": 10.626901626586914, + "loss_alignment": 0.4677734375, + "loss_alignment_w": 0.23388671875, + "loss_sft": 0.4306913949549198, + "loss_total_v6": 0.6641813591122627, + "step": 362 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08371891891891892, + "grad_norm": 25.031566619873047, + "learning_rate": 9.96172642264929e-06, + "loss": 11.68104362487793, + "loss_alignment": 0.47119140625, + "loss_alignment_w": 0.235595703125, + "loss_sft": 0.4943932257592678, + "loss_total_v6": 0.7300652116537094, + "step": 363 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08394954954954954, + "grad_norm": 24.17753791809082, + "learning_rate": 9.961255976959473e-06, + "loss": 11.558430671691895, + "loss_alignment": 0.470947265625, + "loss_alignment_w": 0.2354736328125, + "loss_sft": 0.4867757149040699, + "loss_total_v6": 0.722401924431324, + "step": 364 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08418018018018018, + "grad_norm": 25.441730499267578, + "learning_rate": 9.960782668869811e-06, + "loss": 10.862478256225586, + "loss_alignment": 0.468505859375, + "loss_alignment_w": 0.2342529296875, + "loss_sft": 0.44420943409204483, + "loss_total_v6": 0.6789048835635185, + "step": 365 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08441081081081081, + "grad_norm": 28.27293586730957, + "learning_rate": 9.96030649865338e-06, + "loss": 10.80552864074707, + "loss_alignment": 0.4716796875, + "loss_alignment_w": 0.23583984375, + "loss_sft": 0.4389869049191475, + "loss_total_v6": 0.6753455400466919, + "step": 366 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08464144144144144, + "grad_norm": 29.366703033447266, + "learning_rate": 9.959827466584907e-06, + "loss": 11.295354843139648, + "loss_alignment": 0.47509765625, + "loss_alignment_w": 0.237548828125, + "loss_sft": 0.4682430066168308, + "loss_total_v6": 0.7059596702456474, + "step": 367 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08487207207207208, + "grad_norm": 23.448400497436523, + "learning_rate": 9.959345572940771e-06, + "loss": 10.946962356567383, + "loss_alignment": 0.466064453125, + "loss_alignment_w": 0.2330322265625, + "loss_sft": 0.4513360261917114, + "loss_total_v6": 0.6841851621866226, + "step": 368 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0851027027027027, + "grad_norm": 24.55569076538086, + "learning_rate": 9.958860817999002e-06, + "loss": 11.204721450805664, + "loss_alignment": 0.473876953125, + "loss_alignment_w": 0.2369384765625, + "loss_sft": 0.4632955230772495, + "loss_total_v6": 0.7002950310707092, + "step": 369 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08533333333333333, + "grad_norm": 25.7351131439209, + "learning_rate": 9.958373202039278e-06, + "loss": 11.350851058959961, + "loss_alignment": 0.474365234375, + "loss_alignment_w": 0.2371826171875, + "loss_sft": 0.4723981283605099, + "loss_total_v6": 0.7094281762838364, + "step": 370 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08556396396396396, + "grad_norm": 24.081661224365234, + "learning_rate": 9.957882725342926e-06, + "loss": 10.783031463623047, + "loss_alignment": 0.469970703125, + "loss_alignment_w": 0.2349853515625, + "loss_sft": 0.4391219727694988, + "loss_total_v6": 0.6739394664764404, + "step": 371 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0857945945945946, + "grad_norm": 21.938745498657227, + "learning_rate": 9.957389388192935e-06, + "loss": 11.0560941696167, + "loss_alignment": 0.47216796875, + "loss_alignment_w": 0.236083984375, + "loss_sft": 0.4551050178706646, + "loss_total_v6": 0.6910059005022049, + "step": 372 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08602522522522522, + "grad_norm": 23.45677375793457, + "learning_rate": 9.956893190873928e-06, + "loss": 10.593063354492188, + "loss_alignment": 0.465576171875, + "loss_alignment_w": 0.2327880859375, + "loss_sft": 0.42932411283254623, + "loss_total_v6": 0.6620664149522781, + "step": 373 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08625585585585585, + "grad_norm": 25.202787399291992, + "learning_rate": 9.956394133672193e-06, + "loss": 10.900467872619629, + "loss_alignment": 0.4697265625, + "loss_alignment_w": 0.23486328125, + "loss_sft": 0.4466448128223419, + "loss_total_v6": 0.6812792271375656, + "step": 374 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08648648648648649, + "grad_norm": 25.122669219970703, + "learning_rate": 9.955892216875656e-06, + "loss": 11.705443382263184, + "loss_alignment": 0.4697265625, + "loss_alignment_w": 0.23486328125, + "loss_sft": 0.4971999451518059, + "loss_total_v6": 0.7315901964902878, + "step": 375 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08671711711711712, + "grad_norm": 22.63596534729004, + "learning_rate": 9.955387440773902e-06, + "loss": 11.105514526367188, + "loss_alignment": 0.473876953125, + "loss_alignment_w": 0.2369384765625, + "loss_sft": 0.4573697857558727, + "loss_total_v6": 0.6940946355462074, + "step": 376 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08694774774774774, + "grad_norm": 24.28704261779785, + "learning_rate": 9.95487980565816e-06, + "loss": 11.335485458374023, + "loss_alignment": 0.46728515625, + "loss_alignment_w": 0.233642578125, + "loss_sft": 0.475084625184536, + "loss_total_v6": 0.7084678262472153, + "step": 377 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08717837837837838, + "grad_norm": 25.378664016723633, + "learning_rate": 9.95436931182131e-06, + "loss": 11.057291030883789, + "loss_alignment": 0.48095703125, + "loss_alignment_w": 0.240478515625, + "loss_sft": 0.4508920758962631, + "loss_total_v6": 0.6910806819796562, + "step": 378 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08740900900900901, + "grad_norm": 27.53331756591797, + "learning_rate": 9.953855959557883e-06, + "loss": 11.092002868652344, + "loss_alignment": 0.46630859375, + "loss_alignment_w": 0.233154296875, + "loss_sft": 0.4603705182671547, + "loss_total_v6": 0.6932501494884491, + "step": 379 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08763963963963964, + "grad_norm": 24.46952247619629, + "learning_rate": 9.953339749164057e-06, + "loss": 10.753085136413574, + "loss_alignment": 0.468994140625, + "loss_alignment_w": 0.2344970703125, + "loss_sft": 0.4373723901808262, + "loss_total_v6": 0.6720678210258484, + "step": 380 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08787027027027026, + "grad_norm": 24.535783767700195, + "learning_rate": 9.95282068093766e-06, + "loss": 11.48320484161377, + "loss_alignment": 0.474609375, + "loss_alignment_w": 0.2373046875, + "loss_sft": 0.4797699972987175, + "loss_total_v6": 0.7177003100514412, + "step": 381 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0881009009009009, + "grad_norm": 24.813201904296875, + "learning_rate": 9.952298755178171e-06, + "loss": 11.127164840698242, + "loss_alignment": 0.473876953125, + "loss_alignment_w": 0.2369384765625, + "loss_sft": 0.45831096917390823, + "loss_total_v6": 0.6954478323459625, + "step": 382 + }, + { + "action_cond_aligned": 0.9765625, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08833153153153153, + "grad_norm": 24.353971481323242, + "learning_rate": 9.951773972186712e-06, + "loss": 11.155618667602539, + "loss_alignment": 0.464599609375, + "loss_alignment_w": 0.2322998046875, + "loss_sft": 0.4651552699506283, + "loss_total_v6": 0.6972261890769005, + "step": 383 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08856216216216216, + "grad_norm": 26.22604751586914, + "learning_rate": 9.951246332266058e-06, + "loss": 11.011947631835938, + "loss_alignment": 0.46875, + "loss_alignment_w": 0.234375, + "loss_sft": 0.4540090784430504, + "loss_total_v6": 0.6882467493414879, + "step": 384 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0887927927927928, + "grad_norm": 26.76230812072754, + "learning_rate": 9.950715835720633e-06, + "loss": 10.974649429321289, + "loss_alignment": 0.470458984375, + "loss_alignment_w": 0.2352294921875, + "loss_sft": 0.4504724554717541, + "loss_total_v6": 0.6859155893325806, + "step": 385 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08902342342342343, + "grad_norm": 24.36956214904785, + "learning_rate": 9.950182482856506e-06, + "loss": 10.889663696289062, + "loss_alignment": 0.470703125, + "loss_alignment_w": 0.2353515625, + "loss_sft": 0.4456643871963024, + "loss_total_v6": 0.6806039586663246, + "step": 386 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08925405405405405, + "grad_norm": 26.124401092529297, + "learning_rate": 9.949646273981394e-06, + "loss": 11.040702819824219, + "loss_alignment": 0.481689453125, + "loss_alignment_w": 0.2408447265625, + "loss_sft": 0.4495043270289898, + "loss_total_v6": 0.6900438815355301, + "step": 387 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08948468468468468, + "grad_norm": 26.455060958862305, + "learning_rate": 9.949107209404664e-06, + "loss": 11.37448501586914, + "loss_alignment": 0.4716796875, + "loss_alignment_w": 0.23583984375, + "loss_sft": 0.4750044383108616, + "loss_total_v6": 0.7109053209424019, + "step": 388 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08971531531531532, + "grad_norm": 24.945594787597656, + "learning_rate": 9.94856528943733e-06, + "loss": 11.168741226196289, + "loss_alignment": 0.47265625, + "loss_alignment_w": 0.236328125, + "loss_sft": 0.46141307428479195, + "loss_total_v6": 0.6980463713407516, + "step": 389 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.08994594594594595, + "grad_norm": 24.7276611328125, + "learning_rate": 9.948020514392055e-06, + "loss": 11.158744812011719, + "loss_alignment": 0.456298828125, + "loss_alignment_w": 0.2281494140625, + "loss_sft": 0.4692111127078533, + "loss_total_v6": 0.6974215656518936, + "step": 390 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09017657657657657, + "grad_norm": 25.324430465698242, + "learning_rate": 9.947472884583145e-06, + "loss": 11.119882583618164, + "loss_alignment": 0.4716796875, + "loss_alignment_w": 0.23583984375, + "loss_sft": 0.45851195231080055, + "loss_total_v6": 0.6949926465749741, + "step": 391 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09040720720720721, + "grad_norm": 26.631824493408203, + "learning_rate": 9.946922400326554e-06, + "loss": 11.374500274658203, + "loss_alignment": 0.4697265625, + "loss_alignment_w": 0.23486328125, + "loss_sft": 0.47608882561326027, + "loss_total_v6": 0.7109063267707825, + "step": 392 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09063783783783784, + "grad_norm": 24.0897159576416, + "learning_rate": 9.946369061939886e-06, + "loss": 10.977721214294434, + "loss_alignment": 0.466552734375, + "loss_alignment_w": 0.2332763671875, + "loss_sft": 0.4529227092862129, + "loss_total_v6": 0.6861075684428215, + "step": 393 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09086846846846847, + "grad_norm": 25.47414779663086, + "learning_rate": 9.94581286974239e-06, + "loss": 11.119277000427246, + "loss_alignment": 0.464599609375, + "loss_alignment_w": 0.2322998046875, + "loss_sft": 0.46277710050344467, + "loss_total_v6": 0.6949548050761223, + "step": 394 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09109909909909909, + "grad_norm": 23.19736671447754, + "learning_rate": 9.945253824054964e-06, + "loss": 10.773561477661133, + "loss_alignment": 0.458984375, + "loss_alignment_w": 0.2294921875, + "loss_sft": 0.4439469948410988, + "loss_total_v6": 0.6733476221561432, + "step": 395 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09132972972972973, + "grad_norm": 25.557607650756836, + "learning_rate": 9.944691925200145e-06, + "loss": 10.588091850280762, + "loss_alignment": 0.46533203125, + "loss_alignment_w": 0.232666015625, + "loss_sft": 0.4295169413089752, + "loss_total_v6": 0.6617557257413864, + "step": 396 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09156036036036036, + "grad_norm": 26.0633544921875, + "learning_rate": 9.944127173502123e-06, + "loss": 11.077411651611328, + "loss_alignment": 0.476318359375, + "loss_alignment_w": 0.2381591796875, + "loss_sft": 0.4542095251381397, + "loss_total_v6": 0.692338190972805, + "step": 397 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09179099099099099, + "grad_norm": 26.480632781982422, + "learning_rate": 9.943559569286731e-06, + "loss": 10.999720573425293, + "loss_alignment": 0.464599609375, + "loss_alignment_w": 0.2322998046875, + "loss_sft": 0.4554268643260002, + "loss_total_v6": 0.6874825358390808, + "step": 398 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09202162162162163, + "grad_norm": 22.22087860107422, + "learning_rate": 9.942989112881451e-06, + "loss": 11.024212837219238, + "loss_alignment": 0.4775390625, + "loss_alignment_w": 0.23876953125, + "loss_sft": 0.4500301368534565, + "loss_total_v6": 0.6890133246779442, + "step": 399 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09225225225225225, + "grad_norm": 25.295692443847656, + "learning_rate": 9.942415804615407e-06, + "loss": 10.7208251953125, + "loss_alignment": 0.46240234375, + "loss_alignment_w": 0.231201171875, + "loss_sft": 0.43822478875517845, + "loss_total_v6": 0.6700515523552895, + "step": 400 + }, + { + "epoch": 0.09225225225225225, + "eval_loss": 0.7194551229476929, + "eval_loss_alignment": 0.4538786030251142, + "eval_loss_alignment_w": 0.2269393015125571, + "eval_loss_sft": 0.4925157951794939, + "eval_loss_total_v6": 0.7194550965474621, + "eval_loss_wm_recon": 0.0, + "eval_loss_wm_recon_w": 0.0, + "eval_runtime": 209.8694, + "eval_samples_per_second": 8.315, + "eval_steps_per_second": 1.044, + "step": 400 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09248288288288288, + "grad_norm": 24.94394302368164, + "learning_rate": 9.941839644819366e-06, + "loss": 10.481552124023438, + "loss_alignment": 0.470458984375, + "loss_alignment_w": 0.2352294921875, + "loss_sft": 0.41994384676218033, + "loss_total_v6": 0.6550970152020454, + "step": 401 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0927135135135135, + "grad_norm": 25.565290451049805, + "learning_rate": 9.941260633825752e-06, + "loss": 10.937538146972656, + "loss_alignment": 0.46875, + "loss_alignment_w": 0.234375, + "loss_sft": 0.449709415435791, + "loss_total_v6": 0.6835961267352104, + "step": 402 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09294414414414415, + "grad_norm": 26.49386978149414, + "learning_rate": 9.940678771968617e-06, + "loss": 10.91921329498291, + "loss_alignment": 0.479248046875, + "loss_alignment_w": 0.2396240234375, + "loss_sft": 0.4428878352046013, + "loss_total_v6": 0.6824508383870125, + "step": 403 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09317477477477477, + "grad_norm": 23.625497817993164, + "learning_rate": 9.940094059583672e-06, + "loss": 10.959511756896973, + "loss_alignment": 0.4794921875, + "loss_alignment_w": 0.23974609375, + "loss_sft": 0.4459557868540287, + "loss_total_v6": 0.6849694773554802, + "step": 404 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0934054054054054, + "grad_norm": 23.206157684326172, + "learning_rate": 9.939506497008264e-06, + "loss": 10.708620071411133, + "loss_alignment": 0.46533203125, + "loss_alignment_w": 0.232666015625, + "loss_sft": 0.43605812266469, + "loss_total_v6": 0.6692887097597122, + "step": 405 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09363603603603604, + "grad_norm": 26.904699325561523, + "learning_rate": 9.938916084581391e-06, + "loss": 11.2586030960083, + "loss_alignment": 0.468994140625, + "loss_alignment_w": 0.2344970703125, + "loss_sft": 0.4692419357597828, + "loss_total_v6": 0.7036626935005188, + "step": 406 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09386666666666667, + "grad_norm": 22.19285011291504, + "learning_rate": 9.938322822643688e-06, + "loss": 10.634665489196777, + "loss_alignment": 0.46923828125, + "loss_alignment_w": 0.234619140625, + "loss_sft": 0.4301542416214943, + "loss_total_v6": 0.6646665632724762, + "step": 407 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0940972972972973, + "grad_norm": 23.99490737915039, + "learning_rate": 9.937726711537441e-06, + "loss": 11.02810001373291, + "loss_alignment": 0.462890625, + "loss_alignment_w": 0.2314453125, + "loss_sft": 0.4581923969089985, + "loss_total_v6": 0.6892562806606293, + "step": 408 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09432792792792793, + "grad_norm": 25.338985443115234, + "learning_rate": 9.937127751606577e-06, + "loss": 10.412419319152832, + "loss_alignment": 0.468505859375, + "loss_alignment_w": 0.2342529296875, + "loss_sft": 0.4164622575044632, + "loss_total_v6": 0.6507762223482132, + "step": 409 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09455855855855856, + "grad_norm": 26.554367065429688, + "learning_rate": 9.936525943196663e-06, + "loss": 10.859957695007324, + "loss_alignment": 0.4775390625, + "loss_alignment_w": 0.23876953125, + "loss_sft": 0.43965744227170944, + "loss_total_v6": 0.6787473931908607, + "step": 410 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09478918918918919, + "grad_norm": 44.71291732788086, + "learning_rate": 9.935921286654914e-06, + "loss": 11.309735298156738, + "loss_alignment": 0.4658203125, + "loss_alignment_w": 0.23291015625, + "loss_sft": 0.47509270161390305, + "loss_total_v6": 0.7068584561347961, + "step": 411 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09501981981981981, + "grad_norm": 22.989051818847656, + "learning_rate": 9.93531378233019e-06, + "loss": 10.868724822998047, + "loss_alignment": 0.473876953125, + "loss_alignment_w": 0.2369384765625, + "loss_sft": 0.4420974738895893, + "loss_total_v6": 0.6792953535914421, + "step": 412 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09525045045045046, + "grad_norm": 22.819204330444336, + "learning_rate": 9.934703430572988e-06, + "loss": 10.760719299316406, + "loss_alignment": 0.46630859375, + "loss_alignment_w": 0.233154296875, + "loss_sft": 0.4395585171878338, + "loss_total_v6": 0.6725449934601784, + "step": 413 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09548108108108108, + "grad_norm": 21.730939865112305, + "learning_rate": 9.93409023173545e-06, + "loss": 10.751691818237305, + "loss_alignment": 0.46435546875, + "loss_alignment_w": 0.232177734375, + "loss_sft": 0.43954358994960785, + "loss_total_v6": 0.6719807088375092, + "step": 414 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09571171171171171, + "grad_norm": 24.625465393066406, + "learning_rate": 9.933474186171363e-06, + "loss": 10.572286605834961, + "loss_alignment": 0.470458984375, + "loss_alignment_w": 0.2352294921875, + "loss_sft": 0.42552314326167107, + "loss_total_v6": 0.6607678905129433, + "step": 415 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09594234234234235, + "grad_norm": 25.0935001373291, + "learning_rate": 9.932855294236154e-06, + "loss": 11.451169967651367, + "loss_alignment": 0.4736328125, + "loss_alignment_w": 0.23681640625, + "loss_sft": 0.4785766080021858, + "loss_total_v6": 0.7156981602311134, + "step": 416 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09617297297297298, + "grad_norm": 24.318086624145508, + "learning_rate": 9.932233556286897e-06, + "loss": 11.019262313842773, + "loss_alignment": 0.46435546875, + "loss_alignment_w": 0.232177734375, + "loss_sft": 0.45646511763334274, + "loss_total_v6": 0.6887039244174957, + "step": 417 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0964036036036036, + "grad_norm": 24.848920822143555, + "learning_rate": 9.931608972682298e-06, + "loss": 11.271371841430664, + "loss_alignment": 0.464599609375, + "loss_alignment_w": 0.2322998046875, + "loss_sft": 0.4718709848821163, + "loss_total_v6": 0.7044606953859329, + "step": 418 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09663423423423423, + "grad_norm": 23.89292335510254, + "learning_rate": 9.930981543782716e-06, + "loss": 11.417171478271484, + "loss_alignment": 0.470703125, + "loss_alignment_w": 0.2353515625, + "loss_sft": 0.4786335937678814, + "loss_total_v6": 0.7135731726884842, + "step": 419 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09686486486486487, + "grad_norm": 23.5177059173584, + "learning_rate": 9.930351269950144e-06, + "loss": 11.32206916809082, + "loss_alignment": 0.4794921875, + "loss_alignment_w": 0.23974609375, + "loss_sft": 0.4682189002633095, + "loss_total_v6": 0.7076292932033539, + "step": 420 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0970954954954955, + "grad_norm": 26.188718795776367, + "learning_rate": 9.929718151548218e-06, + "loss": 10.99237060546875, + "loss_alignment": 0.46142578125, + "loss_alignment_w": 0.230712890625, + "loss_sft": 0.4560203067958355, + "loss_total_v6": 0.6870231181383133, + "step": 421 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09732612612612612, + "grad_norm": 24.84207534790039, + "learning_rate": 9.929082188942219e-06, + "loss": 10.795680046081543, + "loss_alignment": 0.4609375, + "loss_alignment_w": 0.23046875, + "loss_sft": 0.4446122348308563, + "loss_total_v6": 0.6747300177812576, + "step": 422 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09755675675675676, + "grad_norm": 22.353849411010742, + "learning_rate": 9.928443382499063e-06, + "loss": 10.731430053710938, + "loss_alignment": 0.46875, + "loss_alignment_w": 0.234375, + "loss_sft": 0.4366598539054394, + "loss_total_v6": 0.6707143783569336, + "step": 423 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09778738738738739, + "grad_norm": 22.532855987548828, + "learning_rate": 9.927801732587311e-06, + "loss": 10.619050025939941, + "loss_alignment": 0.467041015625, + "loss_alignment_w": 0.2335205078125, + "loss_sft": 0.43059737980365753, + "loss_total_v6": 0.6636906415224075, + "step": 424 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09801801801801802, + "grad_norm": 22.689729690551758, + "learning_rate": 9.927157239577165e-06, + "loss": 11.33890151977539, + "loss_alignment": 0.4638671875, + "loss_alignment_w": 0.23193359375, + "loss_sft": 0.4767782911658287, + "loss_total_v6": 0.7086813598871231, + "step": 425 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09824864864864864, + "grad_norm": 23.40890121459961, + "learning_rate": 9.926509903840464e-06, + "loss": 11.422761917114258, + "loss_alignment": 0.466064453125, + "loss_alignment_w": 0.2330322265625, + "loss_sft": 0.4812413118779659, + "loss_total_v6": 0.7139225825667381, + "step": 426 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09847927927927928, + "grad_norm": 24.186742782592773, + "learning_rate": 9.92585972575069e-06, + "loss": 10.980646133422852, + "loss_alignment": 0.468017578125, + "loss_alignment_w": 0.2340087890625, + "loss_sft": 0.4519154019653797, + "loss_total_v6": 0.6862904131412506, + "step": 427 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09870990990990991, + "grad_norm": 24.189449310302734, + "learning_rate": 9.925206705682961e-06, + "loss": 11.791481971740723, + "loss_alignment": 0.4765625, + "loss_alignment_w": 0.23828125, + "loss_sft": 0.49922045320272446, + "loss_total_v6": 0.7369675934314728, + "step": 428 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09894054054054054, + "grad_norm": 21.983205795288086, + "learning_rate": 9.924550844014039e-06, + "loss": 10.903364181518555, + "loss_alignment": 0.46728515625, + "loss_alignment_w": 0.233642578125, + "loss_sft": 0.4477413408458233, + "loss_total_v6": 0.6814602166414261, + "step": 429 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09917117117117118, + "grad_norm": 22.52764129638672, + "learning_rate": 9.923892141122324e-06, + "loss": 10.75400161743164, + "loss_alignment": 0.4677734375, + "loss_alignment_w": 0.23388671875, + "loss_sft": 0.43816204741597176, + "loss_total_v6": 0.6721250787377357, + "step": 430 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.0994018018018018, + "grad_norm": 20.946805953979492, + "learning_rate": 9.923230597387856e-06, + "loss": 10.646282196044922, + "loss_alignment": 0.459228515625, + "loss_alignment_w": 0.2296142578125, + "loss_sft": 0.4356410317122936, + "loss_total_v6": 0.6653926372528076, + "step": 431 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09963243243243243, + "grad_norm": 22.686975479125977, + "learning_rate": 9.92256621319231e-06, + "loss": 10.973611831665039, + "loss_alignment": 0.47021484375, + "loss_alignment_w": 0.235107421875, + "loss_sft": 0.45107903331518173, + "loss_total_v6": 0.6858507618308067, + "step": 432 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.09986306306306306, + "grad_norm": 23.62600326538086, + "learning_rate": 9.921898988919006e-06, + "loss": 10.9442720413208, + "loss_alignment": 0.466552734375, + "loss_alignment_w": 0.2332763671875, + "loss_sft": 0.45043543726205826, + "loss_total_v6": 0.6840169653296471, + "step": 433 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.1000936936936937, + "grad_norm": 23.826017379760742, + "learning_rate": 9.9212289249529e-06, + "loss": 11.258280754089355, + "loss_alignment": 0.46875, + "loss_alignment_w": 0.234375, + "loss_sft": 0.4694658927619457, + "loss_total_v6": 0.7036425396800041, + "step": 434 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10032432432432432, + "grad_norm": 21.677005767822266, + "learning_rate": 9.92055602168058e-06, + "loss": 11.405101776123047, + "loss_alignment": 0.470947265625, + "loss_alignment_w": 0.2354736328125, + "loss_sft": 0.47781822085380554, + "loss_total_v6": 0.7128188386559486, + "step": 435 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10055495495495495, + "grad_norm": 22.411161422729492, + "learning_rate": 9.919880279490286e-06, + "loss": 11.285571098327637, + "loss_alignment": 0.466064453125, + "loss_alignment_w": 0.2330322265625, + "loss_sft": 0.4729568250477314, + "loss_total_v6": 0.7053481936454773, + "step": 436 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10078558558558559, + "grad_norm": 22.169029235839844, + "learning_rate": 9.919201698771882e-06, + "loss": 10.787910461425781, + "loss_alignment": 0.46923828125, + "loss_alignment_w": 0.234619140625, + "loss_sft": 0.43938111513853073, + "loss_total_v6": 0.6742443740367889, + "step": 437 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10101621621621622, + "grad_norm": 23.377365112304688, + "learning_rate": 9.918520279916879e-06, + "loss": 10.544861793518066, + "loss_alignment": 0.46533203125, + "loss_alignment_w": 0.232666015625, + "loss_sft": 0.42637256160378456, + "loss_total_v6": 0.659053847193718, + "step": 438 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10124684684684684, + "grad_norm": 25.30510711669922, + "learning_rate": 9.91783602331842e-06, + "loss": 10.905317306518555, + "loss_alignment": 0.46533203125, + "loss_alignment_w": 0.232666015625, + "loss_sft": 0.4486110806465149, + "loss_total_v6": 0.6815822869539261, + "step": 439 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10147747747747747, + "grad_norm": 23.370729446411133, + "learning_rate": 9.917148929371287e-06, + "loss": 10.542648315429688, + "loss_alignment": 0.4658203125, + "loss_alignment_w": 0.23291015625, + "loss_sft": 0.4258527308702469, + "loss_total_v6": 0.6589155048131943, + "step": 440 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10170810810810811, + "grad_norm": 23.257204055786133, + "learning_rate": 9.916458998471902e-06, + "loss": 10.848541259765625, + "loss_alignment": 0.46826171875, + "loss_alignment_w": 0.234130859375, + "loss_sft": 0.44407085701823235, + "loss_total_v6": 0.6780338808894157, + "step": 441 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10193873873873874, + "grad_norm": 25.300785064697266, + "learning_rate": 9.915766231018317e-06, + "loss": 10.127312660217285, + "loss_alignment": 0.462158203125, + "loss_alignment_w": 0.2310791015625, + "loss_sft": 0.4018932282924652, + "loss_total_v6": 0.6329570487141609, + "step": 442 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10216936936936936, + "grad_norm": 23.565044403076172, + "learning_rate": 9.915070627410229e-06, + "loss": 10.759785652160645, + "loss_alignment": 0.462158203125, + "loss_alignment_w": 0.2310791015625, + "loss_sft": 0.4413770064711571, + "loss_total_v6": 0.6724866256117821, + "step": 443 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.1024, + "grad_norm": 25.046476364135742, + "learning_rate": 9.914372188048964e-06, + "loss": 11.540979385375977, + "loss_alignment": 0.4638671875, + "loss_alignment_w": 0.23193359375, + "loss_sft": 0.48924029618501663, + "loss_total_v6": 0.7213111966848373, + "step": 444 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10263063063063063, + "grad_norm": 23.507482528686523, + "learning_rate": 9.91367091333749e-06, + "loss": 10.785213470458984, + "loss_alignment": 0.465576171875, + "loss_alignment_w": 0.2327880859375, + "loss_sft": 0.4414098598062992, + "loss_total_v6": 0.6740758791565895, + "step": 445 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10286126126126126, + "grad_norm": 23.55121421813965, + "learning_rate": 9.912966803680404e-06, + "loss": 10.66843032836914, + "loss_alignment": 0.463134765625, + "loss_alignment_w": 0.2315673828125, + "loss_sft": 0.4348585568368435, + "loss_total_v6": 0.6667769104242325, + "step": 446 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.1030918918918919, + "grad_norm": 22.710813522338867, + "learning_rate": 9.912259859483943e-06, + "loss": 10.554765701293945, + "loss_alignment": 0.46826171875, + "loss_alignment_w": 0.234130859375, + "loss_sft": 0.4259997755289078, + "loss_total_v6": 0.6596728414297104, + "step": 447 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10332252252252253, + "grad_norm": 23.726688385009766, + "learning_rate": 9.911550081155983e-06, + "loss": 10.771045684814453, + "loss_alignment": 0.467041015625, + "loss_alignment_w": 0.2335205078125, + "loss_sft": 0.4394715204834938, + "loss_total_v6": 0.6731903776526451, + "step": 448 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10355315315315315, + "grad_norm": 23.651884078979492, + "learning_rate": 9.910837469106028e-06, + "loss": 11.265338897705078, + "loss_alignment": 0.4677734375, + "loss_alignment_w": 0.23388671875, + "loss_sft": 0.47045638784766197, + "loss_total_v6": 0.7040837109088898, + "step": 449 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10378378378378378, + "grad_norm": 22.328561782836914, + "learning_rate": 9.910122023745219e-06, + "loss": 10.864168167114258, + "loss_alignment": 0.460693359375, + "loss_alignment_w": 0.2303466796875, + "loss_sft": 0.4482061043381691, + "loss_total_v6": 0.6790105625987053, + "step": 450 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10401441441441442, + "grad_norm": 24.680404663085938, + "learning_rate": 9.909403745486335e-06, + "loss": 11.207143783569336, + "loss_alignment": 0.46240234375, + "loss_alignment_w": 0.231201171875, + "loss_sft": 0.4689859002828598, + "loss_total_v6": 0.7004464417695999, + "step": 451 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10424504504504505, + "grad_norm": 53.7684440612793, + "learning_rate": 9.908682634743785e-06, + "loss": 10.545040130615234, + "loss_alignment": 0.46875, + "loss_alignment_w": 0.234375, + "loss_sft": 0.42458321154117584, + "loss_total_v6": 0.6590649858117104, + "step": 452 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10447567567567567, + "grad_norm": 26.217926025390625, + "learning_rate": 9.907958691933616e-06, + "loss": 10.970144271850586, + "loss_alignment": 0.45654296875, + "loss_alignment_w": 0.228271484375, + "loss_sft": 0.4570115767419338, + "loss_total_v6": 0.685634009540081, + "step": 453 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10470630630630631, + "grad_norm": 23.562213897705078, + "learning_rate": 9.907231917473507e-06, + "loss": 11.092033386230469, + "loss_alignment": 0.466796875, + "loss_alignment_w": 0.2333984375, + "loss_sft": 0.4598536193370819, + "loss_total_v6": 0.6932520493865013, + "step": 454 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10493693693693694, + "grad_norm": 22.80263900756836, + "learning_rate": 9.90650231178277e-06, + "loss": 11.075626373291016, + "loss_alignment": 0.46484375, + "loss_alignment_w": 0.232421875, + "loss_sft": 0.4598810747265816, + "loss_total_v6": 0.6922266483306885, + "step": 455 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10516756756756757, + "grad_norm": 26.141660690307617, + "learning_rate": 9.905769875282351e-06, + "loss": 10.776735305786133, + "loss_alignment": 0.467529296875, + "loss_alignment_w": 0.2337646484375, + "loss_sft": 0.4397661052644253, + "loss_total_v6": 0.6735459938645363, + "step": 456 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10539819819819819, + "grad_norm": 23.490503311157227, + "learning_rate": 9.905034608394835e-06, + "loss": 10.894706726074219, + "loss_alignment": 0.465087890625, + "loss_alignment_w": 0.2325439453125, + "loss_sft": 0.44823792949318886, + "loss_total_v6": 0.6809191703796387, + "step": 457 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10562882882882883, + "grad_norm": 23.6722469329834, + "learning_rate": 9.904296511544427e-06, + "loss": 10.732725143432617, + "loss_alignment": 0.46337890625, + "loss_alignment_w": 0.231689453125, + "loss_sft": 0.4388617053627968, + "loss_total_v6": 0.6707952916622162, + "step": 458 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10585945945945946, + "grad_norm": 25.30912971496582, + "learning_rate": 9.903555585156977e-06, + "loss": 11.462015151977539, + "loss_alignment": 0.462158203125, + "loss_alignment_w": 0.2310791015625, + "loss_sft": 0.4845491573214531, + "loss_total_v6": 0.716375932097435, + "step": 459 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10609009009009009, + "grad_norm": 21.056793212890625, + "learning_rate": 9.902811829659962e-06, + "loss": 11.2574462890625, + "loss_alignment": 0.46728515625, + "loss_alignment_w": 0.233642578125, + "loss_sft": 0.4695969261229038, + "loss_total_v6": 0.7035904452204704, + "step": 460 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10632072072072073, + "grad_norm": 22.523239135742188, + "learning_rate": 9.902065245482493e-06, + "loss": 11.171540260314941, + "loss_alignment": 0.47119140625, + "loss_alignment_w": 0.235595703125, + "loss_sft": 0.46239667013287544, + "loss_total_v6": 0.6982212737202644, + "step": 461 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10655135135135135, + "grad_norm": 22.005508422851562, + "learning_rate": 9.90131583305531e-06, + "loss": 10.44137191772461, + "loss_alignment": 0.46142578125, + "loss_alignment_w": 0.230712890625, + "loss_sft": 0.422284796833992, + "loss_total_v6": 0.6525857150554657, + "step": 462 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10678198198198198, + "grad_norm": 23.120498657226562, + "learning_rate": 9.900563592810789e-06, + "loss": 10.783355712890625, + "loss_alignment": 0.4619140625, + "loss_alignment_w": 0.23095703125, + "loss_sft": 0.44298744946718216, + "loss_total_v6": 0.6739597544074059, + "step": 463 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.1070126126126126, + "grad_norm": 23.64076042175293, + "learning_rate": 9.899808525182935e-06, + "loss": 11.24630069732666, + "loss_alignment": 0.469970703125, + "loss_alignment_w": 0.2349853515625, + "loss_sft": 0.46787794306874275, + "loss_total_v6": 0.7028938010334969, + "step": 464 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10724324324324325, + "grad_norm": 24.56892204284668, + "learning_rate": 9.899050630607386e-06, + "loss": 10.630768775939941, + "loss_alignment": 0.46728515625, + "loss_alignment_w": 0.233642578125, + "loss_sft": 0.43059738352894783, + "loss_total_v6": 0.6644230633974075, + "step": 465 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10747387387387387, + "grad_norm": 25.06197166442871, + "learning_rate": 9.898289909521405e-06, + "loss": 10.660741806030273, + "loss_alignment": 0.458984375, + "loss_alignment_w": 0.2294921875, + "loss_sft": 0.43681950494647026, + "loss_total_v6": 0.6662964075803757, + "step": 466 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.1077045045045045, + "grad_norm": 25.77151870727539, + "learning_rate": 9.897526362363896e-06, + "loss": 11.260348320007324, + "loss_alignment": 0.466796875, + "loss_alignment_w": 0.2333984375, + "loss_sft": 0.47054116800427437, + "loss_total_v6": 0.7037717550992966, + "step": 467 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10793513513513514, + "grad_norm": 23.865127563476562, + "learning_rate": 9.896759989575386e-06, + "loss": 11.09048843383789, + "loss_alignment": 0.46435546875, + "loss_alignment_w": 0.232177734375, + "loss_sft": 0.4611913859844208, + "loss_total_v6": 0.693155474960804, + "step": 468 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10816576576576577, + "grad_norm": 23.187761306762695, + "learning_rate": 9.895990791598037e-06, + "loss": 10.385616302490234, + "loss_alignment": 0.470947265625, + "loss_alignment_w": 0.2354736328125, + "loss_sft": 0.41374944150447845, + "loss_total_v6": 0.6491010338068008, + "step": 469 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.1083963963963964, + "grad_norm": 25.664648056030273, + "learning_rate": 9.89521876887563e-06, + "loss": 11.470087051391602, + "loss_alignment": 0.459716796875, + "loss_alignment_w": 0.2298583984375, + "loss_sft": 0.4868541769683361, + "loss_total_v6": 0.7168804183602333, + "step": 470 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10862702702702702, + "grad_norm": 22.11395263671875, + "learning_rate": 9.894443921853594e-06, + "loss": 11.059349060058594, + "loss_alignment": 0.4716796875, + "loss_alignment_w": 0.23583984375, + "loss_sft": 0.4552779495716095, + "loss_total_v6": 0.6912093386054039, + "step": 471 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10885765765765766, + "grad_norm": 23.912078857421875, + "learning_rate": 9.893666250978971e-06, + "loss": 10.871261596679688, + "loss_alignment": 0.458740234375, + "loss_alignment_w": 0.2293701171875, + "loss_sft": 0.45019053667783737, + "loss_total_v6": 0.6794538721442223, + "step": 472 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10908828828828829, + "grad_norm": 24.913379669189453, + "learning_rate": 9.892885756700441e-06, + "loss": 10.732388496398926, + "loss_alignment": 0.45703125, + "loss_alignment_w": 0.228515625, + "loss_sft": 0.44213660806417465, + "loss_total_v6": 0.670774295926094, + "step": 473 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10931891891891891, + "grad_norm": 22.169631958007812, + "learning_rate": 9.89210243946831e-06, + "loss": 10.82347297668457, + "loss_alignment": 0.474609375, + "loss_alignment_w": 0.2373046875, + "loss_sft": 0.43955906853079796, + "loss_total_v6": 0.6764670386910439, + "step": 474 + }, + { + "action_cond_aligned": 0.984375, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10954954954954955, + "grad_norm": 23.640565872192383, + "learning_rate": 9.891316299734514e-06, + "loss": 10.660772323608398, + "loss_alignment": 0.45556640625, + "loss_alignment_w": 0.227783203125, + "loss_sft": 0.4381793849170208, + "loss_total_v6": 0.6662982553243637, + "step": 475 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.10978018018018018, + "grad_norm": 23.740629196166992, + "learning_rate": 9.890527337952617e-06, + "loss": 10.758347511291504, + "loss_alignment": 0.462158203125, + "loss_alignment_w": 0.2310791015625, + "loss_sft": 0.4416074976325035, + "loss_total_v6": 0.6723966747522354, + "step": 476 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11001081081081081, + "grad_norm": 23.36722755432129, + "learning_rate": 9.88973555457781e-06, + "loss": 10.70438003540039, + "loss_alignment": 0.4658203125, + "loss_alignment_w": 0.23291015625, + "loss_sft": 0.43594571575522423, + "loss_total_v6": 0.669023722410202, + "step": 477 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11024144144144143, + "grad_norm": 21.863033294677734, + "learning_rate": 9.888940950066915e-06, + "loss": 11.246312141418457, + "loss_alignment": 0.46337890625, + "loss_alignment_w": 0.231689453125, + "loss_sft": 0.47047263383865356, + "loss_total_v6": 0.702894501388073, + "step": 478 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11047207207207208, + "grad_norm": 22.770244598388672, + "learning_rate": 9.888143524878378e-06, + "loss": 10.43455696105957, + "loss_alignment": 0.46923828125, + "loss_alignment_w": 0.234619140625, + "loss_sft": 0.4173117130994797, + "loss_total_v6": 0.6521597430109978, + "step": 479 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.1107027027027027, + "grad_norm": 22.883928298950195, + "learning_rate": 9.887343279472272e-06, + "loss": 10.81997299194336, + "loss_alignment": 0.468505859375, + "loss_alignment_w": 0.2342529296875, + "loss_sft": 0.44193436577916145, + "loss_total_v6": 0.6762483268976212, + "step": 480 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11093333333333333, + "grad_norm": 23.1566162109375, + "learning_rate": 9.886540214310303e-06, + "loss": 10.664705276489258, + "loss_alignment": 0.471923828125, + "loss_alignment_w": 0.2359619140625, + "loss_sft": 0.4307500459253788, + "loss_total_v6": 0.6665441170334816, + "step": 481 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11116396396396397, + "grad_norm": 23.15321922302246, + "learning_rate": 9.885734329855798e-06, + "loss": 11.004983901977539, + "loss_alignment": 0.46875, + "loss_alignment_w": 0.234375, + "loss_sft": 0.4539247639477253, + "loss_total_v6": 0.6878114938735962, + "step": 482 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.1113945945945946, + "grad_norm": 25.8681697845459, + "learning_rate": 9.884925626573714e-06, + "loss": 10.900215148925781, + "loss_alignment": 0.46142578125, + "loss_alignment_w": 0.230712890625, + "loss_sft": 0.45030637830495834, + "loss_total_v6": 0.6812634021043777, + "step": 483 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11162522522522522, + "grad_norm": 25.794992446899414, + "learning_rate": 9.884114104930629e-06, + "loss": 10.907711029052734, + "loss_alignment": 0.46630859375, + "loss_alignment_w": 0.233154296875, + "loss_sft": 0.4490659609436989, + "loss_total_v6": 0.6817319542169571, + "step": 484 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11185585585585586, + "grad_norm": 22.76789665222168, + "learning_rate": 9.883299765394756e-06, + "loss": 11.095043182373047, + "loss_alignment": 0.468017578125, + "loss_alignment_w": 0.2340087890625, + "loss_sft": 0.45918727293610573, + "loss_total_v6": 0.6934402137994766, + "step": 485 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11208648648648649, + "grad_norm": 22.36644744873047, + "learning_rate": 9.882482608435924e-06, + "loss": 10.827279090881348, + "loss_alignment": 0.460693359375, + "loss_alignment_w": 0.2303466796875, + "loss_sft": 0.44661770388484, + "loss_total_v6": 0.6767049804329872, + "step": 486 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11231711711711712, + "grad_norm": 23.028827667236328, + "learning_rate": 9.881662634525596e-06, + "loss": 10.604097366333008, + "loss_alignment": 0.455810546875, + "loss_alignment_w": 0.2279052734375, + "loss_sft": 0.4350186698138714, + "loss_total_v6": 0.6627560928463936, + "step": 487 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11254774774774774, + "grad_norm": 23.679529190063477, + "learning_rate": 9.880839844136854e-06, + "loss": 10.997076034545898, + "loss_alignment": 0.458740234375, + "loss_alignment_w": 0.2293701171875, + "loss_sft": 0.45813027024269104, + "loss_total_v6": 0.6873172745108604, + "step": 488 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11277837837837838, + "grad_norm": 21.273813247680664, + "learning_rate": 9.88001423774441e-06, + "loss": 11.124725341796875, + "loss_alignment": 0.470947265625, + "loss_alignment_w": 0.2354736328125, + "loss_sft": 0.4595318101346493, + "loss_total_v6": 0.6952953562140465, + "step": 489 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11300900900900901, + "grad_norm": 23.441883087158203, + "learning_rate": 9.879185815824595e-06, + "loss": 10.884700775146484, + "loss_alignment": 0.46044921875, + "loss_alignment_w": 0.230224609375, + "loss_sft": 0.4497334621846676, + "loss_total_v6": 0.6802937984466553, + "step": 490 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11323963963963964, + "grad_norm": 26.165691375732422, + "learning_rate": 9.87835457885537e-06, + "loss": 10.745962142944336, + "loss_alignment": 0.45751953125, + "loss_alignment_w": 0.228759765625, + "loss_sft": 0.442649208009243, + "loss_total_v6": 0.6716225892305374, + "step": 491 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11347027027027028, + "grad_norm": 24.133716583251953, + "learning_rate": 9.877520527316317e-06, + "loss": 10.726882934570312, + "loss_alignment": 0.4658203125, + "loss_alignment_w": 0.23291015625, + "loss_sft": 0.4369402341544628, + "loss_total_v6": 0.6704302206635475, + "step": 492 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.1137009009009009, + "grad_norm": 22.75576400756836, + "learning_rate": 9.876683661688642e-06, + "loss": 10.771232604980469, + "loss_alignment": 0.4658203125, + "loss_alignment_w": 0.23291015625, + "loss_sft": 0.4405970349907875, + "loss_total_v6": 0.6732020750641823, + "step": 493 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11393153153153153, + "grad_norm": 24.46021270751953, + "learning_rate": 9.875843982455176e-06, + "loss": 11.010108947753906, + "loss_alignment": 0.467529296875, + "loss_alignment_w": 0.2337646484375, + "loss_sft": 0.45387889072299004, + "loss_total_v6": 0.6881318166851997, + "step": 494 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11416216216216216, + "grad_norm": 25.227407455444336, + "learning_rate": 9.875001490100372e-06, + "loss": 10.517667770385742, + "loss_alignment": 0.4541015625, + "loss_alignment_w": 0.22705078125, + "loss_sft": 0.4303797595202923, + "loss_total_v6": 0.6573542505502701, + "step": 495 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.1143927927927928, + "grad_norm": 21.991230010986328, + "learning_rate": 9.874156185110307e-06, + "loss": 10.776002883911133, + "loss_alignment": 0.474853515625, + "loss_alignment_w": 0.2374267578125, + "loss_sft": 0.4362259954214096, + "loss_total_v6": 0.6735001727938652, + "step": 496 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11462342342342342, + "grad_norm": 23.572975158691406, + "learning_rate": 9.873308067972679e-06, + "loss": 11.135774612426758, + "loss_alignment": 0.471923828125, + "loss_alignment_w": 0.2359619140625, + "loss_sft": 0.4601765610277653, + "loss_total_v6": 0.6959858685731888, + "step": 497 + }, + { + "action_cond_aligned": 1.0, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11485405405405405, + "grad_norm": 22.884092330932617, + "learning_rate": 9.872457139176812e-06, + "loss": 11.033153533935547, + "loss_alignment": 0.45947265625, + "loss_alignment_w": 0.229736328125, + "loss_sft": 0.4604002982378006, + "loss_total_v6": 0.6895720511674881, + "step": 498 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11508468468468469, + "grad_norm": 27.144725799560547, + "learning_rate": 9.871603399213647e-06, + "loss": 11.32634449005127, + "loss_alignment": 0.472900390625, + "loss_alignment_w": 0.2364501953125, + "loss_sft": 0.47242290526628494, + "loss_total_v6": 0.7078965455293655, + "step": 499 + }, + { + "action_cond_aligned": 0.9921875, + "action_cond_missing": 0.0, + "action_cond_skipped": 0.0, + "epoch": 0.11531531531531532, + "grad_norm": 22.507171630859375, + "learning_rate": 9.870746848575751e-06, + "loss": 10.520195007324219, + "loss_alignment": 0.457763671875, + "loss_alignment_w": 0.2288818359375, + "loss_sft": 0.4286761023104191, + "loss_total_v6": 0.6575121507048607, + "step": 500 + }, + { + "epoch": 0.11531531531531532, + "eval_loss": 0.7139609456062317, + "eval_loss_alignment": 0.4486502033390411, + "eval_loss_alignment_w": 0.22432510166952055, + "eval_loss_sft": 0.4896359051148233, + "eval_loss_total_v6": 0.7139610070650164, + "eval_loss_wm_recon": 0.0, + "eval_loss_wm_recon_w": 0.0, + "eval_runtime": 209.6577, + "eval_samples_per_second": 8.323, + "eval_steps_per_second": 1.045, + "step": 500 + } + ], + "logging_steps": 1, + "max_steps": 4336, + "num_input_tokens_seen": 0, + "num_train_epochs": 1, + "save_steps": 100, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6.2765121036752e+18, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/zero_to_fp32.py b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/zero_to_fp32.py new file mode 100644 index 0000000000000000000000000000000000000000..5995d6e6f04e43b989587aa9022a3aef0c66d694 --- /dev/null +++ b/qwen-cua__runs__ocu_exact_rt_sig_ac_clean_gcon_cmp__v0-20260613-023235__checkpoint-500__evalweights/zero_to_fp32.py @@ -0,0 +1,760 @@ +#!/usr/bin/env python + +# Copyright (c) Microsoft Corporation. +# SPDX-License-Identifier: Apache-2.0 + +# DeepSpeed Team + +# This script extracts fp32 consolidated weights from a zero 1, 2 and 3 DeepSpeed checkpoints. It gets +# copied into the top level checkpoint dir, so the user can easily do the conversion at any point in +# the future. Once extracted, the weights don't require DeepSpeed and can be used in any +# application. +# +# example: +# python zero_to_fp32.py . output_dir/ +# or +# python zero_to_fp32.py . output_dir/ --safe_serialization + +import argparse +import torch +import glob +import math +import os +import re +import gc +import json +import numpy as np +from tqdm import tqdm +from collections import OrderedDict +from dataclasses import dataclass + +# while this script doesn't use deepspeed to recover data, since the checkpoints are pickled with +# DeepSpeed data structures it has to be available in the current python environment. +from deepspeed.utils import logger +from deepspeed.checkpoint.constants import (DS_VERSION, OPTIMIZER_STATE_DICT, SINGLE_PARTITION_OF_FP32_GROUPS, + FP32_FLAT_GROUPS, ZERO_STAGE, PARTITION_COUNT, PARAM_SHAPES, BUFFER_NAMES, + FROZEN_PARAM_SHAPES, FROZEN_PARAM_FRAGMENTS) + + +@dataclass +class zero_model_state: + buffers: dict() + param_shapes: dict() + shared_params: list + ds_version: int + frozen_param_shapes: dict() + frozen_param_fragments: dict() + + +debug = 0 + +# load to cpu +device = torch.device('cpu') + + +def atoi(text): + return int(text) if text.isdigit() else text + + +def natural_keys(text): + ''' + alist.sort(key=natural_keys) sorts in human order + http://nedbatchelder.com/blog/200712/human_sorting.html + (See Toothy's implementation in the comments) + ''' + return [atoi(c) for c in re.split(r'(\d+)', text)] + + +def get_model_state_file(checkpoint_dir, zero_stage): + if not os.path.isdir(checkpoint_dir): + raise FileNotFoundError(f"Directory '{checkpoint_dir}' doesn't exist") + + # there should be only one file + if zero_stage <= 2: + file = os.path.join(checkpoint_dir, "mp_rank_00_model_states.pt") + elif zero_stage == 3: + file = os.path.join(checkpoint_dir, "zero_pp_rank_0_mp_rank_00_model_states.pt") + + if not os.path.exists(file): + raise FileNotFoundError(f"can't find model states file at '{file}'") + + return file + + +def get_checkpoint_files(checkpoint_dir, glob_pattern): + # XXX: need to test that this simple glob rule works for multi-node setup too + ckpt_files = sorted(glob.glob(os.path.join(checkpoint_dir, glob_pattern)), key=natural_keys) + + if len(ckpt_files) == 0: + raise FileNotFoundError(f"can't find {glob_pattern} files in directory '{checkpoint_dir}'") + + return ckpt_files + + +def get_optim_files(checkpoint_dir): + return get_checkpoint_files(checkpoint_dir, "*_optim_states.pt") + + +def get_model_state_files(checkpoint_dir): + return get_checkpoint_files(checkpoint_dir, "*_model_states.pt") + + +def parse_model_states(files): + zero_model_states = [] + for file in files: + state_dict = torch.load(file, map_location=device, weights_only=False) + + if BUFFER_NAMES not in state_dict: + raise ValueError(f"{file} is not a model state checkpoint") + buffer_names = state_dict[BUFFER_NAMES] + if debug: + print("Found buffers:", buffer_names) + + # recover just the buffers while restoring them to fp32 if they were saved in fp16 + buffers = {k: v.float() for k, v in state_dict["module"].items() if k in buffer_names} + param_shapes = state_dict[PARAM_SHAPES] + + # collect parameters that are included in param_shapes + param_names = [] + for s in param_shapes: + for name in s.keys(): + param_names.append(name) + + # update with frozen parameters + frozen_param_shapes = state_dict.get(FROZEN_PARAM_SHAPES, None) + if frozen_param_shapes is not None: + if debug: + print(f"Found frozen_param_shapes: {frozen_param_shapes}") + param_names += list(frozen_param_shapes.keys()) + + # handle shared params + shared_params = [[k, v] for k, v in state_dict["shared_params"].items()] + + ds_version = state_dict.get(DS_VERSION, None) + + frozen_param_fragments = state_dict.get(FROZEN_PARAM_FRAGMENTS, None) + + z_model_state = zero_model_state(buffers=buffers, + param_shapes=param_shapes, + shared_params=shared_params, + ds_version=ds_version, + frozen_param_shapes=frozen_param_shapes, + frozen_param_fragments=frozen_param_fragments) + zero_model_states.append(z_model_state) + + return zero_model_states + + +def parse_optim_states(files, ds_checkpoint_dir): + total_files = len(files) + state_dicts = [] + for f in tqdm(files, desc='Loading checkpoint shards'): + state_dict = torch.load(f, map_location=device, mmap=True, weights_only=False) + # immediately discard the potentially huge 2 optimizer states as we only care for fp32 master weights + # and also handle the case where it was already removed by another helper script + state_dict["optimizer_state_dict"].pop("optimizer_state_dict", None) + state_dicts.append(state_dict) + + if ZERO_STAGE not in state_dicts[0][OPTIMIZER_STATE_DICT]: + raise ValueError(f"{files[0]} is not a zero checkpoint") + zero_stage = state_dicts[0][OPTIMIZER_STATE_DICT][ZERO_STAGE] + world_size = state_dicts[0][OPTIMIZER_STATE_DICT][PARTITION_COUNT] + + # For ZeRO-2 each param group can have different partition_count as data parallelism for expert + # parameters can be different from data parallelism for non-expert parameters. So we can just + # use the max of the partition_count to get the dp world_size. + + if type(world_size) is list: + world_size = max(world_size) + + if world_size != total_files: + raise ValueError( + f"Expected {world_size} of '*_optim_states.pt' under '{ds_checkpoint_dir}' but found {total_files} files. " + "Possibly due to an overwrite of an old checkpoint, or a checkpoint didn't get saved by one or more processes." + ) + + # the groups are named differently in each stage + if zero_stage <= 2: + fp32_groups_key = SINGLE_PARTITION_OF_FP32_GROUPS + elif zero_stage == 3: + fp32_groups_key = FP32_FLAT_GROUPS + else: + raise ValueError(f"unknown zero stage {zero_stage}") + + fp32_flat_groups = [state_dicts[i][OPTIMIZER_STATE_DICT][fp32_groups_key] for i in range(len(state_dicts))] + return zero_stage, world_size, fp32_flat_groups + + +def _get_fp32_state_dict_from_zero_checkpoint(ds_checkpoint_dir, exclude_frozen_parameters): + """ + Returns fp32 state_dict reconstructed from ds checkpoint + + Args: + - ``ds_checkpoint_dir``: path to the deepspeed checkpoint folder (where the optimizer files are) + + """ + print(f"Processing zero checkpoint '{ds_checkpoint_dir}'") + + optim_files = get_optim_files(ds_checkpoint_dir) + zero_stage, world_size, fp32_flat_groups = parse_optim_states(optim_files, ds_checkpoint_dir) + print(f"Detected checkpoint of type zero stage {zero_stage}, world_size: {world_size}") + + model_files = get_model_state_files(ds_checkpoint_dir) + + zero_model_states = parse_model_states(model_files) + print(f'Parsing checkpoint created by deepspeed=={zero_model_states[0].ds_version}') + + if zero_stage <= 2: + return _get_fp32_state_dict_from_zero2_checkpoint(world_size, fp32_flat_groups, zero_model_states, + exclude_frozen_parameters) + elif zero_stage == 3: + return _get_fp32_state_dict_from_zero3_checkpoint(world_size, fp32_flat_groups, zero_model_states, + exclude_frozen_parameters) + + +def _zero2_merge_frozen_params(state_dict, zero_model_states): + if zero_model_states[0].frozen_param_shapes is None or len(zero_model_states[0].frozen_param_shapes) == 0: + return + + frozen_param_shapes = zero_model_states[0].frozen_param_shapes + frozen_param_fragments = zero_model_states[0].frozen_param_fragments + + if debug: + num_elem = sum(s.numel() for s in frozen_param_shapes.values()) + print(f'rank 0: {FROZEN_PARAM_SHAPES}.numel = {num_elem}') + + wanted_params = len(frozen_param_shapes) + wanted_numel = sum(s.numel() for s in frozen_param_shapes.values()) + avail_numel = sum([p.numel() for p in frozen_param_fragments.values()]) + print(f'Frozen params: Have {avail_numel} numels to process.') + print(f'Frozen params: Need {wanted_numel} numels in {wanted_params} params') + + total_params = 0 + total_numel = 0 + for name, shape in frozen_param_shapes.items(): + total_params += 1 + unpartitioned_numel = shape.numel() + total_numel += unpartitioned_numel + + state_dict[name] = frozen_param_fragments[name] + + if debug: + print(f"{name} full shape: {shape} unpartitioned numel {unpartitioned_numel} ") + + print(f"Reconstructed Frozen fp32 state dict with {total_params} params {total_numel} elements") + + +def _has_callable(obj, fn): + attr = getattr(obj, fn, None) + return callable(attr) + + +def _zero2_merge_trainable_params(state_dict, world_size, fp32_flat_groups, zero_model_states): + param_shapes = zero_model_states[0].param_shapes + + # Reconstruction protocol: + # + # XXX: document this + + if debug: + for i in range(world_size): + for j in range(len(fp32_flat_groups[0])): + print(f"{FP32_FLAT_GROUPS}[{i}][{j}].shape={fp32_flat_groups[i][j].shape}") + + # XXX: memory usage doubles here (zero2) + num_param_groups = len(fp32_flat_groups[0]) + merged_single_partition_of_fp32_groups = [] + for i in range(num_param_groups): + merged_partitions = [sd[i] for sd in fp32_flat_groups] + full_single_fp32_vector = torch.cat(merged_partitions, 0) + merged_single_partition_of_fp32_groups.append(full_single_fp32_vector) + avail_numel = sum( + [full_single_fp32_vector.numel() for full_single_fp32_vector in merged_single_partition_of_fp32_groups]) + + if debug: + wanted_params = sum([len(shapes) for shapes in param_shapes]) + wanted_numel = sum([sum(shape.numel() for shape in shapes.values()) for shapes in param_shapes]) + # not asserting if there is a mismatch due to possible padding + print(f"Have {avail_numel} numels to process.") + print(f"Need {wanted_numel} numels in {wanted_params} params.") + + # params + # XXX: for huge models that can't fit into the host's RAM we will have to recode this to support + # out-of-core computing solution + total_numel = 0 + total_params = 0 + for shapes, full_single_fp32_vector in zip(param_shapes, merged_single_partition_of_fp32_groups): + offset = 0 + avail_numel = full_single_fp32_vector.numel() + for name, shape in shapes.items(): + + unpartitioned_numel = shape.numel() if _has_callable(shape, 'numel') else math.prod(shape) + total_numel += unpartitioned_numel + total_params += 1 + + if debug: + print(f"{name} full shape: {shape} unpartitioned numel {unpartitioned_numel} ") + state_dict[name] = full_single_fp32_vector.narrow(0, offset, unpartitioned_numel).view(shape) + offset += unpartitioned_numel + + # Z2 started to align to 2*world_size to improve nccl performance. Therefore both offset and + # avail_numel can differ by anywhere between 0..2*world_size. Due to two unrelated complex + # paddings performed in the code it's almost impossible to predict the exact numbers w/o the + # live optimizer object, so we are checking that the numbers are within the right range + align_to = 2 * world_size + + def zero2_align(x): + return align_to * math.ceil(x / align_to) + + if debug: + print(f"original offset={offset}, avail_numel={avail_numel}") + + offset = zero2_align(offset) + avail_numel = zero2_align(avail_numel) + + if debug: + print(f"aligned offset={offset}, avail_numel={avail_numel}") + + # Sanity check + if offset != avail_numel: + raise ValueError(f"consumed {offset} numels out of {avail_numel} - something is wrong") + + print(f"Reconstructed fp32 state dict with {total_params} params {total_numel} elements") + + +def _get_fp32_state_dict_from_zero2_checkpoint(world_size, fp32_flat_groups, zero_model_states, + exclude_frozen_parameters): + state_dict = OrderedDict() + + # buffers + buffers = zero_model_states[0].buffers + state_dict.update(buffers) + if debug: + print(f"added {len(buffers)} buffers") + + if not exclude_frozen_parameters: + _zero2_merge_frozen_params(state_dict, zero_model_states) + + _zero2_merge_trainable_params(state_dict, world_size, fp32_flat_groups, zero_model_states) + + # recover shared parameters + for pair in zero_model_states[0].shared_params: + if pair[1] in state_dict: + state_dict[pair[0]] = state_dict[pair[1]] + + return state_dict + + +def zero3_partitioned_param_info(unpartitioned_numel, world_size): + remainder = unpartitioned_numel % world_size + padding_numel = (world_size - remainder) if remainder else 0 + partitioned_numel = math.ceil(unpartitioned_numel / world_size) + return partitioned_numel, padding_numel + + +def _zero3_merge_frozen_params(state_dict, world_size, zero_model_states): + if zero_model_states[0].frozen_param_shapes is None or len(zero_model_states[0].frozen_param_shapes) == 0: + return + + if debug: + for i in range(world_size): + num_elem = sum(s.numel() for s in zero_model_states[i].frozen_param_fragments.values()) + print(f'rank {i}: {FROZEN_PARAM_SHAPES}.numel = {num_elem}') + + frozen_param_shapes = zero_model_states[0].frozen_param_shapes + wanted_params = len(frozen_param_shapes) + wanted_numel = sum(s.numel() for s in frozen_param_shapes.values()) + avail_numel = sum([p.numel() for p in zero_model_states[0].frozen_param_fragments.values()]) * world_size + print(f'Frozen params: Have {avail_numel} numels to process.') + print(f'Frozen params: Need {wanted_numel} numels in {wanted_params} params') + + total_params = 0 + total_numel = 0 + for name, shape in zero_model_states[0].frozen_param_shapes.items(): + total_params += 1 + unpartitioned_numel = shape.numel() + total_numel += unpartitioned_numel + + param_frags = tuple(model_state.frozen_param_fragments[name] for model_state in zero_model_states) + state_dict[name] = torch.cat(param_frags, 0).narrow(0, 0, unpartitioned_numel).view(shape) + + partitioned_numel, partitioned_padding_numel = zero3_partitioned_param_info(unpartitioned_numel, world_size) + + if debug: + print( + f"Frozen params: {total_params} {name} full shape: {shape} partition0 numel={partitioned_numel} partitioned_padding_numel={partitioned_padding_numel}" + ) + + print(f"Reconstructed Frozen fp32 state dict with {total_params} params {total_numel} elements") + + +class GatheredTensor: + """ + A pseudo tensor that collects partitioned weights. + It is more memory efficient when there are multiple groups. + """ + + def __init__(self, flat_groups, flat_groups_offset, offset, partitioned_numel, shape): + self.flat_groups = flat_groups + self.flat_groups_offset = flat_groups_offset + self.offset = offset + self.partitioned_numel = partitioned_numel + self.shape = shape + self.dtype = self.flat_groups[0][0].dtype + + def contiguous(self): + """ + Merge partitioned weights from flat_groups into a single tensor. + """ + end_idx = self.offset + self.partitioned_numel + world_size = len(self.flat_groups) + pad_flat_param_chunks = [] + + for rank_i in range(world_size): + # for each rank, we need to collect weights from related group/groups + flat_groups_at_rank_i = self.flat_groups[rank_i] + start_group_id = None + end_group_id = None + for group_id in range(len(self.flat_groups_offset)): + if self.flat_groups_offset[group_id] <= self.offset < self.flat_groups_offset[group_id + 1]: + start_group_id = group_id + if self.flat_groups_offset[group_id] < end_idx <= self.flat_groups_offset[group_id + 1]: + end_group_id = group_id + break + # collect weights from related group/groups + for group_id in range(start_group_id, end_group_id + 1): + flat_tensor = flat_groups_at_rank_i[group_id] + start_offset = self.offset - self.flat_groups_offset[group_id] + end_offset = min(end_idx, self.flat_groups_offset[group_id + 1]) - self.flat_groups_offset[group_id] + pad_flat_param_chunks.append(flat_tensor[start_offset:end_offset]) + + # collect weights from all ranks + pad_flat_param = torch.cat(pad_flat_param_chunks, dim=0) + param = pad_flat_param[:self.shape.numel()].view(self.shape).contiguous() + return param + + +def _zero3_merge_trainable_params(state_dict, world_size, fp32_flat_groups, zero_model_states): + param_shapes = zero_model_states[0].param_shapes + avail_numel = sum([flat_group.numel() for flat_group in fp32_flat_groups[0]]) * world_size + + # Reconstruction protocol: For zero3 we need to zip the partitions together at boundary of each + # param, re-consolidating each param, while dealing with padding if any + + # merge list of dicts, preserving order + param_shapes = {k: v for d in param_shapes for k, v in d.items()} + + if debug: + for i in range(world_size): + print(f"{FP32_FLAT_GROUPS}[{i}].shape={fp32_flat_groups[i].shape}") + + wanted_params = len(param_shapes) + wanted_numel = sum(shape.numel() for shape in param_shapes.values()) + # not asserting if there is a mismatch due to possible padding + avail_numel = fp32_flat_groups[0].numel() * world_size + print(f"Trainable params: Have {avail_numel} numels to process.") + print(f"Trainable params: Need {wanted_numel} numels in {wanted_params} params.") + + # params + # XXX: for huge models that can't fit into the host's RAM we will have to recode this to support + # out-of-core computing solution + offset = 0 + total_numel = 0 + total_params = 0 + flat_groups_offset = [0] + list(np.cumsum([flat_tensor.numel() for flat_tensor in fp32_flat_groups[0]])) + for name, shape in tqdm(param_shapes.items(), desc='Gathering sharded weights'): + unpartitioned_numel = shape.numel() + total_numel += unpartitioned_numel + total_params += 1 + partitioned_numel, partitioned_padding_numel = zero3_partitioned_param_info(unpartitioned_numel, world_size) + + if debug: + print( + f"Trainable params: {total_params} {name} full shape: {shape} partition0 numel={partitioned_numel} partitioned_padding_numel={partitioned_padding_numel}" + ) + + # memory efficient tensor + tensor = GatheredTensor(fp32_flat_groups, flat_groups_offset, offset, partitioned_numel, shape) + state_dict[name] = tensor + offset += partitioned_numel + + offset *= world_size + + # Sanity check + if offset != avail_numel: + raise ValueError(f"consumed {offset} numels out of {avail_numel} - something is wrong") + + print(f"Reconstructed Trainable fp32 state dict with {total_params} params {total_numel} elements") + + +def _get_fp32_state_dict_from_zero3_checkpoint(world_size, fp32_flat_groups, zero_model_states, + exclude_frozen_parameters): + state_dict = OrderedDict() + + # buffers + buffers = zero_model_states[0].buffers + state_dict.update(buffers) + if debug: + print(f"added {len(buffers)} buffers") + + if not exclude_frozen_parameters: + _zero3_merge_frozen_params(state_dict, world_size, zero_model_states) + + _zero3_merge_trainable_params(state_dict, world_size, fp32_flat_groups, zero_model_states) + + # recover shared parameters + for pair in zero_model_states[0].shared_params: + if pair[1] in state_dict: + state_dict[pair[0]] = state_dict[pair[1]] + + return state_dict + + +def to_torch_tensor(state_dict, return_empty_tensor=False): + """ + Convert state_dict of GatheredTensor to torch tensor + """ + torch_state_dict = {} + converted_tensors = {} + for name, tensor in state_dict.items(): + tensor_id = id(tensor) + if tensor_id in converted_tensors: # shared tensors + shared_tensor = torch_state_dict[converted_tensors[tensor_id]] + torch_state_dict[name] = shared_tensor + else: + converted_tensors[tensor_id] = name + if return_empty_tensor: + torch_state_dict[name] = torch.empty(tensor.shape, dtype=tensor.dtype) + else: + torch_state_dict[name] = tensor.contiguous() + return torch_state_dict + + +def get_fp32_state_dict_from_zero_checkpoint(checkpoint_dir, + tag=None, + exclude_frozen_parameters=False, + lazy_mode=False): + """ + Convert ZeRO 2 or 3 checkpoint into a single fp32 consolidated state_dict that can be loaded with + ``load_state_dict()`` and used for training without DeepSpeed or shared with others, for example + via a model hub. + + Args: + - ``checkpoint_dir``: path to the desired checkpoint folder + - ``tag``: checkpoint tag used as a unique identifier for checkpoint. If not provided will attempt to load tag in 'latest' file. e.g., ``global_step14`` + - ``exclude_frozen_parameters``: exclude frozen parameters + - ``lazy_mode``: get state_dict in lazy mode. It returns a dict of pesduo tensor instead of torch tensor, which is more memory efficient. + Convert the pesduo tensor to torch tensor by ``.contiguous()`` + + Returns: + - pytorch ``state_dict`` + + A typical usage might be :: + + from deepspeed.utils.zero_to_fp32 import get_fp32_state_dict_from_zero_checkpoint + # do the training and checkpoint saving + state_dict = get_fp32_state_dict_from_zero_checkpoint(checkpoint_dir) # already on cpu + model = model.cpu() # move to cpu + model.load_state_dict(state_dict) + # submit to model hub or save the model to share with others + + In this example the ``model`` will no longer be usable in the deepspeed context of the same + application. i.e. you will need to re-initialize the deepspeed engine, since + ``model.load_state_dict(state_dict)`` will remove all the deepspeed magic from it. + + If you want it all done for you, use ``load_state_dict_from_zero_checkpoint`` instead. + + Note: the above usage may not work if your application doesn't have sufficient free CPU memory. + You may need to use the offline approach using the ``zero_to_fp32.py`` script that is saved with + the checkpoint. Or you can load state_dict in lazy mode :: + + from deepspeed.utils.zero_to_fp32 import get_fp32_state_dict_from_zero_checkpoint + state_dict = get_fp32_state_dict_from_zero_checkpoint(checkpoint_dir, lazy_mode=True) # not on cpu + for name, lazy_tensor in state_dict.item(): + tensor = lazy_tensor.contiguous() # to cpu + print(name, tensor) + # del tensor to release memory if it no longer in use + """ + if tag is None: + latest_path = os.path.join(checkpoint_dir, 'latest') + if os.path.isfile(latest_path): + with open(latest_path, 'r') as fd: + tag = fd.read().strip() + else: + raise ValueError(f"Unable to find 'latest' file at {latest_path}") + + ds_checkpoint_dir = os.path.join(checkpoint_dir, tag) + + if not os.path.isdir(ds_checkpoint_dir): + raise FileNotFoundError(f"Directory '{ds_checkpoint_dir}' doesn't exist") + + state_dict = _get_fp32_state_dict_from_zero_checkpoint(ds_checkpoint_dir, exclude_frozen_parameters) + if lazy_mode: + return state_dict + else: + return to_torch_tensor(state_dict) + + +def convert_zero_checkpoint_to_fp32_state_dict(checkpoint_dir, + output_dir, + max_shard_size="5GB", + safe_serialization=False, + tag=None, + exclude_frozen_parameters=False): + """ + Convert ZeRO 2 or 3 checkpoint into a single fp32 consolidated ``state_dict`` file that can be + loaded with ``torch.load(file)`` + ``load_state_dict()`` and used for training without DeepSpeed. + + Args: + - ``checkpoint_dir``: path to the desired checkpoint folder. (one that contains the tag-folder, like ``global_step14``) + - ``output_dir``: directory to the pytorch fp32 state_dict output files + - ``max_shard_size``: the maximum size for a checkpoint before being sharded, default value is 5GB + - ``safe_serialization``: whether to save the model using `safetensors` or the traditional PyTorch way (that uses `pickle`). + - ``tag``: checkpoint tag used as a unique identifier for checkpoint. If not provided will attempt to load tag in the file named ``latest`` in the checkpoint folder, e.g., ``global_step14`` + - ``exclude_frozen_parameters``: exclude frozen parameters + """ + + # Dependency pre-check + if safe_serialization: + try: + from safetensors.torch import save_file + except ImportError: + print('If you want to use `safe_serialization`, please `pip install safetensors`') + raise + if max_shard_size is not None: + try: + from huggingface_hub import split_torch_state_dict_into_shards + except ImportError: + print('If you want to use `max_shard_size`, please `pip install huggingface_hub`') + raise + + # Convert zero checkpoint to state_dict + state_dict = get_fp32_state_dict_from_zero_checkpoint(checkpoint_dir, + tag, + exclude_frozen_parameters, + lazy_mode=True) + + # Shard the model if it is too big. + weights_name = "model.safetensors" if safe_serialization else "pytorch_model.bin" + if max_shard_size is not None: + filename_pattern = weights_name.replace(".bin", "{suffix}.bin").replace(".safetensors", "{suffix}.safetensors") + # an memory-efficient approach for sharding + empty_state_dict = to_torch_tensor(state_dict, return_empty_tensor=True) + state_dict_split = split_torch_state_dict_into_shards(empty_state_dict, + filename_pattern=filename_pattern, + max_shard_size=max_shard_size) + else: + from collections import namedtuple + StateDictSplit = namedtuple("StateDictSplit", ["is_sharded", "filename_to_tensors"]) + state_dict_split = StateDictSplit(is_sharded=False, + filename_to_tensors={weights_name: list(state_dict.keys())}) + + # Save the model by shard + os.makedirs(output_dir, exist_ok=True) + filename_to_tensors = state_dict_split.filename_to_tensors.items() + for shard_file, tensors in tqdm(filename_to_tensors, desc="Saving checkpoint shards"): + shard_state_dict = {tensor_name: state_dict[tensor_name] for tensor_name in tensors} + shard_state_dict = to_torch_tensor(shard_state_dict) + output_path = os.path.join(output_dir, shard_file) + if safe_serialization: + save_file(shard_state_dict, output_path, metadata={"format": "pt"}) + else: + torch.save(shard_state_dict, output_path) + # release the memory of current shard + for tensor_name in list(shard_state_dict.keys()): + del state_dict[tensor_name] + del shard_state_dict[tensor_name] + del shard_state_dict + gc.collect() + + # Save index if sharded + if state_dict_split.is_sharded: + index = { + "metadata": state_dict_split.metadata, + "weight_map": state_dict_split.tensor_to_filename, + } + save_index_file = "model.safetensors.index.json" if safe_serialization else "pytorch_model.bin.index.json" + save_index_file = os.path.join(output_dir, save_index_file) + with open(save_index_file, "w", encoding="utf-8") as f: + content = json.dumps(index, indent=2, sort_keys=True) + "\n" + f.write(content) + + +def load_state_dict_from_zero_checkpoint(model, checkpoint_dir, tag=None): + """ + 1. Put the provided model to cpu + 2. Convert ZeRO 2 or 3 checkpoint into a single fp32 consolidated ``state_dict`` + 3. Load it into the provided model + + Args: + - ``model``: the model object to update + - ``checkpoint_dir``: path to the desired checkpoint folder. (one that contains the tag-folder, like ``global_step14``) + - ``tag``: checkpoint tag used as a unique identifier for checkpoint. If not provided will attempt to load tag in the file named ``latest`` in the checkpoint folder, e.g., ``global_step14`` + + Returns: + - ``model`: modified model + + Make sure you have plenty of CPU memory available before you call this function. If you don't + have enough use the ``zero_to_fp32.py`` utility to do the conversion. You will find it + conveniently placed for you in the checkpoint folder. + + A typical usage might be :: + + from deepspeed.utils.zero_to_fp32 import load_state_dict_from_zero_checkpoint + model = load_state_dict_from_zero_checkpoint(trainer.model, checkpoint_dir) + # submit to model hub or save the model to share with others + + Note, that once this was run, the ``model`` will no longer be usable in the deepspeed context + of the same application. i.e. you will need to re-initialize the deepspeed engine, since + ``model.load_state_dict(state_dict)`` will remove all the deepspeed magic from it. + + """ + logger.info("Extracting fp32 weights") + state_dict = get_fp32_state_dict_from_zero_checkpoint(checkpoint_dir, tag) + + logger.info("Overwriting model with fp32 weights") + model = model.cpu() + model.load_state_dict(state_dict, strict=False) + + return model + + +if __name__ == "__main__": + parser = argparse.ArgumentParser() + parser.add_argument("checkpoint_dir", + type=str, + help="path to the desired checkpoint folder, e.g., path/checkpoint-12") + parser.add_argument("output_dir", + type=str, + help="directory to the pytorch fp32 state_dict output files" + "(e.g. path/checkpoint-12-output/)") + parser.add_argument( + "--max_shard_size", + type=str, + default="5GB", + help="The maximum size for a checkpoint before being sharded. Checkpoints shard will then be each of size" + "lower than this size. If expressed as a string, needs to be digits followed by a unit (like `5MB`" + "We default it to 5GB in order for models to be able to run easily on free-tier google colab instances" + "without CPU OOM issues.") + parser.add_argument( + "--safe_serialization", + default=False, + action='store_true', + help="Whether to save the model using `safetensors` or the traditional PyTorch way (that uses `pickle`).") + parser.add_argument("-t", + "--tag", + type=str, + default=None, + help="checkpoint tag used as a unique identifier for checkpoint. e.g., global_step1") + parser.add_argument("--exclude_frozen_parameters", action='store_true', help="exclude frozen parameters") + parser.add_argument("-d", "--debug", action='store_true', help="enable debug") + args = parser.parse_args() + + debug = args.debug + + convert_zero_checkpoint_to_fp32_state_dict(args.checkpoint_dir, + args.output_dir, + max_shard_size=args.max_shard_size, + safe_serialization=args.safe_serialization, + tag=args.tag, + exclude_frozen_parameters=args.exclude_frozen_parameters)