Download run.json from JonesLin/next-jev-stage2-last-loop: direct link, hf CLI and curl.
- Browser
- Download file 2.52 kB
-
https://huggingface.co/JonesLin/next-jev-stage2-last-loop/resolve/main/run.json
- Command line
-
hf download hf://JonesLin/next-jev-stage2-last-loop/run.json
-
curl -L -o run.json https://huggingface.co/JonesLin/next-jev-stage2-last-loop/resolve/main/run.json
2.52 kB
| { | |
| "run_id": "f4acad907e444bf79d25f05e1d61aab2", | |
| "model": { | |
| "model_name_or_path": "models/Qwen3.5-0.8B", | |
| "revision": "2fc06364715b967f1860aea9cf38778875588b17", | |
| "dtype": "bfloat16", | |
| "attn_implementation": "sdpa", | |
| "workspace_tokens": 128, | |
| "loops": 3, | |
| "loop_scope": "middle_workspace", | |
| "loop_start_block": 1, | |
| "loop_end_block": 24, | |
| "detach_loop_state": false, | |
| "classification_loss_scope": "last_loop", | |
| "aux_memory_mode": "last_loop", | |
| "aux_weight": 0.3, | |
| "aux_dim": 256, | |
| "aux_heads": 8, | |
| "aux_max_tokens": 4096, | |
| "aux_chunk_size": 1024, | |
| "aux_response_chunk_size": 16, | |
| "aux_layers": 1, | |
| "aux_self_attn_layers": null, | |
| "aux_self_attn_window": null, | |
| "freeze_vision": false, | |
| "gradient_checkpointing": true, | |
| "training_stage": 2, | |
| "stage1_objective": "answer_and_cot" | |
| }, | |
| "train": { | |
| "train_file": "data/phase123-multi-response/train.jsonl", | |
| "validation_file": "data/phase123-multi-response/validation.jsonl", | |
| "output_dir": "runs/stage2-last-loop", | |
| "epochs": 1, | |
| "batch_size": 32, | |
| "gradient_accumulation": 1, | |
| "learning_rate": 8e-06, | |
| "lr_schedule": "warmup_cosine", | |
| "warmup_ratio": 0.0, | |
| "min_learning_rate": 8e-07, | |
| "weight_decay": 0.01, | |
| "max_grad_norm": 1.0, | |
| "max_length": 2048, | |
| "max_image_pixels": 262144, | |
| "seed": 42, | |
| "device": "cuda", | |
| "max_steps": null, | |
| "save_every": 100, | |
| "keep_checkpoints": 2, | |
| "num_workers": 4, | |
| "classifier_learning_rate": 8e-05, | |
| "init_checkpoint": "runs/stage1-multi-response/checkpoint-00006347", | |
| "answer_max_tokens": 256 | |
| }, | |
| "source_hashes": { | |
| "train_file": "52e712f8bf847f952515a4b377332c6765b97bcf4dfa8a82a06336fb56844408", | |
| "validation_file": "18e90d614a6da718fc9acec5571adcfb2c0003afbbf55d93b75d3fd47ad0e927", | |
| "train_images": "842bba11f1084b9c31cf9fc069c6e8310fd89e106e36c0c627e18a8c96ca8a1f", | |
| "validation_images": "b3a1754a3160fd19c88082a8294381c44a58058dfcdaea104e644eec0ccf30e8" | |
| }, | |
| "learning_rate_plan": { | |
| "type": "warmup_cosine", | |
| "total_steps": 5921, | |
| "warmup_steps": 0, | |
| "initial_learning_rate": 8e-06, | |
| "min_learning_rate": 8e-07 | |
| }, | |
| "torch_version": "2.14.0+cu130", | |
| "transformers_version": "5.17.0", | |
| "parameter_dtype": "float32", | |
| "initialization": { | |
| "checkpoint": "/scratch/255028/next_jev/runs/stage1-multi-response/checkpoint-00006347", | |
| "sha256": "5fda55205f1855792ea60a8dfb5368766e6c861860e08f7e55609e18270c76f8", | |
| "reset_classifier": true | |
| } | |
| } | |