Qwen3-4B-Countdown-RLVR / initial_phase_contract.json
NotoriousH2's picture
Update Countdown RLVR 3072-token training artifacts
1a3685c verified
Raw
History Blame Contribute Delete
5.35 kB
{
"schema_version": 1,
"task": "countdown_rlvr_initial_phase",
"model": {
"kind": "hub",
"repo_id": "Qwen/Qwen3-4B",
"revision": "1cfa9a7208912126459214e8b04321603b3df60c"
},
"training_data": {
"path": "data/countdown/train.jsonl",
"samples": 1024,
"hash_basis": "file_bytes",
"sha256": "aa7abb6242d8ada72e55a6d8d0917e3618473ddf2f0f880b288814394b231131"
},
"system_prompt": "제시된 숫자는 각 항목을 정확히 한 번씩 사용하여 목표값을 만드세요. 허용 연산은 +, -, *, /이며 괄호를 사용할 수 있습니다. 연산 기호는 두 수 사이에만 사용하세요. 숫자를 이어 붙여 새 수를 만들지 마세요. 풀이를 간결하게 검토한 뒤 <answer>와 </answer> 사이에는 등호와 계산 결과 없이 수식만 작성하세요.",
"phase": {
"name": "초기 구간",
"start_step": 1,
"end_step": 120,
"scheduler_end_step": 480,
"max_completion_length": 3072,
"vllm_gpu_memory_utilization": 0.25,
"resume_from_checkpoint": null
},
"training_phases": [
{
"name": "초기 구간",
"start_step": 1,
"end_step": 120,
"scheduler_end_step": 480,
"max_completion_length": 3072,
"vllm_gpu_memory_utilization": 0.25,
"resume_from_checkpoint": null
},
{
"name": "재개 구간",
"start_step": 121,
"end_step": 480,
"scheduler_end_step": 480,
"max_completion_length": 3072,
"vllm_gpu_memory_utilization": 0.25,
"resume_from_checkpoint": "checkpoint-120"
}
],
"runtime": {
"python": "3.12.3",
"packages": {
"torch": "2.11.0+cu130",
"transformers": "5.14.1",
"trl": "1.9.2",
"peft": "0.20.0",
"vllm": "0.24.0"
},
"allocator_environment": {
"PYTORCH_ALLOC_CONF": "expandable_segments:True",
"PYTORCH_CUDA_ALLOC_CONF": "expandable_segments:True"
},
"training_script_sha256": "4886fc449a84929704516cc67eae9196d531afdc8cc4d35fac92799038955fb4",
"dependency_lock": {
"path": "requirements-lock.txt",
"sha256": "e8d7d01893ceed1278caee3b21bed24ed3350ce4e649f103c91570e4c61b4562"
}
},
"lora": {
"r": 16,
"lora_alpha": 32,
"lora_dropout": 0.05,
"bias": "none",
"target_modules": [
"q_proj",
"k_proj",
"v_proj",
"o_proj",
"gate_proj",
"up_proj",
"down_proj"
],
"task_type": "CAUSAL_LM"
},
"rewards": {
"functions": [
"format_rewards",
"expression_rewards",
"correctness_rewards"
],
"weights": [
0.3,
0.2,
0.5
],
"implementation_sha256": "d779da290b9c7376298160a4e9c722e17af6e60d9e45a7bfdb4952ccf19b07f3"
},
"grpo": {
"seed": 42,
"per_device_train_batch_size": 2,
"gradient_accumulation_steps": 4,
"steps_per_generation": 4,
"generation_batch_size": 8,
"num_generations": 8,
"max_completion_length": 3072,
"temperature": 0.6,
"top_p": 0.95,
"top_k": 20,
"mask_truncated_completions": true,
"use_vllm": true,
"vllm_mode": "colocate",
"vllm_gpu_memory_utilization": 0.25,
"learning_rate": 5e-06,
"lr_scheduler_type": "SchedulerType.COSINE",
"warmup_steps": 0.05,
"optim": "OptimizerNames.PAGED_ADAMW_8BIT",
"beta": 0.001,
"epsilon": 0.2,
"loss_type": "grpo",
"bf16": true,
"gradient_checkpointing": true,
"chat_template_kwargs": {
"enable_thinking": true
},
"model_init_kwargs": {
"dtype": "bfloat16",
"revision": "1cfa9a7208912126459214e8b04321603b3df60c"
},
"max_steps": 480
},
"checkpoint": {
"label": "checkpoint-120",
"global_step": 120,
"files": {
"README.md": {
"bytes": 5193,
"sha256": "37e16a61fcdb14836debbbe0d1d63c181cd3e98ca2049d22b513c82bf2b55fd1"
},
"adapter_config.json": {
"bytes": 1146,
"sha256": "04f4e9988d744c123dd84078aa823e2afc9b78c9cdedae44edeea72fe876089e"
},
"adapter_model.safetensors": {
"bytes": 132187888,
"sha256": "e422bfa8c0c6088002925a85f57d554771757b890036639ad8d9345c9287562c"
},
"chat_template.jinja": {
"bytes": 4168,
"sha256": "a55ee1b1660128b7098723e0abcd92caa0788061051c62d51cbe87d9cf1974d8"
},
"optimizer.pt": {
"bytes": 67694085,
"sha256": "880ef606fe577d606b01fea31837f72ca083ed1710427cd64c0f2cbd8c2db2f1"
},
"rng_state.pth": {
"bytes": 14645,
"sha256": "c329ead9ac4b1f5a08c6d75afef1853aca4b1d7a9242bb4a9224b2682fd2e1ef"
},
"scheduler.pt": {
"bytes": 1465,
"sha256": "b21770590c70fe1297817882fd4f61724612625618a8c998fe2fcbd51ceeb8aa"
},
"tokenizer.json": {
"bytes": 11422650,
"sha256": "be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506"
},
"tokenizer_config.json": {
"bytes": 694,
"sha256": "04b1682c59acbd057f4c9072297faa73d56fc9de053094c659cdb4c464f58f86"
},
"trainer_state.json": {
"bytes": 40791,
"sha256": "a41022ab39cc63b52b47dfe5e00be141b14fb788058f0994ec3bea739ec9f4c8"
},
"training_args.bin": {
"bytes": 7761,
"sha256": "7722ce538c162271ac3dcf8ef53ba88fe9b2eb0d02a4347cd5ac692162992ad9"
}
}
}
}