diff --git a/.gitattributes b/.gitattributes index ef31f897da86a9672886e97ca78c2748119f2658..23461ede29fe63f267081264bb345d826ed0818f 100644 --- a/.gitattributes +++ b/.gitattributes @@ -70,3 +70,16 @@ videos/round002/f4f8079f6637_k000_a403ac96.mp4 filter=lfs diff=lfs merge=lfs -te videos/round002/7ebc8b336779_k001_585674db.mp4 filter=lfs diff=lfs merge=lfs -text videos/round002/f4f8079f6637_k001_a40bb712.mp4 filter=lfs diff=lfs merge=lfs -text videos/round002/f4f8079f6637_k001_45c1d63e.mp4 filter=lfs diff=lfs merge=lfs -text +reward_debug/round000/1b67584bff87_k000_ddafd33c_CLEAN_ms1.0641.mp4 filter=lfs diff=lfs merge=lfs -text +reward_debug/round000/1b67584bff87_k000_32666f10_CLEAN_ms1.1054.mp4 filter=lfs diff=lfs merge=lfs -text +reward_debug/round000/1b67584bff87_k000_89927d91_CLEAN_ms1.3450.mp4 filter=lfs diff=lfs merge=lfs -text +reward_debug/round000/1b67584bff87_k001_62e8daff_FAIL_ms1.0768.mp4 filter=lfs diff=lfs merge=lfs -text +reward_debug/round000/1b67584bff87_k000_140f07dc_CLEAN_ms1.1054.mp4 filter=lfs diff=lfs merge=lfs -text +reward_debug/round000/1b67584bff87_k001_6615c2c4_CLEAN_ms1.1069.mp4 filter=lfs diff=lfs merge=lfs -text +reward_debug/round000/1b67584bff87_k000_2d3023c7_CLEAN_ms1.0881.mp4 filter=lfs diff=lfs merge=lfs -text +reward_debug/round000/1b67584bff87_k000_4dd03d13_CLEAN_ms1.0985.mp4 filter=lfs diff=lfs merge=lfs -text +reward_debug/round000/1b67584bff87_k000_1c91e481_CLEAN_ms1.0826.mp4 filter=lfs diff=lfs merge=lfs -text +reward_debug/round000/1b67584bff87_k000_c304d5a9_CLEAN_ms1.0826.mp4 filter=lfs diff=lfs merge=lfs -text +reward_debug/round000/1b67584bff87_k000_98e9a7bc_CLEAN_ms1.3450.mp4 filter=lfs diff=lfs merge=lfs -text +reward_debug/round000/1b67584bff87_k001_65f6114e_CLEAN_ms1.0744.mp4 filter=lfs diff=lfs merge=lfs -text +reward_debug/round000/1b67584bff87_k000_6843a418_CLEAN_ms1.0641.mp4 filter=lfs diff=lfs merge=lfs -text diff --git a/reward_debug/round000/1b67584bff87_k000_140f07dc_CLEAN_ms1.1054.mp4 b/reward_debug/round000/1b67584bff87_k000_140f07dc_CLEAN_ms1.1054.mp4 new file mode 100644 index 0000000000000000000000000000000000000000..6d70f44f6fd678b98b159cee6c4e05c1b48dda82 --- /dev/null +++ b/reward_debug/round000/1b67584bff87_k000_140f07dc_CLEAN_ms1.1054.mp4 @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fbf507778ce75b6bb56bab71c7494c67addb54ce20d32ef56bcb822da25c3ab4 +size 2361972 diff --git a/reward_debug/round000/1b67584bff87_k000_1c91e481_CLEAN_ms1.0826.mp4 b/reward_debug/round000/1b67584bff87_k000_1c91e481_CLEAN_ms1.0826.mp4 new file mode 100644 index 0000000000000000000000000000000000000000..d43a173b907adef61651d1a3609a42bd87bcd95e --- /dev/null +++ b/reward_debug/round000/1b67584bff87_k000_1c91e481_CLEAN_ms1.0826.mp4 @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7b9bb67e3c7ac5da546a5ca12c4449a4effeb623d45eaf344a42e7ea8261152f +size 2339614 diff --git a/reward_debug/round000/1b67584bff87_k000_2d3023c7_CLEAN_ms1.0881.mp4 b/reward_debug/round000/1b67584bff87_k000_2d3023c7_CLEAN_ms1.0881.mp4 new file mode 100644 index 0000000000000000000000000000000000000000..fa749988a046b24b7135aa087016266f52bc6184 --- /dev/null +++ b/reward_debug/round000/1b67584bff87_k000_2d3023c7_CLEAN_ms1.0881.mp4 @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ab2fab2f0908c2fb038d679cb3415aabc412b67e8ce4ac2abfcadab5a4c489cd +size 2360584 diff --git a/reward_debug/round000/1b67584bff87_k000_32666f10_CLEAN_ms1.1054.mp4 b/reward_debug/round000/1b67584bff87_k000_32666f10_CLEAN_ms1.1054.mp4 new file mode 100644 index 0000000000000000000000000000000000000000..6d70f44f6fd678b98b159cee6c4e05c1b48dda82 --- /dev/null +++ b/reward_debug/round000/1b67584bff87_k000_32666f10_CLEAN_ms1.1054.mp4 @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fbf507778ce75b6bb56bab71c7494c67addb54ce20d32ef56bcb822da25c3ab4 +size 2361972 diff --git a/reward_debug/round000/1b67584bff87_k000_4dd03d13_CLEAN_ms1.0985.mp4 b/reward_debug/round000/1b67584bff87_k000_4dd03d13_CLEAN_ms1.0985.mp4 new file mode 100644 index 0000000000000000000000000000000000000000..e2302f9de7718abfbaf44978665d179cca734a7f --- /dev/null +++ b/reward_debug/round000/1b67584bff87_k000_4dd03d13_CLEAN_ms1.0985.mp4 @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:788fdbf6c6081f1182cf9b162937f6dfd58e217d9691989f7b277e6c3732936f +size 2356905 diff --git a/reward_debug/round000/1b67584bff87_k000_6843a418_CLEAN_ms1.0641.mp4 b/reward_debug/round000/1b67584bff87_k000_6843a418_CLEAN_ms1.0641.mp4 new file mode 100644 index 0000000000000000000000000000000000000000..b2a848930fdb2f1e1a8e790cd46d26603b60f28a --- /dev/null +++ b/reward_debug/round000/1b67584bff87_k000_6843a418_CLEAN_ms1.0641.mp4 @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2473aab17063a10360651d35b4785e4993ececb5ccce50874616ee87a90c7c70 +size 2392799 diff --git a/reward_debug/round000/1b67584bff87_k000_89927d91_CLEAN_ms1.3450.mp4 b/reward_debug/round000/1b67584bff87_k000_89927d91_CLEAN_ms1.3450.mp4 new file mode 100644 index 0000000000000000000000000000000000000000..09a7750af3103ba0b67f4f0de5ec77c6fb5c37fc --- /dev/null +++ b/reward_debug/round000/1b67584bff87_k000_89927d91_CLEAN_ms1.3450.mp4 @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7fa9070952b043c3da2268df805a52ff3df7058b04e83d9b7e00e2285871fec1 +size 2400922 diff --git a/reward_debug/round000/1b67584bff87_k000_98e9a7bc_CLEAN_ms1.3450.mp4 b/reward_debug/round000/1b67584bff87_k000_98e9a7bc_CLEAN_ms1.3450.mp4 new file mode 100644 index 0000000000000000000000000000000000000000..09a7750af3103ba0b67f4f0de5ec77c6fb5c37fc --- /dev/null +++ b/reward_debug/round000/1b67584bff87_k000_98e9a7bc_CLEAN_ms1.3450.mp4 @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7fa9070952b043c3da2268df805a52ff3df7058b04e83d9b7e00e2285871fec1 +size 2400922 diff --git a/reward_debug/round000/1b67584bff87_k000_c304d5a9_CLEAN_ms1.0826.mp4 b/reward_debug/round000/1b67584bff87_k000_c304d5a9_CLEAN_ms1.0826.mp4 new file mode 100644 index 0000000000000000000000000000000000000000..d43a173b907adef61651d1a3609a42bd87bcd95e --- /dev/null +++ b/reward_debug/round000/1b67584bff87_k000_c304d5a9_CLEAN_ms1.0826.mp4 @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7b9bb67e3c7ac5da546a5ca12c4449a4effeb623d45eaf344a42e7ea8261152f +size 2339614 diff --git a/reward_debug/round000/1b67584bff87_k000_ddafd33c_CLEAN_ms1.0641.mp4 b/reward_debug/round000/1b67584bff87_k000_ddafd33c_CLEAN_ms1.0641.mp4 new file mode 100644 index 0000000000000000000000000000000000000000..b2a848930fdb2f1e1a8e790cd46d26603b60f28a --- /dev/null +++ b/reward_debug/round000/1b67584bff87_k000_ddafd33c_CLEAN_ms1.0641.mp4 @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2473aab17063a10360651d35b4785e4993ececb5ccce50874616ee87a90c7c70 +size 2392799 diff --git a/reward_debug/round000/1b67584bff87_k001_62e8daff_FAIL_ms1.0768.mp4 b/reward_debug/round000/1b67584bff87_k001_62e8daff_FAIL_ms1.0768.mp4 new file mode 100644 index 0000000000000000000000000000000000000000..04f16e711595fb7143c4f3010eaa75fde1bc66b3 --- /dev/null +++ b/reward_debug/round000/1b67584bff87_k001_62e8daff_FAIL_ms1.0768.mp4 @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1fed03bdb01e0194771940b43a20058bd46224a3e36273f761fe78e5ff5e52a6 +size 2376223 diff --git a/reward_debug/round000/1b67584bff87_k001_65f6114e_CLEAN_ms1.0744.mp4 b/reward_debug/round000/1b67584bff87_k001_65f6114e_CLEAN_ms1.0744.mp4 new file mode 100644 index 0000000000000000000000000000000000000000..fad6c5e3ee265890f518c9d3e0220ed31bf66c17 --- /dev/null +++ b/reward_debug/round000/1b67584bff87_k001_65f6114e_CLEAN_ms1.0744.mp4 @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3eafe2bfbd7107ccb9fbd1b1838b70a3d7ddee382f6d04f9e8fb058f8578db96 +size 2323915 diff --git a/reward_debug/round000/1b67584bff87_k001_6615c2c4_CLEAN_ms1.1069.mp4 b/reward_debug/round000/1b67584bff87_k001_6615c2c4_CLEAN_ms1.1069.mp4 new file mode 100644 index 0000000000000000000000000000000000000000..a1c32ab47e7007546d96df74a780ec6dc9b5511d --- /dev/null +++ b/reward_debug/round000/1b67584bff87_k001_6615c2c4_CLEAN_ms1.1069.mp4 @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:10ee7b335cad92e16e4ce034607ffb5d7db382e1e78f381389822fbc4b21667f +size 2350533 diff --git a/wandb/run-20260408_000908-i6zi4vwm/files/config.yaml b/wandb/run-20260408_000908-i6zi4vwm/files/config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b5875adecedbb41900bc7cd38d2cc6c4684240e --- /dev/null +++ b/wandb/run-20260408_000908-i6zi4vwm/files/config.yaml @@ -0,0 +1,145 @@ +_wandb: + value: + cli_version: 0.18.5 + m: [] + python_version: 3.11.2 + t: + "1": + - 1 + - 11 + - 41 + - 49 + - 55 + - 71 + - 83 + - 105 + "2": + - 1 + - 11 + - 41 + - 49 + - 55 + - 63 + - 71 + - 83 + - 98 + - 105 + "3": + - 13 + - 16 + - 23 + - 55 + "4": 3.11.2 + "5": 0.18.5 + "6": 4.46.1 + "8": + - 5 + "12": 0.18.5 + "13": linux-x86_64 +K: + value: 16 +bsv2_check_window_frac: + value: 0.2 +bsv2_dup_max_frames: + value: 0 +bsv2_expected_grip_changes: + value: 6 +bsv2_idm_ckpt_path: + value: ckpts/vidar_ckpt/idm.pt +bsv2_mj_hi: + value: 1.3 +bsv2_mj_lo: + value: 0.6 +bsv2_pick_thr: + value: 0.05 +bsv2_place_thr: + value: 0.05 +bsv2_prompts: + value: + - red block + - green block + - blue block +bsv2_vertical_sep_thr: + value: 0.03 +bsv2_x_align_thr: + value: 0.025 +checkpointing_steps: + value: 10 +ckpt_dir: + value: ckpts/Wan2.2-TI2V-5B +convert_model_dtype: + value: true +dataset_json: + value: data/rl_train/robotwin_stack_blocks_three.json +effective_batch_size: + value: 16 +epochs_per_round: + value: 100 +frame_num: + value: 121 +gradient_checkpointing: + value: true +hallucination_crop_top_ratio: + value: 0.6667 +lambda_gripper: + value: 1 +learning_rate: + value: 1e-05 +lora_alpha: + value: 64 +lora_rank: + value: 64 +lora_target_modules: + value: null +max_grad_norm: + value: 2 +max_per_condition: + value: 64 +max_samples: + value: -1 +neg_prompt: + value: 色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走 +num_ode_steps: + value: 20 +num_rounds: + value: 10 +num_train_timesteps: + value: 1000 +offload_model: + value: false +output_dir: + value: data/outputs/creflow_stack_blocks_three +pt_dir: + value: ckpts/vidar_ckpt/merged_vidar_lora.pt +reset_optimizer_per_round: + value: false +resume_from_lora_checkpoint: + value: null +reward_backend: + value: blocks_stack_v2 +reward_config: + value: blocks_stack_v2 +sample_guide_scale: + value: 5 +sample_shift: + value: 5 +seed: + value: 42 +size: + value: 640*736 +skip_reward_debug_video: + value: true +task: + value: ti2v-5B +use_8bit_adam: + value: true +vidar_root: + value: "" +w_bad: + value: 1 +wandb_project: + value: Corrective-Reflow +wandb_run_name: + value: creflow_stack_blocks_three +weight_decay: + value: 0.01 diff --git a/wandb/run-20260408_000908-i6zi4vwm/files/output.log b/wandb/run-20260408_000908-i6zi4vwm/files/output.log new file mode 100644 index 0000000000000000000000000000000000000000..97f38fc3a7283c8753433d7cb7c2465d1e13069b --- /dev/null +++ b/wandb/run-20260408_000908-i6zi4vwm/files/output.log @@ -0,0 +1,139 @@ +Building Wan2.2 TI2V model ... (DDP=False, world_size=1) + Moving T5 encoder to GPU permanently (offload_model=false) ... +Building reward scorer ... + Loading IDM from ckpts/vidar_ckpt/idm.pt on cuda:0 ... + IDM loaded. + Loading SAM3 VideoPredictor (blocks_stack_v2) on cuda:0 ... +INFO 2026-04-08 00:10:37,404 17550 sam3_video_base.py: 125: setting max_num_objects=10000 and num_obj_for_compile=16 + SAM3 VideoPredictor (blocks_stack_v2) loaded. +Applying LoRA (single adapter) for CReflow ... + LoRA injected: rank=64, alpha=64, target_modules=['self_attn.q', 'self_attn.k', 'self_attn.v', 'self_attn.o', 'cross_attn.q', 'cross_attn.k', 'cross_attn.v', 'cross_attn.o'] + Trainable: 94.4 M / 5094.2 M total (1.85%) + Gradient checkpointing enabled on 30 DiT blocks +Trainable parameters: 94.4 M + Using 8-bit AdamW (bitsandbytes) +Dataset: 10 conditions +Encoding 10 conditions ... + [1/10] 6b20973f10ef... encoded + [2/10] cff33035b28a... encoded + [3/10] 7ebc8b336779... encoded + [4/10] 8077c679fb62... encoded + [5/10] 41855438a8e0... encoded + [6/10] a67b02f4e7ce... encoded + [7/10] 4568a9603f3b... encoded + [8/10] 1b67584bff87... encoded + [9/10] f4f8079f6637... encoded + [10/10] 1b70247c6412... encoded +All 10 conditions encoded. + +Starting Corrective Reflow | rounds=10 K=16 K_local=16 epochs/round=100 effective_bs=16 lr=1e-05 conditions=10 + +====================================================================== +ROUND 1/10 +====================================================================== +[Phase 1] ODE rollout ... + [CReflow rollout] condition 0/10: K=16, id=6b20973f10ef... +Traceback (most recent call last): + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1034, in + main() + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 866, in main + rollouts = creflow_rollout( + ^^^^^^^^^^^^^^^^ + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/creflow/rollout.py", line 267, in creflow_rollout + x0_preds = _ode_rollout_batch( + ^^^^^^^^^^^^^^^^^^^ + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/creflow/rollout.py", line 139, in _ode_rollout_batch + outputs = dit(x_list, t=ts_batch, context=context_list, seq_len=seq_len) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl + return self._call_impl(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl + return forward_call(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/home/tiger/.local/lib/python3.11/site-packages/peft/peft_model.py", line 812, in forward + return self.get_base_model()(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl + return self._call_impl(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl + return forward_call(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/models/wan/modules/model.py", line 498, in forward + x = block(x, **kwargs) + ^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl + return self._call_impl(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl + return forward_call(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 126, in _ckpt_fwd + return torch_ckpt.checkpoint(orig_fwd, *args, use_reentrant=False, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/_compile.py", line 53, in inner + return disable_fn(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/_dynamo/eval_frame.py", line 1044, in _fn + return fn(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/utils/checkpoint.py", line 503, in checkpoint + ret = function(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/models/wan/modules/model.py", line 241, in forward + e = (self.modulation.unsqueeze(0) + e).chunk(6, dim=2) + ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^~~ +torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 31.33 GiB. GPU 0 has a total capacity of 79.11 GiB of which 4.63 GiB is free. Process 3796 has 0 bytes memory in use. Process 8629 has 0 bytes memory in use. Including non-PyTorch memory, this process has 0 bytes memory in use. Of the allocated memory 66.44 GiB is allocated by PyTorch, and 5.42 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables) +Traceback (most recent call last): + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1034, in + main() + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 866, in main + rollouts = creflow_rollout( + ^^^^^^^^^^^^^^^^ + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/creflow/rollout.py", line 267, in creflow_rollout + x0_preds = _ode_rollout_batch( + ^^^^^^^^^^^^^^^^^^^ + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/creflow/rollout.py", line 139, in _ode_rollout_batch + outputs = dit(x_list, t=ts_batch, context=context_list, seq_len=seq_len) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl + return self._call_impl(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl + return forward_call(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/home/tiger/.local/lib/python3.11/site-packages/peft/peft_model.py", line 812, in forward + return self.get_base_model()(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl + return self._call_impl(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl + return forward_call(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/models/wan/modules/model.py", line 498, in forward + x = block(x, **kwargs) + ^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl + return self._call_impl(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl + return forward_call(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 126, in _ckpt_fwd + return torch_ckpt.checkpoint(orig_fwd, *args, use_reentrant=False, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/_compile.py", line 53, in inner + return disable_fn(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/_dynamo/eval_frame.py", line 1044, in _fn + return fn(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/utils/checkpoint.py", line 503, in checkpoint + ret = function(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/models/wan/modules/model.py", line 241, in forward + e = (self.modulation.unsqueeze(0) + e).chunk(6, dim=2) + ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^~~ +torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 31.33 GiB. GPU 0 has a total capacity of 79.11 GiB of which 4.63 GiB is free. Process 3796 has 0 bytes memory in use. Process 8629 has 0 bytes memory in use. Including non-PyTorch memory, this process has 0 bytes memory in use. Of the allocated memory 66.44 GiB is allocated by PyTorch, and 5.42 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables) diff --git a/wandb/run-20260408_000908-i6zi4vwm/files/wandb-metadata.json b/wandb/run-20260408_000908-i6zi4vwm/files/wandb-metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..5a5236ba99cdf7ea32743d9d3c1b512cb85fa168 --- /dev/null +++ b/wandb/run-20260408_000908-i6zi4vwm/files/wandb-metadata.json @@ -0,0 +1,144 @@ +{ + "os": "Linux-5.15.120.bsk.6-amd64-x86_64-with-glibc2.36", + "python": "3.11.2", + "startedAt": "2026-04-08T00:09:08.281832Z", + "args": [ + "--task", + "ti2v-5B", + "--size", + "640*736", + "--frame_num", + "121", + "--ckpt_dir", + "ckpts/Wan2.2-TI2V-5B", + "--pt_dir", + "ckpts/vidar_ckpt/merged_vidar_lora.pt", + "--dataset_json", + "data/rl_train/robotwin_stack_blocks_three.json", + "--output_dir", + "data/outputs/creflow_stack_blocks_three", + "--num_ode_steps", + "20", + "--sample_shift", + "5.0", + "--sample_guide_scale", + "5.0", + "--K", + "16", + "--num_rounds", + "10", + "--epochs_per_round", + "100", + "--max_per_condition", + "64", + "--effective_batch_size", + "16", + "--w_bad", + "1.0", + "--lambda_gripper", + "1.0", + "--seed", + "42", + "--reward_config", + "blocks_stack_v2", + "--convert_model_dtype", + "--offload_model", + "false", + "--learning_rate", + "1e-5", + "--weight_decay", + "0.01", + "--max_grad_norm", + "2.0", + "--checkpointing_steps", + "10", + "--lora_rank", + "64", + "--lora_alpha", + "64", + "--wandb_project", + "Corrective-Reflow", + "--wandb_run_name", + "creflow_stack_blocks_three" + ], + "program": "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", + "codePath": "fastvideo/train_creflow_wan_2_2_ti2v.py", + "git": { + "remote": "git@github.com:VincentNi0107/EmbodiedVideoRL.git", + "commit": "912b225d838164e0178805670881ee201ad81ff0" + }, + "email": "1163051845@qq.com", + "root": "data/outputs/creflow_stack_blocks_three", + "host": "n124-104-180", + "username": "tiger", + "executable": "/usr/bin/python", + "codePathLocal": "fastvideo/train_creflow_wan_2_2_ti2v.py", + "cpu_count": 104, + "cpu_count_logical": 208, + "gpu": "NVIDIA H100 80GB HBM3", + "gpu_count": 8, + "disk": { + "/": { + "total": "1055740600320", + "used": "477545373696" + } + }, + "memory": { + "total": "1977660338176" + }, + "cpu": { + "count": 104, + "countLogical": 208 + }, + "gpu_nvidia": [ + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + } + ], + "cudaVersion": "12.9" +} \ No newline at end of file diff --git a/wandb/run-20260408_000908-i6zi4vwm/files/wandb-summary.json b/wandb/run-20260408_000908-i6zi4vwm/files/wandb-summary.json new file mode 100644 index 0000000000000000000000000000000000000000..2d0dce4fbf52b32689fa336f9f595b7e109f207c --- /dev/null +++ b/wandb/run-20260408_000908-i6zi4vwm/files/wandb-summary.json @@ -0,0 +1 @@ +{"_wandb":{"runtime":121}} \ No newline at end of file diff --git a/wandb/run-20260408_000908-i6zi4vwm/logs/debug.log b/wandb/run-20260408_000908-i6zi4vwm/logs/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..958cf7babe6f7aa452771e6dd36085952656e882 --- /dev/null +++ b/wandb/run-20260408_000908-i6zi4vwm/logs/debug.log @@ -0,0 +1,26 @@ +2026-04-08 00:09:08,247 INFO MainThread:17550 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5 +2026-04-08 00:09:08,247 INFO MainThread:17550 [wandb_setup.py:_flush():79] Configure stats pid to 17550 +2026-04-08 00:09:08,247 INFO MainThread:17550 [wandb_setup.py:_flush():79] Loading settings from /home/tiger/.config/wandb/settings +2026-04-08 00:09:08,247 INFO MainThread:17550 [wandb_setup.py:_flush():79] Loading settings from /mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/wandb/settings +2026-04-08 00:09:08,247 INFO MainThread:17550 [wandb_setup.py:_flush():79] Loading settings from environment variables: {'disabled': 'false', 'project': 'Corrective-Reflow', 'api_key': '***REDACTED***', 'mode': 'online'} +2026-04-08 00:09:08,247 INFO MainThread:17550 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': 'online', '_disable_service': None} +2026-04-08 00:09:08,248 INFO MainThread:17550 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'fastvideo/train_creflow_wan_2_2_ti2v.py', 'program_abspath': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py', 'program': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py'} +2026-04-08 00:09:08,248 INFO MainThread:17550 [wandb_setup.py:_flush():79] Applying login settings: {} +2026-04-08 00:09:08,249 INFO MainThread:17550 [wandb_init.py:_log_setup():534] Logging user logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_000908-i6zi4vwm/logs/debug.log +2026-04-08 00:09:08,251 INFO MainThread:17550 [wandb_init.py:_log_setup():535] Logging internal logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_000908-i6zi4vwm/logs/debug-internal.log +2026-04-08 00:09:08,251 INFO MainThread:17550 [wandb_init.py:init():621] calling init triggers +2026-04-08 00:09:08,251 INFO MainThread:17550 [wandb_init.py:init():628] wandb.init called with sweep_config: {} +config: {'vidar_root': '', 'task': 'ti2v-5B', 'size': '640*736', 'frame_num': 121, 'ckpt_dir': 'ckpts/Wan2.2-TI2V-5B', 'pt_dir': 'ckpts/vidar_ckpt/merged_vidar_lora.pt', 'convert_model_dtype': True, 'offload_model': False, 'dataset_json': 'data/rl_train/robotwin_stack_blocks_three.json', 'max_samples': -1, 'output_dir': 'data/outputs/creflow_stack_blocks_three', 'num_ode_steps': 20, 'sample_shift': 5.0, 'sample_guide_scale': 5.0, 'neg_prompt': '色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走', 'seed': 42, 'K': 16, 'num_rounds': 10, 'epochs_per_round': 100, 'effective_batch_size': 16, 'w_bad': 1.0, 'max_per_condition': 64, 'lambda_gripper': 1.0, 'reset_optimizer_per_round': False, 'num_train_timesteps': 1000, 'learning_rate': 1e-05, 'weight_decay': 0.01, 'max_grad_norm': 2.0, 'checkpointing_steps': 10, 'gradient_checkpointing': True, 'use_8bit_adam': True, 'lora_rank': 64, 'lora_alpha': 64, 'lora_target_modules': None, 'resume_from_lora_checkpoint': None, 'reward_config': 'blocks_stack_v2', 'reward_backend': 'blocks_stack_v2', 'skip_reward_debug_video': True, 'wandb_project': 'Corrective-Reflow', 'wandb_run_name': 'creflow_stack_blocks_three', 'hallucination_crop_top_ratio': 0.6667, 'bsv2_prompts': ['red block', 'green block', 'blue block'], 'bsv2_pick_thr': 0.05, 'bsv2_place_thr': 0.05, 'bsv2_vertical_sep_thr': 0.03, 'bsv2_x_align_thr': 0.025, 'bsv2_check_window_frac': 0.2, 'bsv2_dup_max_frames': 0, 'bsv2_expected_grip_changes': 6, 'bsv2_mj_lo': 0.6, 'bsv2_mj_hi': 1.3, 'bsv2_idm_ckpt_path': 'ckpts/vidar_ckpt/idm.pt'} +2026-04-08 00:09:08,251 INFO MainThread:17550 [wandb_init.py:init():671] starting backend +2026-04-08 00:09:08,251 INFO MainThread:17550 [wandb_init.py:init():675] sending inform_init request +2026-04-08 00:09:08,279 INFO MainThread:17550 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn +2026-04-08 00:09:08,279 INFO MainThread:17550 [wandb_init.py:init():688] backend started and connected +2026-04-08 00:09:08,337 INFO MainThread:17550 [wandb_init.py:init():783] updated telemetry +2026-04-08 00:09:08,505 INFO MainThread:17550 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout +2026-04-08 00:09:08,884 INFO MainThread:17550 [wandb_init.py:init():867] starting run threads in backend +2026-04-08 00:09:09,247 INFO MainThread:17550 [wandb_run.py:_console_start():2463] atexit reg +2026-04-08 00:09:09,247 INFO MainThread:17550 [wandb_run.py:_redirect():2311] redirect: wrap_raw +2026-04-08 00:09:09,247 INFO MainThread:17550 [wandb_run.py:_redirect():2376] Wrapping output streams. +2026-04-08 00:09:09,247 INFO MainThread:17550 [wandb_run.py:_redirect():2401] Redirects installed. +2026-04-08 00:09:09,252 INFO MainThread:17550 [wandb_init.py:init():911] run started, returning control to user process +2026-04-08 00:11:09,780 WARNING MsgRouterThr:17550 [router.py:message_loop():77] message_loop has been closed diff --git a/wandb/run-20260408_001541-g1xdtpwt/files/output.log b/wandb/run-20260408_001541-g1xdtpwt/files/output.log new file mode 100644 index 0000000000000000000000000000000000000000..fc06fe06354ed6cfb82d227490fc416329cbee15 --- /dev/null +++ b/wandb/run-20260408_001541-g1xdtpwt/files/output.log @@ -0,0 +1,286 @@ +Building Wan2.2 TI2V model ... (DDP=True, world_size=8) +/usr/local/lib/python3.11/dist-packages/torch/distributed/distributed_c10d.py:4876: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning. + warnings.warn( # warn only once + Moving T5 encoder to GPU permanently (offload_model=false) ... +Building reward scorer ... + Loading IDM from ckpts/vidar_ckpt/idm.pt on cuda:0 ... + IDM loaded. + Loading SAM3 VideoPredictor (blocks_stack_v2) on cuda:0 ... +INFO 2026-04-08 00:16:58,025 18968 sam3_video_base.py: 125: setting max_num_objects=10000 and num_obj_for_compile=16 + SAM3 VideoPredictor (blocks_stack_v2) loaded. +Applying LoRA (single adapter) for CReflow ... + LoRA injected: rank=64, alpha=64, target_modules=['self_attn.q', 'self_attn.k', 'self_attn.v', 'self_attn.o', 'cross_attn.q', 'cross_attn.k', 'cross_attn.v', 'cross_attn.o'] + Trainable: 94.4 M / 5094.2 M total (1.85%) + Gradient checkpointing enabled on 30 DiT blocks +Trainable parameters: 94.4 M + Using 8-bit AdamW (bitsandbytes) +Dataset: 10 conditions +Encoding 10 conditions ... + [1/10] 6b20973f10ef... encoded + [2/10] cff33035b28a... encoded + [3/10] 7ebc8b336779... encoded + [4/10] 8077c679fb62... encoded + [5/10] 41855438a8e0... encoded + [6/10] a67b02f4e7ce... encoded + [7/10] 4568a9603f3b... encoded + [8/10] 1b67584bff87... encoded + [9/10] f4f8079f6637... encoded + [10/10] 1b70247c6412... encoded +All 10 conditions encoded. + +Starting Corrective Reflow | rounds=10 K=16 K_local=2 epochs/round=100 effective_bs=16 lr=1e-05 conditions=10 + +====================================================================== +ROUND 1/10 +====================================================================== +[Phase 1] ODE rollout ... + [CReflow rollout] condition 0/10: K=2, id=6b20973f10ef... + rollout: 2 videos in 56.5s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.17it/s] +propagate_in_video: 100%|██████████| 121/121 [00:12<00:00, 9.53it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.55it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.19it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.09it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.96it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 72.76it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.60it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.24it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.41it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.98it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.12it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 1/10: K=2, id=cff33035b28a... + rollout: 2 videos in 56.9s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.44it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.01it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.92it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.90it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.73it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.73it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.54it/s] +propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.75it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.42it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.50it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.13it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.15it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 2/10: K=2, id=7ebc8b336779... + rollout: 2 videos in 57.5s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.66it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.60it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.08it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.02it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.85it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.32it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.15it/s] +propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.05it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.33it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.21it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.75it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.09it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 3/10: K=2, id=8077c679fb62... + rollout: 2 videos in 57.6s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.11it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.56it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.25it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.81it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 85.87it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.61it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.69it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.20it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.56it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.94it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.82it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.03it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 4/10: K=2, id=41855438a8e0... + rollout: 2 videos in 57.1s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.83it/s] +propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.61it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 77.09it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.07it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.11it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.79it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.75it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.69it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.43it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.37it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.28it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.71it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 5/10: K=2, id=a67b02f4e7ce... + rollout: 2 videos in 57.1s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.96it/s] +propagate_in_video: 100%|██████████| 121/121 [00:11<00:00, 10.48it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.64it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.48it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.33it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.43it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 77.50it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.21it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.01it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.30it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.11it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.23it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 6/10: K=2, id=4568a9603f3b... + rollout: 2 videos in 57.7s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.97it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.58it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.65it/s] +propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.75it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.19it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.97it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.75it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.32it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.90it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.36it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.29it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.29it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 7/10: K=2, id=1b67584bff87... + rollout: 2 videos in 57.4s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.97it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.02it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 76.90it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.12it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.20it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.42it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.36it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.12it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.63it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.37it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.10it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.13it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 8/10: K=2, id=f4f8079f6637... + rollout: 2 videos in 57.6s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.98it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.02it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.98it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.34it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.75it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.74it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.41it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.15it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.98it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.29it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.62it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.98it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 9/10: K=2, id=1b70247c6412... + rollout: 2 videos in 57.6s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.88it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.21it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 85.68it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.16it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.28it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.22it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.94it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.26it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.23it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.00it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.12it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.45it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] done: 20 samples, rewards = [1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 0.00, 0.00] + Rollout done: 20 samples, good=18, bad=2, avg_reward=0.900 + 1b67584bff87: mean=1.00, good=2/2 + 1b70247c6412: mean=0.00, good=0/2 + 41855438a8e0: mean=1.00, good=2/2 + 4568a9603f3b: mean=1.00, good=2/2 + 6b20973f10ef: mean=1.00, good=2/2 + 7ebc8b336779: mean=1.00, good=2/2 + 8077c679fb62: mean=1.00, good=2/2 + a67b02f4e7ce: mean=1.00, good=2/2 + cff33035b28a: mean=1.00, good=2/2 + f4f8079f6637: mean=1.00, good=2/2 +[Phase 2] Library update ... + [Library] rank=0 has 18 good samples + [Library] after update: SuccessLibrary(total=127, coverage=10/10, max_per_condition=64) +[Phase 3] Re-pairing ... + [Re-pair] train_set: 20 samples (anchor=18, correction=2, skipped=0) +[Phase 4] Training (100 epochs, 20 samples) ... +Traceback (most recent call last): + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1034, in + main() + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 934, in main + train_metrics = train_one_round( + ^^^^^^^^^^^^^^^^ + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 574, in train_one_round + loss.backward() + File "/usr/local/lib/python3.11/dist-packages/torch/_tensor.py", line 625, in backward + torch.autograd.backward( + File "/usr/local/lib/python3.11/dist-packages/torch/autograd/__init__.py", line 354, in backward + _engine_run_backward( + File "/usr/local/lib/python3.11/dist-packages/torch/autograd/graph.py", line 841, in _engine_run_backward + return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +RuntimeError: element 0 of tensors does not require grad and does not have a grad_fn +[rank0]: Traceback (most recent call last): +[rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1034, in +[rank0]: main() +[rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 934, in main +[rank0]: train_metrics = train_one_round( +[rank0]: ^^^^^^^^^^^^^^^^ +[rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 574, in train_one_round +[rank0]: loss.backward() +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/_tensor.py", line 625, in backward +[rank0]: torch.autograd.backward( +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/autograd/__init__.py", line 354, in backward +[rank0]: _engine_run_backward( +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/autograd/graph.py", line 841, in _engine_run_backward +[rank0]: return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: RuntimeError: element 0 of tensors does not require grad and does not have a grad_fn diff --git a/wandb/run-20260408_001541-g1xdtpwt/files/wandb-metadata.json b/wandb/run-20260408_001541-g1xdtpwt/files/wandb-metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..e7c79f26c975321b64b848281be286a0304b072e --- /dev/null +++ b/wandb/run-20260408_001541-g1xdtpwt/files/wandb-metadata.json @@ -0,0 +1,144 @@ +{ + "os": "Linux-5.15.120.bsk.6-amd64-x86_64-with-glibc2.36", + "python": "3.11.2", + "startedAt": "2026-04-08T00:15:41.604548Z", + "args": [ + "--task", + "ti2v-5B", + "--size", + "640*736", + "--frame_num", + "121", + "--ckpt_dir", + "ckpts/Wan2.2-TI2V-5B", + "--pt_dir", + "ckpts/vidar_ckpt/merged_vidar_lora.pt", + "--dataset_json", + "data/rl_train/robotwin_stack_blocks_three.json", + "--output_dir", + "data/outputs/creflow_stack_blocks_three", + "--num_ode_steps", + "20", + "--sample_shift", + "5.0", + "--sample_guide_scale", + "5.0", + "--K", + "16", + "--num_rounds", + "10", + "--epochs_per_round", + "100", + "--max_per_condition", + "64", + "--effective_batch_size", + "16", + "--w_bad", + "1.0", + "--lambda_gripper", + "1.0", + "--seed", + "42", + "--reward_config", + "blocks_stack_v2", + "--convert_model_dtype", + "--offload_model", + "false", + "--learning_rate", + "1e-5", + "--weight_decay", + "0.01", + "--max_grad_norm", + "2.0", + "--checkpointing_steps", + "10", + "--lora_rank", + "64", + "--lora_alpha", + "64", + "--wandb_project", + "Corrective-Reflow", + "--wandb_run_name", + "creflow_stack_blocks_three" + ], + "program": "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", + "codePath": "fastvideo/train_creflow_wan_2_2_ti2v.py", + "git": { + "remote": "git@github.com:VincentNi0107/EmbodiedVideoRL.git", + "commit": "912b225d838164e0178805670881ee201ad81ff0" + }, + "email": "1163051845@qq.com", + "root": "data/outputs/creflow_stack_blocks_three", + "host": "n124-104-180", + "username": "tiger", + "executable": "/usr/bin/python", + "codePathLocal": "fastvideo/train_creflow_wan_2_2_ti2v.py", + "cpu_count": 104, + "cpu_count_logical": 208, + "gpu": "NVIDIA H100 80GB HBM3", + "gpu_count": 8, + "disk": { + "/": { + "total": "1055740600320", + "used": "481117626368" + } + }, + "memory": { + "total": "1977660338176" + }, + "cpu": { + "count": 104, + "countLogical": 208 + }, + "gpu_nvidia": [ + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + } + ], + "cudaVersion": "12.9" +} \ No newline at end of file diff --git a/wandb/run-20260408_001541-g1xdtpwt/files/wandb-summary.json b/wandb/run-20260408_001541-g1xdtpwt/files/wandb-summary.json new file mode 100644 index 0000000000000000000000000000000000000000..304f6858c0ba737e3033111fb701a256995ec60d --- /dev/null +++ b/wandb/run-20260408_001541-g1xdtpwt/files/wandb-summary.json @@ -0,0 +1 @@ +{"_wandb":{"runtime":1601}} \ No newline at end of file diff --git a/wandb/run-20260408_001541-g1xdtpwt/logs/debug-internal.log b/wandb/run-20260408_001541-g1xdtpwt/logs/debug-internal.log new file mode 100644 index 0000000000000000000000000000000000000000..34b0561ffae6b9b8eb1e06023fcdaeee9cce03c3 --- /dev/null +++ b/wandb/run-20260408_001541-g1xdtpwt/logs/debug-internal.log @@ -0,0 +1,16 @@ +{"time":"2026-04-08T00:15:41.614211928Z","level":"INFO","msg":"using version","core version":"0.18.5"} +{"time":"2026-04-08T00:15:41.614246115Z","level":"INFO","msg":"created symlink","path":"data/outputs/creflow_stack_blocks_three/wandb/run-20260408_001541-g1xdtpwt/logs/debug-core.log"} +{"time":"2026-04-08T00:15:41.932223148Z","level":"INFO","msg":"created new stream","id":"g1xdtpwt"} +{"time":"2026-04-08T00:15:41.932414446Z","level":"INFO","msg":"stream: started","id":"g1xdtpwt"} +{"time":"2026-04-08T00:15:41.932510608Z","level":"INFO","msg":"sender: started","stream_id":"g1xdtpwt"} +{"time":"2026-04-08T00:15:41.932482435Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"g1xdtpwt"}} +{"time":"2026-04-08T00:15:41.932549715Z","level":"INFO","msg":"handler: started","stream_id":{"value":"g1xdtpwt"}} +{"time":"2026-04-08T00:15:42.915509387Z","level":"INFO","msg":"Starting system monitor"} +{"time":"2026-04-08T00:42:23.439351832Z","level":"INFO","msg":"stream: closing","id":"g1xdtpwt"} +{"time":"2026-04-08T00:42:23.439437558Z","level":"INFO","msg":"Stopping system monitor"} +{"time":"2026-04-08T00:42:23.445282629Z","level":"INFO","msg":"Stopped system monitor"} +{"time":"2026-04-08T00:42:24.081303366Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"} +{"time":"2026-04-08T00:42:24.367520127Z","level":"INFO","msg":"handler: closed","stream_id":{"value":"g1xdtpwt"}} +{"time":"2026-04-08T00:42:24.367585187Z","level":"INFO","msg":"writer: Close: closed","stream_id":{"value":"g1xdtpwt"}} +{"time":"2026-04-08T00:42:24.367601446Z","level":"INFO","msg":"sender: closed","stream_id":"g1xdtpwt"} +{"time":"2026-04-08T00:42:24.371256063Z","level":"INFO","msg":"stream: closed","id":"g1xdtpwt"} diff --git a/wandb/run-20260408_001541-g1xdtpwt/logs/debug.log b/wandb/run-20260408_001541-g1xdtpwt/logs/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..4c85465f85dc9ab5ffcde08c148710c937dcf4e7 --- /dev/null +++ b/wandb/run-20260408_001541-g1xdtpwt/logs/debug.log @@ -0,0 +1,26 @@ +2026-04-08 00:15:41,570 INFO MainThread:18968 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5 +2026-04-08 00:15:41,571 INFO MainThread:18968 [wandb_setup.py:_flush():79] Configure stats pid to 18968 +2026-04-08 00:15:41,571 INFO MainThread:18968 [wandb_setup.py:_flush():79] Loading settings from /home/tiger/.config/wandb/settings +2026-04-08 00:15:41,571 INFO MainThread:18968 [wandb_setup.py:_flush():79] Loading settings from /mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/wandb/settings +2026-04-08 00:15:41,571 INFO MainThread:18968 [wandb_setup.py:_flush():79] Loading settings from environment variables: {'disabled': 'false', 'project': 'Corrective-Reflow', 'api_key': '***REDACTED***', 'mode': 'online'} +2026-04-08 00:15:41,571 INFO MainThread:18968 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': 'online', '_disable_service': None} +2026-04-08 00:15:41,571 INFO MainThread:18968 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'fastvideo/train_creflow_wan_2_2_ti2v.py', 'program_abspath': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py', 'program': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py'} +2026-04-08 00:15:41,571 INFO MainThread:18968 [wandb_setup.py:_flush():79] Applying login settings: {} +2026-04-08 00:15:41,573 INFO MainThread:18968 [wandb_init.py:_log_setup():534] Logging user logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_001541-g1xdtpwt/logs/debug.log +2026-04-08 00:15:41,575 INFO MainThread:18968 [wandb_init.py:_log_setup():535] Logging internal logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_001541-g1xdtpwt/logs/debug-internal.log +2026-04-08 00:15:41,575 INFO MainThread:18968 [wandb_init.py:init():621] calling init triggers +2026-04-08 00:15:41,575 INFO MainThread:18968 [wandb_init.py:init():628] wandb.init called with sweep_config: {} +config: {'vidar_root': '', 'task': 'ti2v-5B', 'size': '640*736', 'frame_num': 121, 'ckpt_dir': 'ckpts/Wan2.2-TI2V-5B', 'pt_dir': 'ckpts/vidar_ckpt/merged_vidar_lora.pt', 'convert_model_dtype': True, 'offload_model': False, 'dataset_json': 'data/rl_train/robotwin_stack_blocks_three.json', 'max_samples': -1, 'output_dir': 'data/outputs/creflow_stack_blocks_three', 'num_ode_steps': 20, 'sample_shift': 5.0, 'sample_guide_scale': 5.0, 'neg_prompt': '色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走', 'seed': 42, 'K': 16, 'num_rounds': 10, 'epochs_per_round': 100, 'effective_batch_size': 16, 'w_bad': 1.0, 'max_per_condition': 64, 'lambda_gripper': 1.0, 'reset_optimizer_per_round': False, 'num_train_timesteps': 1000, 'learning_rate': 1e-05, 'weight_decay': 0.01, 'max_grad_norm': 2.0, 'checkpointing_steps': 10, 'gradient_checkpointing': True, 'use_8bit_adam': True, 'lora_rank': 64, 'lora_alpha': 64, 'lora_target_modules': None, 'resume_from_lora_checkpoint': None, 'reward_config': 'blocks_stack_v2', 'reward_backend': 'blocks_stack_v2', 'skip_reward_debug_video': True, 'wandb_project': 'Corrective-Reflow', 'wandb_run_name': 'creflow_stack_blocks_three', 'hallucination_crop_top_ratio': 0.6667, 'bsv2_prompts': ['red block', 'green block', 'blue block'], 'bsv2_pick_thr': 0.05, 'bsv2_place_thr': 0.05, 'bsv2_vertical_sep_thr': 0.03, 'bsv2_x_align_thr': 0.025, 'bsv2_check_window_frac': 0.2, 'bsv2_dup_max_frames': 0, 'bsv2_expected_grip_changes': 6, 'bsv2_mj_lo': 0.6, 'bsv2_mj_hi': 1.3, 'bsv2_idm_ckpt_path': 'ckpts/vidar_ckpt/idm.pt'} +2026-04-08 00:15:41,575 INFO MainThread:18968 [wandb_init.py:init():671] starting backend +2026-04-08 00:15:41,575 INFO MainThread:18968 [wandb_init.py:init():675] sending inform_init request +2026-04-08 00:15:41,602 INFO MainThread:18968 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn +2026-04-08 00:15:41,602 INFO MainThread:18968 [wandb_init.py:init():688] backend started and connected +2026-04-08 00:15:41,658 INFO MainThread:18968 [wandb_init.py:init():783] updated telemetry +2026-04-08 00:15:41,824 INFO MainThread:18968 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout +2026-04-08 00:15:42,864 INFO MainThread:18968 [wandb_init.py:init():867] starting run threads in backend +2026-04-08 00:15:43,291 INFO MainThread:18968 [wandb_run.py:_console_start():2463] atexit reg +2026-04-08 00:15:43,291 INFO MainThread:18968 [wandb_run.py:_redirect():2311] redirect: wrap_raw +2026-04-08 00:15:43,291 INFO MainThread:18968 [wandb_run.py:_redirect():2376] Wrapping output streams. +2026-04-08 00:15:43,291 INFO MainThread:18968 [wandb_run.py:_redirect():2401] Redirects installed. +2026-04-08 00:15:43,297 INFO MainThread:18968 [wandb_init.py:init():911] run started, returning control to user process +2026-04-08 00:42:23,439 WARNING MsgRouterThr:18968 [router.py:message_loop():77] message_loop has been closed diff --git a/wandb/run-20260408_004547-x9lalqi6/files/config.yaml b/wandb/run-20260408_004547-x9lalqi6/files/config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b5875adecedbb41900bc7cd38d2cc6c4684240e --- /dev/null +++ b/wandb/run-20260408_004547-x9lalqi6/files/config.yaml @@ -0,0 +1,145 @@ +_wandb: + value: + cli_version: 0.18.5 + m: [] + python_version: 3.11.2 + t: + "1": + - 1 + - 11 + - 41 + - 49 + - 55 + - 71 + - 83 + - 105 + "2": + - 1 + - 11 + - 41 + - 49 + - 55 + - 63 + - 71 + - 83 + - 98 + - 105 + "3": + - 13 + - 16 + - 23 + - 55 + "4": 3.11.2 + "5": 0.18.5 + "6": 4.46.1 + "8": + - 5 + "12": 0.18.5 + "13": linux-x86_64 +K: + value: 16 +bsv2_check_window_frac: + value: 0.2 +bsv2_dup_max_frames: + value: 0 +bsv2_expected_grip_changes: + value: 6 +bsv2_idm_ckpt_path: + value: ckpts/vidar_ckpt/idm.pt +bsv2_mj_hi: + value: 1.3 +bsv2_mj_lo: + value: 0.6 +bsv2_pick_thr: + value: 0.05 +bsv2_place_thr: + value: 0.05 +bsv2_prompts: + value: + - red block + - green block + - blue block +bsv2_vertical_sep_thr: + value: 0.03 +bsv2_x_align_thr: + value: 0.025 +checkpointing_steps: + value: 10 +ckpt_dir: + value: ckpts/Wan2.2-TI2V-5B +convert_model_dtype: + value: true +dataset_json: + value: data/rl_train/robotwin_stack_blocks_three.json +effective_batch_size: + value: 16 +epochs_per_round: + value: 100 +frame_num: + value: 121 +gradient_checkpointing: + value: true +hallucination_crop_top_ratio: + value: 0.6667 +lambda_gripper: + value: 1 +learning_rate: + value: 1e-05 +lora_alpha: + value: 64 +lora_rank: + value: 64 +lora_target_modules: + value: null +max_grad_norm: + value: 2 +max_per_condition: + value: 64 +max_samples: + value: -1 +neg_prompt: + value: 色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走 +num_ode_steps: + value: 20 +num_rounds: + value: 10 +num_train_timesteps: + value: 1000 +offload_model: + value: false +output_dir: + value: data/outputs/creflow_stack_blocks_three +pt_dir: + value: ckpts/vidar_ckpt/merged_vidar_lora.pt +reset_optimizer_per_round: + value: false +resume_from_lora_checkpoint: + value: null +reward_backend: + value: blocks_stack_v2 +reward_config: + value: blocks_stack_v2 +sample_guide_scale: + value: 5 +sample_shift: + value: 5 +seed: + value: 42 +size: + value: 640*736 +skip_reward_debug_video: + value: true +task: + value: ti2v-5B +use_8bit_adam: + value: true +vidar_root: + value: "" +w_bad: + value: 1 +wandb_project: + value: Corrective-Reflow +wandb_run_name: + value: creflow_stack_blocks_three +weight_decay: + value: 0.01 diff --git a/wandb/run-20260408_004547-x9lalqi6/files/output.log b/wandb/run-20260408_004547-x9lalqi6/files/output.log new file mode 100644 index 0000000000000000000000000000000000000000..6a008bb005c72987208d8af1b1f6ec82cc1afd04 --- /dev/null +++ b/wandb/run-20260408_004547-x9lalqi6/files/output.log @@ -0,0 +1,89 @@ +Building Wan2.2 TI2V model ... (DDP=True, world_size=8) +/usr/local/lib/python3.11/dist-packages/torch/distributed/distributed_c10d.py:4876: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning. + warnings.warn( # warn only once + Moving T5 encoder to GPU permanently (offload_model=false) ... +Building reward scorer ... + Loading IDM from ckpts/vidar_ckpt/idm.pt on cuda:0 ... + IDM loaded. + Loading SAM3 VideoPredictor (blocks_stack_v2) on cuda:0 ... +INFO 2026-04-08 00:47:04,252 107287 sam3_video_base.py: 125: setting max_num_objects=10000 and num_obj_for_compile=16 + SAM3 VideoPredictor (blocks_stack_v2) loaded. +Applying LoRA (single adapter) for CReflow ... + LoRA injected: rank=64, alpha=64, target_modules=['self_attn.q', 'self_attn.k', 'self_attn.v', 'self_attn.o', 'cross_attn.q', 'cross_attn.k', 'cross_attn.v', 'cross_attn.o'] + Trainable: 94.4 M / 5094.2 M total (1.85%) +Traceback (most recent call last): + File "/home/tiger/.local/lib/python3.11/site-packages/peft/peft_model.py", line 787, in __getattr__ + return super().__getattr__(name) # defer to nn.Module's logic + ^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1964, in __getattr__ + raise AttributeError( +AttributeError: 'PeftModel' object has no attribute 'enable_input_require_grads' + +During handling of the above exception, another exception occurred: + +Traceback (most recent call last): + File "/home/tiger/.local/lib/python3.11/site-packages/peft/tuners/lora/model.py", line 360, in __getattr__ + return super().__getattr__(name) # defer to nn.Module's logic + ^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1964, in __getattr__ + raise AttributeError( +AttributeError: 'LoraModel' object has no attribute 'enable_input_require_grads' + +During handling of the above exception, another exception occurred: + +Traceback (most recent call last): + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1035, in + main() + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 770, in main + dit.enable_input_require_grads() + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/home/tiger/.local/lib/python3.11/site-packages/peft/peft_model.py", line 791, in __getattr__ + return getattr(self.base_model, name) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/home/tiger/.local/lib/python3.11/site-packages/peft/tuners/lora/model.py", line 364, in __getattr__ + return getattr(self.model, name) + ^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/home/tiger/.local/lib/python3.11/site-packages/diffusers/models/modeling_utils.py", line 173, in __getattr__ + return super().__getattr__(name) + ^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1964, in __getattr__ + raise AttributeError( +AttributeError: 'WanModel' object has no attribute 'enable_input_require_grads' +[rank0]: Traceback (most recent call last): +[rank0]: File "/home/tiger/.local/lib/python3.11/site-packages/peft/peft_model.py", line 787, in __getattr__ +[rank0]: return super().__getattr__(name) # defer to nn.Module's logic +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1964, in __getattr__ +[rank0]: raise AttributeError( +[rank0]: AttributeError: 'PeftModel' object has no attribute 'enable_input_require_grads' + +[rank0]: During handling of the above exception, another exception occurred: + +[rank0]: Traceback (most recent call last): +[rank0]: File "/home/tiger/.local/lib/python3.11/site-packages/peft/tuners/lora/model.py", line 360, in __getattr__ +[rank0]: return super().__getattr__(name) # defer to nn.Module's logic +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1964, in __getattr__ +[rank0]: raise AttributeError( +[rank0]: AttributeError: 'LoraModel' object has no attribute 'enable_input_require_grads' + +[rank0]: During handling of the above exception, another exception occurred: + +[rank0]: Traceback (most recent call last): +[rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1035, in +[rank0]: main() +[rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 770, in main +[rank0]: dit.enable_input_require_grads() +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/home/tiger/.local/lib/python3.11/site-packages/peft/peft_model.py", line 791, in __getattr__ +[rank0]: return getattr(self.base_model, name) +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/home/tiger/.local/lib/python3.11/site-packages/peft/tuners/lora/model.py", line 364, in __getattr__ +[rank0]: return getattr(self.model, name) +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/home/tiger/.local/lib/python3.11/site-packages/diffusers/models/modeling_utils.py", line 173, in __getattr__ +[rank0]: return super().__getattr__(name) +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1964, in __getattr__ +[rank0]: raise AttributeError( +[rank0]: AttributeError: 'WanModel' object has no attribute 'enable_input_require_grads' diff --git a/wandb/run-20260408_004547-x9lalqi6/files/wandb-metadata.json b/wandb/run-20260408_004547-x9lalqi6/files/wandb-metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..afc23fc4332c400daf6dca42436a555831c1a5ed --- /dev/null +++ b/wandb/run-20260408_004547-x9lalqi6/files/wandb-metadata.json @@ -0,0 +1,144 @@ +{ + "os": "Linux-5.15.120.bsk.6-amd64-x86_64-with-glibc2.36", + "python": "3.11.2", + "startedAt": "2026-04-08T00:45:47.718687Z", + "args": [ + "--task", + "ti2v-5B", + "--size", + "640*736", + "--frame_num", + "121", + "--ckpt_dir", + "ckpts/Wan2.2-TI2V-5B", + "--pt_dir", + "ckpts/vidar_ckpt/merged_vidar_lora.pt", + "--dataset_json", + "data/rl_train/robotwin_stack_blocks_three.json", + "--output_dir", + "data/outputs/creflow_stack_blocks_three", + "--num_ode_steps", + "20", + "--sample_shift", + "5.0", + "--sample_guide_scale", + "5.0", + "--K", + "16", + "--num_rounds", + "10", + "--epochs_per_round", + "100", + "--max_per_condition", + "64", + "--effective_batch_size", + "16", + "--w_bad", + "1.0", + "--lambda_gripper", + "1.0", + "--seed", + "42", + "--reward_config", + "blocks_stack_v2", + "--convert_model_dtype", + "--offload_model", + "false", + "--learning_rate", + "1e-5", + "--weight_decay", + "0.01", + "--max_grad_norm", + "2.0", + "--checkpointing_steps", + "10", + "--lora_rank", + "64", + "--lora_alpha", + "64", + "--wandb_project", + "Corrective-Reflow", + "--wandb_run_name", + "creflow_stack_blocks_three" + ], + "program": "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", + "codePath": "fastvideo/train_creflow_wan_2_2_ti2v.py", + "git": { + "remote": "git@github.com:VincentNi0107/EmbodiedVideoRL.git", + "commit": "a8f5b170f0ef7d865af7e8f32c4aebb46efec2c1" + }, + "email": "1163051845@qq.com", + "root": "data/outputs/creflow_stack_blocks_three", + "host": "n124-104-180", + "username": "tiger", + "executable": "/usr/bin/python", + "codePathLocal": "fastvideo/train_creflow_wan_2_2_ti2v.py", + "cpu_count": 104, + "cpu_count_logical": 208, + "gpu": "NVIDIA H100 80GB HBM3", + "gpu_count": 8, + "disk": { + "/": { + "total": "1055740600320", + "used": "481625993216" + } + }, + "memory": { + "total": "1977660338176" + }, + "cpu": { + "count": 104, + "countLogical": 208 + }, + "gpu_nvidia": [ + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + } + ], + "cudaVersion": "12.9" +} \ No newline at end of file diff --git a/wandb/run-20260408_004547-x9lalqi6/files/wandb-summary.json b/wandb/run-20260408_004547-x9lalqi6/files/wandb-summary.json new file mode 100644 index 0000000000000000000000000000000000000000..def9e63764bfca2799cdd25200c3e9cecb7b07ce --- /dev/null +++ b/wandb/run-20260408_004547-x9lalqi6/files/wandb-summary.json @@ -0,0 +1 @@ +{"_wandb":{"runtime":80}} \ No newline at end of file diff --git a/wandb/run-20260408_004547-x9lalqi6/logs/debug-internal.log b/wandb/run-20260408_004547-x9lalqi6/logs/debug-internal.log new file mode 100644 index 0000000000000000000000000000000000000000..d4884f157abf85ff172f2bac4cbb69b156015f58 --- /dev/null +++ b/wandb/run-20260408_004547-x9lalqi6/logs/debug-internal.log @@ -0,0 +1,16 @@ +{"time":"2026-04-08T00:45:47.72832013Z","level":"INFO","msg":"using version","core version":"0.18.5"} +{"time":"2026-04-08T00:45:47.72833167Z","level":"INFO","msg":"created symlink","path":"data/outputs/creflow_stack_blocks_three/wandb/run-20260408_004547-x9lalqi6/logs/debug-core.log"} +{"time":"2026-04-08T00:45:47.945807814Z","level":"INFO","msg":"created new stream","id":"x9lalqi6"} +{"time":"2026-04-08T00:45:47.945941311Z","level":"INFO","msg":"stream: started","id":"x9lalqi6"} +{"time":"2026-04-08T00:45:47.945997568Z","level":"INFO","msg":"handler: started","stream_id":{"value":"x9lalqi6"}} +{"time":"2026-04-08T00:45:47.946052958Z","level":"INFO","msg":"sender: started","stream_id":"x9lalqi6"} +{"time":"2026-04-08T00:45:47.945978269Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"x9lalqi6"}} +{"time":"2026-04-08T00:45:48.359921482Z","level":"INFO","msg":"Starting system monitor"} +{"time":"2026-04-08T00:47:08.634395008Z","level":"INFO","msg":"stream: closing","id":"x9lalqi6"} +{"time":"2026-04-08T00:47:08.634488596Z","level":"INFO","msg":"Stopping system monitor"} +{"time":"2026-04-08T00:47:08.639585554Z","level":"INFO","msg":"Stopped system monitor"} +{"time":"2026-04-08T00:47:10.458241724Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"} +{"time":"2026-04-08T00:47:10.582858773Z","level":"INFO","msg":"handler: closed","stream_id":{"value":"x9lalqi6"}} +{"time":"2026-04-08T00:47:10.582906095Z","level":"INFO","msg":"writer: Close: closed","stream_id":{"value":"x9lalqi6"}} +{"time":"2026-04-08T00:47:10.582938514Z","level":"INFO","msg":"sender: closed","stream_id":"x9lalqi6"} +{"time":"2026-04-08T00:47:10.586475232Z","level":"INFO","msg":"stream: closed","id":"x9lalqi6"} diff --git a/wandb/run-20260408_004547-x9lalqi6/logs/debug.log b/wandb/run-20260408_004547-x9lalqi6/logs/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..f38c29949fc8e4071854a5d246098c873ae5714b --- /dev/null +++ b/wandb/run-20260408_004547-x9lalqi6/logs/debug.log @@ -0,0 +1,26 @@ +2026-04-08 00:45:47,684 INFO MainThread:107287 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5 +2026-04-08 00:45:47,684 INFO MainThread:107287 [wandb_setup.py:_flush():79] Configure stats pid to 107287 +2026-04-08 00:45:47,684 INFO MainThread:107287 [wandb_setup.py:_flush():79] Loading settings from /home/tiger/.config/wandb/settings +2026-04-08 00:45:47,684 INFO MainThread:107287 [wandb_setup.py:_flush():79] Loading settings from /mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/wandb/settings +2026-04-08 00:45:47,684 INFO MainThread:107287 [wandb_setup.py:_flush():79] Loading settings from environment variables: {'disabled': 'false', 'project': 'Corrective-Reflow', 'api_key': '***REDACTED***', 'mode': 'online'} +2026-04-08 00:45:47,684 INFO MainThread:107287 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': 'online', '_disable_service': None} +2026-04-08 00:45:47,684 INFO MainThread:107287 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'fastvideo/train_creflow_wan_2_2_ti2v.py', 'program_abspath': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py', 'program': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py'} +2026-04-08 00:45:47,684 INFO MainThread:107287 [wandb_setup.py:_flush():79] Applying login settings: {} +2026-04-08 00:45:47,686 INFO MainThread:107287 [wandb_init.py:_log_setup():534] Logging user logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_004547-x9lalqi6/logs/debug.log +2026-04-08 00:45:47,688 INFO MainThread:107287 [wandb_init.py:_log_setup():535] Logging internal logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_004547-x9lalqi6/logs/debug-internal.log +2026-04-08 00:45:47,688 INFO MainThread:107287 [wandb_init.py:init():621] calling init triggers +2026-04-08 00:45:47,688 INFO MainThread:107287 [wandb_init.py:init():628] wandb.init called with sweep_config: {} +config: {'vidar_root': '', 'task': 'ti2v-5B', 'size': '640*736', 'frame_num': 121, 'ckpt_dir': 'ckpts/Wan2.2-TI2V-5B', 'pt_dir': 'ckpts/vidar_ckpt/merged_vidar_lora.pt', 'convert_model_dtype': True, 'offload_model': False, 'dataset_json': 'data/rl_train/robotwin_stack_blocks_three.json', 'max_samples': -1, 'output_dir': 'data/outputs/creflow_stack_blocks_three', 'num_ode_steps': 20, 'sample_shift': 5.0, 'sample_guide_scale': 5.0, 'neg_prompt': '色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走', 'seed': 42, 'K': 16, 'num_rounds': 10, 'epochs_per_round': 100, 'effective_batch_size': 16, 'w_bad': 1.0, 'max_per_condition': 64, 'lambda_gripper': 1.0, 'reset_optimizer_per_round': False, 'num_train_timesteps': 1000, 'learning_rate': 1e-05, 'weight_decay': 0.01, 'max_grad_norm': 2.0, 'checkpointing_steps': 10, 'gradient_checkpointing': True, 'use_8bit_adam': True, 'lora_rank': 64, 'lora_alpha': 64, 'lora_target_modules': None, 'resume_from_lora_checkpoint': None, 'reward_config': 'blocks_stack_v2', 'reward_backend': 'blocks_stack_v2', 'skip_reward_debug_video': True, 'wandb_project': 'Corrective-Reflow', 'wandb_run_name': 'creflow_stack_blocks_three', 'hallucination_crop_top_ratio': 0.6667, 'bsv2_prompts': ['red block', 'green block', 'blue block'], 'bsv2_pick_thr': 0.05, 'bsv2_place_thr': 0.05, 'bsv2_vertical_sep_thr': 0.03, 'bsv2_x_align_thr': 0.025, 'bsv2_check_window_frac': 0.2, 'bsv2_dup_max_frames': 0, 'bsv2_expected_grip_changes': 6, 'bsv2_mj_lo': 0.6, 'bsv2_mj_hi': 1.3, 'bsv2_idm_ckpt_path': 'ckpts/vidar_ckpt/idm.pt'} +2026-04-08 00:45:47,688 INFO MainThread:107287 [wandb_init.py:init():671] starting backend +2026-04-08 00:45:47,688 INFO MainThread:107287 [wandb_init.py:init():675] sending inform_init request +2026-04-08 00:45:47,716 INFO MainThread:107287 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn +2026-04-08 00:45:47,716 INFO MainThread:107287 [wandb_init.py:init():688] backend started and connected +2026-04-08 00:45:47,773 INFO MainThread:107287 [wandb_init.py:init():783] updated telemetry +2026-04-08 00:45:47,947 INFO MainThread:107287 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout +2026-04-08 00:45:48,310 INFO MainThread:107287 [wandb_init.py:init():867] starting run threads in backend +2026-04-08 00:45:48,678 INFO MainThread:107287 [wandb_run.py:_console_start():2463] atexit reg +2026-04-08 00:45:48,678 INFO MainThread:107287 [wandb_run.py:_redirect():2311] redirect: wrap_raw +2026-04-08 00:45:48,678 INFO MainThread:107287 [wandb_run.py:_redirect():2376] Wrapping output streams. +2026-04-08 00:45:48,678 INFO MainThread:107287 [wandb_run.py:_redirect():2401] Redirects installed. +2026-04-08 00:45:48,683 INFO MainThread:107287 [wandb_init.py:init():911] run started, returning control to user process +2026-04-08 00:47:08,634 WARNING MsgRouterThr:107287 [router.py:message_loop():77] message_loop has been closed diff --git a/wandb/run-20260408_004931-ku6kqemc/files/config.yaml b/wandb/run-20260408_004931-ku6kqemc/files/config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b5875adecedbb41900bc7cd38d2cc6c4684240e --- /dev/null +++ b/wandb/run-20260408_004931-ku6kqemc/files/config.yaml @@ -0,0 +1,145 @@ +_wandb: + value: + cli_version: 0.18.5 + m: [] + python_version: 3.11.2 + t: + "1": + - 1 + - 11 + - 41 + - 49 + - 55 + - 71 + - 83 + - 105 + "2": + - 1 + - 11 + - 41 + - 49 + - 55 + - 63 + - 71 + - 83 + - 98 + - 105 + "3": + - 13 + - 16 + - 23 + - 55 + "4": 3.11.2 + "5": 0.18.5 + "6": 4.46.1 + "8": + - 5 + "12": 0.18.5 + "13": linux-x86_64 +K: + value: 16 +bsv2_check_window_frac: + value: 0.2 +bsv2_dup_max_frames: + value: 0 +bsv2_expected_grip_changes: + value: 6 +bsv2_idm_ckpt_path: + value: ckpts/vidar_ckpt/idm.pt +bsv2_mj_hi: + value: 1.3 +bsv2_mj_lo: + value: 0.6 +bsv2_pick_thr: + value: 0.05 +bsv2_place_thr: + value: 0.05 +bsv2_prompts: + value: + - red block + - green block + - blue block +bsv2_vertical_sep_thr: + value: 0.03 +bsv2_x_align_thr: + value: 0.025 +checkpointing_steps: + value: 10 +ckpt_dir: + value: ckpts/Wan2.2-TI2V-5B +convert_model_dtype: + value: true +dataset_json: + value: data/rl_train/robotwin_stack_blocks_three.json +effective_batch_size: + value: 16 +epochs_per_round: + value: 100 +frame_num: + value: 121 +gradient_checkpointing: + value: true +hallucination_crop_top_ratio: + value: 0.6667 +lambda_gripper: + value: 1 +learning_rate: + value: 1e-05 +lora_alpha: + value: 64 +lora_rank: + value: 64 +lora_target_modules: + value: null +max_grad_norm: + value: 2 +max_per_condition: + value: 64 +max_samples: + value: -1 +neg_prompt: + value: 色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走 +num_ode_steps: + value: 20 +num_rounds: + value: 10 +num_train_timesteps: + value: 1000 +offload_model: + value: false +output_dir: + value: data/outputs/creflow_stack_blocks_three +pt_dir: + value: ckpts/vidar_ckpt/merged_vidar_lora.pt +reset_optimizer_per_round: + value: false +resume_from_lora_checkpoint: + value: null +reward_backend: + value: blocks_stack_v2 +reward_config: + value: blocks_stack_v2 +sample_guide_scale: + value: 5 +sample_shift: + value: 5 +seed: + value: 42 +size: + value: 640*736 +skip_reward_debug_video: + value: true +task: + value: ti2v-5B +use_8bit_adam: + value: true +vidar_root: + value: "" +w_bad: + value: 1 +wandb_project: + value: Corrective-Reflow +wandb_run_name: + value: creflow_stack_blocks_three +weight_decay: + value: 0.01 diff --git a/wandb/run-20260408_004931-ku6kqemc/files/output.log b/wandb/run-20260408_004931-ku6kqemc/files/output.log new file mode 100644 index 0000000000000000000000000000000000000000..08e95e0f0e776a5e82b612b6e3a76cd0c4cf4293 --- /dev/null +++ b/wandb/run-20260408_004931-ku6kqemc/files/output.log @@ -0,0 +1,685 @@ +Building Wan2.2 TI2V model ... (DDP=True, world_size=8) +/usr/local/lib/python3.11/dist-packages/torch/distributed/distributed_c10d.py:4876: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning. + warnings.warn( # warn only once + Moving T5 encoder to GPU permanently (offload_model=false) ... +Building reward scorer ... + Loading IDM from ckpts/vidar_ckpt/idm.pt on cuda:0 ... + IDM loaded. + Loading SAM3 VideoPredictor (blocks_stack_v2) on cuda:0 ... +INFO 2026-04-08 00:50:46,165 108291 sam3_video_base.py: 125: setting max_num_objects=10000 and num_obj_for_compile=16 + SAM3 VideoPredictor (blocks_stack_v2) loaded. +Applying LoRA (single adapter) for CReflow ... + LoRA injected: rank=64, alpha=64, target_modules=['self_attn.q', 'self_attn.k', 'self_attn.v', 'self_attn.o', 'cross_attn.q', 'cross_attn.k', 'cross_attn.v', 'cross_attn.o'] + Trainable: 94.4 M / 5094.2 M total (1.85%) + Gradient checkpointing enabled on 30 DiT blocks +Trainable parameters: 94.4 M + Using 8-bit AdamW (bitsandbytes) +Dataset: 10 conditions +Encoding 10 conditions ... + [1/10] 6b20973f10ef... encoded + [2/10] cff33035b28a... encoded + [3/10] 7ebc8b336779... encoded + [4/10] 8077c679fb62... encoded + [5/10] 41855438a8e0... encoded + [6/10] a67b02f4e7ce... encoded + [7/10] 4568a9603f3b... encoded + [8/10] 1b67584bff87... encoded + [9/10] f4f8079f6637... encoded + [10/10] 1b70247c6412... encoded +All 10 conditions encoded. + +Starting Corrective Reflow | rounds=10 K=16 K_local=2 epochs/round=100 effective_bs=16 lr=1e-05 conditions=10 + +====================================================================== +ROUND 1/10 +====================================================================== +[Phase 1] ODE rollout ... + [CReflow rollout] condition 0/10: K=2, id=6b20973f10ef... + rollout: 2 videos in 56.5s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 77.15it/s] +propagate_in_video: 100%|██████████| 121/121 [00:13<00:00, 9.10it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.59it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.01it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.27it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.04it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 77.55it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.01it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.27it/s] +propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.86it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.24it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.54it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 1/10: K=2, id=cff33035b28a... + rollout: 2 videos in 57.1s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.62it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.20it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.25it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.35it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.43it/s] +propagate_in_video: 100%|██████████| 121/121 [00:11<00:00, 10.55it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.77it/s] +propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 12.06it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.19it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.33it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.24it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.20it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 2/10: K=2, id=7ebc8b336779... + rollout: 2 videos in 57.0s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.12it/s] +propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.78it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.92it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.17it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.49it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.90it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.84it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.68it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.59it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.49it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.49it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.21it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 3/10: K=2, id=8077c679fb62... + rollout: 2 videos in 57.5s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.59it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.09it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.19it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.13it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.23it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.08it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.03it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.33it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.08it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.59it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.56it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.20it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 4/10: K=2, id=41855438a8e0... + rollout: 2 videos in 57.6s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.67it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.03it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.84it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.26it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.09it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.33it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.97it/s] +propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.96it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.08it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.43it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.49it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.21it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 5/10: K=2, id=a67b02f4e7ce... + rollout: 2 videos in 57.6s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.45it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.21it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.39it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.50it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.89it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.22it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.10it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.58it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.92it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.58it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.27it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.24it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 6/10: K=2, id=4568a9603f3b... + rollout: 2 videos in 57.5s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.48it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.14it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.29it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.42it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.61it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.95it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.03it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.22it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.11it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.16it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.35it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.00it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 7/10: K=2, id=1b67584bff87... + rollout: 2 videos in 57.5s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.24it/s] +propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.66it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.27it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.37it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.42it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.54it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.09it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.94it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.10it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.71it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.78it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.89it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 8/10: K=2, id=f4f8079f6637... + rollout: 2 videos in 57.3s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.78it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.30it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.94it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.79it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.10it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.94it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.40it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.37it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 74.81it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.53it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.78it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.22it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 9/10: K=2, id=1b70247c6412... + rollout: 2 videos in 57.5s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.12it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.43it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.20it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.76it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.34it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.95it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.77it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.76it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.45it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.27it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 85.50it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.21it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] done: 20 samples, rewards = [1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 0.00, 0.00] + Rollout done: 20 samples, good=18, bad=2, avg_reward=0.900 + 1b67584bff87: mean=1.00, good=2/2 + 1b70247c6412: mean=0.00, good=0/2 + 41855438a8e0: mean=1.00, good=2/2 + 4568a9603f3b: mean=1.00, good=2/2 + 6b20973f10ef: mean=1.00, good=2/2 + 7ebc8b336779: mean=1.00, good=2/2 + 8077c679fb62: mean=1.00, good=2/2 + a67b02f4e7ce: mean=1.00, good=2/2 + cff33035b28a: mean=1.00, good=2/2 + f4f8079f6637: mean=1.00, good=2/2 +[Phase 2] Library update ... + [Library] rank=0 has 18 good samples + [Library] after update: SuccessLibrary(total=127, coverage=10/10, max_per_condition=64) +[Phase 3] Re-pairing ... + [Re-pair] train_set: 20 samples (anchor=18, correction=2, skipped=0) +[Phase 4] Training (100 epochs, 20 samples) ... +Traceback (most recent call last): + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1037, in + main() + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 937, in main + train_metrics = train_one_round( + ^^^^^^^^^^^^^^^^ + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 577, in train_one_round + loss.backward() + File "/usr/local/lib/python3.11/dist-packages/torch/_tensor.py", line 625, in backward + torch.autograd.backward( + File "/usr/local/lib/python3.11/dist-packages/torch/autograd/__init__.py", line 354, in backward + _engine_run_backward( + File "/usr/local/lib/python3.11/dist-packages/torch/autograd/graph.py", line 841, in _engine_run_backward + return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/utils/checkpoint.py", line 1158, in unpack_hook + frame.check_recomputed_tensors_match(gid) + File "/usr/local/lib/python3.11/dist-packages/torch/utils/checkpoint.py", line 911, in check_recomputed_tensors_match + raise CheckpointError( +torch.utils.checkpoint.CheckpointError: torch.utils.checkpoint: Recomputed values for the following tensors have different metadata than during the forward pass. +tensor at position 6: +saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 7: +saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 8: +saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 9: +saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 10: +saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 11: +saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 12: +saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 13: +saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 14: +saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 15: +saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 16: +saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 17: +saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 18: +saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 19: +saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 20: +saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 21: +saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 22: +saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 23: +saved metadata: {'shape': torch.Size([14260, 1, 64]), 'dtype': torch.complex128, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 24: +saved metadata: {'shape': torch.Size([14260, 1, 64]), 'dtype': torch.complex128, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 25: +saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 26: +saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 27: +saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 28: +saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 29: +saved metadata: {'shape': torch.Size([24, 14260]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([14260, 1, 64]), 'dtype': torch.complex128, 'device': device(type='cuda', index=0)} +tensor at position 30: +saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([14260, 1, 64]), 'dtype': torch.complex128, 'device': device(type='cuda', index=0)} +tensor at position 31: +saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 32: +saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int64, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 33: +saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 34: +saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 35: +saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([24, 14260]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 36: +saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)} +tensor at position 37: +saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)} +tensor at position 38: +saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([2]), 'dtype': torch.int64, 'device': device(type='cuda', index=0)} +tensor at position 39: +saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 40: +saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 41: +saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 42: +saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 43: +saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 44: +saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 45: +saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 46: +saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 47: +saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 48: +saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 49: +saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 50: +saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 51: +saved metadata: {'shape': torch.Size([512, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 52: +saved metadata: {'shape': torch.Size([512, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 53: +saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 54: +saved metadata: {'shape': torch.Size([24, 14260]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 55: +saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 56: +saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 57: +saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int64, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 58: +saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 59: +saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 60: +saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([512, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 61: +saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 62: +saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([512, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 63: +saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 512, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 64: +saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 512, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 65: +saved metadata: {'shape': torch.Size([3072, 14336]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 512, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 66: +saved metadata: {'shape': torch.Size([1, 14260, 14336]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([1, 512, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +tensor at position 67: +saved metadata: {'shape': torch.Size([14336, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +tensor at position 68: +saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +recomputed metadata: {'shape': torch.Size([512, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +. + +Tip: To see a more detailed error message, either pass `debug=True` to +`torch.utils.checkpoint.checkpoint(...)` or wrap the code block +with `with torch.utils.checkpoint.set_checkpoint_debug_enabled(True):` to +enable checkpoint‑debug mode globally. + +[rank0]: Traceback (most recent call last): +[rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1037, in +[rank0]: main() +[rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 937, in main +[rank0]: train_metrics = train_one_round( +[rank0]: ^^^^^^^^^^^^^^^^ +[rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 577, in train_one_round +[rank0]: loss.backward() +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/_tensor.py", line 625, in backward +[rank0]: torch.autograd.backward( +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/autograd/__init__.py", line 354, in backward +[rank0]: _engine_run_backward( +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/autograd/graph.py", line 841, in _engine_run_backward +[rank0]: return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/utils/checkpoint.py", line 1158, in unpack_hook +[rank0]: frame.check_recomputed_tensors_match(gid) +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/utils/checkpoint.py", line 911, in check_recomputed_tensors_match +[rank0]: raise CheckpointError( +[rank0]: torch.utils.checkpoint.CheckpointError: torch.utils.checkpoint: Recomputed values for the following tensors have different metadata than during the forward pass. +[rank0]: tensor at position 6: +[rank0]: saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 7: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 8: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 9: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 10: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 11: +[rank0]: saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 12: +[rank0]: saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 13: +[rank0]: saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 14: +[rank0]: saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 15: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 16: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 17: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 18: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 19: +[rank0]: saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 20: +[rank0]: saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 21: +[rank0]: saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 22: +[rank0]: saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 23: +[rank0]: saved metadata: {'shape': torch.Size([14260, 1, 64]), 'dtype': torch.complex128, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 24: +[rank0]: saved metadata: {'shape': torch.Size([14260, 1, 64]), 'dtype': torch.complex128, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 25: +[rank0]: saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 26: +[rank0]: saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 27: +[rank0]: saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 28: +[rank0]: saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 29: +[rank0]: saved metadata: {'shape': torch.Size([24, 14260]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([14260, 1, 64]), 'dtype': torch.complex128, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 30: +[rank0]: saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([14260, 1, 64]), 'dtype': torch.complex128, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 31: +[rank0]: saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 32: +[rank0]: saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int64, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 33: +[rank0]: saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 34: +[rank0]: saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 35: +[rank0]: saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([24, 14260]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 36: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 37: +[rank0]: saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 38: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([2]), 'dtype': torch.int64, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 39: +[rank0]: saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 40: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 41: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 42: +[rank0]: saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 43: +[rank0]: saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 44: +[rank0]: saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 45: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 46: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 47: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 48: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 49: +[rank0]: saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 50: +[rank0]: saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 51: +[rank0]: saved metadata: {'shape': torch.Size([512, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 52: +[rank0]: saved metadata: {'shape': torch.Size([512, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 53: +[rank0]: saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 54: +[rank0]: saved metadata: {'shape': torch.Size([24, 14260]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 55: +[rank0]: saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 56: +[rank0]: saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 57: +[rank0]: saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int64, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 58: +[rank0]: saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 59: +[rank0]: saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 60: +[rank0]: saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([512, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 61: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 62: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([512, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 63: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 512, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 64: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 512, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 65: +[rank0]: saved metadata: {'shape': torch.Size([3072, 14336]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 512, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 66: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 14336]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([1, 512, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 67: +[rank0]: saved metadata: {'shape': torch.Size([14336, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: tensor at position 68: +[rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)} +[rank0]: recomputed metadata: {'shape': torch.Size([512, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)} +[rank0]: . + +[rank0]: Tip: To see a more detailed error message, either pass `debug=True` to +[rank0]: `torch.utils.checkpoint.checkpoint(...)` or wrap the code block +[rank0]: with `with torch.utils.checkpoint.set_checkpoint_debug_enabled(True):` to +[rank0]: enable checkpoint‑debug mode globally. diff --git a/wandb/run-20260408_004931-ku6kqemc/files/wandb-metadata.json b/wandb/run-20260408_004931-ku6kqemc/files/wandb-metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..91dd73c6ae280a24b94d0e9a754c5e6accb2370f --- /dev/null +++ b/wandb/run-20260408_004931-ku6kqemc/files/wandb-metadata.json @@ -0,0 +1,144 @@ +{ + "os": "Linux-5.15.120.bsk.6-amd64-x86_64-with-glibc2.36", + "python": "3.11.2", + "startedAt": "2026-04-08T00:49:31.229785Z", + "args": [ + "--task", + "ti2v-5B", + "--size", + "640*736", + "--frame_num", + "121", + "--ckpt_dir", + "ckpts/Wan2.2-TI2V-5B", + "--pt_dir", + "ckpts/vidar_ckpt/merged_vidar_lora.pt", + "--dataset_json", + "data/rl_train/robotwin_stack_blocks_three.json", + "--output_dir", + "data/outputs/creflow_stack_blocks_three", + "--num_ode_steps", + "20", + "--sample_shift", + "5.0", + "--sample_guide_scale", + "5.0", + "--K", + "16", + "--num_rounds", + "10", + "--epochs_per_round", + "100", + "--max_per_condition", + "64", + "--effective_batch_size", + "16", + "--w_bad", + "1.0", + "--lambda_gripper", + "1.0", + "--seed", + "42", + "--reward_config", + "blocks_stack_v2", + "--convert_model_dtype", + "--offload_model", + "false", + "--learning_rate", + "1e-5", + "--weight_decay", + "0.01", + "--max_grad_norm", + "2.0", + "--checkpointing_steps", + "10", + "--lora_rank", + "64", + "--lora_alpha", + "64", + "--wandb_project", + "Corrective-Reflow", + "--wandb_run_name", + "creflow_stack_blocks_three" + ], + "program": "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", + "codePath": "fastvideo/train_creflow_wan_2_2_ti2v.py", + "git": { + "remote": "git@github.com:VincentNi0107/EmbodiedVideoRL.git", + "commit": "a8f5b170f0ef7d865af7e8f32c4aebb46efec2c1" + }, + "email": "1163051845@qq.com", + "root": "data/outputs/creflow_stack_blocks_three", + "host": "n124-104-180", + "username": "tiger", + "executable": "/usr/bin/python", + "codePathLocal": "fastvideo/train_creflow_wan_2_2_ti2v.py", + "cpu_count": 104, + "cpu_count_logical": 208, + "gpu": "NVIDIA H100 80GB HBM3", + "gpu_count": 8, + "disk": { + "/": { + "total": "1055740600320", + "used": "481703223296" + } + }, + "memory": { + "total": "1977660338176" + }, + "cpu": { + "count": 104, + "countLogical": 208 + }, + "gpu_nvidia": [ + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + } + ], + "cudaVersion": "12.9" +} \ No newline at end of file diff --git a/wandb/run-20260408_004931-ku6kqemc/files/wandb-summary.json b/wandb/run-20260408_004931-ku6kqemc/files/wandb-summary.json new file mode 100644 index 0000000000000000000000000000000000000000..b651c50a8a33fdffa2f0e172af147c76c55e484c --- /dev/null +++ b/wandb/run-20260408_004931-ku6kqemc/files/wandb-summary.json @@ -0,0 +1 @@ +{"_wandb":{"runtime":1584}} \ No newline at end of file diff --git a/wandb/run-20260408_004931-ku6kqemc/logs/debug-internal.log b/wandb/run-20260408_004931-ku6kqemc/logs/debug-internal.log new file mode 100644 index 0000000000000000000000000000000000000000..c09938bc490a999e7b1efb525ea8f3e26b2e3a6f --- /dev/null +++ b/wandb/run-20260408_004931-ku6kqemc/logs/debug-internal.log @@ -0,0 +1,19 @@ +{"time":"2026-04-08T00:49:31.238719294Z","level":"INFO","msg":"using version","core version":"0.18.5"} +{"time":"2026-04-08T00:49:31.23873382Z","level":"INFO","msg":"created symlink","path":"data/outputs/creflow_stack_blocks_three/wandb/run-20260408_004931-ku6kqemc/logs/debug-core.log"} +{"time":"2026-04-08T00:49:31.455165146Z","level":"INFO","msg":"created new stream","id":"ku6kqemc"} +{"time":"2026-04-08T00:49:31.455371786Z","level":"INFO","msg":"stream: started","id":"ku6kqemc"} +{"time":"2026-04-08T00:49:31.455479963Z","level":"INFO","msg":"sender: started","stream_id":"ku6kqemc"} +{"time":"2026-04-08T00:49:31.455431707Z","level":"INFO","msg":"handler: started","stream_id":{"value":"ku6kqemc"}} +{"time":"2026-04-08T00:49:31.455410767Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"ku6kqemc"}} +{"time":"2026-04-08T00:49:31.831542842Z","level":"INFO","msg":"Starting system monitor"} +{"time":"2026-04-08T01:00:32.624028988Z","level":"INFO","msg":"api: retrying error","error":"Post \"https://api.wandb.ai/graphql\": context deadline exceeded"} +{"time":"2026-04-08T01:01:05.122882271Z","level":"INFO","msg":"api: retrying error","error":"Post \"https://api.wandb.ai/graphql\": net/http: request canceled (Client.Timeout exceeded while awaiting headers)"} +{"time":"2026-04-08T01:09:32.404506925Z","level":"INFO","msg":"api: retrying error","error":"Post \"https://api.wandb.ai/files/vincentni/Corrective-Reflow/ku6kqemc/file_stream\": dial tcp 35.186.228.49:443: connect: connection timed out"} +{"time":"2026-04-08T01:15:56.227072593Z","level":"INFO","msg":"stream: closing","id":"ku6kqemc"} +{"time":"2026-04-08T01:15:56.227162615Z","level":"INFO","msg":"Stopping system monitor"} +{"time":"2026-04-08T01:15:56.232220483Z","level":"INFO","msg":"Stopped system monitor"} +{"time":"2026-04-08T01:15:56.791766357Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"} +{"time":"2026-04-08T01:15:56.926286921Z","level":"INFO","msg":"handler: closed","stream_id":{"value":"ku6kqemc"}} +{"time":"2026-04-08T01:15:56.926354182Z","level":"INFO","msg":"sender: closed","stream_id":"ku6kqemc"} +{"time":"2026-04-08T01:15:56.926329752Z","level":"INFO","msg":"writer: Close: closed","stream_id":{"value":"ku6kqemc"}} +{"time":"2026-04-08T01:15:56.930384735Z","level":"INFO","msg":"stream: closed","id":"ku6kqemc"} diff --git a/wandb/run-20260408_004931-ku6kqemc/logs/debug.log b/wandb/run-20260408_004931-ku6kqemc/logs/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..7b721561f5873ebb6a7f55d2fb1128be83012987 --- /dev/null +++ b/wandb/run-20260408_004931-ku6kqemc/logs/debug.log @@ -0,0 +1,26 @@ +2026-04-08 00:49:31,195 INFO MainThread:108291 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5 +2026-04-08 00:49:31,195 INFO MainThread:108291 [wandb_setup.py:_flush():79] Configure stats pid to 108291 +2026-04-08 00:49:31,195 INFO MainThread:108291 [wandb_setup.py:_flush():79] Loading settings from /home/tiger/.config/wandb/settings +2026-04-08 00:49:31,196 INFO MainThread:108291 [wandb_setup.py:_flush():79] Loading settings from /mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/wandb/settings +2026-04-08 00:49:31,196 INFO MainThread:108291 [wandb_setup.py:_flush():79] Loading settings from environment variables: {'disabled': 'false', 'project': 'Corrective-Reflow', 'api_key': '***REDACTED***', 'mode': 'online'} +2026-04-08 00:49:31,196 INFO MainThread:108291 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': 'online', '_disable_service': None} +2026-04-08 00:49:31,196 INFO MainThread:108291 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'fastvideo/train_creflow_wan_2_2_ti2v.py', 'program_abspath': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py', 'program': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py'} +2026-04-08 00:49:31,196 INFO MainThread:108291 [wandb_setup.py:_flush():79] Applying login settings: {} +2026-04-08 00:49:31,198 INFO MainThread:108291 [wandb_init.py:_log_setup():534] Logging user logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_004931-ku6kqemc/logs/debug.log +2026-04-08 00:49:31,199 INFO MainThread:108291 [wandb_init.py:_log_setup():535] Logging internal logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_004931-ku6kqemc/logs/debug-internal.log +2026-04-08 00:49:31,200 INFO MainThread:108291 [wandb_init.py:init():621] calling init triggers +2026-04-08 00:49:31,200 INFO MainThread:108291 [wandb_init.py:init():628] wandb.init called with sweep_config: {} +config: {'vidar_root': '', 'task': 'ti2v-5B', 'size': '640*736', 'frame_num': 121, 'ckpt_dir': 'ckpts/Wan2.2-TI2V-5B', 'pt_dir': 'ckpts/vidar_ckpt/merged_vidar_lora.pt', 'convert_model_dtype': True, 'offload_model': False, 'dataset_json': 'data/rl_train/robotwin_stack_blocks_three.json', 'max_samples': -1, 'output_dir': 'data/outputs/creflow_stack_blocks_three', 'num_ode_steps': 20, 'sample_shift': 5.0, 'sample_guide_scale': 5.0, 'neg_prompt': '色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走', 'seed': 42, 'K': 16, 'num_rounds': 10, 'epochs_per_round': 100, 'effective_batch_size': 16, 'w_bad': 1.0, 'max_per_condition': 64, 'lambda_gripper': 1.0, 'reset_optimizer_per_round': False, 'num_train_timesteps': 1000, 'learning_rate': 1e-05, 'weight_decay': 0.01, 'max_grad_norm': 2.0, 'checkpointing_steps': 10, 'gradient_checkpointing': True, 'use_8bit_adam': True, 'lora_rank': 64, 'lora_alpha': 64, 'lora_target_modules': None, 'resume_from_lora_checkpoint': None, 'reward_config': 'blocks_stack_v2', 'reward_backend': 'blocks_stack_v2', 'skip_reward_debug_video': True, 'wandb_project': 'Corrective-Reflow', 'wandb_run_name': 'creflow_stack_blocks_three', 'hallucination_crop_top_ratio': 0.6667, 'bsv2_prompts': ['red block', 'green block', 'blue block'], 'bsv2_pick_thr': 0.05, 'bsv2_place_thr': 0.05, 'bsv2_vertical_sep_thr': 0.03, 'bsv2_x_align_thr': 0.025, 'bsv2_check_window_frac': 0.2, 'bsv2_dup_max_frames': 0, 'bsv2_expected_grip_changes': 6, 'bsv2_mj_lo': 0.6, 'bsv2_mj_hi': 1.3, 'bsv2_idm_ckpt_path': 'ckpts/vidar_ckpt/idm.pt'} +2026-04-08 00:49:31,200 INFO MainThread:108291 [wandb_init.py:init():671] starting backend +2026-04-08 00:49:31,200 INFO MainThread:108291 [wandb_init.py:init():675] sending inform_init request +2026-04-08 00:49:31,227 INFO MainThread:108291 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn +2026-04-08 00:49:31,227 INFO MainThread:108291 [wandb_init.py:init():688] backend started and connected +2026-04-08 00:49:31,285 INFO MainThread:108291 [wandb_init.py:init():783] updated telemetry +2026-04-08 00:49:31,449 INFO MainThread:108291 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout +2026-04-08 00:49:31,780 INFO MainThread:108291 [wandb_init.py:init():867] starting run threads in backend +2026-04-08 00:49:32,146 INFO MainThread:108291 [wandb_run.py:_console_start():2463] atexit reg +2026-04-08 00:49:32,146 INFO MainThread:108291 [wandb_run.py:_redirect():2311] redirect: wrap_raw +2026-04-08 00:49:32,146 INFO MainThread:108291 [wandb_run.py:_redirect():2376] Wrapping output streams. +2026-04-08 00:49:32,146 INFO MainThread:108291 [wandb_run.py:_redirect():2401] Redirects installed. +2026-04-08 00:49:32,151 INFO MainThread:108291 [wandb_init.py:init():911] run started, returning control to user process +2026-04-08 01:15:56,227 WARNING MsgRouterThr:108291 [router.py:message_loop():77] message_loop has been closed diff --git a/wandb/run-20260408_012032-idkfwxb0/files/config.yaml b/wandb/run-20260408_012032-idkfwxb0/files/config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b5875adecedbb41900bc7cd38d2cc6c4684240e --- /dev/null +++ b/wandb/run-20260408_012032-idkfwxb0/files/config.yaml @@ -0,0 +1,145 @@ +_wandb: + value: + cli_version: 0.18.5 + m: [] + python_version: 3.11.2 + t: + "1": + - 1 + - 11 + - 41 + - 49 + - 55 + - 71 + - 83 + - 105 + "2": + - 1 + - 11 + - 41 + - 49 + - 55 + - 63 + - 71 + - 83 + - 98 + - 105 + "3": + - 13 + - 16 + - 23 + - 55 + "4": 3.11.2 + "5": 0.18.5 + "6": 4.46.1 + "8": + - 5 + "12": 0.18.5 + "13": linux-x86_64 +K: + value: 16 +bsv2_check_window_frac: + value: 0.2 +bsv2_dup_max_frames: + value: 0 +bsv2_expected_grip_changes: + value: 6 +bsv2_idm_ckpt_path: + value: ckpts/vidar_ckpt/idm.pt +bsv2_mj_hi: + value: 1.3 +bsv2_mj_lo: + value: 0.6 +bsv2_pick_thr: + value: 0.05 +bsv2_place_thr: + value: 0.05 +bsv2_prompts: + value: + - red block + - green block + - blue block +bsv2_vertical_sep_thr: + value: 0.03 +bsv2_x_align_thr: + value: 0.025 +checkpointing_steps: + value: 10 +ckpt_dir: + value: ckpts/Wan2.2-TI2V-5B +convert_model_dtype: + value: true +dataset_json: + value: data/rl_train/robotwin_stack_blocks_three.json +effective_batch_size: + value: 16 +epochs_per_round: + value: 100 +frame_num: + value: 121 +gradient_checkpointing: + value: true +hallucination_crop_top_ratio: + value: 0.6667 +lambda_gripper: + value: 1 +learning_rate: + value: 1e-05 +lora_alpha: + value: 64 +lora_rank: + value: 64 +lora_target_modules: + value: null +max_grad_norm: + value: 2 +max_per_condition: + value: 64 +max_samples: + value: -1 +neg_prompt: + value: 色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走 +num_ode_steps: + value: 20 +num_rounds: + value: 10 +num_train_timesteps: + value: 1000 +offload_model: + value: false +output_dir: + value: data/outputs/creflow_stack_blocks_three +pt_dir: + value: ckpts/vidar_ckpt/merged_vidar_lora.pt +reset_optimizer_per_round: + value: false +resume_from_lora_checkpoint: + value: null +reward_backend: + value: blocks_stack_v2 +reward_config: + value: blocks_stack_v2 +sample_guide_scale: + value: 5 +sample_shift: + value: 5 +seed: + value: 42 +size: + value: 640*736 +skip_reward_debug_video: + value: true +task: + value: ti2v-5B +use_8bit_adam: + value: true +vidar_root: + value: "" +w_bad: + value: 1 +wandb_project: + value: Corrective-Reflow +wandb_run_name: + value: creflow_stack_blocks_three +weight_decay: + value: 0.01 diff --git a/wandb/run-20260408_012032-idkfwxb0/files/output.log b/wandb/run-20260408_012032-idkfwxb0/files/output.log new file mode 100644 index 0000000000000000000000000000000000000000..197767ecfcb77865db4321f7d0d20f1f8639eb02 --- /dev/null +++ b/wandb/run-20260408_012032-idkfwxb0/files/output.log @@ -0,0 +1,133 @@ +Building Wan2.2 TI2V model ... (DDP=True, world_size=8) +/usr/local/lib/python3.11/dist-packages/torch/distributed/distributed_c10d.py:4876: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning. + warnings.warn( # warn only once + Moving T5 encoder to GPU permanently (offload_model=false) ... +Building reward scorer ... + Loading IDM from ckpts/vidar_ckpt/idm.pt on cuda:0 ... + IDM loaded. + Loading SAM3 VideoPredictor (blocks_stack_v2) on cuda:0 ... +INFO 2026-04-08 01:21:50,257 195712 sam3_video_base.py: 125: setting max_num_objects=10000 and num_obj_for_compile=16 + SAM3 VideoPredictor (blocks_stack_v2) loaded. +Applying LoRA (single adapter) for CReflow ... + LoRA injected: rank=64, alpha=64, target_modules=['self_attn.q', 'self_attn.k', 'self_attn.v', 'self_attn.o', 'cross_attn.q', 'cross_attn.k', 'cross_attn.v', 'cross_attn.o'] + Trainable: 94.4 M / 5094.2 M total (1.85%) + Gradient checkpointing enabled on 30 DiT blocks +Trainable parameters: 94.4 M + Using 8-bit AdamW (bitsandbytes) +Dataset: 10 conditions +Encoding 10 conditions ... + [1/10] 6b20973f10ef... encoded + [2/10] cff33035b28a... encoded + [3/10] 7ebc8b336779... encoded + [4/10] 8077c679fb62... encoded + [5/10] 41855438a8e0... encoded + [6/10] a67b02f4e7ce... encoded + [7/10] 4568a9603f3b... encoded + [8/10] 1b67584bff87... encoded + [9/10] f4f8079f6637... encoded + [10/10] 1b70247c6412... encoded +All 10 conditions encoded. + +Starting Corrective Reflow | rounds=10 K=16 K_local=2 epochs/round=100 effective_bs=16 lr=1e-05 conditions=10 + +====================================================================== +ROUND 1/10 +====================================================================== +[Phase 1] ODE rollout ... + [CReflow rollout] condition 0/10: K=2, id=6b20973f10ef... +Traceback (most recent call last): + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1038, in + main() + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 870, in main + rollouts = creflow_rollout( + ^^^^^^^^^^^^^^^^ + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/creflow/rollout.py", line 267, in creflow_rollout + x0_preds = _ode_rollout_batch( + ^^^^^^^^^^^^^^^^^^^ + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/creflow/rollout.py", line 139, in _ode_rollout_batch + outputs = dit(x_list, t=ts_batch, context=context_list, seq_len=seq_len) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl + return self._call_impl(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl + return forward_call(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/home/tiger/.local/lib/python3.11/site-packages/peft/peft_model.py", line 812, in forward + return self.get_base_model()(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl + return self._call_impl(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl + return forward_call(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/models/wan/modules/model.py", line 498, in forward + x = block(x, **kwargs) + ^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl + return self._call_impl(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl + return forward_call(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 130, in _ckpt_fwd + return torch_ckpt.checkpoint(orig_fwd, *args, use_reentrant=True, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/_compile.py", line 53, in inner + return disable_fn(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/_dynamo/eval_frame.py", line 1044, in _fn + return fn(*args, **kwargs) + ^^^^^^^^^^^^^^^^^^^ + File "/usr/local/lib/python3.11/dist-packages/torch/utils/checkpoint.py", line 486, in checkpoint + raise ValueError( +ValueError: Unexpected keyword arguments: e,seq_lens,grid_sizes,freqs,context,context_lens +[rank0]: Traceback (most recent call last): +[rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1038, in +[rank0]: main() +[rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 870, in main +[rank0]: rollouts = creflow_rollout( +[rank0]: ^^^^^^^^^^^^^^^^ +[rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/creflow/rollout.py", line 267, in creflow_rollout +[rank0]: x0_preds = _ode_rollout_batch( +[rank0]: ^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/creflow/rollout.py", line 139, in _ode_rollout_batch +[rank0]: outputs = dit(x_list, t=ts_batch, context=context_list, seq_len=seq_len) +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl +[rank0]: return self._call_impl(*args, **kwargs) +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl +[rank0]: return forward_call(*args, **kwargs) +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/home/tiger/.local/lib/python3.11/site-packages/peft/peft_model.py", line 812, in forward +[rank0]: return self.get_base_model()(*args, **kwargs) +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl +[rank0]: return self._call_impl(*args, **kwargs) +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl +[rank0]: return forward_call(*args, **kwargs) +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/models/wan/modules/model.py", line 498, in forward +[rank0]: x = block(x, **kwargs) +[rank0]: ^^^^^^^^^^^^^^^^^^ +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl +[rank0]: return self._call_impl(*args, **kwargs) +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl +[rank0]: return forward_call(*args, **kwargs) +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 130, in _ckpt_fwd +[rank0]: return torch_ckpt.checkpoint(orig_fwd, *args, use_reentrant=True, **kwargs) +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/_compile.py", line 53, in inner +[rank0]: return disable_fn(*args, **kwargs) +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/_dynamo/eval_frame.py", line 1044, in _fn +[rank0]: return fn(*args, **kwargs) +[rank0]: ^^^^^^^^^^^^^^^^^^^ +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/utils/checkpoint.py", line 486, in checkpoint +[rank0]: raise ValueError( +[rank0]: ValueError: Unexpected keyword arguments: e,seq_lens,grid_sizes,freqs,context,context_lens diff --git a/wandb/run-20260408_012032-idkfwxb0/files/wandb-metadata.json b/wandb/run-20260408_012032-idkfwxb0/files/wandb-metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..934e851a9ccef5d4d89d49423c56f1aa7680f2dd --- /dev/null +++ b/wandb/run-20260408_012032-idkfwxb0/files/wandb-metadata.json @@ -0,0 +1,144 @@ +{ + "os": "Linux-5.15.120.bsk.6-amd64-x86_64-with-glibc2.36", + "python": "3.11.2", + "startedAt": "2026-04-08T01:20:32.358609Z", + "args": [ + "--task", + "ti2v-5B", + "--size", + "640*736", + "--frame_num", + "121", + "--ckpt_dir", + "ckpts/Wan2.2-TI2V-5B", + "--pt_dir", + "ckpts/vidar_ckpt/merged_vidar_lora.pt", + "--dataset_json", + "data/rl_train/robotwin_stack_blocks_three.json", + "--output_dir", + "data/outputs/creflow_stack_blocks_three", + "--num_ode_steps", + "20", + "--sample_shift", + "5.0", + "--sample_guide_scale", + "5.0", + "--K", + "16", + "--num_rounds", + "10", + "--epochs_per_round", + "100", + "--max_per_condition", + "64", + "--effective_batch_size", + "16", + "--w_bad", + "1.0", + "--lambda_gripper", + "1.0", + "--seed", + "42", + "--reward_config", + "blocks_stack_v2", + "--convert_model_dtype", + "--offload_model", + "false", + "--learning_rate", + "1e-5", + "--weight_decay", + "0.01", + "--max_grad_norm", + "2.0", + "--checkpointing_steps", + "10", + "--lora_rank", + "64", + "--lora_alpha", + "64", + "--wandb_project", + "Corrective-Reflow", + "--wandb_run_name", + "creflow_stack_blocks_three" + ], + "program": "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", + "codePath": "fastvideo/train_creflow_wan_2_2_ti2v.py", + "git": { + "remote": "git@github.com:VincentNi0107/EmbodiedVideoRL.git", + "commit": "a8f5b170f0ef7d865af7e8f32c4aebb46efec2c1" + }, + "email": "1163051845@qq.com", + "root": "data/outputs/creflow_stack_blocks_three", + "host": "n124-104-180", + "username": "tiger", + "executable": "/usr/bin/python", + "codePathLocal": "fastvideo/train_creflow_wan_2_2_ti2v.py", + "cpu_count": 104, + "cpu_count_logical": 208, + "gpu": "NVIDIA H100 80GB HBM3", + "gpu_count": 8, + "disk": { + "/": { + "total": "1055740600320", + "used": "482171138048" + } + }, + "memory": { + "total": "1977660338176" + }, + "cpu": { + "count": 104, + "countLogical": 208 + }, + "gpu_nvidia": [ + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + } + ], + "cudaVersion": "12.9" +} \ No newline at end of file diff --git a/wandb/run-20260408_012032-idkfwxb0/files/wandb-summary.json b/wandb/run-20260408_012032-idkfwxb0/files/wandb-summary.json new file mode 100644 index 0000000000000000000000000000000000000000..11cf6f33ec04c95a2bfa77fc360aaacd4b5b9a6f --- /dev/null +++ b/wandb/run-20260408_012032-idkfwxb0/files/wandb-summary.json @@ -0,0 +1 @@ +{"_wandb":{"runtime":84}} \ No newline at end of file diff --git a/wandb/run-20260408_012032-idkfwxb0/logs/debug-internal.log b/wandb/run-20260408_012032-idkfwxb0/logs/debug-internal.log new file mode 100644 index 0000000000000000000000000000000000000000..42c16d8687662f9e6e74cbba17f39f679d731509 --- /dev/null +++ b/wandb/run-20260408_012032-idkfwxb0/logs/debug-internal.log @@ -0,0 +1,16 @@ +{"time":"2026-04-08T01:20:32.367150627Z","level":"INFO","msg":"using version","core version":"0.18.5"} +{"time":"2026-04-08T01:20:32.367162484Z","level":"INFO","msg":"created symlink","path":"data/outputs/creflow_stack_blocks_three/wandb/run-20260408_012032-idkfwxb0/logs/debug-core.log"} +{"time":"2026-04-08T01:20:32.581603896Z","level":"INFO","msg":"created new stream","id":"idkfwxb0"} +{"time":"2026-04-08T01:20:32.581769665Z","level":"INFO","msg":"stream: started","id":"idkfwxb0"} +{"time":"2026-04-08T01:20:32.5818252Z","level":"INFO","msg":"sender: started","stream_id":"idkfwxb0"} +{"time":"2026-04-08T01:20:32.581798295Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"idkfwxb0"}} +{"time":"2026-04-08T01:20:32.581839505Z","level":"INFO","msg":"handler: started","stream_id":{"value":"idkfwxb0"}} +{"time":"2026-04-08T01:20:32.990097512Z","level":"INFO","msg":"Starting system monitor"} +{"time":"2026-04-08T01:21:57.340247846Z","level":"INFO","msg":"stream: closing","id":"idkfwxb0"} +{"time":"2026-04-08T01:21:57.340446284Z","level":"INFO","msg":"Stopping system monitor"} +{"time":"2026-04-08T01:21:57.346621494Z","level":"INFO","msg":"Stopped system monitor"} +{"time":"2026-04-08T01:21:58.004298171Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"} +{"time":"2026-04-08T01:21:58.157765962Z","level":"INFO","msg":"handler: closed","stream_id":{"value":"idkfwxb0"}} +{"time":"2026-04-08T01:21:58.157800232Z","level":"INFO","msg":"writer: Close: closed","stream_id":{"value":"idkfwxb0"}} +{"time":"2026-04-08T01:21:58.157836905Z","level":"INFO","msg":"sender: closed","stream_id":"idkfwxb0"} +{"time":"2026-04-08T01:21:58.16571198Z","level":"INFO","msg":"stream: closed","id":"idkfwxb0"} diff --git a/wandb/run-20260408_012032-idkfwxb0/logs/debug.log b/wandb/run-20260408_012032-idkfwxb0/logs/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..0e385544efc1dafb71e342d0818f1462baca7de6 --- /dev/null +++ b/wandb/run-20260408_012032-idkfwxb0/logs/debug.log @@ -0,0 +1,26 @@ +2026-04-08 01:20:32,325 INFO MainThread:195712 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5 +2026-04-08 01:20:32,325 INFO MainThread:195712 [wandb_setup.py:_flush():79] Configure stats pid to 195712 +2026-04-08 01:20:32,325 INFO MainThread:195712 [wandb_setup.py:_flush():79] Loading settings from /home/tiger/.config/wandb/settings +2026-04-08 01:20:32,325 INFO MainThread:195712 [wandb_setup.py:_flush():79] Loading settings from /mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/wandb/settings +2026-04-08 01:20:32,325 INFO MainThread:195712 [wandb_setup.py:_flush():79] Loading settings from environment variables: {'disabled': 'false', 'project': 'Corrective-Reflow', 'api_key': '***REDACTED***', 'mode': 'online'} +2026-04-08 01:20:32,325 INFO MainThread:195712 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': 'online', '_disable_service': None} +2026-04-08 01:20:32,325 INFO MainThread:195712 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'fastvideo/train_creflow_wan_2_2_ti2v.py', 'program_abspath': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py', 'program': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py'} +2026-04-08 01:20:32,325 INFO MainThread:195712 [wandb_setup.py:_flush():79] Applying login settings: {} +2026-04-08 01:20:32,327 INFO MainThread:195712 [wandb_init.py:_log_setup():534] Logging user logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_012032-idkfwxb0/logs/debug.log +2026-04-08 01:20:32,329 INFO MainThread:195712 [wandb_init.py:_log_setup():535] Logging internal logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_012032-idkfwxb0/logs/debug-internal.log +2026-04-08 01:20:32,329 INFO MainThread:195712 [wandb_init.py:init():621] calling init triggers +2026-04-08 01:20:32,329 INFO MainThread:195712 [wandb_init.py:init():628] wandb.init called with sweep_config: {} +config: {'vidar_root': '', 'task': 'ti2v-5B', 'size': '640*736', 'frame_num': 121, 'ckpt_dir': 'ckpts/Wan2.2-TI2V-5B', 'pt_dir': 'ckpts/vidar_ckpt/merged_vidar_lora.pt', 'convert_model_dtype': True, 'offload_model': False, 'dataset_json': 'data/rl_train/robotwin_stack_blocks_three.json', 'max_samples': -1, 'output_dir': 'data/outputs/creflow_stack_blocks_three', 'num_ode_steps': 20, 'sample_shift': 5.0, 'sample_guide_scale': 5.0, 'neg_prompt': '色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走', 'seed': 42, 'K': 16, 'num_rounds': 10, 'epochs_per_round': 100, 'effective_batch_size': 16, 'w_bad': 1.0, 'max_per_condition': 64, 'lambda_gripper': 1.0, 'reset_optimizer_per_round': False, 'num_train_timesteps': 1000, 'learning_rate': 1e-05, 'weight_decay': 0.01, 'max_grad_norm': 2.0, 'checkpointing_steps': 10, 'gradient_checkpointing': True, 'use_8bit_adam': True, 'lora_rank': 64, 'lora_alpha': 64, 'lora_target_modules': None, 'resume_from_lora_checkpoint': None, 'reward_config': 'blocks_stack_v2', 'reward_backend': 'blocks_stack_v2', 'skip_reward_debug_video': True, 'wandb_project': 'Corrective-Reflow', 'wandb_run_name': 'creflow_stack_blocks_three', 'hallucination_crop_top_ratio': 0.6667, 'bsv2_prompts': ['red block', 'green block', 'blue block'], 'bsv2_pick_thr': 0.05, 'bsv2_place_thr': 0.05, 'bsv2_vertical_sep_thr': 0.03, 'bsv2_x_align_thr': 0.025, 'bsv2_check_window_frac': 0.2, 'bsv2_dup_max_frames': 0, 'bsv2_expected_grip_changes': 6, 'bsv2_mj_lo': 0.6, 'bsv2_mj_hi': 1.3, 'bsv2_idm_ckpt_path': 'ckpts/vidar_ckpt/idm.pt'} +2026-04-08 01:20:32,329 INFO MainThread:195712 [wandb_init.py:init():671] starting backend +2026-04-08 01:20:32,329 INFO MainThread:195712 [wandb_init.py:init():675] sending inform_init request +2026-04-08 01:20:32,356 INFO MainThread:195712 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn +2026-04-08 01:20:32,356 INFO MainThread:195712 [wandb_init.py:init():688] backend started and connected +2026-04-08 01:20:32,412 INFO MainThread:195712 [wandb_init.py:init():783] updated telemetry +2026-04-08 01:20:32,575 INFO MainThread:195712 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout +2026-04-08 01:20:32,942 INFO MainThread:195712 [wandb_init.py:init():867] starting run threads in backend +2026-04-08 01:20:33,301 INFO MainThread:195712 [wandb_run.py:_console_start():2463] atexit reg +2026-04-08 01:20:33,301 INFO MainThread:195712 [wandb_run.py:_redirect():2311] redirect: wrap_raw +2026-04-08 01:20:33,301 INFO MainThread:195712 [wandb_run.py:_redirect():2376] Wrapping output streams. +2026-04-08 01:20:33,301 INFO MainThread:195712 [wandb_run.py:_redirect():2401] Redirects installed. +2026-04-08 01:20:33,306 INFO MainThread:195712 [wandb_init.py:init():911] run started, returning control to user process +2026-04-08 01:21:57,340 WARNING MsgRouterThr:195712 [router.py:message_loop():77] message_loop has been closed diff --git a/wandb/run-20260408_014250-d3p1n5w7/files/config.yaml b/wandb/run-20260408_014250-d3p1n5w7/files/config.yaml new file mode 100644 index 0000000000000000000000000000000000000000..7b5875adecedbb41900bc7cd38d2cc6c4684240e --- /dev/null +++ b/wandb/run-20260408_014250-d3p1n5w7/files/config.yaml @@ -0,0 +1,145 @@ +_wandb: + value: + cli_version: 0.18.5 + m: [] + python_version: 3.11.2 + t: + "1": + - 1 + - 11 + - 41 + - 49 + - 55 + - 71 + - 83 + - 105 + "2": + - 1 + - 11 + - 41 + - 49 + - 55 + - 63 + - 71 + - 83 + - 98 + - 105 + "3": + - 13 + - 16 + - 23 + - 55 + "4": 3.11.2 + "5": 0.18.5 + "6": 4.46.1 + "8": + - 5 + "12": 0.18.5 + "13": linux-x86_64 +K: + value: 16 +bsv2_check_window_frac: + value: 0.2 +bsv2_dup_max_frames: + value: 0 +bsv2_expected_grip_changes: + value: 6 +bsv2_idm_ckpt_path: + value: ckpts/vidar_ckpt/idm.pt +bsv2_mj_hi: + value: 1.3 +bsv2_mj_lo: + value: 0.6 +bsv2_pick_thr: + value: 0.05 +bsv2_place_thr: + value: 0.05 +bsv2_prompts: + value: + - red block + - green block + - blue block +bsv2_vertical_sep_thr: + value: 0.03 +bsv2_x_align_thr: + value: 0.025 +checkpointing_steps: + value: 10 +ckpt_dir: + value: ckpts/Wan2.2-TI2V-5B +convert_model_dtype: + value: true +dataset_json: + value: data/rl_train/robotwin_stack_blocks_three.json +effective_batch_size: + value: 16 +epochs_per_round: + value: 100 +frame_num: + value: 121 +gradient_checkpointing: + value: true +hallucination_crop_top_ratio: + value: 0.6667 +lambda_gripper: + value: 1 +learning_rate: + value: 1e-05 +lora_alpha: + value: 64 +lora_rank: + value: 64 +lora_target_modules: + value: null +max_grad_norm: + value: 2 +max_per_condition: + value: 64 +max_samples: + value: -1 +neg_prompt: + value: 色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走 +num_ode_steps: + value: 20 +num_rounds: + value: 10 +num_train_timesteps: + value: 1000 +offload_model: + value: false +output_dir: + value: data/outputs/creflow_stack_blocks_three +pt_dir: + value: ckpts/vidar_ckpt/merged_vidar_lora.pt +reset_optimizer_per_round: + value: false +resume_from_lora_checkpoint: + value: null +reward_backend: + value: blocks_stack_v2 +reward_config: + value: blocks_stack_v2 +sample_guide_scale: + value: 5 +sample_shift: + value: 5 +seed: + value: 42 +size: + value: 640*736 +skip_reward_debug_video: + value: true +task: + value: ti2v-5B +use_8bit_adam: + value: true +vidar_root: + value: "" +w_bad: + value: 1 +wandb_project: + value: Corrective-Reflow +wandb_run_name: + value: creflow_stack_blocks_three +weight_decay: + value: 0.01 diff --git a/wandb/run-20260408_014250-d3p1n5w7/files/output.log b/wandb/run-20260408_014250-d3p1n5w7/files/output.log new file mode 100644 index 0000000000000000000000000000000000000000..e70104493b5c51c26bdb4f13b35af3b26622b2b6 --- /dev/null +++ b/wandb/run-20260408_014250-d3p1n5w7/files/output.log @@ -0,0 +1,287 @@ +Building Wan2.2 TI2V model ... (DDP=True, world_size=8) +/usr/local/lib/python3.11/dist-packages/torch/distributed/distributed_c10d.py:4876: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning. + warnings.warn( # warn only once + Moving T5 encoder to GPU permanently (offload_model=false) ... +Building reward scorer ... + Loading IDM from ckpts/vidar_ckpt/idm.pt on cuda:0 ... + IDM loaded. + Loading SAM3 VideoPredictor (blocks_stack_v2) on cuda:0 ... +INFO 2026-04-08 01:44:06,994 198734 sam3_video_base.py: 125: setting max_num_objects=10000 and num_obj_for_compile=16 + SAM3 VideoPredictor (blocks_stack_v2) loaded. +Applying LoRA (single adapter) for CReflow ... + LoRA injected: rank=64, alpha=64, target_modules=['self_attn.q', 'self_attn.k', 'self_attn.v', 'self_attn.o', 'cross_attn.q', 'cross_attn.k', 'cross_attn.v', 'cross_attn.o'] + Trainable: 94.4 M / 5094.2 M total (1.85%) + Gradient checkpointing enabled on 30 DiT blocks +Trainable parameters: 94.4 M + Using 8-bit AdamW (bitsandbytes) +Dataset: 10 conditions +Encoding 10 conditions ... + [1/10] 6b20973f10ef... encoded + [2/10] cff33035b28a... encoded + [3/10] 7ebc8b336779... encoded + [4/10] 8077c679fb62... encoded + [5/10] 41855438a8e0... encoded + [6/10] a67b02f4e7ce... encoded + [7/10] 4568a9603f3b... encoded + [8/10] 1b67584bff87... encoded + [9/10] f4f8079f6637... encoded + [10/10] 1b70247c6412... encoded +All 10 conditions encoded. + +Starting Corrective Reflow | rounds=10 K=16 K_local=2 epochs/round=100 effective_bs=16 lr=1e-05 conditions=10 + +====================================================================== +ROUND 1/10 +====================================================================== +[Phase 1] ODE rollout ... + [CReflow rollout] condition 0/10: K=2, id=6b20973f10ef... + rollout: 2 videos in 56.8s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.44it/s] +propagate_in_video: 100%|██████████| 121/121 [00:11<00:00, 10.30it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 77.75it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.78it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.49it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.74it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 72.64it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.57it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.74it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.79it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.37it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.02it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 1/10: K=2, id=cff33035b28a... + rollout: 2 videos in 57.5s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 77.04it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.14it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.48it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.66it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.72it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.64it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 77.93it/s] +propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.33it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.69it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.32it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.65it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.69it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 2/10: K=2, id=7ebc8b336779... + rollout: 2 videos in 57.1s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.80it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.71it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.51it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.17it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.31it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.71it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 77.59it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.19it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.99it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.72it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.19it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.99it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 3/10: K=2, id=8077c679fb62... + rollout: 2 videos in 57.3s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 75.88it/s] +propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.54it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.33it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.03it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.71it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.20it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.97it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.94it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.80it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.48it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.06it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.86it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 4/10: K=2, id=41855438a8e0... + rollout: 2 videos in 57.6s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.59it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.88it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.29it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.15it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.15it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.88it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.97it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.74it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.05it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.30it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.51it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.83it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 5/10: K=2, id=a67b02f4e7ce... + rollout: 2 videos in 57.5s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.02it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.57it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.85it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.65it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.17it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.58it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.90it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.12it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.12it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.62it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.90it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.09it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 6/10: K=2, id=4568a9603f3b... + rollout: 2 videos in 57.5s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.94it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.46it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.68it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.73it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.83it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.45it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.64it/s] +propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.76it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.14it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.60it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.70it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.80it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 7/10: K=2, id=1b67584bff87... + rollout: 2 videos in 57.0s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 75.90it/s] +propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.75it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.26it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.18it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.63it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.95it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.39it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.61it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.79it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.57it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.72it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.06it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 8/10: K=2, id=f4f8079f6637... + rollout: 2 videos in 57.5s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.04it/s] +propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 12.04it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.37it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.40it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.08it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.58it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.47it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.20it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.48it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.30it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.85it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.52it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] condition 9/10: K=2, id=1b70247c6412... + rollout: 2 videos in 56.8s +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.06it/s] +propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.54it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.74it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.27it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.74it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.43it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.00it/s] +propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 12.07it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.11it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.98it/s] +propagate_in_video: 0it [00:00, ?it/s] +frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.99it/s] +propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.26it/s] +propagate_in_video: 0it [00:00, ?it/s] + [CReflow rollout] done: 20 samples, rewards = [1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 0.00, 0.00] + Rollout done: 20 samples, good=18, bad=2, avg_reward=0.900 + 1b67584bff87: mean=1.00, good=2/2 + 1b70247c6412: mean=0.00, good=0/2 + 41855438a8e0: mean=1.00, good=2/2 + 4568a9603f3b: mean=1.00, good=2/2 + 6b20973f10ef: mean=1.00, good=2/2 + 7ebc8b336779: mean=1.00, good=2/2 + 8077c679fb62: mean=1.00, good=2/2 + a67b02f4e7ce: mean=1.00, good=2/2 + cff33035b28a: mean=1.00, good=2/2 + f4f8079f6637: mean=1.00, good=2/2 +[Phase 2] Library update ... + [Library] rank=0 has 18 good samples + [Library] after update: SuccessLibrary(total=127, coverage=10/10, max_per_condition=64) +[Phase 3] Re-pairing ... + [Re-pair] train_set: 20 samples (anchor=18, correction=2, skipped=0) +[Phase 4] Training (100 epochs, 20 samples) ... + [DEBUG] v_pred requires_grad=False, grad_fn=False +Traceback (most recent call last): + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1039, in + main() + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 939, in main + train_metrics = train_one_round( + ^^^^^^^^^^^^^^^^ + File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 579, in train_one_round + loss.backward() + File "/usr/local/lib/python3.11/dist-packages/torch/_tensor.py", line 625, in backward + torch.autograd.backward( + File "/usr/local/lib/python3.11/dist-packages/torch/autograd/__init__.py", line 354, in backward + _engine_run_backward( + File "/usr/local/lib/python3.11/dist-packages/torch/autograd/graph.py", line 841, in _engine_run_backward + return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass + ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +RuntimeError: element 0 of tensors does not require grad and does not have a grad_fn +[rank0]: Traceback (most recent call last): +[rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1039, in +[rank0]: main() +[rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 939, in main +[rank0]: train_metrics = train_one_round( +[rank0]: ^^^^^^^^^^^^^^^^ +[rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 579, in train_one_round +[rank0]: loss.backward() +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/_tensor.py", line 625, in backward +[rank0]: torch.autograd.backward( +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/autograd/__init__.py", line 354, in backward +[rank0]: _engine_run_backward( +[rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/autograd/graph.py", line 841, in _engine_run_backward +[rank0]: return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass +[rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +[rank0]: RuntimeError: element 0 of tensors does not require grad and does not have a grad_fn diff --git a/wandb/run-20260408_014250-d3p1n5w7/files/wandb-metadata.json b/wandb/run-20260408_014250-d3p1n5w7/files/wandb-metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..a6e7f8137baedb2de8467fe6b3d7f84dad56ce3e --- /dev/null +++ b/wandb/run-20260408_014250-d3p1n5w7/files/wandb-metadata.json @@ -0,0 +1,144 @@ +{ + "os": "Linux-5.15.120.bsk.6-amd64-x86_64-with-glibc2.36", + "python": "3.11.2", + "startedAt": "2026-04-08T01:42:50.836955Z", + "args": [ + "--task", + "ti2v-5B", + "--size", + "640*736", + "--frame_num", + "121", + "--ckpt_dir", + "ckpts/Wan2.2-TI2V-5B", + "--pt_dir", + "ckpts/vidar_ckpt/merged_vidar_lora.pt", + "--dataset_json", + "data/rl_train/robotwin_stack_blocks_three.json", + "--output_dir", + "data/outputs/creflow_stack_blocks_three", + "--num_ode_steps", + "20", + "--sample_shift", + "5.0", + "--sample_guide_scale", + "5.0", + "--K", + "16", + "--num_rounds", + "10", + "--epochs_per_round", + "100", + "--max_per_condition", + "64", + "--effective_batch_size", + "16", + "--w_bad", + "1.0", + "--lambda_gripper", + "1.0", + "--seed", + "42", + "--reward_config", + "blocks_stack_v2", + "--convert_model_dtype", + "--offload_model", + "false", + "--learning_rate", + "1e-5", + "--weight_decay", + "0.01", + "--max_grad_norm", + "2.0", + "--checkpointing_steps", + "10", + "--lora_rank", + "64", + "--lora_alpha", + "64", + "--wandb_project", + "Corrective-Reflow", + "--wandb_run_name", + "creflow_stack_blocks_three" + ], + "program": "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", + "codePath": "fastvideo/train_creflow_wan_2_2_ti2v.py", + "git": { + "remote": "git@github.com:VincentNi0107/EmbodiedVideoRL.git", + "commit": "a8f5b170f0ef7d865af7e8f32c4aebb46efec2c1" + }, + "email": "1163051845@qq.com", + "root": "data/outputs/creflow_stack_blocks_three", + "host": "n124-104-180", + "username": "tiger", + "executable": "/usr/bin/python", + "codePathLocal": "fastvideo/train_creflow_wan_2_2_ti2v.py", + "cpu_count": 104, + "cpu_count_logical": 208, + "gpu": "NVIDIA H100 80GB HBM3", + "gpu_count": 8, + "disk": { + "/": { + "total": "1055740600320", + "used": "482529443840" + } + }, + "memory": { + "total": "1977660338176" + }, + "cpu": { + "count": 104, + "countLogical": 208 + }, + "gpu_nvidia": [ + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + } + ], + "cudaVersion": "12.9" +} \ No newline at end of file diff --git a/wandb/run-20260408_014250-d3p1n5w7/files/wandb-summary.json b/wandb/run-20260408_014250-d3p1n5w7/files/wandb-summary.json new file mode 100644 index 0000000000000000000000000000000000000000..410bcc0539b0ea29a1a54d11700205615ed9ee95 --- /dev/null +++ b/wandb/run-20260408_014250-d3p1n5w7/files/wandb-summary.json @@ -0,0 +1 @@ +{"_wandb":{"runtime":1586}} \ No newline at end of file diff --git a/wandb/run-20260408_014250-d3p1n5w7/logs/debug-internal.log b/wandb/run-20260408_014250-d3p1n5w7/logs/debug-internal.log new file mode 100644 index 0000000000000000000000000000000000000000..2c991f38ee6c316ec10ef341ad79ce5bc199e982 --- /dev/null +++ b/wandb/run-20260408_014250-d3p1n5w7/logs/debug-internal.log @@ -0,0 +1,18 @@ +{"time":"2026-04-08T01:42:50.846912727Z","level":"INFO","msg":"using version","core version":"0.18.5"} +{"time":"2026-04-08T01:42:50.846924973Z","level":"INFO","msg":"created symlink","path":"data/outputs/creflow_stack_blocks_three/wandb/run-20260408_014250-d3p1n5w7/logs/debug-core.log"} +{"time":"2026-04-08T01:42:51.065325741Z","level":"INFO","msg":"created new stream","id":"d3p1n5w7"} +{"time":"2026-04-08T01:42:51.06552384Z","level":"INFO","msg":"stream: started","id":"d3p1n5w7"} +{"time":"2026-04-08T01:42:51.065611464Z","level":"INFO","msg":"sender: started","stream_id":"d3p1n5w7"} +{"time":"2026-04-08T01:42:51.06558469Z","level":"INFO","msg":"handler: started","stream_id":{"value":"d3p1n5w7"}} +{"time":"2026-04-08T01:42:51.065564877Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"d3p1n5w7"}} +{"time":"2026-04-08T01:42:51.47008492Z","level":"INFO","msg":"Starting system monitor"} +{"time":"2026-04-08T02:00:36.824476956Z","level":"INFO","msg":"api: retrying error","error":"Post \"https://api.wandb.ai/graphql\": net/http: request canceled (Client.Timeout exceeded while awaiting headers)"} +{"time":"2026-04-08T02:01:09.269479394Z","level":"INFO","msg":"api: retrying error","error":"Post \"https://api.wandb.ai/graphql\": context deadline exceeded"} +{"time":"2026-04-08T02:09:17.196524875Z","level":"INFO","msg":"stream: closing","id":"d3p1n5w7"} +{"time":"2026-04-08T02:09:17.196574831Z","level":"INFO","msg":"Stopping system monitor"} +{"time":"2026-04-08T02:09:17.201147257Z","level":"INFO","msg":"Stopped system monitor"} +{"time":"2026-04-08T02:09:18.01646151Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"} +{"time":"2026-04-08T02:09:18.177322397Z","level":"INFO","msg":"handler: closed","stream_id":{"value":"d3p1n5w7"}} +{"time":"2026-04-08T02:09:18.177361768Z","level":"INFO","msg":"sender: closed","stream_id":"d3p1n5w7"} +{"time":"2026-04-08T02:09:18.177361574Z","level":"INFO","msg":"writer: Close: closed","stream_id":{"value":"d3p1n5w7"}} +{"time":"2026-04-08T02:09:18.183256398Z","level":"INFO","msg":"stream: closed","id":"d3p1n5w7"} diff --git a/wandb/run-20260408_014250-d3p1n5w7/logs/debug.log b/wandb/run-20260408_014250-d3p1n5w7/logs/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..ca724622251d642f8113003501cabd09efba921b --- /dev/null +++ b/wandb/run-20260408_014250-d3p1n5w7/logs/debug.log @@ -0,0 +1,26 @@ +2026-04-08 01:42:50,803 INFO MainThread:198734 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5 +2026-04-08 01:42:50,803 INFO MainThread:198734 [wandb_setup.py:_flush():79] Configure stats pid to 198734 +2026-04-08 01:42:50,803 INFO MainThread:198734 [wandb_setup.py:_flush():79] Loading settings from /home/tiger/.config/wandb/settings +2026-04-08 01:42:50,803 INFO MainThread:198734 [wandb_setup.py:_flush():79] Loading settings from /mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/wandb/settings +2026-04-08 01:42:50,803 INFO MainThread:198734 [wandb_setup.py:_flush():79] Loading settings from environment variables: {'disabled': 'false', 'project': 'Corrective-Reflow', 'api_key': '***REDACTED***', 'mode': 'online'} +2026-04-08 01:42:50,803 INFO MainThread:198734 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': 'online', '_disable_service': None} +2026-04-08 01:42:50,803 INFO MainThread:198734 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'fastvideo/train_creflow_wan_2_2_ti2v.py', 'program_abspath': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py', 'program': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py'} +2026-04-08 01:42:50,803 INFO MainThread:198734 [wandb_setup.py:_flush():79] Applying login settings: {} +2026-04-08 01:42:50,805 INFO MainThread:198734 [wandb_init.py:_log_setup():534] Logging user logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_014250-d3p1n5w7/logs/debug.log +2026-04-08 01:42:50,807 INFO MainThread:198734 [wandb_init.py:_log_setup():535] Logging internal logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_014250-d3p1n5w7/logs/debug-internal.log +2026-04-08 01:42:50,807 INFO MainThread:198734 [wandb_init.py:init():621] calling init triggers +2026-04-08 01:42:50,807 INFO MainThread:198734 [wandb_init.py:init():628] wandb.init called with sweep_config: {} +config: {'vidar_root': '', 'task': 'ti2v-5B', 'size': '640*736', 'frame_num': 121, 'ckpt_dir': 'ckpts/Wan2.2-TI2V-5B', 'pt_dir': 'ckpts/vidar_ckpt/merged_vidar_lora.pt', 'convert_model_dtype': True, 'offload_model': False, 'dataset_json': 'data/rl_train/robotwin_stack_blocks_three.json', 'max_samples': -1, 'output_dir': 'data/outputs/creflow_stack_blocks_three', 'num_ode_steps': 20, 'sample_shift': 5.0, 'sample_guide_scale': 5.0, 'neg_prompt': '色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走', 'seed': 42, 'K': 16, 'num_rounds': 10, 'epochs_per_round': 100, 'effective_batch_size': 16, 'w_bad': 1.0, 'max_per_condition': 64, 'lambda_gripper': 1.0, 'reset_optimizer_per_round': False, 'num_train_timesteps': 1000, 'learning_rate': 1e-05, 'weight_decay': 0.01, 'max_grad_norm': 2.0, 'checkpointing_steps': 10, 'gradient_checkpointing': True, 'use_8bit_adam': True, 'lora_rank': 64, 'lora_alpha': 64, 'lora_target_modules': None, 'resume_from_lora_checkpoint': None, 'reward_config': 'blocks_stack_v2', 'reward_backend': 'blocks_stack_v2', 'skip_reward_debug_video': True, 'wandb_project': 'Corrective-Reflow', 'wandb_run_name': 'creflow_stack_blocks_three', 'hallucination_crop_top_ratio': 0.6667, 'bsv2_prompts': ['red block', 'green block', 'blue block'], 'bsv2_pick_thr': 0.05, 'bsv2_place_thr': 0.05, 'bsv2_vertical_sep_thr': 0.03, 'bsv2_x_align_thr': 0.025, 'bsv2_check_window_frac': 0.2, 'bsv2_dup_max_frames': 0, 'bsv2_expected_grip_changes': 6, 'bsv2_mj_lo': 0.6, 'bsv2_mj_hi': 1.3, 'bsv2_idm_ckpt_path': 'ckpts/vidar_ckpt/idm.pt'} +2026-04-08 01:42:50,807 INFO MainThread:198734 [wandb_init.py:init():671] starting backend +2026-04-08 01:42:50,807 INFO MainThread:198734 [wandb_init.py:init():675] sending inform_init request +2026-04-08 01:42:50,834 INFO MainThread:198734 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn +2026-04-08 01:42:50,835 INFO MainThread:198734 [wandb_init.py:init():688] backend started and connected +2026-04-08 01:42:50,889 INFO MainThread:198734 [wandb_init.py:init():783] updated telemetry +2026-04-08 01:42:51,055 INFO MainThread:198734 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout +2026-04-08 01:42:51,420 INFO MainThread:198734 [wandb_init.py:init():867] starting run threads in backend +2026-04-08 01:42:51,784 INFO MainThread:198734 [wandb_run.py:_console_start():2463] atexit reg +2026-04-08 01:42:51,785 INFO MainThread:198734 [wandb_run.py:_redirect():2311] redirect: wrap_raw +2026-04-08 01:42:51,785 INFO MainThread:198734 [wandb_run.py:_redirect():2376] Wrapping output streams. +2026-04-08 01:42:51,785 INFO MainThread:198734 [wandb_run.py:_redirect():2401] Redirects installed. +2026-04-08 01:42:51,790 INFO MainThread:198734 [wandb_init.py:init():911] run started, returning control to user process +2026-04-08 02:09:17,196 WARNING MsgRouterThr:198734 [router.py:message_loop():77] message_loop has been closed diff --git a/wandb/run-20260408_025012-pf74wm1f/logs/debug-internal.log b/wandb/run-20260408_025012-pf74wm1f/logs/debug-internal.log new file mode 100644 index 0000000000000000000000000000000000000000..92b7e953eac91123c48e9696af442a1bea189979 --- /dev/null +++ b/wandb/run-20260408_025012-pf74wm1f/logs/debug-internal.log @@ -0,0 +1,8 @@ +{"time":"2026-04-08T02:50:12.360136031Z","level":"INFO","msg":"using version","core version":"0.18.5"} +{"time":"2026-04-08T02:50:12.360147Z","level":"INFO","msg":"created symlink","path":"data/outputs/creflow_stack_blocks_three/wandb/run-20260408_025012-pf74wm1f/logs/debug-core.log"} +{"time":"2026-04-08T02:50:12.578898735Z","level":"INFO","msg":"created new stream","id":"pf74wm1f"} +{"time":"2026-04-08T02:50:12.579095021Z","level":"INFO","msg":"stream: started","id":"pf74wm1f"} +{"time":"2026-04-08T02:50:12.5791654Z","level":"INFO","msg":"sender: started","stream_id":"pf74wm1f"} +{"time":"2026-04-08T02:50:12.579140373Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"pf74wm1f"}} +{"time":"2026-04-08T02:50:12.579178443Z","level":"INFO","msg":"handler: started","stream_id":{"value":"pf74wm1f"}} +{"time":"2026-04-08T02:50:12.901012044Z","level":"INFO","msg":"Starting system monitor"} diff --git a/wandb/run-20260408_164919-x63spn6n/files/wandb-metadata.json b/wandb/run-20260408_164919-x63spn6n/files/wandb-metadata.json new file mode 100644 index 0000000000000000000000000000000000000000..d4abd7d8acc5240ce42cb067cde48b0a363a16fb --- /dev/null +++ b/wandb/run-20260408_164919-x63spn6n/files/wandb-metadata.json @@ -0,0 +1,144 @@ +{ + "os": "Linux-5.15.120.bsk.6-amd64-x86_64-with-glibc2.36", + "python": "3.11.2", + "startedAt": "2026-04-08T16:49:19.417199Z", + "args": [ + "--task", + "ti2v-5B", + "--size", + "640*736", + "--frame_num", + "121", + "--ckpt_dir", + "ckpts/Wan2.2-TI2V-5B", + "--pt_dir", + "ckpts/vidar_ckpt/merged_vidar_lora.pt", + "--dataset_json", + "data/rl_train/robotwin_stack_blocks_three.json", + "--output_dir", + "data/outputs/creflow_stack_blocks_three", + "--num_ode_steps", + "20", + "--sample_shift", + "5.0", + "--sample_guide_scale", + "5.0", + "--K", + "16", + "--num_rounds", + "10", + "--epochs_per_round", + "100", + "--max_per_condition", + "64", + "--effective_batch_size", + "16", + "--w_bad", + "1.0", + "--lambda_gripper", + "1.0", + "--seed", + "42", + "--reward_config", + "blocks_stack_v2", + "--convert_model_dtype", + "--offload_model", + "false", + "--learning_rate", + "1e-5", + "--weight_decay", + "0.01", + "--max_grad_norm", + "2.0", + "--checkpointing_steps", + "10", + "--lora_rank", + "64", + "--lora_alpha", + "64", + "--wandb_project", + "Corrective-Reflow", + "--wandb_run_name", + "creflow_stack_blocks_three" + ], + "program": "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", + "codePath": "fastvideo/train_creflow_wan_2_2_ti2v.py", + "git": { + "remote": "git@github.com:VincentNi0107/EmbodiedVideoRL.git", + "commit": "d5e0e14ab1e94676078e078720b47c2281e8446e" + }, + "email": "1163051845@qq.com", + "root": "data/outputs/creflow_stack_blocks_three", + "host": "n124-106-114", + "username": "tiger", + "executable": "/usr/bin/python", + "codePathLocal": "fastvideo/train_creflow_wan_2_2_ti2v.py", + "cpu_count": 104, + "cpu_count_logical": 208, + "gpu": "NVIDIA H100 80GB HBM3", + "gpu_count": 8, + "disk": { + "/": { + "total": "1055740600320", + "used": "290213711872" + } + }, + "memory": { + "total": "1977660334080" + }, + "cpu": { + "count": 104, + "countLogical": 208 + }, + "gpu_nvidia": [ + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + }, + { + "name": "NVIDIA H100 80GB HBM3", + "memoryTotal": "85520809984", + "cudaCores": 16896, + "architecture": "Hopper" + } + ], + "cudaVersion": "12.9" +} \ No newline at end of file diff --git a/wandb/run-20260408_164919-x63spn6n/logs/debug.log b/wandb/run-20260408_164919-x63spn6n/logs/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..c970259f28b9aad7af44a421b7b07c052ea49175 --- /dev/null +++ b/wandb/run-20260408_164919-x63spn6n/logs/debug.log @@ -0,0 +1,25 @@ +2026-04-08 16:49:19,387 INFO MainThread:1168332 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5 +2026-04-08 16:49:19,387 INFO MainThread:1168332 [wandb_setup.py:_flush():79] Configure stats pid to 1168332 +2026-04-08 16:49:19,387 INFO MainThread:1168332 [wandb_setup.py:_flush():79] Loading settings from /home/tiger/.config/wandb/settings +2026-04-08 16:49:19,387 INFO MainThread:1168332 [wandb_setup.py:_flush():79] Loading settings from /mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/wandb/settings +2026-04-08 16:49:19,387 INFO MainThread:1168332 [wandb_setup.py:_flush():79] Loading settings from environment variables: {'disabled': 'false', 'project': 'Corrective-Reflow', 'api_key': '***REDACTED***', 'mode': 'online'} +2026-04-08 16:49:19,387 INFO MainThread:1168332 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': 'online', '_disable_service': None} +2026-04-08 16:49:19,387 INFO MainThread:1168332 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'fastvideo/train_creflow_wan_2_2_ti2v.py', 'program_abspath': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py', 'program': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py'} +2026-04-08 16:49:19,387 INFO MainThread:1168332 [wandb_setup.py:_flush():79] Applying login settings: {} +2026-04-08 16:49:19,389 INFO MainThread:1168332 [wandb_init.py:_log_setup():534] Logging user logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_164919-x63spn6n/logs/debug.log +2026-04-08 16:49:19,390 INFO MainThread:1168332 [wandb_init.py:_log_setup():535] Logging internal logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_164919-x63spn6n/logs/debug-internal.log +2026-04-08 16:49:19,390 INFO MainThread:1168332 [wandb_init.py:init():621] calling init triggers +2026-04-08 16:49:19,390 INFO MainThread:1168332 [wandb_init.py:init():628] wandb.init called with sweep_config: {} +config: {'vidar_root': '', 'task': 'ti2v-5B', 'size': '640*736', 'frame_num': 121, 'ckpt_dir': 'ckpts/Wan2.2-TI2V-5B', 'pt_dir': 'ckpts/vidar_ckpt/merged_vidar_lora.pt', 'convert_model_dtype': True, 'offload_model': False, 'dataset_json': 'data/rl_train/robotwin_stack_blocks_three.json', 'max_samples': -1, 'output_dir': 'data/outputs/creflow_stack_blocks_three', 'num_ode_steps': 20, 'sample_shift': 5.0, 'sample_guide_scale': 5.0, 'neg_prompt': '色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走', 'seed': 42, 'K': 16, 'num_rounds': 10, 'epochs_per_round': 100, 'effective_batch_size': 16, 'w_bad': 1.0, 'max_per_condition': 64, 'lambda_gripper': 1.0, 'reset_optimizer_per_round': False, 'num_train_timesteps': 1000, 'learning_rate': 1e-05, 'weight_decay': 0.01, 'max_grad_norm': 2.0, 'checkpointing_steps': 10, 'gradient_checkpointing': True, 'use_8bit_adam': True, 'lora_rank': 64, 'lora_alpha': 64, 'lora_target_modules': None, 'resume_from_lora_checkpoint': None, 'reward_config': 'blocks_stack_v2', 'reward_backend': 'blocks_stack_v2', 'skip_reward_debug_video': True, 'wandb_project': 'Corrective-Reflow', 'wandb_run_name': 'creflow_stack_blocks_three', 'hallucination_crop_top_ratio': 0.6667, 'bsv2_prompts': ['red block', 'green block', 'blue block'], 'bsv2_pick_thr': 0.05, 'bsv2_place_thr': 0.05, 'bsv2_vertical_sep_thr': 0.03, 'bsv2_x_align_thr': 0.025, 'bsv2_check_window_frac': 0.2, 'bsv2_dup_max_frames': 0, 'bsv2_expected_grip_changes': 6, 'bsv2_mj_lo': 0.6, 'bsv2_mj_hi': 1.3, 'bsv2_idm_ckpt_path': 'ckpts/vidar_ckpt/idm.pt'} +2026-04-08 16:49:19,390 INFO MainThread:1168332 [wandb_init.py:init():671] starting backend +2026-04-08 16:49:19,390 INFO MainThread:1168332 [wandb_init.py:init():675] sending inform_init request +2026-04-08 16:49:19,415 INFO MainThread:1168332 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn +2026-04-08 16:49:19,415 INFO MainThread:1168332 [wandb_init.py:init():688] backend started and connected +2026-04-08 16:49:19,465 INFO MainThread:1168332 [wandb_init.py:init():783] updated telemetry +2026-04-08 16:49:19,620 INFO MainThread:1168332 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout +2026-04-08 16:49:20,302 INFO MainThread:1168332 [wandb_init.py:init():867] starting run threads in backend +2026-04-08 16:49:20,654 INFO MainThread:1168332 [wandb_run.py:_console_start():2463] atexit reg +2026-04-08 16:49:20,655 INFO MainThread:1168332 [wandb_run.py:_redirect():2311] redirect: wrap_raw +2026-04-08 16:49:20,655 INFO MainThread:1168332 [wandb_run.py:_redirect():2376] Wrapping output streams. +2026-04-08 16:49:20,655 INFO MainThread:1168332 [wandb_run.py:_redirect():2401] Redirects installed. +2026-04-08 16:49:20,659 INFO MainThread:1168332 [wandb_init.py:init():911] run started, returning control to user process