VincentNi commited on
Commit
44068c2
·
verified ·
1 Parent(s): b75f519

Add files using upload-large-folder tool

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +13 -0
  2. reward_debug/round000/1b67584bff87_k000_140f07dc_CLEAN_ms1.1054.mp4 +3 -0
  3. reward_debug/round000/1b67584bff87_k000_1c91e481_CLEAN_ms1.0826.mp4 +3 -0
  4. reward_debug/round000/1b67584bff87_k000_2d3023c7_CLEAN_ms1.0881.mp4 +3 -0
  5. reward_debug/round000/1b67584bff87_k000_32666f10_CLEAN_ms1.1054.mp4 +3 -0
  6. reward_debug/round000/1b67584bff87_k000_4dd03d13_CLEAN_ms1.0985.mp4 +3 -0
  7. reward_debug/round000/1b67584bff87_k000_6843a418_CLEAN_ms1.0641.mp4 +3 -0
  8. reward_debug/round000/1b67584bff87_k000_89927d91_CLEAN_ms1.3450.mp4 +3 -0
  9. reward_debug/round000/1b67584bff87_k000_98e9a7bc_CLEAN_ms1.3450.mp4 +3 -0
  10. reward_debug/round000/1b67584bff87_k000_c304d5a9_CLEAN_ms1.0826.mp4 +3 -0
  11. reward_debug/round000/1b67584bff87_k000_ddafd33c_CLEAN_ms1.0641.mp4 +3 -0
  12. reward_debug/round000/1b67584bff87_k001_62e8daff_FAIL_ms1.0768.mp4 +3 -0
  13. reward_debug/round000/1b67584bff87_k001_65f6114e_CLEAN_ms1.0744.mp4 +3 -0
  14. reward_debug/round000/1b67584bff87_k001_6615c2c4_CLEAN_ms1.1069.mp4 +3 -0
  15. wandb/run-20260408_000908-i6zi4vwm/files/config.yaml +145 -0
  16. wandb/run-20260408_000908-i6zi4vwm/files/output.log +139 -0
  17. wandb/run-20260408_000908-i6zi4vwm/files/wandb-metadata.json +144 -0
  18. wandb/run-20260408_000908-i6zi4vwm/files/wandb-summary.json +1 -0
  19. wandb/run-20260408_000908-i6zi4vwm/logs/debug.log +26 -0
  20. wandb/run-20260408_001541-g1xdtpwt/files/output.log +286 -0
  21. wandb/run-20260408_001541-g1xdtpwt/files/wandb-metadata.json +144 -0
  22. wandb/run-20260408_001541-g1xdtpwt/files/wandb-summary.json +1 -0
  23. wandb/run-20260408_001541-g1xdtpwt/logs/debug-internal.log +16 -0
  24. wandb/run-20260408_001541-g1xdtpwt/logs/debug.log +26 -0
  25. wandb/run-20260408_004547-x9lalqi6/files/config.yaml +145 -0
  26. wandb/run-20260408_004547-x9lalqi6/files/output.log +89 -0
  27. wandb/run-20260408_004547-x9lalqi6/files/wandb-metadata.json +144 -0
  28. wandb/run-20260408_004547-x9lalqi6/files/wandb-summary.json +1 -0
  29. wandb/run-20260408_004547-x9lalqi6/logs/debug-internal.log +16 -0
  30. wandb/run-20260408_004547-x9lalqi6/logs/debug.log +26 -0
  31. wandb/run-20260408_004931-ku6kqemc/files/config.yaml +145 -0
  32. wandb/run-20260408_004931-ku6kqemc/files/output.log +685 -0
  33. wandb/run-20260408_004931-ku6kqemc/files/wandb-metadata.json +144 -0
  34. wandb/run-20260408_004931-ku6kqemc/files/wandb-summary.json +1 -0
  35. wandb/run-20260408_004931-ku6kqemc/logs/debug-internal.log +19 -0
  36. wandb/run-20260408_004931-ku6kqemc/logs/debug.log +26 -0
  37. wandb/run-20260408_012032-idkfwxb0/files/config.yaml +145 -0
  38. wandb/run-20260408_012032-idkfwxb0/files/output.log +133 -0
  39. wandb/run-20260408_012032-idkfwxb0/files/wandb-metadata.json +144 -0
  40. wandb/run-20260408_012032-idkfwxb0/files/wandb-summary.json +1 -0
  41. wandb/run-20260408_012032-idkfwxb0/logs/debug-internal.log +16 -0
  42. wandb/run-20260408_012032-idkfwxb0/logs/debug.log +26 -0
  43. wandb/run-20260408_014250-d3p1n5w7/files/config.yaml +145 -0
  44. wandb/run-20260408_014250-d3p1n5w7/files/output.log +287 -0
  45. wandb/run-20260408_014250-d3p1n5w7/files/wandb-metadata.json +144 -0
  46. wandb/run-20260408_014250-d3p1n5w7/files/wandb-summary.json +1 -0
  47. wandb/run-20260408_014250-d3p1n5w7/logs/debug-internal.log +18 -0
  48. wandb/run-20260408_014250-d3p1n5w7/logs/debug.log +26 -0
  49. wandb/run-20260408_025012-pf74wm1f/logs/debug-internal.log +8 -0
  50. wandb/run-20260408_164919-x63spn6n/files/wandb-metadata.json +144 -0
.gitattributes CHANGED
@@ -70,3 +70,16 @@ videos/round002/f4f8079f6637_k000_a403ac96.mp4 filter=lfs diff=lfs merge=lfs -te
70
  videos/round002/7ebc8b336779_k001_585674db.mp4 filter=lfs diff=lfs merge=lfs -text
71
  videos/round002/f4f8079f6637_k001_a40bb712.mp4 filter=lfs diff=lfs merge=lfs -text
72
  videos/round002/f4f8079f6637_k001_45c1d63e.mp4 filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
70
  videos/round002/7ebc8b336779_k001_585674db.mp4 filter=lfs diff=lfs merge=lfs -text
71
  videos/round002/f4f8079f6637_k001_a40bb712.mp4 filter=lfs diff=lfs merge=lfs -text
72
  videos/round002/f4f8079f6637_k001_45c1d63e.mp4 filter=lfs diff=lfs merge=lfs -text
73
+ reward_debug/round000/1b67584bff87_k000_ddafd33c_CLEAN_ms1.0641.mp4 filter=lfs diff=lfs merge=lfs -text
74
+ reward_debug/round000/1b67584bff87_k000_32666f10_CLEAN_ms1.1054.mp4 filter=lfs diff=lfs merge=lfs -text
75
+ reward_debug/round000/1b67584bff87_k000_89927d91_CLEAN_ms1.3450.mp4 filter=lfs diff=lfs merge=lfs -text
76
+ reward_debug/round000/1b67584bff87_k001_62e8daff_FAIL_ms1.0768.mp4 filter=lfs diff=lfs merge=lfs -text
77
+ reward_debug/round000/1b67584bff87_k000_140f07dc_CLEAN_ms1.1054.mp4 filter=lfs diff=lfs merge=lfs -text
78
+ reward_debug/round000/1b67584bff87_k001_6615c2c4_CLEAN_ms1.1069.mp4 filter=lfs diff=lfs merge=lfs -text
79
+ reward_debug/round000/1b67584bff87_k000_2d3023c7_CLEAN_ms1.0881.mp4 filter=lfs diff=lfs merge=lfs -text
80
+ reward_debug/round000/1b67584bff87_k000_4dd03d13_CLEAN_ms1.0985.mp4 filter=lfs diff=lfs merge=lfs -text
81
+ reward_debug/round000/1b67584bff87_k000_1c91e481_CLEAN_ms1.0826.mp4 filter=lfs diff=lfs merge=lfs -text
82
+ reward_debug/round000/1b67584bff87_k000_c304d5a9_CLEAN_ms1.0826.mp4 filter=lfs diff=lfs merge=lfs -text
83
+ reward_debug/round000/1b67584bff87_k000_98e9a7bc_CLEAN_ms1.3450.mp4 filter=lfs diff=lfs merge=lfs -text
84
+ reward_debug/round000/1b67584bff87_k001_65f6114e_CLEAN_ms1.0744.mp4 filter=lfs diff=lfs merge=lfs -text
85
+ reward_debug/round000/1b67584bff87_k000_6843a418_CLEAN_ms1.0641.mp4 filter=lfs diff=lfs merge=lfs -text
reward_debug/round000/1b67584bff87_k000_140f07dc_CLEAN_ms1.1054.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fbf507778ce75b6bb56bab71c7494c67addb54ce20d32ef56bcb822da25c3ab4
3
+ size 2361972
reward_debug/round000/1b67584bff87_k000_1c91e481_CLEAN_ms1.0826.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7b9bb67e3c7ac5da546a5ca12c4449a4effeb623d45eaf344a42e7ea8261152f
3
+ size 2339614
reward_debug/round000/1b67584bff87_k000_2d3023c7_CLEAN_ms1.0881.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ab2fab2f0908c2fb038d679cb3415aabc412b67e8ce4ac2abfcadab5a4c489cd
3
+ size 2360584
reward_debug/round000/1b67584bff87_k000_32666f10_CLEAN_ms1.1054.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fbf507778ce75b6bb56bab71c7494c67addb54ce20d32ef56bcb822da25c3ab4
3
+ size 2361972
reward_debug/round000/1b67584bff87_k000_4dd03d13_CLEAN_ms1.0985.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:788fdbf6c6081f1182cf9b162937f6dfd58e217d9691989f7b277e6c3732936f
3
+ size 2356905
reward_debug/round000/1b67584bff87_k000_6843a418_CLEAN_ms1.0641.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2473aab17063a10360651d35b4785e4993ececb5ccce50874616ee87a90c7c70
3
+ size 2392799
reward_debug/round000/1b67584bff87_k000_89927d91_CLEAN_ms1.3450.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7fa9070952b043c3da2268df805a52ff3df7058b04e83d9b7e00e2285871fec1
3
+ size 2400922
reward_debug/round000/1b67584bff87_k000_98e9a7bc_CLEAN_ms1.3450.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7fa9070952b043c3da2268df805a52ff3df7058b04e83d9b7e00e2285871fec1
3
+ size 2400922
reward_debug/round000/1b67584bff87_k000_c304d5a9_CLEAN_ms1.0826.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7b9bb67e3c7ac5da546a5ca12c4449a4effeb623d45eaf344a42e7ea8261152f
3
+ size 2339614
reward_debug/round000/1b67584bff87_k000_ddafd33c_CLEAN_ms1.0641.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2473aab17063a10360651d35b4785e4993ececb5ccce50874616ee87a90c7c70
3
+ size 2392799
reward_debug/round000/1b67584bff87_k001_62e8daff_FAIL_ms1.0768.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1fed03bdb01e0194771940b43a20058bd46224a3e36273f761fe78e5ff5e52a6
3
+ size 2376223
reward_debug/round000/1b67584bff87_k001_65f6114e_CLEAN_ms1.0744.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3eafe2bfbd7107ccb9fbd1b1838b70a3d7ddee382f6d04f9e8fb058f8578db96
3
+ size 2323915
reward_debug/round000/1b67584bff87_k001_6615c2c4_CLEAN_ms1.1069.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:10ee7b335cad92e16e4ce034607ffb5d7db382e1e78f381389822fbc4b21667f
3
+ size 2350533
wandb/run-20260408_000908-i6zi4vwm/files/config.yaml ADDED
@@ -0,0 +1,145 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ _wandb:
2
+ value:
3
+ cli_version: 0.18.5
4
+ m: []
5
+ python_version: 3.11.2
6
+ t:
7
+ "1":
8
+ - 1
9
+ - 11
10
+ - 41
11
+ - 49
12
+ - 55
13
+ - 71
14
+ - 83
15
+ - 105
16
+ "2":
17
+ - 1
18
+ - 11
19
+ - 41
20
+ - 49
21
+ - 55
22
+ - 63
23
+ - 71
24
+ - 83
25
+ - 98
26
+ - 105
27
+ "3":
28
+ - 13
29
+ - 16
30
+ - 23
31
+ - 55
32
+ "4": 3.11.2
33
+ "5": 0.18.5
34
+ "6": 4.46.1
35
+ "8":
36
+ - 5
37
+ "12": 0.18.5
38
+ "13": linux-x86_64
39
+ K:
40
+ value: 16
41
+ bsv2_check_window_frac:
42
+ value: 0.2
43
+ bsv2_dup_max_frames:
44
+ value: 0
45
+ bsv2_expected_grip_changes:
46
+ value: 6
47
+ bsv2_idm_ckpt_path:
48
+ value: ckpts/vidar_ckpt/idm.pt
49
+ bsv2_mj_hi:
50
+ value: 1.3
51
+ bsv2_mj_lo:
52
+ value: 0.6
53
+ bsv2_pick_thr:
54
+ value: 0.05
55
+ bsv2_place_thr:
56
+ value: 0.05
57
+ bsv2_prompts:
58
+ value:
59
+ - red block
60
+ - green block
61
+ - blue block
62
+ bsv2_vertical_sep_thr:
63
+ value: 0.03
64
+ bsv2_x_align_thr:
65
+ value: 0.025
66
+ checkpointing_steps:
67
+ value: 10
68
+ ckpt_dir:
69
+ value: ckpts/Wan2.2-TI2V-5B
70
+ convert_model_dtype:
71
+ value: true
72
+ dataset_json:
73
+ value: data/rl_train/robotwin_stack_blocks_three.json
74
+ effective_batch_size:
75
+ value: 16
76
+ epochs_per_round:
77
+ value: 100
78
+ frame_num:
79
+ value: 121
80
+ gradient_checkpointing:
81
+ value: true
82
+ hallucination_crop_top_ratio:
83
+ value: 0.6667
84
+ lambda_gripper:
85
+ value: 1
86
+ learning_rate:
87
+ value: 1e-05
88
+ lora_alpha:
89
+ value: 64
90
+ lora_rank:
91
+ value: 64
92
+ lora_target_modules:
93
+ value: null
94
+ max_grad_norm:
95
+ value: 2
96
+ max_per_condition:
97
+ value: 64
98
+ max_samples:
99
+ value: -1
100
+ neg_prompt:
101
+ value: 色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走
102
+ num_ode_steps:
103
+ value: 20
104
+ num_rounds:
105
+ value: 10
106
+ num_train_timesteps:
107
+ value: 1000
108
+ offload_model:
109
+ value: false
110
+ output_dir:
111
+ value: data/outputs/creflow_stack_blocks_three
112
+ pt_dir:
113
+ value: ckpts/vidar_ckpt/merged_vidar_lora.pt
114
+ reset_optimizer_per_round:
115
+ value: false
116
+ resume_from_lora_checkpoint:
117
+ value: null
118
+ reward_backend:
119
+ value: blocks_stack_v2
120
+ reward_config:
121
+ value: blocks_stack_v2
122
+ sample_guide_scale:
123
+ value: 5
124
+ sample_shift:
125
+ value: 5
126
+ seed:
127
+ value: 42
128
+ size:
129
+ value: 640*736
130
+ skip_reward_debug_video:
131
+ value: true
132
+ task:
133
+ value: ti2v-5B
134
+ use_8bit_adam:
135
+ value: true
136
+ vidar_root:
137
+ value: ""
138
+ w_bad:
139
+ value: 1
140
+ wandb_project:
141
+ value: Corrective-Reflow
142
+ wandb_run_name:
143
+ value: creflow_stack_blocks_three
144
+ weight_decay:
145
+ value: 0.01
wandb/run-20260408_000908-i6zi4vwm/files/output.log ADDED
@@ -0,0 +1,139 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Building Wan2.2 TI2V model ... (DDP=False, world_size=1)
2
+ Moving T5 encoder to GPU permanently (offload_model=false) ...
3
+ Building reward scorer ...
4
+ Loading IDM from ckpts/vidar_ckpt/idm.pt on cuda:0 ...
5
+ IDM loaded.
6
+ Loading SAM3 VideoPredictor (blocks_stack_v2) on cuda:0 ...
7
+ INFO 2026-04-08 00:10:37,404 17550 sam3_video_base.py: 125: setting max_num_objects=10000 and num_obj_for_compile=16
8
+ SAM3 VideoPredictor (blocks_stack_v2) loaded.
9
+ Applying LoRA (single adapter) for CReflow ...
10
+ LoRA injected: rank=64, alpha=64, target_modules=['self_attn.q', 'self_attn.k', 'self_attn.v', 'self_attn.o', 'cross_attn.q', 'cross_attn.k', 'cross_attn.v', 'cross_attn.o']
11
+ Trainable: 94.4 M / 5094.2 M total (1.85%)
12
+ Gradient checkpointing enabled on 30 DiT blocks
13
+ Trainable parameters: 94.4 M
14
+ Using 8-bit AdamW (bitsandbytes)
15
+ Dataset: 10 conditions
16
+ Encoding 10 conditions ...
17
+ [1/10] 6b20973f10ef... encoded
18
+ [2/10] cff33035b28a... encoded
19
+ [3/10] 7ebc8b336779... encoded
20
+ [4/10] 8077c679fb62... encoded
21
+ [5/10] 41855438a8e0... encoded
22
+ [6/10] a67b02f4e7ce... encoded
23
+ [7/10] 4568a9603f3b... encoded
24
+ [8/10] 1b67584bff87... encoded
25
+ [9/10] f4f8079f6637... encoded
26
+ [10/10] 1b70247c6412... encoded
27
+ All 10 conditions encoded.
28
+
29
+ Starting Corrective Reflow | rounds=10 K=16 K_local=16 epochs/round=100 effective_bs=16 lr=1e-05 conditions=10
30
+
31
+ ======================================================================
32
+ ROUND 1/10
33
+ ======================================================================
34
+ [Phase 1] ODE rollout ...
35
+ [CReflow rollout] condition 0/10: K=16, id=6b20973f10ef...
36
+ Traceback (most recent call last):
37
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1034, in <module>
38
+ main()
39
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 866, in main
40
+ rollouts = creflow_rollout(
41
+ ^^^^^^^^^^^^^^^^
42
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/creflow/rollout.py", line 267, in creflow_rollout
43
+ x0_preds = _ode_rollout_batch(
44
+ ^^^^^^^^^^^^^^^^^^^
45
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/creflow/rollout.py", line 139, in _ode_rollout_batch
46
+ outputs = dit(x_list, t=ts_batch, context=context_list, seq_len=seq_len)
47
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
48
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl
49
+ return self._call_impl(*args, **kwargs)
50
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
51
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl
52
+ return forward_call(*args, **kwargs)
53
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
54
+ File "/home/tiger/.local/lib/python3.11/site-packages/peft/peft_model.py", line 812, in forward
55
+ return self.get_base_model()(*args, **kwargs)
56
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
57
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl
58
+ return self._call_impl(*args, **kwargs)
59
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
60
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl
61
+ return forward_call(*args, **kwargs)
62
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
63
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/models/wan/modules/model.py", line 498, in forward
64
+ x = block(x, **kwargs)
65
+ ^^^^^^^^^^^^^^^^^^
66
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl
67
+ return self._call_impl(*args, **kwargs)
68
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
69
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl
70
+ return forward_call(*args, **kwargs)
71
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
72
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 126, in _ckpt_fwd
73
+ return torch_ckpt.checkpoint(orig_fwd, *args, use_reentrant=False, **kwargs)
74
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
75
+ File "/usr/local/lib/python3.11/dist-packages/torch/_compile.py", line 53, in inner
76
+ return disable_fn(*args, **kwargs)
77
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
78
+ File "/usr/local/lib/python3.11/dist-packages/torch/_dynamo/eval_frame.py", line 1044, in _fn
79
+ return fn(*args, **kwargs)
80
+ ^^^^^^^^^^^^^^^^^^^
81
+ File "/usr/local/lib/python3.11/dist-packages/torch/utils/checkpoint.py", line 503, in checkpoint
82
+ ret = function(*args, **kwargs)
83
+ ^^^^^^^^^^^^^^^^^^^^^^^^^
84
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/models/wan/modules/model.py", line 241, in forward
85
+ e = (self.modulation.unsqueeze(0) + e).chunk(6, dim=2)
86
+ ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^~~
87
+ torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 31.33 GiB. GPU 0 has a total capacity of 79.11 GiB of which 4.63 GiB is free. Process 3796 has 0 bytes memory in use. Process 8629 has 0 bytes memory in use. Including non-PyTorch memory, this process has 0 bytes memory in use. Of the allocated memory 66.44 GiB is allocated by PyTorch, and 5.42 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
88
+ Traceback (most recent call last):
89
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1034, in <module>
90
+ main()
91
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 866, in main
92
+ rollouts = creflow_rollout(
93
+ ^^^^^^^^^^^^^^^^
94
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/creflow/rollout.py", line 267, in creflow_rollout
95
+ x0_preds = _ode_rollout_batch(
96
+ ^^^^^^^^^^^^^^^^^^^
97
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/creflow/rollout.py", line 139, in _ode_rollout_batch
98
+ outputs = dit(x_list, t=ts_batch, context=context_list, seq_len=seq_len)
99
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
100
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl
101
+ return self._call_impl(*args, **kwargs)
102
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
103
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl
104
+ return forward_call(*args, **kwargs)
105
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
106
+ File "/home/tiger/.local/lib/python3.11/site-packages/peft/peft_model.py", line 812, in forward
107
+ return self.get_base_model()(*args, **kwargs)
108
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
109
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl
110
+ return self._call_impl(*args, **kwargs)
111
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
112
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl
113
+ return forward_call(*args, **kwargs)
114
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
115
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/models/wan/modules/model.py", line 498, in forward
116
+ x = block(x, **kwargs)
117
+ ^^^^^^^^^^^^^^^^^^
118
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl
119
+ return self._call_impl(*args, **kwargs)
120
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
121
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl
122
+ return forward_call(*args, **kwargs)
123
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
124
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 126, in _ckpt_fwd
125
+ return torch_ckpt.checkpoint(orig_fwd, *args, use_reentrant=False, **kwargs)
126
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
127
+ File "/usr/local/lib/python3.11/dist-packages/torch/_compile.py", line 53, in inner
128
+ return disable_fn(*args, **kwargs)
129
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
130
+ File "/usr/local/lib/python3.11/dist-packages/torch/_dynamo/eval_frame.py", line 1044, in _fn
131
+ return fn(*args, **kwargs)
132
+ ^^^^^^^^^^^^^^^^^^^
133
+ File "/usr/local/lib/python3.11/dist-packages/torch/utils/checkpoint.py", line 503, in checkpoint
134
+ ret = function(*args, **kwargs)
135
+ ^^^^^^^^^^^^^^^^^^^^^^^^^
136
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/models/wan/modules/model.py", line 241, in forward
137
+ e = (self.modulation.unsqueeze(0) + e).chunk(6, dim=2)
138
+ ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~^~~
139
+ torch.OutOfMemoryError: CUDA out of memory. Tried to allocate 31.33 GiB. GPU 0 has a total capacity of 79.11 GiB of which 4.63 GiB is free. Process 3796 has 0 bytes memory in use. Process 8629 has 0 bytes memory in use. Including non-PyTorch memory, this process has 0 bytes memory in use. Of the allocated memory 66.44 GiB is allocated by PyTorch, and 5.42 GiB is reserved by PyTorch but unallocated. If reserved but unallocated memory is large try setting PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True to avoid fragmentation. See documentation for Memory Management (https://pytorch.org/docs/stable/notes/cuda.html#environment-variables)
wandb/run-20260408_000908-i6zi4vwm/files/wandb-metadata.json ADDED
@@ -0,0 +1,144 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-5.15.120.bsk.6-amd64-x86_64-with-glibc2.36",
3
+ "python": "3.11.2",
4
+ "startedAt": "2026-04-08T00:09:08.281832Z",
5
+ "args": [
6
+ "--task",
7
+ "ti2v-5B",
8
+ "--size",
9
+ "640*736",
10
+ "--frame_num",
11
+ "121",
12
+ "--ckpt_dir",
13
+ "ckpts/Wan2.2-TI2V-5B",
14
+ "--pt_dir",
15
+ "ckpts/vidar_ckpt/merged_vidar_lora.pt",
16
+ "--dataset_json",
17
+ "data/rl_train/robotwin_stack_blocks_three.json",
18
+ "--output_dir",
19
+ "data/outputs/creflow_stack_blocks_three",
20
+ "--num_ode_steps",
21
+ "20",
22
+ "--sample_shift",
23
+ "5.0",
24
+ "--sample_guide_scale",
25
+ "5.0",
26
+ "--K",
27
+ "16",
28
+ "--num_rounds",
29
+ "10",
30
+ "--epochs_per_round",
31
+ "100",
32
+ "--max_per_condition",
33
+ "64",
34
+ "--effective_batch_size",
35
+ "16",
36
+ "--w_bad",
37
+ "1.0",
38
+ "--lambda_gripper",
39
+ "1.0",
40
+ "--seed",
41
+ "42",
42
+ "--reward_config",
43
+ "blocks_stack_v2",
44
+ "--convert_model_dtype",
45
+ "--offload_model",
46
+ "false",
47
+ "--learning_rate",
48
+ "1e-5",
49
+ "--weight_decay",
50
+ "0.01",
51
+ "--max_grad_norm",
52
+ "2.0",
53
+ "--checkpointing_steps",
54
+ "10",
55
+ "--lora_rank",
56
+ "64",
57
+ "--lora_alpha",
58
+ "64",
59
+ "--wandb_project",
60
+ "Corrective-Reflow",
61
+ "--wandb_run_name",
62
+ "creflow_stack_blocks_three"
63
+ ],
64
+ "program": "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py",
65
+ "codePath": "fastvideo/train_creflow_wan_2_2_ti2v.py",
66
+ "git": {
67
+ "remote": "git@github.com:VincentNi0107/EmbodiedVideoRL.git",
68
+ "commit": "912b225d838164e0178805670881ee201ad81ff0"
69
+ },
70
+ "email": "1163051845@qq.com",
71
+ "root": "data/outputs/creflow_stack_blocks_three",
72
+ "host": "n124-104-180",
73
+ "username": "tiger",
74
+ "executable": "/usr/bin/python",
75
+ "codePathLocal": "fastvideo/train_creflow_wan_2_2_ti2v.py",
76
+ "cpu_count": 104,
77
+ "cpu_count_logical": 208,
78
+ "gpu": "NVIDIA H100 80GB HBM3",
79
+ "gpu_count": 8,
80
+ "disk": {
81
+ "/": {
82
+ "total": "1055740600320",
83
+ "used": "477545373696"
84
+ }
85
+ },
86
+ "memory": {
87
+ "total": "1977660338176"
88
+ },
89
+ "cpu": {
90
+ "count": 104,
91
+ "countLogical": 208
92
+ },
93
+ "gpu_nvidia": [
94
+ {
95
+ "name": "NVIDIA H100 80GB HBM3",
96
+ "memoryTotal": "85520809984",
97
+ "cudaCores": 16896,
98
+ "architecture": "Hopper"
99
+ },
100
+ {
101
+ "name": "NVIDIA H100 80GB HBM3",
102
+ "memoryTotal": "85520809984",
103
+ "cudaCores": 16896,
104
+ "architecture": "Hopper"
105
+ },
106
+ {
107
+ "name": "NVIDIA H100 80GB HBM3",
108
+ "memoryTotal": "85520809984",
109
+ "cudaCores": 16896,
110
+ "architecture": "Hopper"
111
+ },
112
+ {
113
+ "name": "NVIDIA H100 80GB HBM3",
114
+ "memoryTotal": "85520809984",
115
+ "cudaCores": 16896,
116
+ "architecture": "Hopper"
117
+ },
118
+ {
119
+ "name": "NVIDIA H100 80GB HBM3",
120
+ "memoryTotal": "85520809984",
121
+ "cudaCores": 16896,
122
+ "architecture": "Hopper"
123
+ },
124
+ {
125
+ "name": "NVIDIA H100 80GB HBM3",
126
+ "memoryTotal": "85520809984",
127
+ "cudaCores": 16896,
128
+ "architecture": "Hopper"
129
+ },
130
+ {
131
+ "name": "NVIDIA H100 80GB HBM3",
132
+ "memoryTotal": "85520809984",
133
+ "cudaCores": 16896,
134
+ "architecture": "Hopper"
135
+ },
136
+ {
137
+ "name": "NVIDIA H100 80GB HBM3",
138
+ "memoryTotal": "85520809984",
139
+ "cudaCores": 16896,
140
+ "architecture": "Hopper"
141
+ }
142
+ ],
143
+ "cudaVersion": "12.9"
144
+ }
wandb/run-20260408_000908-i6zi4vwm/files/wandb-summary.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"_wandb":{"runtime":121}}
wandb/run-20260408_000908-i6zi4vwm/logs/debug.log ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 2026-04-08 00:09:08,247 INFO MainThread:17550 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5
2
+ 2026-04-08 00:09:08,247 INFO MainThread:17550 [wandb_setup.py:_flush():79] Configure stats pid to 17550
3
+ 2026-04-08 00:09:08,247 INFO MainThread:17550 [wandb_setup.py:_flush():79] Loading settings from /home/tiger/.config/wandb/settings
4
+ 2026-04-08 00:09:08,247 INFO MainThread:17550 [wandb_setup.py:_flush():79] Loading settings from /mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/wandb/settings
5
+ 2026-04-08 00:09:08,247 INFO MainThread:17550 [wandb_setup.py:_flush():79] Loading settings from environment variables: {'disabled': 'false', 'project': 'Corrective-Reflow', 'api_key': '***REDACTED***', 'mode': 'online'}
6
+ 2026-04-08 00:09:08,247 INFO MainThread:17550 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': 'online', '_disable_service': None}
7
+ 2026-04-08 00:09:08,248 INFO MainThread:17550 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'fastvideo/train_creflow_wan_2_2_ti2v.py', 'program_abspath': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py', 'program': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py'}
8
+ 2026-04-08 00:09:08,248 INFO MainThread:17550 [wandb_setup.py:_flush():79] Applying login settings: {}
9
+ 2026-04-08 00:09:08,249 INFO MainThread:17550 [wandb_init.py:_log_setup():534] Logging user logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_000908-i6zi4vwm/logs/debug.log
10
+ 2026-04-08 00:09:08,251 INFO MainThread:17550 [wandb_init.py:_log_setup():535] Logging internal logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_000908-i6zi4vwm/logs/debug-internal.log
11
+ 2026-04-08 00:09:08,251 INFO MainThread:17550 [wandb_init.py:init():621] calling init triggers
12
+ 2026-04-08 00:09:08,251 INFO MainThread:17550 [wandb_init.py:init():628] wandb.init called with sweep_config: {}
13
+ config: {'vidar_root': '', 'task': 'ti2v-5B', 'size': '640*736', 'frame_num': 121, 'ckpt_dir': 'ckpts/Wan2.2-TI2V-5B', 'pt_dir': 'ckpts/vidar_ckpt/merged_vidar_lora.pt', 'convert_model_dtype': True, 'offload_model': False, 'dataset_json': 'data/rl_train/robotwin_stack_blocks_three.json', 'max_samples': -1, 'output_dir': 'data/outputs/creflow_stack_blocks_three', 'num_ode_steps': 20, 'sample_shift': 5.0, 'sample_guide_scale': 5.0, 'neg_prompt': '色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走', 'seed': 42, 'K': 16, 'num_rounds': 10, 'epochs_per_round': 100, 'effective_batch_size': 16, 'w_bad': 1.0, 'max_per_condition': 64, 'lambda_gripper': 1.0, 'reset_optimizer_per_round': False, 'num_train_timesteps': 1000, 'learning_rate': 1e-05, 'weight_decay': 0.01, 'max_grad_norm': 2.0, 'checkpointing_steps': 10, 'gradient_checkpointing': True, 'use_8bit_adam': True, 'lora_rank': 64, 'lora_alpha': 64, 'lora_target_modules': None, 'resume_from_lora_checkpoint': None, 'reward_config': 'blocks_stack_v2', 'reward_backend': 'blocks_stack_v2', 'skip_reward_debug_video': True, 'wandb_project': 'Corrective-Reflow', 'wandb_run_name': 'creflow_stack_blocks_three', 'hallucination_crop_top_ratio': 0.6667, 'bsv2_prompts': ['red block', 'green block', 'blue block'], 'bsv2_pick_thr': 0.05, 'bsv2_place_thr': 0.05, 'bsv2_vertical_sep_thr': 0.03, 'bsv2_x_align_thr': 0.025, 'bsv2_check_window_frac': 0.2, 'bsv2_dup_max_frames': 0, 'bsv2_expected_grip_changes': 6, 'bsv2_mj_lo': 0.6, 'bsv2_mj_hi': 1.3, 'bsv2_idm_ckpt_path': 'ckpts/vidar_ckpt/idm.pt'}
14
+ 2026-04-08 00:09:08,251 INFO MainThread:17550 [wandb_init.py:init():671] starting backend
15
+ 2026-04-08 00:09:08,251 INFO MainThread:17550 [wandb_init.py:init():675] sending inform_init request
16
+ 2026-04-08 00:09:08,279 INFO MainThread:17550 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
17
+ 2026-04-08 00:09:08,279 INFO MainThread:17550 [wandb_init.py:init():688] backend started and connected
18
+ 2026-04-08 00:09:08,337 INFO MainThread:17550 [wandb_init.py:init():783] updated telemetry
19
+ 2026-04-08 00:09:08,505 INFO MainThread:17550 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout
20
+ 2026-04-08 00:09:08,884 INFO MainThread:17550 [wandb_init.py:init():867] starting run threads in backend
21
+ 2026-04-08 00:09:09,247 INFO MainThread:17550 [wandb_run.py:_console_start():2463] atexit reg
22
+ 2026-04-08 00:09:09,247 INFO MainThread:17550 [wandb_run.py:_redirect():2311] redirect: wrap_raw
23
+ 2026-04-08 00:09:09,247 INFO MainThread:17550 [wandb_run.py:_redirect():2376] Wrapping output streams.
24
+ 2026-04-08 00:09:09,247 INFO MainThread:17550 [wandb_run.py:_redirect():2401] Redirects installed.
25
+ 2026-04-08 00:09:09,252 INFO MainThread:17550 [wandb_init.py:init():911] run started, returning control to user process
26
+ 2026-04-08 00:11:09,780 WARNING MsgRouterThr:17550 [router.py:message_loop():77] message_loop has been closed
wandb/run-20260408_001541-g1xdtpwt/files/output.log ADDED
@@ -0,0 +1,286 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Building Wan2.2 TI2V model ... (DDP=True, world_size=8)
2
+ /usr/local/lib/python3.11/dist-packages/torch/distributed/distributed_c10d.py:4876: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
3
+ warnings.warn( # warn only once
4
+ Moving T5 encoder to GPU permanently (offload_model=false) ...
5
+ Building reward scorer ...
6
+ Loading IDM from ckpts/vidar_ckpt/idm.pt on cuda:0 ...
7
+ IDM loaded.
8
+ Loading SAM3 VideoPredictor (blocks_stack_v2) on cuda:0 ...
9
+ INFO 2026-04-08 00:16:58,025 18968 sam3_video_base.py: 125: setting max_num_objects=10000 and num_obj_for_compile=16
10
+ SAM3 VideoPredictor (blocks_stack_v2) loaded.
11
+ Applying LoRA (single adapter) for CReflow ...
12
+ LoRA injected: rank=64, alpha=64, target_modules=['self_attn.q', 'self_attn.k', 'self_attn.v', 'self_attn.o', 'cross_attn.q', 'cross_attn.k', 'cross_attn.v', 'cross_attn.o']
13
+ Trainable: 94.4 M / 5094.2 M total (1.85%)
14
+ Gradient checkpointing enabled on 30 DiT blocks
15
+ Trainable parameters: 94.4 M
16
+ Using 8-bit AdamW (bitsandbytes)
17
+ Dataset: 10 conditions
18
+ Encoding 10 conditions ...
19
+ [1/10] 6b20973f10ef... encoded
20
+ [2/10] cff33035b28a... encoded
21
+ [3/10] 7ebc8b336779... encoded
22
+ [4/10] 8077c679fb62... encoded
23
+ [5/10] 41855438a8e0... encoded
24
+ [6/10] a67b02f4e7ce... encoded
25
+ [7/10] 4568a9603f3b... encoded
26
+ [8/10] 1b67584bff87... encoded
27
+ [9/10] f4f8079f6637... encoded
28
+ [10/10] 1b70247c6412... encoded
29
+ All 10 conditions encoded.
30
+
31
+ Starting Corrective Reflow | rounds=10 K=16 K_local=2 epochs/round=100 effective_bs=16 lr=1e-05 conditions=10
32
+
33
+ ======================================================================
34
+ ROUND 1/10
35
+ ======================================================================
36
+ [Phase 1] ODE rollout ...
37
+ [CReflow rollout] condition 0/10: K=2, id=6b20973f10ef...
38
+ rollout: 2 videos in 56.5s
39
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.17it/s]
40
+ propagate_in_video: 100%|██████████| 121/121 [00:12<00:00, 9.53it/s]
41
+ propagate_in_video: 0it [00:00, ?it/s]
42
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.55it/s]
43
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.19it/s]
44
+ propagate_in_video: 0it [00:00, ?it/s]
45
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.09it/s]
46
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.96it/s]
47
+ propagate_in_video: 0it [00:00, ?it/s]
48
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 72.76it/s]
49
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.60it/s]
50
+ propagate_in_video: 0it [00:00, ?it/s]
51
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.24it/s]
52
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.41it/s]
53
+ propagate_in_video: 0it [00:00, ?it/s]
54
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.98it/s]
55
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.12it/s]
56
+ propagate_in_video: 0it [00:00, ?it/s]
57
+ [CReflow rollout] condition 1/10: K=2, id=cff33035b28a...
58
+ rollout: 2 videos in 56.9s
59
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.44it/s]
60
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.01it/s]
61
+ propagate_in_video: 0it [00:00, ?it/s]
62
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.92it/s]
63
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.90it/s]
64
+ propagate_in_video: 0it [00:00, ?it/s]
65
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.73it/s]
66
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.73it/s]
67
+ propagate_in_video: 0it [00:00, ?it/s]
68
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.54it/s]
69
+ propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.75it/s]
70
+ propagate_in_video: 0it [00:00, ?it/s]
71
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.42it/s]
72
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.50it/s]
73
+ propagate_in_video: 0it [00:00, ?it/s]
74
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.13it/s]
75
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.15it/s]
76
+ propagate_in_video: 0it [00:00, ?it/s]
77
+ [CReflow rollout] condition 2/10: K=2, id=7ebc8b336779...
78
+ rollout: 2 videos in 57.5s
79
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.66it/s]
80
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.60it/s]
81
+ propagate_in_video: 0it [00:00, ?it/s]
82
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.08it/s]
83
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.02it/s]
84
+ propagate_in_video: 0it [00:00, ?it/s]
85
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.85it/s]
86
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.32it/s]
87
+ propagate_in_video: 0it [00:00, ?it/s]
88
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.15it/s]
89
+ propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.05it/s]
90
+ propagate_in_video: 0it [00:00, ?it/s]
91
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.33it/s]
92
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.21it/s]
93
+ propagate_in_video: 0it [00:00, ?it/s]
94
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.75it/s]
95
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.09it/s]
96
+ propagate_in_video: 0it [00:00, ?it/s]
97
+ [CReflow rollout] condition 3/10: K=2, id=8077c679fb62...
98
+ rollout: 2 videos in 57.6s
99
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.11it/s]
100
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.56it/s]
101
+ propagate_in_video: 0it [00:00, ?it/s]
102
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.25it/s]
103
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.81it/s]
104
+ propagate_in_video: 0it [00:00, ?it/s]
105
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 85.87it/s]
106
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.61it/s]
107
+ propagate_in_video: 0it [00:00, ?it/s]
108
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.69it/s]
109
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.20it/s]
110
+ propagate_in_video: 0it [00:00, ?it/s]
111
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.56it/s]
112
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.94it/s]
113
+ propagate_in_video: 0it [00:00, ?it/s]
114
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.82it/s]
115
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.03it/s]
116
+ propagate_in_video: 0it [00:00, ?it/s]
117
+ [CReflow rollout] condition 4/10: K=2, id=41855438a8e0...
118
+ rollout: 2 videos in 57.1s
119
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.83it/s]
120
+ propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.61it/s]
121
+ propagate_in_video: 0it [00:00, ?it/s]
122
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 77.09it/s]
123
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.07it/s]
124
+ propagate_in_video: 0it [00:00, ?it/s]
125
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.11it/s]
126
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.79it/s]
127
+ propagate_in_video: 0it [00:00, ?it/s]
128
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.75it/s]
129
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.69it/s]
130
+ propagate_in_video: 0it [00:00, ?it/s]
131
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.43it/s]
132
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.37it/s]
133
+ propagate_in_video: 0it [00:00, ?it/s]
134
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.28it/s]
135
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.71it/s]
136
+ propagate_in_video: 0it [00:00, ?it/s]
137
+ [CReflow rollout] condition 5/10: K=2, id=a67b02f4e7ce...
138
+ rollout: 2 videos in 57.1s
139
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.96it/s]
140
+ propagate_in_video: 100%|██████████| 121/121 [00:11<00:00, 10.48it/s]
141
+ propagate_in_video: 0it [00:00, ?it/s]
142
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.64it/s]
143
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.48it/s]
144
+ propagate_in_video: 0it [00:00, ?it/s]
145
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.33it/s]
146
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.43it/s]
147
+ propagate_in_video: 0it [00:00, ?it/s]
148
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 77.50it/s]
149
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.21it/s]
150
+ propagate_in_video: 0it [00:00, ?it/s]
151
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.01it/s]
152
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.30it/s]
153
+ propagate_in_video: 0it [00:00, ?it/s]
154
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.11it/s]
155
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.23it/s]
156
+ propagate_in_video: 0it [00:00, ?it/s]
157
+ [CReflow rollout] condition 6/10: K=2, id=4568a9603f3b...
158
+ rollout: 2 videos in 57.7s
159
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.97it/s]
160
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.58it/s]
161
+ propagate_in_video: 0it [00:00, ?it/s]
162
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.65it/s]
163
+ propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.75it/s]
164
+ propagate_in_video: 0it [00:00, ?it/s]
165
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.19it/s]
166
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.97it/s]
167
+ propagate_in_video: 0it [00:00, ?it/s]
168
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.75it/s]
169
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.32it/s]
170
+ propagate_in_video: 0it [00:00, ?it/s]
171
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.90it/s]
172
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.36it/s]
173
+ propagate_in_video: 0it [00:00, ?it/s]
174
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.29it/s]
175
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.29it/s]
176
+ propagate_in_video: 0it [00:00, ?it/s]
177
+ [CReflow rollout] condition 7/10: K=2, id=1b67584bff87...
178
+ rollout: 2 videos in 57.4s
179
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.97it/s]
180
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.02it/s]
181
+ propagate_in_video: 0it [00:00, ?it/s]
182
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 76.90it/s]
183
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.12it/s]
184
+ propagate_in_video: 0it [00:00, ?it/s]
185
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.20it/s]
186
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.42it/s]
187
+ propagate_in_video: 0it [00:00, ?it/s]
188
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.36it/s]
189
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.12it/s]
190
+ propagate_in_video: 0it [00:00, ?it/s]
191
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.63it/s]
192
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.37it/s]
193
+ propagate_in_video: 0it [00:00, ?it/s]
194
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.10it/s]
195
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.13it/s]
196
+ propagate_in_video: 0it [00:00, ?it/s]
197
+ [CReflow rollout] condition 8/10: K=2, id=f4f8079f6637...
198
+ rollout: 2 videos in 57.6s
199
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.98it/s]
200
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.02it/s]
201
+ propagate_in_video: 0it [00:00, ?it/s]
202
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.98it/s]
203
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.34it/s]
204
+ propagate_in_video: 0it [00:00, ?it/s]
205
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.75it/s]
206
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.74it/s]
207
+ propagate_in_video: 0it [00:00, ?it/s]
208
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.41it/s]
209
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.15it/s]
210
+ propagate_in_video: 0it [00:00, ?it/s]
211
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.98it/s]
212
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.29it/s]
213
+ propagate_in_video: 0it [00:00, ?it/s]
214
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.62it/s]
215
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.98it/s]
216
+ propagate_in_video: 0it [00:00, ?it/s]
217
+ [CReflow rollout] condition 9/10: K=2, id=1b70247c6412...
218
+ rollout: 2 videos in 57.6s
219
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.88it/s]
220
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.21it/s]
221
+ propagate_in_video: 0it [00:00, ?it/s]
222
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 85.68it/s]
223
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.16it/s]
224
+ propagate_in_video: 0it [00:00, ?it/s]
225
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.28it/s]
226
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.22it/s]
227
+ propagate_in_video: 0it [00:00, ?it/s]
228
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.94it/s]
229
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.26it/s]
230
+ propagate_in_video: 0it [00:00, ?it/s]
231
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.23it/s]
232
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.00it/s]
233
+ propagate_in_video: 0it [00:00, ?it/s]
234
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.12it/s]
235
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.45it/s]
236
+ propagate_in_video: 0it [00:00, ?it/s]
237
+ [CReflow rollout] done: 20 samples, rewards = [1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 0.00, 0.00]
238
+ Rollout done: 20 samples, good=18, bad=2, avg_reward=0.900
239
+ 1b67584bff87: mean=1.00, good=2/2
240
+ 1b70247c6412: mean=0.00, good=0/2
241
+ 41855438a8e0: mean=1.00, good=2/2
242
+ 4568a9603f3b: mean=1.00, good=2/2
243
+ 6b20973f10ef: mean=1.00, good=2/2
244
+ 7ebc8b336779: mean=1.00, good=2/2
245
+ 8077c679fb62: mean=1.00, good=2/2
246
+ a67b02f4e7ce: mean=1.00, good=2/2
247
+ cff33035b28a: mean=1.00, good=2/2
248
+ f4f8079f6637: mean=1.00, good=2/2
249
+ [Phase 2] Library update ...
250
+ [Library] rank=0 has 18 good samples
251
+ [Library] after update: SuccessLibrary(total=127, coverage=10/10, max_per_condition=64)
252
+ [Phase 3] Re-pairing ...
253
+ [Re-pair] train_set: 20 samples (anchor=18, correction=2, skipped=0)
254
+ [Phase 4] Training (100 epochs, 20 samples) ...
255
+ Traceback (most recent call last):
256
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1034, in <module>
257
+ main()
258
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 934, in main
259
+ train_metrics = train_one_round(
260
+ ^^^^^^^^^^^^^^^^
261
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 574, in train_one_round
262
+ loss.backward()
263
+ File "/usr/local/lib/python3.11/dist-packages/torch/_tensor.py", line 625, in backward
264
+ torch.autograd.backward(
265
+ File "/usr/local/lib/python3.11/dist-packages/torch/autograd/__init__.py", line 354, in backward
266
+ _engine_run_backward(
267
+ File "/usr/local/lib/python3.11/dist-packages/torch/autograd/graph.py", line 841, in _engine_run_backward
268
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
269
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
270
+ RuntimeError: element 0 of tensors does not require grad and does not have a grad_fn
271
+ [rank0]: Traceback (most recent call last):
272
+ [rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1034, in <module>
273
+ [rank0]: main()
274
+ [rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 934, in main
275
+ [rank0]: train_metrics = train_one_round(
276
+ [rank0]: ^^^^^^^^^^^^^^^^
277
+ [rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 574, in train_one_round
278
+ [rank0]: loss.backward()
279
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/_tensor.py", line 625, in backward
280
+ [rank0]: torch.autograd.backward(
281
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/autograd/__init__.py", line 354, in backward
282
+ [rank0]: _engine_run_backward(
283
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/autograd/graph.py", line 841, in _engine_run_backward
284
+ [rank0]: return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
285
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
286
+ [rank0]: RuntimeError: element 0 of tensors does not require grad and does not have a grad_fn
wandb/run-20260408_001541-g1xdtpwt/files/wandb-metadata.json ADDED
@@ -0,0 +1,144 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-5.15.120.bsk.6-amd64-x86_64-with-glibc2.36",
3
+ "python": "3.11.2",
4
+ "startedAt": "2026-04-08T00:15:41.604548Z",
5
+ "args": [
6
+ "--task",
7
+ "ti2v-5B",
8
+ "--size",
9
+ "640*736",
10
+ "--frame_num",
11
+ "121",
12
+ "--ckpt_dir",
13
+ "ckpts/Wan2.2-TI2V-5B",
14
+ "--pt_dir",
15
+ "ckpts/vidar_ckpt/merged_vidar_lora.pt",
16
+ "--dataset_json",
17
+ "data/rl_train/robotwin_stack_blocks_three.json",
18
+ "--output_dir",
19
+ "data/outputs/creflow_stack_blocks_three",
20
+ "--num_ode_steps",
21
+ "20",
22
+ "--sample_shift",
23
+ "5.0",
24
+ "--sample_guide_scale",
25
+ "5.0",
26
+ "--K",
27
+ "16",
28
+ "--num_rounds",
29
+ "10",
30
+ "--epochs_per_round",
31
+ "100",
32
+ "--max_per_condition",
33
+ "64",
34
+ "--effective_batch_size",
35
+ "16",
36
+ "--w_bad",
37
+ "1.0",
38
+ "--lambda_gripper",
39
+ "1.0",
40
+ "--seed",
41
+ "42",
42
+ "--reward_config",
43
+ "blocks_stack_v2",
44
+ "--convert_model_dtype",
45
+ "--offload_model",
46
+ "false",
47
+ "--learning_rate",
48
+ "1e-5",
49
+ "--weight_decay",
50
+ "0.01",
51
+ "--max_grad_norm",
52
+ "2.0",
53
+ "--checkpointing_steps",
54
+ "10",
55
+ "--lora_rank",
56
+ "64",
57
+ "--lora_alpha",
58
+ "64",
59
+ "--wandb_project",
60
+ "Corrective-Reflow",
61
+ "--wandb_run_name",
62
+ "creflow_stack_blocks_three"
63
+ ],
64
+ "program": "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py",
65
+ "codePath": "fastvideo/train_creflow_wan_2_2_ti2v.py",
66
+ "git": {
67
+ "remote": "git@github.com:VincentNi0107/EmbodiedVideoRL.git",
68
+ "commit": "912b225d838164e0178805670881ee201ad81ff0"
69
+ },
70
+ "email": "1163051845@qq.com",
71
+ "root": "data/outputs/creflow_stack_blocks_three",
72
+ "host": "n124-104-180",
73
+ "username": "tiger",
74
+ "executable": "/usr/bin/python",
75
+ "codePathLocal": "fastvideo/train_creflow_wan_2_2_ti2v.py",
76
+ "cpu_count": 104,
77
+ "cpu_count_logical": 208,
78
+ "gpu": "NVIDIA H100 80GB HBM3",
79
+ "gpu_count": 8,
80
+ "disk": {
81
+ "/": {
82
+ "total": "1055740600320",
83
+ "used": "481117626368"
84
+ }
85
+ },
86
+ "memory": {
87
+ "total": "1977660338176"
88
+ },
89
+ "cpu": {
90
+ "count": 104,
91
+ "countLogical": 208
92
+ },
93
+ "gpu_nvidia": [
94
+ {
95
+ "name": "NVIDIA H100 80GB HBM3",
96
+ "memoryTotal": "85520809984",
97
+ "cudaCores": 16896,
98
+ "architecture": "Hopper"
99
+ },
100
+ {
101
+ "name": "NVIDIA H100 80GB HBM3",
102
+ "memoryTotal": "85520809984",
103
+ "cudaCores": 16896,
104
+ "architecture": "Hopper"
105
+ },
106
+ {
107
+ "name": "NVIDIA H100 80GB HBM3",
108
+ "memoryTotal": "85520809984",
109
+ "cudaCores": 16896,
110
+ "architecture": "Hopper"
111
+ },
112
+ {
113
+ "name": "NVIDIA H100 80GB HBM3",
114
+ "memoryTotal": "85520809984",
115
+ "cudaCores": 16896,
116
+ "architecture": "Hopper"
117
+ },
118
+ {
119
+ "name": "NVIDIA H100 80GB HBM3",
120
+ "memoryTotal": "85520809984",
121
+ "cudaCores": 16896,
122
+ "architecture": "Hopper"
123
+ },
124
+ {
125
+ "name": "NVIDIA H100 80GB HBM3",
126
+ "memoryTotal": "85520809984",
127
+ "cudaCores": 16896,
128
+ "architecture": "Hopper"
129
+ },
130
+ {
131
+ "name": "NVIDIA H100 80GB HBM3",
132
+ "memoryTotal": "85520809984",
133
+ "cudaCores": 16896,
134
+ "architecture": "Hopper"
135
+ },
136
+ {
137
+ "name": "NVIDIA H100 80GB HBM3",
138
+ "memoryTotal": "85520809984",
139
+ "cudaCores": 16896,
140
+ "architecture": "Hopper"
141
+ }
142
+ ],
143
+ "cudaVersion": "12.9"
144
+ }
wandb/run-20260408_001541-g1xdtpwt/files/wandb-summary.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"_wandb":{"runtime":1601}}
wandb/run-20260408_001541-g1xdtpwt/logs/debug-internal.log ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-04-08T00:15:41.614211928Z","level":"INFO","msg":"using version","core version":"0.18.5"}
2
+ {"time":"2026-04-08T00:15:41.614246115Z","level":"INFO","msg":"created symlink","path":"data/outputs/creflow_stack_blocks_three/wandb/run-20260408_001541-g1xdtpwt/logs/debug-core.log"}
3
+ {"time":"2026-04-08T00:15:41.932223148Z","level":"INFO","msg":"created new stream","id":"g1xdtpwt"}
4
+ {"time":"2026-04-08T00:15:41.932414446Z","level":"INFO","msg":"stream: started","id":"g1xdtpwt"}
5
+ {"time":"2026-04-08T00:15:41.932510608Z","level":"INFO","msg":"sender: started","stream_id":"g1xdtpwt"}
6
+ {"time":"2026-04-08T00:15:41.932482435Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"g1xdtpwt"}}
7
+ {"time":"2026-04-08T00:15:41.932549715Z","level":"INFO","msg":"handler: started","stream_id":{"value":"g1xdtpwt"}}
8
+ {"time":"2026-04-08T00:15:42.915509387Z","level":"INFO","msg":"Starting system monitor"}
9
+ {"time":"2026-04-08T00:42:23.439351832Z","level":"INFO","msg":"stream: closing","id":"g1xdtpwt"}
10
+ {"time":"2026-04-08T00:42:23.439437558Z","level":"INFO","msg":"Stopping system monitor"}
11
+ {"time":"2026-04-08T00:42:23.445282629Z","level":"INFO","msg":"Stopped system monitor"}
12
+ {"time":"2026-04-08T00:42:24.081303366Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
13
+ {"time":"2026-04-08T00:42:24.367520127Z","level":"INFO","msg":"handler: closed","stream_id":{"value":"g1xdtpwt"}}
14
+ {"time":"2026-04-08T00:42:24.367585187Z","level":"INFO","msg":"writer: Close: closed","stream_id":{"value":"g1xdtpwt"}}
15
+ {"time":"2026-04-08T00:42:24.367601446Z","level":"INFO","msg":"sender: closed","stream_id":"g1xdtpwt"}
16
+ {"time":"2026-04-08T00:42:24.371256063Z","level":"INFO","msg":"stream: closed","id":"g1xdtpwt"}
wandb/run-20260408_001541-g1xdtpwt/logs/debug.log ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 2026-04-08 00:15:41,570 INFO MainThread:18968 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5
2
+ 2026-04-08 00:15:41,571 INFO MainThread:18968 [wandb_setup.py:_flush():79] Configure stats pid to 18968
3
+ 2026-04-08 00:15:41,571 INFO MainThread:18968 [wandb_setup.py:_flush():79] Loading settings from /home/tiger/.config/wandb/settings
4
+ 2026-04-08 00:15:41,571 INFO MainThread:18968 [wandb_setup.py:_flush():79] Loading settings from /mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/wandb/settings
5
+ 2026-04-08 00:15:41,571 INFO MainThread:18968 [wandb_setup.py:_flush():79] Loading settings from environment variables: {'disabled': 'false', 'project': 'Corrective-Reflow', 'api_key': '***REDACTED***', 'mode': 'online'}
6
+ 2026-04-08 00:15:41,571 INFO MainThread:18968 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': 'online', '_disable_service': None}
7
+ 2026-04-08 00:15:41,571 INFO MainThread:18968 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'fastvideo/train_creflow_wan_2_2_ti2v.py', 'program_abspath': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py', 'program': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py'}
8
+ 2026-04-08 00:15:41,571 INFO MainThread:18968 [wandb_setup.py:_flush():79] Applying login settings: {}
9
+ 2026-04-08 00:15:41,573 INFO MainThread:18968 [wandb_init.py:_log_setup():534] Logging user logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_001541-g1xdtpwt/logs/debug.log
10
+ 2026-04-08 00:15:41,575 INFO MainThread:18968 [wandb_init.py:_log_setup():535] Logging internal logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_001541-g1xdtpwt/logs/debug-internal.log
11
+ 2026-04-08 00:15:41,575 INFO MainThread:18968 [wandb_init.py:init():621] calling init triggers
12
+ 2026-04-08 00:15:41,575 INFO MainThread:18968 [wandb_init.py:init():628] wandb.init called with sweep_config: {}
13
+ config: {'vidar_root': '', 'task': 'ti2v-5B', 'size': '640*736', 'frame_num': 121, 'ckpt_dir': 'ckpts/Wan2.2-TI2V-5B', 'pt_dir': 'ckpts/vidar_ckpt/merged_vidar_lora.pt', 'convert_model_dtype': True, 'offload_model': False, 'dataset_json': 'data/rl_train/robotwin_stack_blocks_three.json', 'max_samples': -1, 'output_dir': 'data/outputs/creflow_stack_blocks_three', 'num_ode_steps': 20, 'sample_shift': 5.0, 'sample_guide_scale': 5.0, 'neg_prompt': '色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走', 'seed': 42, 'K': 16, 'num_rounds': 10, 'epochs_per_round': 100, 'effective_batch_size': 16, 'w_bad': 1.0, 'max_per_condition': 64, 'lambda_gripper': 1.0, 'reset_optimizer_per_round': False, 'num_train_timesteps': 1000, 'learning_rate': 1e-05, 'weight_decay': 0.01, 'max_grad_norm': 2.0, 'checkpointing_steps': 10, 'gradient_checkpointing': True, 'use_8bit_adam': True, 'lora_rank': 64, 'lora_alpha': 64, 'lora_target_modules': None, 'resume_from_lora_checkpoint': None, 'reward_config': 'blocks_stack_v2', 'reward_backend': 'blocks_stack_v2', 'skip_reward_debug_video': True, 'wandb_project': 'Corrective-Reflow', 'wandb_run_name': 'creflow_stack_blocks_three', 'hallucination_crop_top_ratio': 0.6667, 'bsv2_prompts': ['red block', 'green block', 'blue block'], 'bsv2_pick_thr': 0.05, 'bsv2_place_thr': 0.05, 'bsv2_vertical_sep_thr': 0.03, 'bsv2_x_align_thr': 0.025, 'bsv2_check_window_frac': 0.2, 'bsv2_dup_max_frames': 0, 'bsv2_expected_grip_changes': 6, 'bsv2_mj_lo': 0.6, 'bsv2_mj_hi': 1.3, 'bsv2_idm_ckpt_path': 'ckpts/vidar_ckpt/idm.pt'}
14
+ 2026-04-08 00:15:41,575 INFO MainThread:18968 [wandb_init.py:init():671] starting backend
15
+ 2026-04-08 00:15:41,575 INFO MainThread:18968 [wandb_init.py:init():675] sending inform_init request
16
+ 2026-04-08 00:15:41,602 INFO MainThread:18968 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
17
+ 2026-04-08 00:15:41,602 INFO MainThread:18968 [wandb_init.py:init():688] backend started and connected
18
+ 2026-04-08 00:15:41,658 INFO MainThread:18968 [wandb_init.py:init():783] updated telemetry
19
+ 2026-04-08 00:15:41,824 INFO MainThread:18968 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout
20
+ 2026-04-08 00:15:42,864 INFO MainThread:18968 [wandb_init.py:init():867] starting run threads in backend
21
+ 2026-04-08 00:15:43,291 INFO MainThread:18968 [wandb_run.py:_console_start():2463] atexit reg
22
+ 2026-04-08 00:15:43,291 INFO MainThread:18968 [wandb_run.py:_redirect():2311] redirect: wrap_raw
23
+ 2026-04-08 00:15:43,291 INFO MainThread:18968 [wandb_run.py:_redirect():2376] Wrapping output streams.
24
+ 2026-04-08 00:15:43,291 INFO MainThread:18968 [wandb_run.py:_redirect():2401] Redirects installed.
25
+ 2026-04-08 00:15:43,297 INFO MainThread:18968 [wandb_init.py:init():911] run started, returning control to user process
26
+ 2026-04-08 00:42:23,439 WARNING MsgRouterThr:18968 [router.py:message_loop():77] message_loop has been closed
wandb/run-20260408_004547-x9lalqi6/files/config.yaml ADDED
@@ -0,0 +1,145 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ _wandb:
2
+ value:
3
+ cli_version: 0.18.5
4
+ m: []
5
+ python_version: 3.11.2
6
+ t:
7
+ "1":
8
+ - 1
9
+ - 11
10
+ - 41
11
+ - 49
12
+ - 55
13
+ - 71
14
+ - 83
15
+ - 105
16
+ "2":
17
+ - 1
18
+ - 11
19
+ - 41
20
+ - 49
21
+ - 55
22
+ - 63
23
+ - 71
24
+ - 83
25
+ - 98
26
+ - 105
27
+ "3":
28
+ - 13
29
+ - 16
30
+ - 23
31
+ - 55
32
+ "4": 3.11.2
33
+ "5": 0.18.5
34
+ "6": 4.46.1
35
+ "8":
36
+ - 5
37
+ "12": 0.18.5
38
+ "13": linux-x86_64
39
+ K:
40
+ value: 16
41
+ bsv2_check_window_frac:
42
+ value: 0.2
43
+ bsv2_dup_max_frames:
44
+ value: 0
45
+ bsv2_expected_grip_changes:
46
+ value: 6
47
+ bsv2_idm_ckpt_path:
48
+ value: ckpts/vidar_ckpt/idm.pt
49
+ bsv2_mj_hi:
50
+ value: 1.3
51
+ bsv2_mj_lo:
52
+ value: 0.6
53
+ bsv2_pick_thr:
54
+ value: 0.05
55
+ bsv2_place_thr:
56
+ value: 0.05
57
+ bsv2_prompts:
58
+ value:
59
+ - red block
60
+ - green block
61
+ - blue block
62
+ bsv2_vertical_sep_thr:
63
+ value: 0.03
64
+ bsv2_x_align_thr:
65
+ value: 0.025
66
+ checkpointing_steps:
67
+ value: 10
68
+ ckpt_dir:
69
+ value: ckpts/Wan2.2-TI2V-5B
70
+ convert_model_dtype:
71
+ value: true
72
+ dataset_json:
73
+ value: data/rl_train/robotwin_stack_blocks_three.json
74
+ effective_batch_size:
75
+ value: 16
76
+ epochs_per_round:
77
+ value: 100
78
+ frame_num:
79
+ value: 121
80
+ gradient_checkpointing:
81
+ value: true
82
+ hallucination_crop_top_ratio:
83
+ value: 0.6667
84
+ lambda_gripper:
85
+ value: 1
86
+ learning_rate:
87
+ value: 1e-05
88
+ lora_alpha:
89
+ value: 64
90
+ lora_rank:
91
+ value: 64
92
+ lora_target_modules:
93
+ value: null
94
+ max_grad_norm:
95
+ value: 2
96
+ max_per_condition:
97
+ value: 64
98
+ max_samples:
99
+ value: -1
100
+ neg_prompt:
101
+ value: 色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走
102
+ num_ode_steps:
103
+ value: 20
104
+ num_rounds:
105
+ value: 10
106
+ num_train_timesteps:
107
+ value: 1000
108
+ offload_model:
109
+ value: false
110
+ output_dir:
111
+ value: data/outputs/creflow_stack_blocks_three
112
+ pt_dir:
113
+ value: ckpts/vidar_ckpt/merged_vidar_lora.pt
114
+ reset_optimizer_per_round:
115
+ value: false
116
+ resume_from_lora_checkpoint:
117
+ value: null
118
+ reward_backend:
119
+ value: blocks_stack_v2
120
+ reward_config:
121
+ value: blocks_stack_v2
122
+ sample_guide_scale:
123
+ value: 5
124
+ sample_shift:
125
+ value: 5
126
+ seed:
127
+ value: 42
128
+ size:
129
+ value: 640*736
130
+ skip_reward_debug_video:
131
+ value: true
132
+ task:
133
+ value: ti2v-5B
134
+ use_8bit_adam:
135
+ value: true
136
+ vidar_root:
137
+ value: ""
138
+ w_bad:
139
+ value: 1
140
+ wandb_project:
141
+ value: Corrective-Reflow
142
+ wandb_run_name:
143
+ value: creflow_stack_blocks_three
144
+ weight_decay:
145
+ value: 0.01
wandb/run-20260408_004547-x9lalqi6/files/output.log ADDED
@@ -0,0 +1,89 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Building Wan2.2 TI2V model ... (DDP=True, world_size=8)
2
+ /usr/local/lib/python3.11/dist-packages/torch/distributed/distributed_c10d.py:4876: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
3
+ warnings.warn( # warn only once
4
+ Moving T5 encoder to GPU permanently (offload_model=false) ...
5
+ Building reward scorer ...
6
+ Loading IDM from ckpts/vidar_ckpt/idm.pt on cuda:0 ...
7
+ IDM loaded.
8
+ Loading SAM3 VideoPredictor (blocks_stack_v2) on cuda:0 ...
9
+ INFO 2026-04-08 00:47:04,252 107287 sam3_video_base.py: 125: setting max_num_objects=10000 and num_obj_for_compile=16
10
+ SAM3 VideoPredictor (blocks_stack_v2) loaded.
11
+ Applying LoRA (single adapter) for CReflow ...
12
+ LoRA injected: rank=64, alpha=64, target_modules=['self_attn.q', 'self_attn.k', 'self_attn.v', 'self_attn.o', 'cross_attn.q', 'cross_attn.k', 'cross_attn.v', 'cross_attn.o']
13
+ Trainable: 94.4 M / 5094.2 M total (1.85%)
14
+ Traceback (most recent call last):
15
+ File "/home/tiger/.local/lib/python3.11/site-packages/peft/peft_model.py", line 787, in __getattr__
16
+ return super().__getattr__(name) # defer to nn.Module's logic
17
+ ^^^^^^^^^^^^^^^^^^^^^^^^^
18
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1964, in __getattr__
19
+ raise AttributeError(
20
+ AttributeError: 'PeftModel' object has no attribute 'enable_input_require_grads'
21
+
22
+ During handling of the above exception, another exception occurred:
23
+
24
+ Traceback (most recent call last):
25
+ File "/home/tiger/.local/lib/python3.11/site-packages/peft/tuners/lora/model.py", line 360, in __getattr__
26
+ return super().__getattr__(name) # defer to nn.Module's logic
27
+ ^^^^^^^^^^^^^^^^^^^^^^^^^
28
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1964, in __getattr__
29
+ raise AttributeError(
30
+ AttributeError: 'LoraModel' object has no attribute 'enable_input_require_grads'
31
+
32
+ During handling of the above exception, another exception occurred:
33
+
34
+ Traceback (most recent call last):
35
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1035, in <module>
36
+ main()
37
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 770, in main
38
+ dit.enable_input_require_grads()
39
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
40
+ File "/home/tiger/.local/lib/python3.11/site-packages/peft/peft_model.py", line 791, in __getattr__
41
+ return getattr(self.base_model, name)
42
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
43
+ File "/home/tiger/.local/lib/python3.11/site-packages/peft/tuners/lora/model.py", line 364, in __getattr__
44
+ return getattr(self.model, name)
45
+ ^^^^^^^^^^^^^^^^^^^^^^^^^
46
+ File "/home/tiger/.local/lib/python3.11/site-packages/diffusers/models/modeling_utils.py", line 173, in __getattr__
47
+ return super().__getattr__(name)
48
+ ^^^^^^^^^^^^^^^^^^^^^^^^^
49
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1964, in __getattr__
50
+ raise AttributeError(
51
+ AttributeError: 'WanModel' object has no attribute 'enable_input_require_grads'
52
+ [rank0]: Traceback (most recent call last):
53
+ [rank0]: File "/home/tiger/.local/lib/python3.11/site-packages/peft/peft_model.py", line 787, in __getattr__
54
+ [rank0]: return super().__getattr__(name) # defer to nn.Module's logic
55
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^
56
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1964, in __getattr__
57
+ [rank0]: raise AttributeError(
58
+ [rank0]: AttributeError: 'PeftModel' object has no attribute 'enable_input_require_grads'
59
+
60
+ [rank0]: During handling of the above exception, another exception occurred:
61
+
62
+ [rank0]: Traceback (most recent call last):
63
+ [rank0]: File "/home/tiger/.local/lib/python3.11/site-packages/peft/tuners/lora/model.py", line 360, in __getattr__
64
+ [rank0]: return super().__getattr__(name) # defer to nn.Module's logic
65
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^
66
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1964, in __getattr__
67
+ [rank0]: raise AttributeError(
68
+ [rank0]: AttributeError: 'LoraModel' object has no attribute 'enable_input_require_grads'
69
+
70
+ [rank0]: During handling of the above exception, another exception occurred:
71
+
72
+ [rank0]: Traceback (most recent call last):
73
+ [rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1035, in <module>
74
+ [rank0]: main()
75
+ [rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 770, in main
76
+ [rank0]: dit.enable_input_require_grads()
77
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
78
+ [rank0]: File "/home/tiger/.local/lib/python3.11/site-packages/peft/peft_model.py", line 791, in __getattr__
79
+ [rank0]: return getattr(self.base_model, name)
80
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
81
+ [rank0]: File "/home/tiger/.local/lib/python3.11/site-packages/peft/tuners/lora/model.py", line 364, in __getattr__
82
+ [rank0]: return getattr(self.model, name)
83
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^
84
+ [rank0]: File "/home/tiger/.local/lib/python3.11/site-packages/diffusers/models/modeling_utils.py", line 173, in __getattr__
85
+ [rank0]: return super().__getattr__(name)
86
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^
87
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1964, in __getattr__
88
+ [rank0]: raise AttributeError(
89
+ [rank0]: AttributeError: 'WanModel' object has no attribute 'enable_input_require_grads'
wandb/run-20260408_004547-x9lalqi6/files/wandb-metadata.json ADDED
@@ -0,0 +1,144 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-5.15.120.bsk.6-amd64-x86_64-with-glibc2.36",
3
+ "python": "3.11.2",
4
+ "startedAt": "2026-04-08T00:45:47.718687Z",
5
+ "args": [
6
+ "--task",
7
+ "ti2v-5B",
8
+ "--size",
9
+ "640*736",
10
+ "--frame_num",
11
+ "121",
12
+ "--ckpt_dir",
13
+ "ckpts/Wan2.2-TI2V-5B",
14
+ "--pt_dir",
15
+ "ckpts/vidar_ckpt/merged_vidar_lora.pt",
16
+ "--dataset_json",
17
+ "data/rl_train/robotwin_stack_blocks_three.json",
18
+ "--output_dir",
19
+ "data/outputs/creflow_stack_blocks_three",
20
+ "--num_ode_steps",
21
+ "20",
22
+ "--sample_shift",
23
+ "5.0",
24
+ "--sample_guide_scale",
25
+ "5.0",
26
+ "--K",
27
+ "16",
28
+ "--num_rounds",
29
+ "10",
30
+ "--epochs_per_round",
31
+ "100",
32
+ "--max_per_condition",
33
+ "64",
34
+ "--effective_batch_size",
35
+ "16",
36
+ "--w_bad",
37
+ "1.0",
38
+ "--lambda_gripper",
39
+ "1.0",
40
+ "--seed",
41
+ "42",
42
+ "--reward_config",
43
+ "blocks_stack_v2",
44
+ "--convert_model_dtype",
45
+ "--offload_model",
46
+ "false",
47
+ "--learning_rate",
48
+ "1e-5",
49
+ "--weight_decay",
50
+ "0.01",
51
+ "--max_grad_norm",
52
+ "2.0",
53
+ "--checkpointing_steps",
54
+ "10",
55
+ "--lora_rank",
56
+ "64",
57
+ "--lora_alpha",
58
+ "64",
59
+ "--wandb_project",
60
+ "Corrective-Reflow",
61
+ "--wandb_run_name",
62
+ "creflow_stack_blocks_three"
63
+ ],
64
+ "program": "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py",
65
+ "codePath": "fastvideo/train_creflow_wan_2_2_ti2v.py",
66
+ "git": {
67
+ "remote": "git@github.com:VincentNi0107/EmbodiedVideoRL.git",
68
+ "commit": "a8f5b170f0ef7d865af7e8f32c4aebb46efec2c1"
69
+ },
70
+ "email": "1163051845@qq.com",
71
+ "root": "data/outputs/creflow_stack_blocks_three",
72
+ "host": "n124-104-180",
73
+ "username": "tiger",
74
+ "executable": "/usr/bin/python",
75
+ "codePathLocal": "fastvideo/train_creflow_wan_2_2_ti2v.py",
76
+ "cpu_count": 104,
77
+ "cpu_count_logical": 208,
78
+ "gpu": "NVIDIA H100 80GB HBM3",
79
+ "gpu_count": 8,
80
+ "disk": {
81
+ "/": {
82
+ "total": "1055740600320",
83
+ "used": "481625993216"
84
+ }
85
+ },
86
+ "memory": {
87
+ "total": "1977660338176"
88
+ },
89
+ "cpu": {
90
+ "count": 104,
91
+ "countLogical": 208
92
+ },
93
+ "gpu_nvidia": [
94
+ {
95
+ "name": "NVIDIA H100 80GB HBM3",
96
+ "memoryTotal": "85520809984",
97
+ "cudaCores": 16896,
98
+ "architecture": "Hopper"
99
+ },
100
+ {
101
+ "name": "NVIDIA H100 80GB HBM3",
102
+ "memoryTotal": "85520809984",
103
+ "cudaCores": 16896,
104
+ "architecture": "Hopper"
105
+ },
106
+ {
107
+ "name": "NVIDIA H100 80GB HBM3",
108
+ "memoryTotal": "85520809984",
109
+ "cudaCores": 16896,
110
+ "architecture": "Hopper"
111
+ },
112
+ {
113
+ "name": "NVIDIA H100 80GB HBM3",
114
+ "memoryTotal": "85520809984",
115
+ "cudaCores": 16896,
116
+ "architecture": "Hopper"
117
+ },
118
+ {
119
+ "name": "NVIDIA H100 80GB HBM3",
120
+ "memoryTotal": "85520809984",
121
+ "cudaCores": 16896,
122
+ "architecture": "Hopper"
123
+ },
124
+ {
125
+ "name": "NVIDIA H100 80GB HBM3",
126
+ "memoryTotal": "85520809984",
127
+ "cudaCores": 16896,
128
+ "architecture": "Hopper"
129
+ },
130
+ {
131
+ "name": "NVIDIA H100 80GB HBM3",
132
+ "memoryTotal": "85520809984",
133
+ "cudaCores": 16896,
134
+ "architecture": "Hopper"
135
+ },
136
+ {
137
+ "name": "NVIDIA H100 80GB HBM3",
138
+ "memoryTotal": "85520809984",
139
+ "cudaCores": 16896,
140
+ "architecture": "Hopper"
141
+ }
142
+ ],
143
+ "cudaVersion": "12.9"
144
+ }
wandb/run-20260408_004547-x9lalqi6/files/wandb-summary.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"_wandb":{"runtime":80}}
wandb/run-20260408_004547-x9lalqi6/logs/debug-internal.log ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-04-08T00:45:47.72832013Z","level":"INFO","msg":"using version","core version":"0.18.5"}
2
+ {"time":"2026-04-08T00:45:47.72833167Z","level":"INFO","msg":"created symlink","path":"data/outputs/creflow_stack_blocks_three/wandb/run-20260408_004547-x9lalqi6/logs/debug-core.log"}
3
+ {"time":"2026-04-08T00:45:47.945807814Z","level":"INFO","msg":"created new stream","id":"x9lalqi6"}
4
+ {"time":"2026-04-08T00:45:47.945941311Z","level":"INFO","msg":"stream: started","id":"x9lalqi6"}
5
+ {"time":"2026-04-08T00:45:47.945997568Z","level":"INFO","msg":"handler: started","stream_id":{"value":"x9lalqi6"}}
6
+ {"time":"2026-04-08T00:45:47.946052958Z","level":"INFO","msg":"sender: started","stream_id":"x9lalqi6"}
7
+ {"time":"2026-04-08T00:45:47.945978269Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"x9lalqi6"}}
8
+ {"time":"2026-04-08T00:45:48.359921482Z","level":"INFO","msg":"Starting system monitor"}
9
+ {"time":"2026-04-08T00:47:08.634395008Z","level":"INFO","msg":"stream: closing","id":"x9lalqi6"}
10
+ {"time":"2026-04-08T00:47:08.634488596Z","level":"INFO","msg":"Stopping system monitor"}
11
+ {"time":"2026-04-08T00:47:08.639585554Z","level":"INFO","msg":"Stopped system monitor"}
12
+ {"time":"2026-04-08T00:47:10.458241724Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
13
+ {"time":"2026-04-08T00:47:10.582858773Z","level":"INFO","msg":"handler: closed","stream_id":{"value":"x9lalqi6"}}
14
+ {"time":"2026-04-08T00:47:10.582906095Z","level":"INFO","msg":"writer: Close: closed","stream_id":{"value":"x9lalqi6"}}
15
+ {"time":"2026-04-08T00:47:10.582938514Z","level":"INFO","msg":"sender: closed","stream_id":"x9lalqi6"}
16
+ {"time":"2026-04-08T00:47:10.586475232Z","level":"INFO","msg":"stream: closed","id":"x9lalqi6"}
wandb/run-20260408_004547-x9lalqi6/logs/debug.log ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 2026-04-08 00:45:47,684 INFO MainThread:107287 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5
2
+ 2026-04-08 00:45:47,684 INFO MainThread:107287 [wandb_setup.py:_flush():79] Configure stats pid to 107287
3
+ 2026-04-08 00:45:47,684 INFO MainThread:107287 [wandb_setup.py:_flush():79] Loading settings from /home/tiger/.config/wandb/settings
4
+ 2026-04-08 00:45:47,684 INFO MainThread:107287 [wandb_setup.py:_flush():79] Loading settings from /mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/wandb/settings
5
+ 2026-04-08 00:45:47,684 INFO MainThread:107287 [wandb_setup.py:_flush():79] Loading settings from environment variables: {'disabled': 'false', 'project': 'Corrective-Reflow', 'api_key': '***REDACTED***', 'mode': 'online'}
6
+ 2026-04-08 00:45:47,684 INFO MainThread:107287 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': 'online', '_disable_service': None}
7
+ 2026-04-08 00:45:47,684 INFO MainThread:107287 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'fastvideo/train_creflow_wan_2_2_ti2v.py', 'program_abspath': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py', 'program': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py'}
8
+ 2026-04-08 00:45:47,684 INFO MainThread:107287 [wandb_setup.py:_flush():79] Applying login settings: {}
9
+ 2026-04-08 00:45:47,686 INFO MainThread:107287 [wandb_init.py:_log_setup():534] Logging user logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_004547-x9lalqi6/logs/debug.log
10
+ 2026-04-08 00:45:47,688 INFO MainThread:107287 [wandb_init.py:_log_setup():535] Logging internal logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_004547-x9lalqi6/logs/debug-internal.log
11
+ 2026-04-08 00:45:47,688 INFO MainThread:107287 [wandb_init.py:init():621] calling init triggers
12
+ 2026-04-08 00:45:47,688 INFO MainThread:107287 [wandb_init.py:init():628] wandb.init called with sweep_config: {}
13
+ config: {'vidar_root': '', 'task': 'ti2v-5B', 'size': '640*736', 'frame_num': 121, 'ckpt_dir': 'ckpts/Wan2.2-TI2V-5B', 'pt_dir': 'ckpts/vidar_ckpt/merged_vidar_lora.pt', 'convert_model_dtype': True, 'offload_model': False, 'dataset_json': 'data/rl_train/robotwin_stack_blocks_three.json', 'max_samples': -1, 'output_dir': 'data/outputs/creflow_stack_blocks_three', 'num_ode_steps': 20, 'sample_shift': 5.0, 'sample_guide_scale': 5.0, 'neg_prompt': '色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走', 'seed': 42, 'K': 16, 'num_rounds': 10, 'epochs_per_round': 100, 'effective_batch_size': 16, 'w_bad': 1.0, 'max_per_condition': 64, 'lambda_gripper': 1.0, 'reset_optimizer_per_round': False, 'num_train_timesteps': 1000, 'learning_rate': 1e-05, 'weight_decay': 0.01, 'max_grad_norm': 2.0, 'checkpointing_steps': 10, 'gradient_checkpointing': True, 'use_8bit_adam': True, 'lora_rank': 64, 'lora_alpha': 64, 'lora_target_modules': None, 'resume_from_lora_checkpoint': None, 'reward_config': 'blocks_stack_v2', 'reward_backend': 'blocks_stack_v2', 'skip_reward_debug_video': True, 'wandb_project': 'Corrective-Reflow', 'wandb_run_name': 'creflow_stack_blocks_three', 'hallucination_crop_top_ratio': 0.6667, 'bsv2_prompts': ['red block', 'green block', 'blue block'], 'bsv2_pick_thr': 0.05, 'bsv2_place_thr': 0.05, 'bsv2_vertical_sep_thr': 0.03, 'bsv2_x_align_thr': 0.025, 'bsv2_check_window_frac': 0.2, 'bsv2_dup_max_frames': 0, 'bsv2_expected_grip_changes': 6, 'bsv2_mj_lo': 0.6, 'bsv2_mj_hi': 1.3, 'bsv2_idm_ckpt_path': 'ckpts/vidar_ckpt/idm.pt'}
14
+ 2026-04-08 00:45:47,688 INFO MainThread:107287 [wandb_init.py:init():671] starting backend
15
+ 2026-04-08 00:45:47,688 INFO MainThread:107287 [wandb_init.py:init():675] sending inform_init request
16
+ 2026-04-08 00:45:47,716 INFO MainThread:107287 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
17
+ 2026-04-08 00:45:47,716 INFO MainThread:107287 [wandb_init.py:init():688] backend started and connected
18
+ 2026-04-08 00:45:47,773 INFO MainThread:107287 [wandb_init.py:init():783] updated telemetry
19
+ 2026-04-08 00:45:47,947 INFO MainThread:107287 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout
20
+ 2026-04-08 00:45:48,310 INFO MainThread:107287 [wandb_init.py:init():867] starting run threads in backend
21
+ 2026-04-08 00:45:48,678 INFO MainThread:107287 [wandb_run.py:_console_start():2463] atexit reg
22
+ 2026-04-08 00:45:48,678 INFO MainThread:107287 [wandb_run.py:_redirect():2311] redirect: wrap_raw
23
+ 2026-04-08 00:45:48,678 INFO MainThread:107287 [wandb_run.py:_redirect():2376] Wrapping output streams.
24
+ 2026-04-08 00:45:48,678 INFO MainThread:107287 [wandb_run.py:_redirect():2401] Redirects installed.
25
+ 2026-04-08 00:45:48,683 INFO MainThread:107287 [wandb_init.py:init():911] run started, returning control to user process
26
+ 2026-04-08 00:47:08,634 WARNING MsgRouterThr:107287 [router.py:message_loop():77] message_loop has been closed
wandb/run-20260408_004931-ku6kqemc/files/config.yaml ADDED
@@ -0,0 +1,145 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ _wandb:
2
+ value:
3
+ cli_version: 0.18.5
4
+ m: []
5
+ python_version: 3.11.2
6
+ t:
7
+ "1":
8
+ - 1
9
+ - 11
10
+ - 41
11
+ - 49
12
+ - 55
13
+ - 71
14
+ - 83
15
+ - 105
16
+ "2":
17
+ - 1
18
+ - 11
19
+ - 41
20
+ - 49
21
+ - 55
22
+ - 63
23
+ - 71
24
+ - 83
25
+ - 98
26
+ - 105
27
+ "3":
28
+ - 13
29
+ - 16
30
+ - 23
31
+ - 55
32
+ "4": 3.11.2
33
+ "5": 0.18.5
34
+ "6": 4.46.1
35
+ "8":
36
+ - 5
37
+ "12": 0.18.5
38
+ "13": linux-x86_64
39
+ K:
40
+ value: 16
41
+ bsv2_check_window_frac:
42
+ value: 0.2
43
+ bsv2_dup_max_frames:
44
+ value: 0
45
+ bsv2_expected_grip_changes:
46
+ value: 6
47
+ bsv2_idm_ckpt_path:
48
+ value: ckpts/vidar_ckpt/idm.pt
49
+ bsv2_mj_hi:
50
+ value: 1.3
51
+ bsv2_mj_lo:
52
+ value: 0.6
53
+ bsv2_pick_thr:
54
+ value: 0.05
55
+ bsv2_place_thr:
56
+ value: 0.05
57
+ bsv2_prompts:
58
+ value:
59
+ - red block
60
+ - green block
61
+ - blue block
62
+ bsv2_vertical_sep_thr:
63
+ value: 0.03
64
+ bsv2_x_align_thr:
65
+ value: 0.025
66
+ checkpointing_steps:
67
+ value: 10
68
+ ckpt_dir:
69
+ value: ckpts/Wan2.2-TI2V-5B
70
+ convert_model_dtype:
71
+ value: true
72
+ dataset_json:
73
+ value: data/rl_train/robotwin_stack_blocks_three.json
74
+ effective_batch_size:
75
+ value: 16
76
+ epochs_per_round:
77
+ value: 100
78
+ frame_num:
79
+ value: 121
80
+ gradient_checkpointing:
81
+ value: true
82
+ hallucination_crop_top_ratio:
83
+ value: 0.6667
84
+ lambda_gripper:
85
+ value: 1
86
+ learning_rate:
87
+ value: 1e-05
88
+ lora_alpha:
89
+ value: 64
90
+ lora_rank:
91
+ value: 64
92
+ lora_target_modules:
93
+ value: null
94
+ max_grad_norm:
95
+ value: 2
96
+ max_per_condition:
97
+ value: 64
98
+ max_samples:
99
+ value: -1
100
+ neg_prompt:
101
+ value: 色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走
102
+ num_ode_steps:
103
+ value: 20
104
+ num_rounds:
105
+ value: 10
106
+ num_train_timesteps:
107
+ value: 1000
108
+ offload_model:
109
+ value: false
110
+ output_dir:
111
+ value: data/outputs/creflow_stack_blocks_three
112
+ pt_dir:
113
+ value: ckpts/vidar_ckpt/merged_vidar_lora.pt
114
+ reset_optimizer_per_round:
115
+ value: false
116
+ resume_from_lora_checkpoint:
117
+ value: null
118
+ reward_backend:
119
+ value: blocks_stack_v2
120
+ reward_config:
121
+ value: blocks_stack_v2
122
+ sample_guide_scale:
123
+ value: 5
124
+ sample_shift:
125
+ value: 5
126
+ seed:
127
+ value: 42
128
+ size:
129
+ value: 640*736
130
+ skip_reward_debug_video:
131
+ value: true
132
+ task:
133
+ value: ti2v-5B
134
+ use_8bit_adam:
135
+ value: true
136
+ vidar_root:
137
+ value: ""
138
+ w_bad:
139
+ value: 1
140
+ wandb_project:
141
+ value: Corrective-Reflow
142
+ wandb_run_name:
143
+ value: creflow_stack_blocks_three
144
+ weight_decay:
145
+ value: 0.01
wandb/run-20260408_004931-ku6kqemc/files/output.log ADDED
@@ -0,0 +1,685 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Building Wan2.2 TI2V model ... (DDP=True, world_size=8)
2
+ /usr/local/lib/python3.11/dist-packages/torch/distributed/distributed_c10d.py:4876: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
3
+ warnings.warn( # warn only once
4
+ Moving T5 encoder to GPU permanently (offload_model=false) ...
5
+ Building reward scorer ...
6
+ Loading IDM from ckpts/vidar_ckpt/idm.pt on cuda:0 ...
7
+ IDM loaded.
8
+ Loading SAM3 VideoPredictor (blocks_stack_v2) on cuda:0 ...
9
+ INFO 2026-04-08 00:50:46,165 108291 sam3_video_base.py: 125: setting max_num_objects=10000 and num_obj_for_compile=16
10
+ SAM3 VideoPredictor (blocks_stack_v2) loaded.
11
+ Applying LoRA (single adapter) for CReflow ...
12
+ LoRA injected: rank=64, alpha=64, target_modules=['self_attn.q', 'self_attn.k', 'self_attn.v', 'self_attn.o', 'cross_attn.q', 'cross_attn.k', 'cross_attn.v', 'cross_attn.o']
13
+ Trainable: 94.4 M / 5094.2 M total (1.85%)
14
+ Gradient checkpointing enabled on 30 DiT blocks
15
+ Trainable parameters: 94.4 M
16
+ Using 8-bit AdamW (bitsandbytes)
17
+ Dataset: 10 conditions
18
+ Encoding 10 conditions ...
19
+ [1/10] 6b20973f10ef... encoded
20
+ [2/10] cff33035b28a... encoded
21
+ [3/10] 7ebc8b336779... encoded
22
+ [4/10] 8077c679fb62... encoded
23
+ [5/10] 41855438a8e0... encoded
24
+ [6/10] a67b02f4e7ce... encoded
25
+ [7/10] 4568a9603f3b... encoded
26
+ [8/10] 1b67584bff87... encoded
27
+ [9/10] f4f8079f6637... encoded
28
+ [10/10] 1b70247c6412... encoded
29
+ All 10 conditions encoded.
30
+
31
+ Starting Corrective Reflow | rounds=10 K=16 K_local=2 epochs/round=100 effective_bs=16 lr=1e-05 conditions=10
32
+
33
+ ======================================================================
34
+ ROUND 1/10
35
+ ======================================================================
36
+ [Phase 1] ODE rollout ...
37
+ [CReflow rollout] condition 0/10: K=2, id=6b20973f10ef...
38
+ rollout: 2 videos in 56.5s
39
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 77.15it/s]
40
+ propagate_in_video: 100%|██████████| 121/121 [00:13<00:00, 9.10it/s]
41
+ propagate_in_video: 0it [00:00, ?it/s]
42
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.59it/s]
43
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.01it/s]
44
+ propagate_in_video: 0it [00:00, ?it/s]
45
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.27it/s]
46
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.04it/s]
47
+ propagate_in_video: 0it [00:00, ?it/s]
48
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 77.55it/s]
49
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.01it/s]
50
+ propagate_in_video: 0it [00:00, ?it/s]
51
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.27it/s]
52
+ propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.86it/s]
53
+ propagate_in_video: 0it [00:00, ?it/s]
54
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.24it/s]
55
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.54it/s]
56
+ propagate_in_video: 0it [00:00, ?it/s]
57
+ [CReflow rollout] condition 1/10: K=2, id=cff33035b28a...
58
+ rollout: 2 videos in 57.1s
59
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.62it/s]
60
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.20it/s]
61
+ propagate_in_video: 0it [00:00, ?it/s]
62
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.25it/s]
63
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.35it/s]
64
+ propagate_in_video: 0it [00:00, ?it/s]
65
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.43it/s]
66
+ propagate_in_video: 100%|██████████| 121/121 [00:11<00:00, 10.55it/s]
67
+ propagate_in_video: 0it [00:00, ?it/s]
68
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.77it/s]
69
+ propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 12.06it/s]
70
+ propagate_in_video: 0it [00:00, ?it/s]
71
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.19it/s]
72
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.33it/s]
73
+ propagate_in_video: 0it [00:00, ?it/s]
74
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.24it/s]
75
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.20it/s]
76
+ propagate_in_video: 0it [00:00, ?it/s]
77
+ [CReflow rollout] condition 2/10: K=2, id=7ebc8b336779...
78
+ rollout: 2 videos in 57.0s
79
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.12it/s]
80
+ propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.78it/s]
81
+ propagate_in_video: 0it [00:00, ?it/s]
82
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.92it/s]
83
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.17it/s]
84
+ propagate_in_video: 0it [00:00, ?it/s]
85
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.49it/s]
86
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.90it/s]
87
+ propagate_in_video: 0it [00:00, ?it/s]
88
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.84it/s]
89
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.68it/s]
90
+ propagate_in_video: 0it [00:00, ?it/s]
91
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.59it/s]
92
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.49it/s]
93
+ propagate_in_video: 0it [00:00, ?it/s]
94
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.49it/s]
95
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.21it/s]
96
+ propagate_in_video: 0it [00:00, ?it/s]
97
+ [CReflow rollout] condition 3/10: K=2, id=8077c679fb62...
98
+ rollout: 2 videos in 57.5s
99
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.59it/s]
100
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.09it/s]
101
+ propagate_in_video: 0it [00:00, ?it/s]
102
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.19it/s]
103
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.13it/s]
104
+ propagate_in_video: 0it [00:00, ?it/s]
105
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.23it/s]
106
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.08it/s]
107
+ propagate_in_video: 0it [00:00, ?it/s]
108
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.03it/s]
109
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.33it/s]
110
+ propagate_in_video: 0it [00:00, ?it/s]
111
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.08it/s]
112
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.59it/s]
113
+ propagate_in_video: 0it [00:00, ?it/s]
114
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.56it/s]
115
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.20it/s]
116
+ propagate_in_video: 0it [00:00, ?it/s]
117
+ [CReflow rollout] condition 4/10: K=2, id=41855438a8e0...
118
+ rollout: 2 videos in 57.6s
119
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.67it/s]
120
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.03it/s]
121
+ propagate_in_video: 0it [00:00, ?it/s]
122
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.84it/s]
123
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.26it/s]
124
+ propagate_in_video: 0it [00:00, ?it/s]
125
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.09it/s]
126
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.33it/s]
127
+ propagate_in_video: 0it [00:00, ?it/s]
128
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.97it/s]
129
+ propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.96it/s]
130
+ propagate_in_video: 0it [00:00, ?it/s]
131
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.08it/s]
132
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.43it/s]
133
+ propagate_in_video: 0it [00:00, ?it/s]
134
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.49it/s]
135
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.21it/s]
136
+ propagate_in_video: 0it [00:00, ?it/s]
137
+ [CReflow rollout] condition 5/10: K=2, id=a67b02f4e7ce...
138
+ rollout: 2 videos in 57.6s
139
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.45it/s]
140
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.21it/s]
141
+ propagate_in_video: 0it [00:00, ?it/s]
142
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.39it/s]
143
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.50it/s]
144
+ propagate_in_video: 0it [00:00, ?it/s]
145
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.89it/s]
146
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.22it/s]
147
+ propagate_in_video: 0it [00:00, ?it/s]
148
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.10it/s]
149
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.58it/s]
150
+ propagate_in_video: 0it [00:00, ?it/s]
151
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.92it/s]
152
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.58it/s]
153
+ propagate_in_video: 0it [00:00, ?it/s]
154
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.27it/s]
155
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.24it/s]
156
+ propagate_in_video: 0it [00:00, ?it/s]
157
+ [CReflow rollout] condition 6/10: K=2, id=4568a9603f3b...
158
+ rollout: 2 videos in 57.5s
159
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.48it/s]
160
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.14it/s]
161
+ propagate_in_video: 0it [00:00, ?it/s]
162
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.29it/s]
163
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.42it/s]
164
+ propagate_in_video: 0it [00:00, ?it/s]
165
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.61it/s]
166
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.95it/s]
167
+ propagate_in_video: 0it [00:00, ?it/s]
168
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.03it/s]
169
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.22it/s]
170
+ propagate_in_video: 0it [00:00, ?it/s]
171
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.11it/s]
172
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.16it/s]
173
+ propagate_in_video: 0it [00:00, ?it/s]
174
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.35it/s]
175
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.00it/s]
176
+ propagate_in_video: 0it [00:00, ?it/s]
177
+ [CReflow rollout] condition 7/10: K=2, id=1b67584bff87...
178
+ rollout: 2 videos in 57.5s
179
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.24it/s]
180
+ propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.66it/s]
181
+ propagate_in_video: 0it [00:00, ?it/s]
182
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.27it/s]
183
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.37it/s]
184
+ propagate_in_video: 0it [00:00, ?it/s]
185
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.42it/s]
186
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.54it/s]
187
+ propagate_in_video: 0it [00:00, ?it/s]
188
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.09it/s]
189
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.94it/s]
190
+ propagate_in_video: 0it [00:00, ?it/s]
191
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.10it/s]
192
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.71it/s]
193
+ propagate_in_video: 0it [00:00, ?it/s]
194
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.78it/s]
195
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.89it/s]
196
+ propagate_in_video: 0it [00:00, ?it/s]
197
+ [CReflow rollout] condition 8/10: K=2, id=f4f8079f6637...
198
+ rollout: 2 videos in 57.3s
199
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.78it/s]
200
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.30it/s]
201
+ propagate_in_video: 0it [00:00, ?it/s]
202
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.94it/s]
203
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.79it/s]
204
+ propagate_in_video: 0it [00:00, ?it/s]
205
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.10it/s]
206
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.94it/s]
207
+ propagate_in_video: 0it [00:00, ?it/s]
208
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.40it/s]
209
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.37it/s]
210
+ propagate_in_video: 0it [00:00, ?it/s]
211
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 74.81it/s]
212
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.53it/s]
213
+ propagate_in_video: 0it [00:00, ?it/s]
214
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.78it/s]
215
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.22it/s]
216
+ propagate_in_video: 0it [00:00, ?it/s]
217
+ [CReflow rollout] condition 9/10: K=2, id=1b70247c6412...
218
+ rollout: 2 videos in 57.5s
219
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.12it/s]
220
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.43it/s]
221
+ propagate_in_video: 0it [00:00, ?it/s]
222
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.20it/s]
223
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.76it/s]
224
+ propagate_in_video: 0it [00:00, ?it/s]
225
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.34it/s]
226
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.95it/s]
227
+ propagate_in_video: 0it [00:00, ?it/s]
228
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.77it/s]
229
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.76it/s]
230
+ propagate_in_video: 0it [00:00, ?it/s]
231
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.45it/s]
232
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.27it/s]
233
+ propagate_in_video: 0it [00:00, ?it/s]
234
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 85.50it/s]
235
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.21it/s]
236
+ propagate_in_video: 0it [00:00, ?it/s]
237
+ [CReflow rollout] done: 20 samples, rewards = [1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 0.00, 0.00]
238
+ Rollout done: 20 samples, good=18, bad=2, avg_reward=0.900
239
+ 1b67584bff87: mean=1.00, good=2/2
240
+ 1b70247c6412: mean=0.00, good=0/2
241
+ 41855438a8e0: mean=1.00, good=2/2
242
+ 4568a9603f3b: mean=1.00, good=2/2
243
+ 6b20973f10ef: mean=1.00, good=2/2
244
+ 7ebc8b336779: mean=1.00, good=2/2
245
+ 8077c679fb62: mean=1.00, good=2/2
246
+ a67b02f4e7ce: mean=1.00, good=2/2
247
+ cff33035b28a: mean=1.00, good=2/2
248
+ f4f8079f6637: mean=1.00, good=2/2
249
+ [Phase 2] Library update ...
250
+ [Library] rank=0 has 18 good samples
251
+ [Library] after update: SuccessLibrary(total=127, coverage=10/10, max_per_condition=64)
252
+ [Phase 3] Re-pairing ...
253
+ [Re-pair] train_set: 20 samples (anchor=18, correction=2, skipped=0)
254
+ [Phase 4] Training (100 epochs, 20 samples) ...
255
+ Traceback (most recent call last):
256
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1037, in <module>
257
+ main()
258
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 937, in main
259
+ train_metrics = train_one_round(
260
+ ^^^^^^^^^^^^^^^^
261
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 577, in train_one_round
262
+ loss.backward()
263
+ File "/usr/local/lib/python3.11/dist-packages/torch/_tensor.py", line 625, in backward
264
+ torch.autograd.backward(
265
+ File "/usr/local/lib/python3.11/dist-packages/torch/autograd/__init__.py", line 354, in backward
266
+ _engine_run_backward(
267
+ File "/usr/local/lib/python3.11/dist-packages/torch/autograd/graph.py", line 841, in _engine_run_backward
268
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
269
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
270
+ File "/usr/local/lib/python3.11/dist-packages/torch/utils/checkpoint.py", line 1158, in unpack_hook
271
+ frame.check_recomputed_tensors_match(gid)
272
+ File "/usr/local/lib/python3.11/dist-packages/torch/utils/checkpoint.py", line 911, in check_recomputed_tensors_match
273
+ raise CheckpointError(
274
+ torch.utils.checkpoint.CheckpointError: torch.utils.checkpoint: Recomputed values for the following tensors have different metadata than during the forward pass.
275
+ tensor at position 6:
276
+ saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
277
+ recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
278
+ tensor at position 7:
279
+ saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
280
+ recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
281
+ tensor at position 8:
282
+ saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
283
+ recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
284
+ tensor at position 9:
285
+ saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
286
+ recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
287
+ tensor at position 10:
288
+ saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
289
+ recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
290
+ tensor at position 11:
291
+ saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
292
+ recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
293
+ tensor at position 12:
294
+ saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
295
+ recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
296
+ tensor at position 13:
297
+ saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
298
+ recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
299
+ tensor at position 14:
300
+ saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
301
+ recomputed metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
302
+ tensor at position 15:
303
+ saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
304
+ recomputed metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
305
+ tensor at position 16:
306
+ saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
307
+ recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
308
+ tensor at position 17:
309
+ saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
310
+ recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
311
+ tensor at position 18:
312
+ saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
313
+ recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
314
+ tensor at position 19:
315
+ saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
316
+ recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
317
+ tensor at position 20:
318
+ saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
319
+ recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
320
+ tensor at position 21:
321
+ saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
322
+ recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
323
+ tensor at position 22:
324
+ saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
325
+ recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
326
+ tensor at position 23:
327
+ saved metadata: {'shape': torch.Size([14260, 1, 64]), 'dtype': torch.complex128, 'device': device(type='cuda', index=0)}
328
+ recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
329
+ tensor at position 24:
330
+ saved metadata: {'shape': torch.Size([14260, 1, 64]), 'dtype': torch.complex128, 'device': device(type='cuda', index=0)}
331
+ recomputed metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
332
+ tensor at position 25:
333
+ saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
334
+ recomputed metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
335
+ tensor at position 26:
336
+ saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
337
+ recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
338
+ tensor at position 27:
339
+ saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
340
+ recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
341
+ tensor at position 28:
342
+ saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
343
+ recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
344
+ tensor at position 29:
345
+ saved metadata: {'shape': torch.Size([24, 14260]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
346
+ recomputed metadata: {'shape': torch.Size([14260, 1, 64]), 'dtype': torch.complex128, 'device': device(type='cuda', index=0)}
347
+ tensor at position 30:
348
+ saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)}
349
+ recomputed metadata: {'shape': torch.Size([14260, 1, 64]), 'dtype': torch.complex128, 'device': device(type='cuda', index=0)}
350
+ tensor at position 31:
351
+ saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)}
352
+ recomputed metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
353
+ tensor at position 32:
354
+ saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int64, 'device': device(type='cuda', index=0)}
355
+ recomputed metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
356
+ tensor at position 33:
357
+ saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
358
+ recomputed metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
359
+ tensor at position 34:
360
+ saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
361
+ recomputed metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
362
+ tensor at position 35:
363
+ saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
364
+ recomputed metadata: {'shape': torch.Size([24, 14260]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
365
+ tensor at position 36:
366
+ saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
367
+ recomputed metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)}
368
+ tensor at position 37:
369
+ saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
370
+ recomputed metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)}
371
+ tensor at position 38:
372
+ saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
373
+ recomputed metadata: {'shape': torch.Size([2]), 'dtype': torch.int64, 'device': device(type='cuda', index=0)}
374
+ tensor at position 39:
375
+ saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
376
+ recomputed metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
377
+ tensor at position 40:
378
+ saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
379
+ recomputed metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
380
+ tensor at position 41:
381
+ saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
382
+ recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
383
+ tensor at position 42:
384
+ saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
385
+ recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
386
+ tensor at position 43:
387
+ saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
388
+ recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
389
+ tensor at position 44:
390
+ saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
391
+ recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
392
+ tensor at position 45:
393
+ saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
394
+ recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
395
+ tensor at position 46:
396
+ saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
397
+ recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
398
+ tensor at position 47:
399
+ saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
400
+ recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
401
+ tensor at position 48:
402
+ saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
403
+ recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
404
+ tensor at position 49:
405
+ saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
406
+ recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
407
+ tensor at position 50:
408
+ saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
409
+ recomputed metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
410
+ tensor at position 51:
411
+ saved metadata: {'shape': torch.Size([512, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
412
+ recomputed metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
413
+ tensor at position 52:
414
+ saved metadata: {'shape': torch.Size([512, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
415
+ recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
416
+ tensor at position 53:
417
+ saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
418
+ recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
419
+ tensor at position 54:
420
+ saved metadata: {'shape': torch.Size([24, 14260]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
421
+ recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
422
+ tensor at position 55:
423
+ saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)}
424
+ recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
425
+ tensor at position 56:
426
+ saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)}
427
+ recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
428
+ tensor at position 57:
429
+ saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int64, 'device': device(type='cuda', index=0)}
430
+ recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
431
+ tensor at position 58:
432
+ saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
433
+ recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
434
+ tensor at position 59:
435
+ saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
436
+ recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
437
+ tensor at position 60:
438
+ saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
439
+ recomputed metadata: {'shape': torch.Size([512, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
440
+ tensor at position 61:
441
+ saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
442
+ recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
443
+ tensor at position 62:
444
+ saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
445
+ recomputed metadata: {'shape': torch.Size([512, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
446
+ tensor at position 63:
447
+ saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
448
+ recomputed metadata: {'shape': torch.Size([1, 512, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
449
+ tensor at position 64:
450
+ saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
451
+ recomputed metadata: {'shape': torch.Size([1, 512, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
452
+ tensor at position 65:
453
+ saved metadata: {'shape': torch.Size([3072, 14336]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
454
+ recomputed metadata: {'shape': torch.Size([1, 512, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
455
+ tensor at position 66:
456
+ saved metadata: {'shape': torch.Size([1, 14260, 14336]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
457
+ recomputed metadata: {'shape': torch.Size([1, 512, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
458
+ tensor at position 67:
459
+ saved metadata: {'shape': torch.Size([14336, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
460
+ recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
461
+ tensor at position 68:
462
+ saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
463
+ recomputed metadata: {'shape': torch.Size([512, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
464
+ .
465
+
466
+ Tip: To see a more detailed error message, either pass `debug=True` to
467
+ `torch.utils.checkpoint.checkpoint(...)` or wrap the code block
468
+ with `with torch.utils.checkpoint.set_checkpoint_debug_enabled(True):` to
469
+ enable checkpoint‑debug mode globally.
470
+
471
+ [rank0]: Traceback (most recent call last):
472
+ [rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1037, in <module>
473
+ [rank0]: main()
474
+ [rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 937, in main
475
+ [rank0]: train_metrics = train_one_round(
476
+ [rank0]: ^^^^^^^^^^^^^^^^
477
+ [rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 577, in train_one_round
478
+ [rank0]: loss.backward()
479
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/_tensor.py", line 625, in backward
480
+ [rank0]: torch.autograd.backward(
481
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/autograd/__init__.py", line 354, in backward
482
+ [rank0]: _engine_run_backward(
483
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/autograd/graph.py", line 841, in _engine_run_backward
484
+ [rank0]: return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
485
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
486
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/utils/checkpoint.py", line 1158, in unpack_hook
487
+ [rank0]: frame.check_recomputed_tensors_match(gid)
488
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/utils/checkpoint.py", line 911, in check_recomputed_tensors_match
489
+ [rank0]: raise CheckpointError(
490
+ [rank0]: torch.utils.checkpoint.CheckpointError: torch.utils.checkpoint: Recomputed values for the following tensors have different metadata than during the forward pass.
491
+ [rank0]: tensor at position 6:
492
+ [rank0]: saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
493
+ [rank0]: recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
494
+ [rank0]: tensor at position 7:
495
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
496
+ [rank0]: recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
497
+ [rank0]: tensor at position 8:
498
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
499
+ [rank0]: recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
500
+ [rank0]: tensor at position 9:
501
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
502
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
503
+ [rank0]: tensor at position 10:
504
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
505
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
506
+ [rank0]: tensor at position 11:
507
+ [rank0]: saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
508
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
509
+ [rank0]: tensor at position 12:
510
+ [rank0]: saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
511
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
512
+ [rank0]: tensor at position 13:
513
+ [rank0]: saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
514
+ [rank0]: recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
515
+ [rank0]: tensor at position 14:
516
+ [rank0]: saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
517
+ [rank0]: recomputed metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
518
+ [rank0]: tensor at position 15:
519
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
520
+ [rank0]: recomputed metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
521
+ [rank0]: tensor at position 16:
522
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
523
+ [rank0]: recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
524
+ [rank0]: tensor at position 17:
525
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
526
+ [rank0]: recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
527
+ [rank0]: tensor at position 18:
528
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
529
+ [rank0]: recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
530
+ [rank0]: tensor at position 19:
531
+ [rank0]: saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
532
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
533
+ [rank0]: tensor at position 20:
534
+ [rank0]: saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
535
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
536
+ [rank0]: tensor at position 21:
537
+ [rank0]: saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
538
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
539
+ [rank0]: tensor at position 22:
540
+ [rank0]: saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
541
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
542
+ [rank0]: tensor at position 23:
543
+ [rank0]: saved metadata: {'shape': torch.Size([14260, 1, 64]), 'dtype': torch.complex128, 'device': device(type='cuda', index=0)}
544
+ [rank0]: recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
545
+ [rank0]: tensor at position 24:
546
+ [rank0]: saved metadata: {'shape': torch.Size([14260, 1, 64]), 'dtype': torch.complex128, 'device': device(type='cuda', index=0)}
547
+ [rank0]: recomputed metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
548
+ [rank0]: tensor at position 25:
549
+ [rank0]: saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
550
+ [rank0]: recomputed metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
551
+ [rank0]: tensor at position 26:
552
+ [rank0]: saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
553
+ [rank0]: recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
554
+ [rank0]: tensor at position 27:
555
+ [rank0]: saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
556
+ [rank0]: recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
557
+ [rank0]: tensor at position 28:
558
+ [rank0]: saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
559
+ [rank0]: recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
560
+ [rank0]: tensor at position 29:
561
+ [rank0]: saved metadata: {'shape': torch.Size([24, 14260]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
562
+ [rank0]: recomputed metadata: {'shape': torch.Size([14260, 1, 64]), 'dtype': torch.complex128, 'device': device(type='cuda', index=0)}
563
+ [rank0]: tensor at position 30:
564
+ [rank0]: saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)}
565
+ [rank0]: recomputed metadata: {'shape': torch.Size([14260, 1, 64]), 'dtype': torch.complex128, 'device': device(type='cuda', index=0)}
566
+ [rank0]: tensor at position 31:
567
+ [rank0]: saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)}
568
+ [rank0]: recomputed metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
569
+ [rank0]: tensor at position 32:
570
+ [rank0]: saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int64, 'device': device(type='cuda', index=0)}
571
+ [rank0]: recomputed metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
572
+ [rank0]: tensor at position 33:
573
+ [rank0]: saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
574
+ [rank0]: recomputed metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
575
+ [rank0]: tensor at position 34:
576
+ [rank0]: saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
577
+ [rank0]: recomputed metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
578
+ [rank0]: tensor at position 35:
579
+ [rank0]: saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
580
+ [rank0]: recomputed metadata: {'shape': torch.Size([24, 14260]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
581
+ [rank0]: tensor at position 36:
582
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
583
+ [rank0]: recomputed metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)}
584
+ [rank0]: tensor at position 37:
585
+ [rank0]: saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
586
+ [rank0]: recomputed metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)}
587
+ [rank0]: tensor at position 38:
588
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
589
+ [rank0]: recomputed metadata: {'shape': torch.Size([2]), 'dtype': torch.int64, 'device': device(type='cuda', index=0)}
590
+ [rank0]: tensor at position 39:
591
+ [rank0]: saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
592
+ [rank0]: recomputed metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
593
+ [rank0]: tensor at position 40:
594
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
595
+ [rank0]: recomputed metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
596
+ [rank0]: tensor at position 41:
597
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
598
+ [rank0]: recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
599
+ [rank0]: tensor at position 42:
600
+ [rank0]: saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
601
+ [rank0]: recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
602
+ [rank0]: tensor at position 43:
603
+ [rank0]: saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
604
+ [rank0]: recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
605
+ [rank0]: tensor at position 44:
606
+ [rank0]: saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
607
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
608
+ [rank0]: tensor at position 45:
609
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
610
+ [rank0]: recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
611
+ [rank0]: tensor at position 46:
612
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
613
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
614
+ [rank0]: tensor at position 47:
615
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
616
+ [rank0]: recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
617
+ [rank0]: tensor at position 48:
618
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
619
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
620
+ [rank0]: tensor at position 49:
621
+ [rank0]: saved metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
622
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
623
+ [rank0]: tensor at position 50:
624
+ [rank0]: saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
625
+ [rank0]: recomputed metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
626
+ [rank0]: tensor at position 51:
627
+ [rank0]: saved metadata: {'shape': torch.Size([512, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
628
+ [rank0]: recomputed metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
629
+ [rank0]: tensor at position 52:
630
+ [rank0]: saved metadata: {'shape': torch.Size([512, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
631
+ [rank0]: recomputed metadata: {'shape': torch.Size([14260, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
632
+ [rank0]: tensor at position 53:
633
+ [rank0]: saved metadata: {'shape': torch.Size([14260, 24, 128]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
634
+ [rank0]: recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
635
+ [rank0]: tensor at position 54:
636
+ [rank0]: saved metadata: {'shape': torch.Size([24, 14260]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
637
+ [rank0]: recomputed metadata: {'shape': torch.Size([14260, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
638
+ [rank0]: tensor at position 55:
639
+ [rank0]: saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)}
640
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
641
+ [rank0]: tensor at position 56:
642
+ [rank0]: saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int32, 'device': device(type='cuda', index=0)}
643
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
644
+ [rank0]: tensor at position 57:
645
+ [rank0]: saved metadata: {'shape': torch.Size([2]), 'dtype': torch.int64, 'device': device(type='cuda', index=0)}
646
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
647
+ [rank0]: tensor at position 58:
648
+ [rank0]: saved metadata: {'shape': torch.Size([3072, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
649
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
650
+ [rank0]: tensor at position 59:
651
+ [rank0]: saved metadata: {'shape': torch.Size([3072, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
652
+ [rank0]: recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
653
+ [rank0]: tensor at position 60:
654
+ [rank0]: saved metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
655
+ [rank0]: recomputed metadata: {'shape': torch.Size([512, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
656
+ [rank0]: tensor at position 61:
657
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
658
+ [rank0]: recomputed metadata: {'shape': torch.Size([64, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
659
+ [rank0]: tensor at position 62:
660
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
661
+ [rank0]: recomputed metadata: {'shape': torch.Size([512, 64]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
662
+ [rank0]: tensor at position 63:
663
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
664
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 512, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
665
+ [rank0]: tensor at position 64:
666
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
667
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 512, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
668
+ [rank0]: tensor at position 65:
669
+ [rank0]: saved metadata: {'shape': torch.Size([3072, 14336]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
670
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 512, 1]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
671
+ [rank0]: tensor at position 66:
672
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 14336]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
673
+ [rank0]: recomputed metadata: {'shape': torch.Size([1, 512, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
674
+ [rank0]: tensor at position 67:
675
+ [rank0]: saved metadata: {'shape': torch.Size([14336, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
676
+ [rank0]: recomputed metadata: {'shape': torch.Size([3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
677
+ [rank0]: tensor at position 68:
678
+ [rank0]: saved metadata: {'shape': torch.Size([1, 14260, 3072]), 'dtype': torch.float32, 'device': device(type='cuda', index=0)}
679
+ [rank0]: recomputed metadata: {'shape': torch.Size([512, 3072]), 'dtype': torch.bfloat16, 'device': device(type='cuda', index=0)}
680
+ [rank0]: .
681
+
682
+ [rank0]: Tip: To see a more detailed error message, either pass `debug=True` to
683
+ [rank0]: `torch.utils.checkpoint.checkpoint(...)` or wrap the code block
684
+ [rank0]: with `with torch.utils.checkpoint.set_checkpoint_debug_enabled(True):` to
685
+ [rank0]: enable checkpoint‑debug mode globally.
wandb/run-20260408_004931-ku6kqemc/files/wandb-metadata.json ADDED
@@ -0,0 +1,144 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-5.15.120.bsk.6-amd64-x86_64-with-glibc2.36",
3
+ "python": "3.11.2",
4
+ "startedAt": "2026-04-08T00:49:31.229785Z",
5
+ "args": [
6
+ "--task",
7
+ "ti2v-5B",
8
+ "--size",
9
+ "640*736",
10
+ "--frame_num",
11
+ "121",
12
+ "--ckpt_dir",
13
+ "ckpts/Wan2.2-TI2V-5B",
14
+ "--pt_dir",
15
+ "ckpts/vidar_ckpt/merged_vidar_lora.pt",
16
+ "--dataset_json",
17
+ "data/rl_train/robotwin_stack_blocks_three.json",
18
+ "--output_dir",
19
+ "data/outputs/creflow_stack_blocks_three",
20
+ "--num_ode_steps",
21
+ "20",
22
+ "--sample_shift",
23
+ "5.0",
24
+ "--sample_guide_scale",
25
+ "5.0",
26
+ "--K",
27
+ "16",
28
+ "--num_rounds",
29
+ "10",
30
+ "--epochs_per_round",
31
+ "100",
32
+ "--max_per_condition",
33
+ "64",
34
+ "--effective_batch_size",
35
+ "16",
36
+ "--w_bad",
37
+ "1.0",
38
+ "--lambda_gripper",
39
+ "1.0",
40
+ "--seed",
41
+ "42",
42
+ "--reward_config",
43
+ "blocks_stack_v2",
44
+ "--convert_model_dtype",
45
+ "--offload_model",
46
+ "false",
47
+ "--learning_rate",
48
+ "1e-5",
49
+ "--weight_decay",
50
+ "0.01",
51
+ "--max_grad_norm",
52
+ "2.0",
53
+ "--checkpointing_steps",
54
+ "10",
55
+ "--lora_rank",
56
+ "64",
57
+ "--lora_alpha",
58
+ "64",
59
+ "--wandb_project",
60
+ "Corrective-Reflow",
61
+ "--wandb_run_name",
62
+ "creflow_stack_blocks_three"
63
+ ],
64
+ "program": "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py",
65
+ "codePath": "fastvideo/train_creflow_wan_2_2_ti2v.py",
66
+ "git": {
67
+ "remote": "git@github.com:VincentNi0107/EmbodiedVideoRL.git",
68
+ "commit": "a8f5b170f0ef7d865af7e8f32c4aebb46efec2c1"
69
+ },
70
+ "email": "1163051845@qq.com",
71
+ "root": "data/outputs/creflow_stack_blocks_three",
72
+ "host": "n124-104-180",
73
+ "username": "tiger",
74
+ "executable": "/usr/bin/python",
75
+ "codePathLocal": "fastvideo/train_creflow_wan_2_2_ti2v.py",
76
+ "cpu_count": 104,
77
+ "cpu_count_logical": 208,
78
+ "gpu": "NVIDIA H100 80GB HBM3",
79
+ "gpu_count": 8,
80
+ "disk": {
81
+ "/": {
82
+ "total": "1055740600320",
83
+ "used": "481703223296"
84
+ }
85
+ },
86
+ "memory": {
87
+ "total": "1977660338176"
88
+ },
89
+ "cpu": {
90
+ "count": 104,
91
+ "countLogical": 208
92
+ },
93
+ "gpu_nvidia": [
94
+ {
95
+ "name": "NVIDIA H100 80GB HBM3",
96
+ "memoryTotal": "85520809984",
97
+ "cudaCores": 16896,
98
+ "architecture": "Hopper"
99
+ },
100
+ {
101
+ "name": "NVIDIA H100 80GB HBM3",
102
+ "memoryTotal": "85520809984",
103
+ "cudaCores": 16896,
104
+ "architecture": "Hopper"
105
+ },
106
+ {
107
+ "name": "NVIDIA H100 80GB HBM3",
108
+ "memoryTotal": "85520809984",
109
+ "cudaCores": 16896,
110
+ "architecture": "Hopper"
111
+ },
112
+ {
113
+ "name": "NVIDIA H100 80GB HBM3",
114
+ "memoryTotal": "85520809984",
115
+ "cudaCores": 16896,
116
+ "architecture": "Hopper"
117
+ },
118
+ {
119
+ "name": "NVIDIA H100 80GB HBM3",
120
+ "memoryTotal": "85520809984",
121
+ "cudaCores": 16896,
122
+ "architecture": "Hopper"
123
+ },
124
+ {
125
+ "name": "NVIDIA H100 80GB HBM3",
126
+ "memoryTotal": "85520809984",
127
+ "cudaCores": 16896,
128
+ "architecture": "Hopper"
129
+ },
130
+ {
131
+ "name": "NVIDIA H100 80GB HBM3",
132
+ "memoryTotal": "85520809984",
133
+ "cudaCores": 16896,
134
+ "architecture": "Hopper"
135
+ },
136
+ {
137
+ "name": "NVIDIA H100 80GB HBM3",
138
+ "memoryTotal": "85520809984",
139
+ "cudaCores": 16896,
140
+ "architecture": "Hopper"
141
+ }
142
+ ],
143
+ "cudaVersion": "12.9"
144
+ }
wandb/run-20260408_004931-ku6kqemc/files/wandb-summary.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"_wandb":{"runtime":1584}}
wandb/run-20260408_004931-ku6kqemc/logs/debug-internal.log ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-04-08T00:49:31.238719294Z","level":"INFO","msg":"using version","core version":"0.18.5"}
2
+ {"time":"2026-04-08T00:49:31.23873382Z","level":"INFO","msg":"created symlink","path":"data/outputs/creflow_stack_blocks_three/wandb/run-20260408_004931-ku6kqemc/logs/debug-core.log"}
3
+ {"time":"2026-04-08T00:49:31.455165146Z","level":"INFO","msg":"created new stream","id":"ku6kqemc"}
4
+ {"time":"2026-04-08T00:49:31.455371786Z","level":"INFO","msg":"stream: started","id":"ku6kqemc"}
5
+ {"time":"2026-04-08T00:49:31.455479963Z","level":"INFO","msg":"sender: started","stream_id":"ku6kqemc"}
6
+ {"time":"2026-04-08T00:49:31.455431707Z","level":"INFO","msg":"handler: started","stream_id":{"value":"ku6kqemc"}}
7
+ {"time":"2026-04-08T00:49:31.455410767Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"ku6kqemc"}}
8
+ {"time":"2026-04-08T00:49:31.831542842Z","level":"INFO","msg":"Starting system monitor"}
9
+ {"time":"2026-04-08T01:00:32.624028988Z","level":"INFO","msg":"api: retrying error","error":"Post \"https://api.wandb.ai/graphql\": context deadline exceeded"}
10
+ {"time":"2026-04-08T01:01:05.122882271Z","level":"INFO","msg":"api: retrying error","error":"Post \"https://api.wandb.ai/graphql\": net/http: request canceled (Client.Timeout exceeded while awaiting headers)"}
11
+ {"time":"2026-04-08T01:09:32.404506925Z","level":"INFO","msg":"api: retrying error","error":"Post \"https://api.wandb.ai/files/vincentni/Corrective-Reflow/ku6kqemc/file_stream\": dial tcp 35.186.228.49:443: connect: connection timed out"}
12
+ {"time":"2026-04-08T01:15:56.227072593Z","level":"INFO","msg":"stream: closing","id":"ku6kqemc"}
13
+ {"time":"2026-04-08T01:15:56.227162615Z","level":"INFO","msg":"Stopping system monitor"}
14
+ {"time":"2026-04-08T01:15:56.232220483Z","level":"INFO","msg":"Stopped system monitor"}
15
+ {"time":"2026-04-08T01:15:56.791766357Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
16
+ {"time":"2026-04-08T01:15:56.926286921Z","level":"INFO","msg":"handler: closed","stream_id":{"value":"ku6kqemc"}}
17
+ {"time":"2026-04-08T01:15:56.926354182Z","level":"INFO","msg":"sender: closed","stream_id":"ku6kqemc"}
18
+ {"time":"2026-04-08T01:15:56.926329752Z","level":"INFO","msg":"writer: Close: closed","stream_id":{"value":"ku6kqemc"}}
19
+ {"time":"2026-04-08T01:15:56.930384735Z","level":"INFO","msg":"stream: closed","id":"ku6kqemc"}
wandb/run-20260408_004931-ku6kqemc/logs/debug.log ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 2026-04-08 00:49:31,195 INFO MainThread:108291 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5
2
+ 2026-04-08 00:49:31,195 INFO MainThread:108291 [wandb_setup.py:_flush():79] Configure stats pid to 108291
3
+ 2026-04-08 00:49:31,195 INFO MainThread:108291 [wandb_setup.py:_flush():79] Loading settings from /home/tiger/.config/wandb/settings
4
+ 2026-04-08 00:49:31,196 INFO MainThread:108291 [wandb_setup.py:_flush():79] Loading settings from /mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/wandb/settings
5
+ 2026-04-08 00:49:31,196 INFO MainThread:108291 [wandb_setup.py:_flush():79] Loading settings from environment variables: {'disabled': 'false', 'project': 'Corrective-Reflow', 'api_key': '***REDACTED***', 'mode': 'online'}
6
+ 2026-04-08 00:49:31,196 INFO MainThread:108291 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': 'online', '_disable_service': None}
7
+ 2026-04-08 00:49:31,196 INFO MainThread:108291 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'fastvideo/train_creflow_wan_2_2_ti2v.py', 'program_abspath': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py', 'program': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py'}
8
+ 2026-04-08 00:49:31,196 INFO MainThread:108291 [wandb_setup.py:_flush():79] Applying login settings: {}
9
+ 2026-04-08 00:49:31,198 INFO MainThread:108291 [wandb_init.py:_log_setup():534] Logging user logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_004931-ku6kqemc/logs/debug.log
10
+ 2026-04-08 00:49:31,199 INFO MainThread:108291 [wandb_init.py:_log_setup():535] Logging internal logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_004931-ku6kqemc/logs/debug-internal.log
11
+ 2026-04-08 00:49:31,200 INFO MainThread:108291 [wandb_init.py:init():621] calling init triggers
12
+ 2026-04-08 00:49:31,200 INFO MainThread:108291 [wandb_init.py:init():628] wandb.init called with sweep_config: {}
13
+ config: {'vidar_root': '', 'task': 'ti2v-5B', 'size': '640*736', 'frame_num': 121, 'ckpt_dir': 'ckpts/Wan2.2-TI2V-5B', 'pt_dir': 'ckpts/vidar_ckpt/merged_vidar_lora.pt', 'convert_model_dtype': True, 'offload_model': False, 'dataset_json': 'data/rl_train/robotwin_stack_blocks_three.json', 'max_samples': -1, 'output_dir': 'data/outputs/creflow_stack_blocks_three', 'num_ode_steps': 20, 'sample_shift': 5.0, 'sample_guide_scale': 5.0, 'neg_prompt': '色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走', 'seed': 42, 'K': 16, 'num_rounds': 10, 'epochs_per_round': 100, 'effective_batch_size': 16, 'w_bad': 1.0, 'max_per_condition': 64, 'lambda_gripper': 1.0, 'reset_optimizer_per_round': False, 'num_train_timesteps': 1000, 'learning_rate': 1e-05, 'weight_decay': 0.01, 'max_grad_norm': 2.0, 'checkpointing_steps': 10, 'gradient_checkpointing': True, 'use_8bit_adam': True, 'lora_rank': 64, 'lora_alpha': 64, 'lora_target_modules': None, 'resume_from_lora_checkpoint': None, 'reward_config': 'blocks_stack_v2', 'reward_backend': 'blocks_stack_v2', 'skip_reward_debug_video': True, 'wandb_project': 'Corrective-Reflow', 'wandb_run_name': 'creflow_stack_blocks_three', 'hallucination_crop_top_ratio': 0.6667, 'bsv2_prompts': ['red block', 'green block', 'blue block'], 'bsv2_pick_thr': 0.05, 'bsv2_place_thr': 0.05, 'bsv2_vertical_sep_thr': 0.03, 'bsv2_x_align_thr': 0.025, 'bsv2_check_window_frac': 0.2, 'bsv2_dup_max_frames': 0, 'bsv2_expected_grip_changes': 6, 'bsv2_mj_lo': 0.6, 'bsv2_mj_hi': 1.3, 'bsv2_idm_ckpt_path': 'ckpts/vidar_ckpt/idm.pt'}
14
+ 2026-04-08 00:49:31,200 INFO MainThread:108291 [wandb_init.py:init():671] starting backend
15
+ 2026-04-08 00:49:31,200 INFO MainThread:108291 [wandb_init.py:init():675] sending inform_init request
16
+ 2026-04-08 00:49:31,227 INFO MainThread:108291 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
17
+ 2026-04-08 00:49:31,227 INFO MainThread:108291 [wandb_init.py:init():688] backend started and connected
18
+ 2026-04-08 00:49:31,285 INFO MainThread:108291 [wandb_init.py:init():783] updated telemetry
19
+ 2026-04-08 00:49:31,449 INFO MainThread:108291 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout
20
+ 2026-04-08 00:49:31,780 INFO MainThread:108291 [wandb_init.py:init():867] starting run threads in backend
21
+ 2026-04-08 00:49:32,146 INFO MainThread:108291 [wandb_run.py:_console_start():2463] atexit reg
22
+ 2026-04-08 00:49:32,146 INFO MainThread:108291 [wandb_run.py:_redirect():2311] redirect: wrap_raw
23
+ 2026-04-08 00:49:32,146 INFO MainThread:108291 [wandb_run.py:_redirect():2376] Wrapping output streams.
24
+ 2026-04-08 00:49:32,146 INFO MainThread:108291 [wandb_run.py:_redirect():2401] Redirects installed.
25
+ 2026-04-08 00:49:32,151 INFO MainThread:108291 [wandb_init.py:init():911] run started, returning control to user process
26
+ 2026-04-08 01:15:56,227 WARNING MsgRouterThr:108291 [router.py:message_loop():77] message_loop has been closed
wandb/run-20260408_012032-idkfwxb0/files/config.yaml ADDED
@@ -0,0 +1,145 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ _wandb:
2
+ value:
3
+ cli_version: 0.18.5
4
+ m: []
5
+ python_version: 3.11.2
6
+ t:
7
+ "1":
8
+ - 1
9
+ - 11
10
+ - 41
11
+ - 49
12
+ - 55
13
+ - 71
14
+ - 83
15
+ - 105
16
+ "2":
17
+ - 1
18
+ - 11
19
+ - 41
20
+ - 49
21
+ - 55
22
+ - 63
23
+ - 71
24
+ - 83
25
+ - 98
26
+ - 105
27
+ "3":
28
+ - 13
29
+ - 16
30
+ - 23
31
+ - 55
32
+ "4": 3.11.2
33
+ "5": 0.18.5
34
+ "6": 4.46.1
35
+ "8":
36
+ - 5
37
+ "12": 0.18.5
38
+ "13": linux-x86_64
39
+ K:
40
+ value: 16
41
+ bsv2_check_window_frac:
42
+ value: 0.2
43
+ bsv2_dup_max_frames:
44
+ value: 0
45
+ bsv2_expected_grip_changes:
46
+ value: 6
47
+ bsv2_idm_ckpt_path:
48
+ value: ckpts/vidar_ckpt/idm.pt
49
+ bsv2_mj_hi:
50
+ value: 1.3
51
+ bsv2_mj_lo:
52
+ value: 0.6
53
+ bsv2_pick_thr:
54
+ value: 0.05
55
+ bsv2_place_thr:
56
+ value: 0.05
57
+ bsv2_prompts:
58
+ value:
59
+ - red block
60
+ - green block
61
+ - blue block
62
+ bsv2_vertical_sep_thr:
63
+ value: 0.03
64
+ bsv2_x_align_thr:
65
+ value: 0.025
66
+ checkpointing_steps:
67
+ value: 10
68
+ ckpt_dir:
69
+ value: ckpts/Wan2.2-TI2V-5B
70
+ convert_model_dtype:
71
+ value: true
72
+ dataset_json:
73
+ value: data/rl_train/robotwin_stack_blocks_three.json
74
+ effective_batch_size:
75
+ value: 16
76
+ epochs_per_round:
77
+ value: 100
78
+ frame_num:
79
+ value: 121
80
+ gradient_checkpointing:
81
+ value: true
82
+ hallucination_crop_top_ratio:
83
+ value: 0.6667
84
+ lambda_gripper:
85
+ value: 1
86
+ learning_rate:
87
+ value: 1e-05
88
+ lora_alpha:
89
+ value: 64
90
+ lora_rank:
91
+ value: 64
92
+ lora_target_modules:
93
+ value: null
94
+ max_grad_norm:
95
+ value: 2
96
+ max_per_condition:
97
+ value: 64
98
+ max_samples:
99
+ value: -1
100
+ neg_prompt:
101
+ value: 色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走
102
+ num_ode_steps:
103
+ value: 20
104
+ num_rounds:
105
+ value: 10
106
+ num_train_timesteps:
107
+ value: 1000
108
+ offload_model:
109
+ value: false
110
+ output_dir:
111
+ value: data/outputs/creflow_stack_blocks_three
112
+ pt_dir:
113
+ value: ckpts/vidar_ckpt/merged_vidar_lora.pt
114
+ reset_optimizer_per_round:
115
+ value: false
116
+ resume_from_lora_checkpoint:
117
+ value: null
118
+ reward_backend:
119
+ value: blocks_stack_v2
120
+ reward_config:
121
+ value: blocks_stack_v2
122
+ sample_guide_scale:
123
+ value: 5
124
+ sample_shift:
125
+ value: 5
126
+ seed:
127
+ value: 42
128
+ size:
129
+ value: 640*736
130
+ skip_reward_debug_video:
131
+ value: true
132
+ task:
133
+ value: ti2v-5B
134
+ use_8bit_adam:
135
+ value: true
136
+ vidar_root:
137
+ value: ""
138
+ w_bad:
139
+ value: 1
140
+ wandb_project:
141
+ value: Corrective-Reflow
142
+ wandb_run_name:
143
+ value: creflow_stack_blocks_three
144
+ weight_decay:
145
+ value: 0.01
wandb/run-20260408_012032-idkfwxb0/files/output.log ADDED
@@ -0,0 +1,133 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Building Wan2.2 TI2V model ... (DDP=True, world_size=8)
2
+ /usr/local/lib/python3.11/dist-packages/torch/distributed/distributed_c10d.py:4876: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
3
+ warnings.warn( # warn only once
4
+ Moving T5 encoder to GPU permanently (offload_model=false) ...
5
+ Building reward scorer ...
6
+ Loading IDM from ckpts/vidar_ckpt/idm.pt on cuda:0 ...
7
+ IDM loaded.
8
+ Loading SAM3 VideoPredictor (blocks_stack_v2) on cuda:0 ...
9
+ INFO 2026-04-08 01:21:50,257 195712 sam3_video_base.py: 125: setting max_num_objects=10000 and num_obj_for_compile=16
10
+ SAM3 VideoPredictor (blocks_stack_v2) loaded.
11
+ Applying LoRA (single adapter) for CReflow ...
12
+ LoRA injected: rank=64, alpha=64, target_modules=['self_attn.q', 'self_attn.k', 'self_attn.v', 'self_attn.o', 'cross_attn.q', 'cross_attn.k', 'cross_attn.v', 'cross_attn.o']
13
+ Trainable: 94.4 M / 5094.2 M total (1.85%)
14
+ Gradient checkpointing enabled on 30 DiT blocks
15
+ Trainable parameters: 94.4 M
16
+ Using 8-bit AdamW (bitsandbytes)
17
+ Dataset: 10 conditions
18
+ Encoding 10 conditions ...
19
+ [1/10] 6b20973f10ef... encoded
20
+ [2/10] cff33035b28a... encoded
21
+ [3/10] 7ebc8b336779... encoded
22
+ [4/10] 8077c679fb62... encoded
23
+ [5/10] 41855438a8e0... encoded
24
+ [6/10] a67b02f4e7ce... encoded
25
+ [7/10] 4568a9603f3b... encoded
26
+ [8/10] 1b67584bff87... encoded
27
+ [9/10] f4f8079f6637... encoded
28
+ [10/10] 1b70247c6412... encoded
29
+ All 10 conditions encoded.
30
+
31
+ Starting Corrective Reflow | rounds=10 K=16 K_local=2 epochs/round=100 effective_bs=16 lr=1e-05 conditions=10
32
+
33
+ ======================================================================
34
+ ROUND 1/10
35
+ ======================================================================
36
+ [Phase 1] ODE rollout ...
37
+ [CReflow rollout] condition 0/10: K=2, id=6b20973f10ef...
38
+ Traceback (most recent call last):
39
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1038, in <module>
40
+ main()
41
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 870, in main
42
+ rollouts = creflow_rollout(
43
+ ^^^^^^^^^^^^^^^^
44
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/creflow/rollout.py", line 267, in creflow_rollout
45
+ x0_preds = _ode_rollout_batch(
46
+ ^^^^^^^^^^^^^^^^^^^
47
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/creflow/rollout.py", line 139, in _ode_rollout_batch
48
+ outputs = dit(x_list, t=ts_batch, context=context_list, seq_len=seq_len)
49
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
50
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl
51
+ return self._call_impl(*args, **kwargs)
52
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
53
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl
54
+ return forward_call(*args, **kwargs)
55
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
56
+ File "/home/tiger/.local/lib/python3.11/site-packages/peft/peft_model.py", line 812, in forward
57
+ return self.get_base_model()(*args, **kwargs)
58
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
59
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl
60
+ return self._call_impl(*args, **kwargs)
61
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
62
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl
63
+ return forward_call(*args, **kwargs)
64
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
65
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/models/wan/modules/model.py", line 498, in forward
66
+ x = block(x, **kwargs)
67
+ ^^^^^^^^^^^^^^^^^^
68
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl
69
+ return self._call_impl(*args, **kwargs)
70
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
71
+ File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl
72
+ return forward_call(*args, **kwargs)
73
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
74
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 130, in _ckpt_fwd
75
+ return torch_ckpt.checkpoint(orig_fwd, *args, use_reentrant=True, **kwargs)
76
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
77
+ File "/usr/local/lib/python3.11/dist-packages/torch/_compile.py", line 53, in inner
78
+ return disable_fn(*args, **kwargs)
79
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^
80
+ File "/usr/local/lib/python3.11/dist-packages/torch/_dynamo/eval_frame.py", line 1044, in _fn
81
+ return fn(*args, **kwargs)
82
+ ^^^^^^^^^^^^^^^^^^^
83
+ File "/usr/local/lib/python3.11/dist-packages/torch/utils/checkpoint.py", line 486, in checkpoint
84
+ raise ValueError(
85
+ ValueError: Unexpected keyword arguments: e,seq_lens,grid_sizes,freqs,context,context_lens
86
+ [rank0]: Traceback (most recent call last):
87
+ [rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1038, in <module>
88
+ [rank0]: main()
89
+ [rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 870, in main
90
+ [rank0]: rollouts = creflow_rollout(
91
+ [rank0]: ^^^^^^^^^^^^^^^^
92
+ [rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/creflow/rollout.py", line 267, in creflow_rollout
93
+ [rank0]: x0_preds = _ode_rollout_batch(
94
+ [rank0]: ^^^^^^^^^^^^^^^^^^^
95
+ [rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/creflow/rollout.py", line 139, in _ode_rollout_batch
96
+ [rank0]: outputs = dit(x_list, t=ts_batch, context=context_list, seq_len=seq_len)
97
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
98
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl
99
+ [rank0]: return self._call_impl(*args, **kwargs)
100
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
101
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl
102
+ [rank0]: return forward_call(*args, **kwargs)
103
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
104
+ [rank0]: File "/home/tiger/.local/lib/python3.11/site-packages/peft/peft_model.py", line 812, in forward
105
+ [rank0]: return self.get_base_model()(*args, **kwargs)
106
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
107
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl
108
+ [rank0]: return self._call_impl(*args, **kwargs)
109
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
110
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl
111
+ [rank0]: return forward_call(*args, **kwargs)
112
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
113
+ [rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/models/wan/modules/model.py", line 498, in forward
114
+ [rank0]: x = block(x, **kwargs)
115
+ [rank0]: ^^^^^^^^^^^^^^^^^^
116
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1775, in _wrapped_call_impl
117
+ [rank0]: return self._call_impl(*args, **kwargs)
118
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
119
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/nn/modules/module.py", line 1786, in _call_impl
120
+ [rank0]: return forward_call(*args, **kwargs)
121
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
122
+ [rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 130, in _ckpt_fwd
123
+ [rank0]: return torch_ckpt.checkpoint(orig_fwd, *args, use_reentrant=True, **kwargs)
124
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
125
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/_compile.py", line 53, in inner
126
+ [rank0]: return disable_fn(*args, **kwargs)
127
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^
128
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/_dynamo/eval_frame.py", line 1044, in _fn
129
+ [rank0]: return fn(*args, **kwargs)
130
+ [rank0]: ^^^^^^^^^^^^^^^^^^^
131
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/utils/checkpoint.py", line 486, in checkpoint
132
+ [rank0]: raise ValueError(
133
+ [rank0]: ValueError: Unexpected keyword arguments: e,seq_lens,grid_sizes,freqs,context,context_lens
wandb/run-20260408_012032-idkfwxb0/files/wandb-metadata.json ADDED
@@ -0,0 +1,144 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-5.15.120.bsk.6-amd64-x86_64-with-glibc2.36",
3
+ "python": "3.11.2",
4
+ "startedAt": "2026-04-08T01:20:32.358609Z",
5
+ "args": [
6
+ "--task",
7
+ "ti2v-5B",
8
+ "--size",
9
+ "640*736",
10
+ "--frame_num",
11
+ "121",
12
+ "--ckpt_dir",
13
+ "ckpts/Wan2.2-TI2V-5B",
14
+ "--pt_dir",
15
+ "ckpts/vidar_ckpt/merged_vidar_lora.pt",
16
+ "--dataset_json",
17
+ "data/rl_train/robotwin_stack_blocks_three.json",
18
+ "--output_dir",
19
+ "data/outputs/creflow_stack_blocks_three",
20
+ "--num_ode_steps",
21
+ "20",
22
+ "--sample_shift",
23
+ "5.0",
24
+ "--sample_guide_scale",
25
+ "5.0",
26
+ "--K",
27
+ "16",
28
+ "--num_rounds",
29
+ "10",
30
+ "--epochs_per_round",
31
+ "100",
32
+ "--max_per_condition",
33
+ "64",
34
+ "--effective_batch_size",
35
+ "16",
36
+ "--w_bad",
37
+ "1.0",
38
+ "--lambda_gripper",
39
+ "1.0",
40
+ "--seed",
41
+ "42",
42
+ "--reward_config",
43
+ "blocks_stack_v2",
44
+ "--convert_model_dtype",
45
+ "--offload_model",
46
+ "false",
47
+ "--learning_rate",
48
+ "1e-5",
49
+ "--weight_decay",
50
+ "0.01",
51
+ "--max_grad_norm",
52
+ "2.0",
53
+ "--checkpointing_steps",
54
+ "10",
55
+ "--lora_rank",
56
+ "64",
57
+ "--lora_alpha",
58
+ "64",
59
+ "--wandb_project",
60
+ "Corrective-Reflow",
61
+ "--wandb_run_name",
62
+ "creflow_stack_blocks_three"
63
+ ],
64
+ "program": "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py",
65
+ "codePath": "fastvideo/train_creflow_wan_2_2_ti2v.py",
66
+ "git": {
67
+ "remote": "git@github.com:VincentNi0107/EmbodiedVideoRL.git",
68
+ "commit": "a8f5b170f0ef7d865af7e8f32c4aebb46efec2c1"
69
+ },
70
+ "email": "1163051845@qq.com",
71
+ "root": "data/outputs/creflow_stack_blocks_three",
72
+ "host": "n124-104-180",
73
+ "username": "tiger",
74
+ "executable": "/usr/bin/python",
75
+ "codePathLocal": "fastvideo/train_creflow_wan_2_2_ti2v.py",
76
+ "cpu_count": 104,
77
+ "cpu_count_logical": 208,
78
+ "gpu": "NVIDIA H100 80GB HBM3",
79
+ "gpu_count": 8,
80
+ "disk": {
81
+ "/": {
82
+ "total": "1055740600320",
83
+ "used": "482171138048"
84
+ }
85
+ },
86
+ "memory": {
87
+ "total": "1977660338176"
88
+ },
89
+ "cpu": {
90
+ "count": 104,
91
+ "countLogical": 208
92
+ },
93
+ "gpu_nvidia": [
94
+ {
95
+ "name": "NVIDIA H100 80GB HBM3",
96
+ "memoryTotal": "85520809984",
97
+ "cudaCores": 16896,
98
+ "architecture": "Hopper"
99
+ },
100
+ {
101
+ "name": "NVIDIA H100 80GB HBM3",
102
+ "memoryTotal": "85520809984",
103
+ "cudaCores": 16896,
104
+ "architecture": "Hopper"
105
+ },
106
+ {
107
+ "name": "NVIDIA H100 80GB HBM3",
108
+ "memoryTotal": "85520809984",
109
+ "cudaCores": 16896,
110
+ "architecture": "Hopper"
111
+ },
112
+ {
113
+ "name": "NVIDIA H100 80GB HBM3",
114
+ "memoryTotal": "85520809984",
115
+ "cudaCores": 16896,
116
+ "architecture": "Hopper"
117
+ },
118
+ {
119
+ "name": "NVIDIA H100 80GB HBM3",
120
+ "memoryTotal": "85520809984",
121
+ "cudaCores": 16896,
122
+ "architecture": "Hopper"
123
+ },
124
+ {
125
+ "name": "NVIDIA H100 80GB HBM3",
126
+ "memoryTotal": "85520809984",
127
+ "cudaCores": 16896,
128
+ "architecture": "Hopper"
129
+ },
130
+ {
131
+ "name": "NVIDIA H100 80GB HBM3",
132
+ "memoryTotal": "85520809984",
133
+ "cudaCores": 16896,
134
+ "architecture": "Hopper"
135
+ },
136
+ {
137
+ "name": "NVIDIA H100 80GB HBM3",
138
+ "memoryTotal": "85520809984",
139
+ "cudaCores": 16896,
140
+ "architecture": "Hopper"
141
+ }
142
+ ],
143
+ "cudaVersion": "12.9"
144
+ }
wandb/run-20260408_012032-idkfwxb0/files/wandb-summary.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"_wandb":{"runtime":84}}
wandb/run-20260408_012032-idkfwxb0/logs/debug-internal.log ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-04-08T01:20:32.367150627Z","level":"INFO","msg":"using version","core version":"0.18.5"}
2
+ {"time":"2026-04-08T01:20:32.367162484Z","level":"INFO","msg":"created symlink","path":"data/outputs/creflow_stack_blocks_three/wandb/run-20260408_012032-idkfwxb0/logs/debug-core.log"}
3
+ {"time":"2026-04-08T01:20:32.581603896Z","level":"INFO","msg":"created new stream","id":"idkfwxb0"}
4
+ {"time":"2026-04-08T01:20:32.581769665Z","level":"INFO","msg":"stream: started","id":"idkfwxb0"}
5
+ {"time":"2026-04-08T01:20:32.5818252Z","level":"INFO","msg":"sender: started","stream_id":"idkfwxb0"}
6
+ {"time":"2026-04-08T01:20:32.581798295Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"idkfwxb0"}}
7
+ {"time":"2026-04-08T01:20:32.581839505Z","level":"INFO","msg":"handler: started","stream_id":{"value":"idkfwxb0"}}
8
+ {"time":"2026-04-08T01:20:32.990097512Z","level":"INFO","msg":"Starting system monitor"}
9
+ {"time":"2026-04-08T01:21:57.340247846Z","level":"INFO","msg":"stream: closing","id":"idkfwxb0"}
10
+ {"time":"2026-04-08T01:21:57.340446284Z","level":"INFO","msg":"Stopping system monitor"}
11
+ {"time":"2026-04-08T01:21:57.346621494Z","level":"INFO","msg":"Stopped system monitor"}
12
+ {"time":"2026-04-08T01:21:58.004298171Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
13
+ {"time":"2026-04-08T01:21:58.157765962Z","level":"INFO","msg":"handler: closed","stream_id":{"value":"idkfwxb0"}}
14
+ {"time":"2026-04-08T01:21:58.157800232Z","level":"INFO","msg":"writer: Close: closed","stream_id":{"value":"idkfwxb0"}}
15
+ {"time":"2026-04-08T01:21:58.157836905Z","level":"INFO","msg":"sender: closed","stream_id":"idkfwxb0"}
16
+ {"time":"2026-04-08T01:21:58.16571198Z","level":"INFO","msg":"stream: closed","id":"idkfwxb0"}
wandb/run-20260408_012032-idkfwxb0/logs/debug.log ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 2026-04-08 01:20:32,325 INFO MainThread:195712 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5
2
+ 2026-04-08 01:20:32,325 INFO MainThread:195712 [wandb_setup.py:_flush():79] Configure stats pid to 195712
3
+ 2026-04-08 01:20:32,325 INFO MainThread:195712 [wandb_setup.py:_flush():79] Loading settings from /home/tiger/.config/wandb/settings
4
+ 2026-04-08 01:20:32,325 INFO MainThread:195712 [wandb_setup.py:_flush():79] Loading settings from /mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/wandb/settings
5
+ 2026-04-08 01:20:32,325 INFO MainThread:195712 [wandb_setup.py:_flush():79] Loading settings from environment variables: {'disabled': 'false', 'project': 'Corrective-Reflow', 'api_key': '***REDACTED***', 'mode': 'online'}
6
+ 2026-04-08 01:20:32,325 INFO MainThread:195712 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': 'online', '_disable_service': None}
7
+ 2026-04-08 01:20:32,325 INFO MainThread:195712 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'fastvideo/train_creflow_wan_2_2_ti2v.py', 'program_abspath': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py', 'program': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py'}
8
+ 2026-04-08 01:20:32,325 INFO MainThread:195712 [wandb_setup.py:_flush():79] Applying login settings: {}
9
+ 2026-04-08 01:20:32,327 INFO MainThread:195712 [wandb_init.py:_log_setup():534] Logging user logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_012032-idkfwxb0/logs/debug.log
10
+ 2026-04-08 01:20:32,329 INFO MainThread:195712 [wandb_init.py:_log_setup():535] Logging internal logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_012032-idkfwxb0/logs/debug-internal.log
11
+ 2026-04-08 01:20:32,329 INFO MainThread:195712 [wandb_init.py:init():621] calling init triggers
12
+ 2026-04-08 01:20:32,329 INFO MainThread:195712 [wandb_init.py:init():628] wandb.init called with sweep_config: {}
13
+ config: {'vidar_root': '', 'task': 'ti2v-5B', 'size': '640*736', 'frame_num': 121, 'ckpt_dir': 'ckpts/Wan2.2-TI2V-5B', 'pt_dir': 'ckpts/vidar_ckpt/merged_vidar_lora.pt', 'convert_model_dtype': True, 'offload_model': False, 'dataset_json': 'data/rl_train/robotwin_stack_blocks_three.json', 'max_samples': -1, 'output_dir': 'data/outputs/creflow_stack_blocks_three', 'num_ode_steps': 20, 'sample_shift': 5.0, 'sample_guide_scale': 5.0, 'neg_prompt': '色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走', 'seed': 42, 'K': 16, 'num_rounds': 10, 'epochs_per_round': 100, 'effective_batch_size': 16, 'w_bad': 1.0, 'max_per_condition': 64, 'lambda_gripper': 1.0, 'reset_optimizer_per_round': False, 'num_train_timesteps': 1000, 'learning_rate': 1e-05, 'weight_decay': 0.01, 'max_grad_norm': 2.0, 'checkpointing_steps': 10, 'gradient_checkpointing': True, 'use_8bit_adam': True, 'lora_rank': 64, 'lora_alpha': 64, 'lora_target_modules': None, 'resume_from_lora_checkpoint': None, 'reward_config': 'blocks_stack_v2', 'reward_backend': 'blocks_stack_v2', 'skip_reward_debug_video': True, 'wandb_project': 'Corrective-Reflow', 'wandb_run_name': 'creflow_stack_blocks_three', 'hallucination_crop_top_ratio': 0.6667, 'bsv2_prompts': ['red block', 'green block', 'blue block'], 'bsv2_pick_thr': 0.05, 'bsv2_place_thr': 0.05, 'bsv2_vertical_sep_thr': 0.03, 'bsv2_x_align_thr': 0.025, 'bsv2_check_window_frac': 0.2, 'bsv2_dup_max_frames': 0, 'bsv2_expected_grip_changes': 6, 'bsv2_mj_lo': 0.6, 'bsv2_mj_hi': 1.3, 'bsv2_idm_ckpt_path': 'ckpts/vidar_ckpt/idm.pt'}
14
+ 2026-04-08 01:20:32,329 INFO MainThread:195712 [wandb_init.py:init():671] starting backend
15
+ 2026-04-08 01:20:32,329 INFO MainThread:195712 [wandb_init.py:init():675] sending inform_init request
16
+ 2026-04-08 01:20:32,356 INFO MainThread:195712 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
17
+ 2026-04-08 01:20:32,356 INFO MainThread:195712 [wandb_init.py:init():688] backend started and connected
18
+ 2026-04-08 01:20:32,412 INFO MainThread:195712 [wandb_init.py:init():783] updated telemetry
19
+ 2026-04-08 01:20:32,575 INFO MainThread:195712 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout
20
+ 2026-04-08 01:20:32,942 INFO MainThread:195712 [wandb_init.py:init():867] starting run threads in backend
21
+ 2026-04-08 01:20:33,301 INFO MainThread:195712 [wandb_run.py:_console_start():2463] atexit reg
22
+ 2026-04-08 01:20:33,301 INFO MainThread:195712 [wandb_run.py:_redirect():2311] redirect: wrap_raw
23
+ 2026-04-08 01:20:33,301 INFO MainThread:195712 [wandb_run.py:_redirect():2376] Wrapping output streams.
24
+ 2026-04-08 01:20:33,301 INFO MainThread:195712 [wandb_run.py:_redirect():2401] Redirects installed.
25
+ 2026-04-08 01:20:33,306 INFO MainThread:195712 [wandb_init.py:init():911] run started, returning control to user process
26
+ 2026-04-08 01:21:57,340 WARNING MsgRouterThr:195712 [router.py:message_loop():77] message_loop has been closed
wandb/run-20260408_014250-d3p1n5w7/files/config.yaml ADDED
@@ -0,0 +1,145 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ _wandb:
2
+ value:
3
+ cli_version: 0.18.5
4
+ m: []
5
+ python_version: 3.11.2
6
+ t:
7
+ "1":
8
+ - 1
9
+ - 11
10
+ - 41
11
+ - 49
12
+ - 55
13
+ - 71
14
+ - 83
15
+ - 105
16
+ "2":
17
+ - 1
18
+ - 11
19
+ - 41
20
+ - 49
21
+ - 55
22
+ - 63
23
+ - 71
24
+ - 83
25
+ - 98
26
+ - 105
27
+ "3":
28
+ - 13
29
+ - 16
30
+ - 23
31
+ - 55
32
+ "4": 3.11.2
33
+ "5": 0.18.5
34
+ "6": 4.46.1
35
+ "8":
36
+ - 5
37
+ "12": 0.18.5
38
+ "13": linux-x86_64
39
+ K:
40
+ value: 16
41
+ bsv2_check_window_frac:
42
+ value: 0.2
43
+ bsv2_dup_max_frames:
44
+ value: 0
45
+ bsv2_expected_grip_changes:
46
+ value: 6
47
+ bsv2_idm_ckpt_path:
48
+ value: ckpts/vidar_ckpt/idm.pt
49
+ bsv2_mj_hi:
50
+ value: 1.3
51
+ bsv2_mj_lo:
52
+ value: 0.6
53
+ bsv2_pick_thr:
54
+ value: 0.05
55
+ bsv2_place_thr:
56
+ value: 0.05
57
+ bsv2_prompts:
58
+ value:
59
+ - red block
60
+ - green block
61
+ - blue block
62
+ bsv2_vertical_sep_thr:
63
+ value: 0.03
64
+ bsv2_x_align_thr:
65
+ value: 0.025
66
+ checkpointing_steps:
67
+ value: 10
68
+ ckpt_dir:
69
+ value: ckpts/Wan2.2-TI2V-5B
70
+ convert_model_dtype:
71
+ value: true
72
+ dataset_json:
73
+ value: data/rl_train/robotwin_stack_blocks_three.json
74
+ effective_batch_size:
75
+ value: 16
76
+ epochs_per_round:
77
+ value: 100
78
+ frame_num:
79
+ value: 121
80
+ gradient_checkpointing:
81
+ value: true
82
+ hallucination_crop_top_ratio:
83
+ value: 0.6667
84
+ lambda_gripper:
85
+ value: 1
86
+ learning_rate:
87
+ value: 1e-05
88
+ lora_alpha:
89
+ value: 64
90
+ lora_rank:
91
+ value: 64
92
+ lora_target_modules:
93
+ value: null
94
+ max_grad_norm:
95
+ value: 2
96
+ max_per_condition:
97
+ value: 64
98
+ max_samples:
99
+ value: -1
100
+ neg_prompt:
101
+ value: 色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走
102
+ num_ode_steps:
103
+ value: 20
104
+ num_rounds:
105
+ value: 10
106
+ num_train_timesteps:
107
+ value: 1000
108
+ offload_model:
109
+ value: false
110
+ output_dir:
111
+ value: data/outputs/creflow_stack_blocks_three
112
+ pt_dir:
113
+ value: ckpts/vidar_ckpt/merged_vidar_lora.pt
114
+ reset_optimizer_per_round:
115
+ value: false
116
+ resume_from_lora_checkpoint:
117
+ value: null
118
+ reward_backend:
119
+ value: blocks_stack_v2
120
+ reward_config:
121
+ value: blocks_stack_v2
122
+ sample_guide_scale:
123
+ value: 5
124
+ sample_shift:
125
+ value: 5
126
+ seed:
127
+ value: 42
128
+ size:
129
+ value: 640*736
130
+ skip_reward_debug_video:
131
+ value: true
132
+ task:
133
+ value: ti2v-5B
134
+ use_8bit_adam:
135
+ value: true
136
+ vidar_root:
137
+ value: ""
138
+ w_bad:
139
+ value: 1
140
+ wandb_project:
141
+ value: Corrective-Reflow
142
+ wandb_run_name:
143
+ value: creflow_stack_blocks_three
144
+ weight_decay:
145
+ value: 0.01
wandb/run-20260408_014250-d3p1n5w7/files/output.log ADDED
@@ -0,0 +1,287 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Building Wan2.2 TI2V model ... (DDP=True, world_size=8)
2
+ /usr/local/lib/python3.11/dist-packages/torch/distributed/distributed_c10d.py:4876: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
3
+ warnings.warn( # warn only once
4
+ Moving T5 encoder to GPU permanently (offload_model=false) ...
5
+ Building reward scorer ...
6
+ Loading IDM from ckpts/vidar_ckpt/idm.pt on cuda:0 ...
7
+ IDM loaded.
8
+ Loading SAM3 VideoPredictor (blocks_stack_v2) on cuda:0 ...
9
+ INFO 2026-04-08 01:44:06,994 198734 sam3_video_base.py: 125: setting max_num_objects=10000 and num_obj_for_compile=16
10
+ SAM3 VideoPredictor (blocks_stack_v2) loaded.
11
+ Applying LoRA (single adapter) for CReflow ...
12
+ LoRA injected: rank=64, alpha=64, target_modules=['self_attn.q', 'self_attn.k', 'self_attn.v', 'self_attn.o', 'cross_attn.q', 'cross_attn.k', 'cross_attn.v', 'cross_attn.o']
13
+ Trainable: 94.4 M / 5094.2 M total (1.85%)
14
+ Gradient checkpointing enabled on 30 DiT blocks
15
+ Trainable parameters: 94.4 M
16
+ Using 8-bit AdamW (bitsandbytes)
17
+ Dataset: 10 conditions
18
+ Encoding 10 conditions ...
19
+ [1/10] 6b20973f10ef... encoded
20
+ [2/10] cff33035b28a... encoded
21
+ [3/10] 7ebc8b336779... encoded
22
+ [4/10] 8077c679fb62... encoded
23
+ [5/10] 41855438a8e0... encoded
24
+ [6/10] a67b02f4e7ce... encoded
25
+ [7/10] 4568a9603f3b... encoded
26
+ [8/10] 1b67584bff87... encoded
27
+ [9/10] f4f8079f6637... encoded
28
+ [10/10] 1b70247c6412... encoded
29
+ All 10 conditions encoded.
30
+
31
+ Starting Corrective Reflow | rounds=10 K=16 K_local=2 epochs/round=100 effective_bs=16 lr=1e-05 conditions=10
32
+
33
+ ======================================================================
34
+ ROUND 1/10
35
+ ======================================================================
36
+ [Phase 1] ODE rollout ...
37
+ [CReflow rollout] condition 0/10: K=2, id=6b20973f10ef...
38
+ rollout: 2 videos in 56.8s
39
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.44it/s]
40
+ propagate_in_video: 100%|██████████| 121/121 [00:11<00:00, 10.30it/s]
41
+ propagate_in_video: 0it [00:00, ?it/s]
42
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 77.75it/s]
43
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.78it/s]
44
+ propagate_in_video: 0it [00:00, ?it/s]
45
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.49it/s]
46
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.74it/s]
47
+ propagate_in_video: 0it [00:00, ?it/s]
48
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 72.64it/s]
49
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.57it/s]
50
+ propagate_in_video: 0it [00:00, ?it/s]
51
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.74it/s]
52
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.79it/s]
53
+ propagate_in_video: 0it [00:00, ?it/s]
54
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.37it/s]
55
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.02it/s]
56
+ propagate_in_video: 0it [00:00, ?it/s]
57
+ [CReflow rollout] condition 1/10: K=2, id=cff33035b28a...
58
+ rollout: 2 videos in 57.5s
59
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 77.04it/s]
60
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.14it/s]
61
+ propagate_in_video: 0it [00:00, ?it/s]
62
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.48it/s]
63
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.66it/s]
64
+ propagate_in_video: 0it [00:00, ?it/s]
65
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.72it/s]
66
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.64it/s]
67
+ propagate_in_video: 0it [00:00, ?it/s]
68
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 77.93it/s]
69
+ propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.33it/s]
70
+ propagate_in_video: 0it [00:00, ?it/s]
71
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.69it/s]
72
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.32it/s]
73
+ propagate_in_video: 0it [00:00, ?it/s]
74
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.65it/s]
75
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.69it/s]
76
+ propagate_in_video: 0it [00:00, ?it/s]
77
+ [CReflow rollout] condition 2/10: K=2, id=7ebc8b336779...
78
+ rollout: 2 videos in 57.1s
79
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.80it/s]
80
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.71it/s]
81
+ propagate_in_video: 0it [00:00, ?it/s]
82
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.51it/s]
83
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.17it/s]
84
+ propagate_in_video: 0it [00:00, ?it/s]
85
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.31it/s]
86
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.71it/s]
87
+ propagate_in_video: 0it [00:00, ?it/s]
88
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 77.59it/s]
89
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.19it/s]
90
+ propagate_in_video: 0it [00:00, ?it/s]
91
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.99it/s]
92
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.72it/s]
93
+ propagate_in_video: 0it [00:00, ?it/s]
94
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.19it/s]
95
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.99it/s]
96
+ propagate_in_video: 0it [00:00, ?it/s]
97
+ [CReflow rollout] condition 3/10: K=2, id=8077c679fb62...
98
+ rollout: 2 videos in 57.3s
99
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 75.88it/s]
100
+ propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.54it/s]
101
+ propagate_in_video: 0it [00:00, ?it/s]
102
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.33it/s]
103
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 14.03it/s]
104
+ propagate_in_video: 0it [00:00, ?it/s]
105
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.71it/s]
106
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.20it/s]
107
+ propagate_in_video: 0it [00:00, ?it/s]
108
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.97it/s]
109
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.94it/s]
110
+ propagate_in_video: 0it [00:00, ?it/s]
111
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.80it/s]
112
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.48it/s]
113
+ propagate_in_video: 0it [00:00, ?it/s]
114
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.06it/s]
115
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.86it/s]
116
+ propagate_in_video: 0it [00:00, ?it/s]
117
+ [CReflow rollout] condition 4/10: K=2, id=41855438a8e0...
118
+ rollout: 2 videos in 57.6s
119
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.59it/s]
120
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.88it/s]
121
+ propagate_in_video: 0it [00:00, ?it/s]
122
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.29it/s]
123
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.15it/s]
124
+ propagate_in_video: 0it [00:00, ?it/s]
125
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.15it/s]
126
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.88it/s]
127
+ propagate_in_video: 0it [00:00, ?it/s]
128
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.97it/s]
129
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.74it/s]
130
+ propagate_in_video: 0it [00:00, ?it/s]
131
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.05it/s]
132
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.30it/s]
133
+ propagate_in_video: 0it [00:00, ?it/s]
134
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.51it/s]
135
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.83it/s]
136
+ propagate_in_video: 0it [00:00, ?it/s]
137
+ [CReflow rollout] condition 5/10: K=2, id=a67b02f4e7ce...
138
+ rollout: 2 videos in 57.5s
139
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.02it/s]
140
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.57it/s]
141
+ propagate_in_video: 0it [00:00, ?it/s]
142
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.85it/s]
143
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.65it/s]
144
+ propagate_in_video: 0it [00:00, ?it/s]
145
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.17it/s]
146
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.58it/s]
147
+ propagate_in_video: 0it [00:00, ?it/s]
148
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.90it/s]
149
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.12it/s]
150
+ propagate_in_video: 0it [00:00, ?it/s]
151
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.12it/s]
152
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.62it/s]
153
+ propagate_in_video: 0it [00:00, ?it/s]
154
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.90it/s]
155
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.09it/s]
156
+ propagate_in_video: 0it [00:00, ?it/s]
157
+ [CReflow rollout] condition 6/10: K=2, id=4568a9603f3b...
158
+ rollout: 2 videos in 57.5s
159
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.94it/s]
160
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.46it/s]
161
+ propagate_in_video: 0it [00:00, ?it/s]
162
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.68it/s]
163
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.73it/s]
164
+ propagate_in_video: 0it [00:00, ?it/s]
165
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.83it/s]
166
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.45it/s]
167
+ propagate_in_video: 0it [00:00, ?it/s]
168
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.64it/s]
169
+ propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.76it/s]
170
+ propagate_in_video: 0it [00:00, ?it/s]
171
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.14it/s]
172
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.60it/s]
173
+ propagate_in_video: 0it [00:00, ?it/s]
174
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.70it/s]
175
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.80it/s]
176
+ propagate_in_video: 0it [00:00, ?it/s]
177
+ [CReflow rollout] condition 7/10: K=2, id=1b67584bff87...
178
+ rollout: 2 videos in 57.0s
179
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 75.90it/s]
180
+ propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 11.75it/s]
181
+ propagate_in_video: 0it [00:00, ?it/s]
182
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.26it/s]
183
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.18it/s]
184
+ propagate_in_video: 0it [00:00, ?it/s]
185
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.63it/s]
186
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.95it/s]
187
+ propagate_in_video: 0it [00:00, ?it/s]
188
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 79.39it/s]
189
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.61it/s]
190
+ propagate_in_video: 0it [00:00, ?it/s]
191
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.79it/s]
192
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.57it/s]
193
+ propagate_in_video: 0it [00:00, ?it/s]
194
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.72it/s]
195
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.06it/s]
196
+ propagate_in_video: 0it [00:00, ?it/s]
197
+ [CReflow rollout] condition 8/10: K=2, id=f4f8079f6637...
198
+ rollout: 2 videos in 57.5s
199
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.04it/s]
200
+ propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 12.04it/s]
201
+ propagate_in_video: 0it [00:00, ?it/s]
202
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 82.37it/s]
203
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.40it/s]
204
+ propagate_in_video: 0it [00:00, ?it/s]
205
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.08it/s]
206
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.58it/s]
207
+ propagate_in_video: 0it [00:00, ?it/s]
208
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.47it/s]
209
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.20it/s]
210
+ propagate_in_video: 0it [00:00, ?it/s]
211
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.48it/s]
212
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.30it/s]
213
+ propagate_in_video: 0it [00:00, ?it/s]
214
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 78.85it/s]
215
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.52it/s]
216
+ propagate_in_video: 0it [00:00, ?it/s]
217
+ [CReflow rollout] condition 9/10: K=2, id=1b70247c6412...
218
+ rollout: 2 videos in 56.8s
219
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.06it/s]
220
+ propagate_in_video: 100%|██████████| 121/121 [00:08<00:00, 13.54it/s]
221
+ propagate_in_video: 0it [00:00, ?it/s]
222
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 83.74it/s]
223
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.27it/s]
224
+ propagate_in_video: 0it [00:00, ?it/s]
225
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 81.74it/s]
226
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.43it/s]
227
+ propagate_in_video: 0it [00:00, ?it/s]
228
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.00it/s]
229
+ propagate_in_video: 100%|██████████| 121/121 [00:10<00:00, 12.07it/s]
230
+ propagate_in_video: 0it [00:00, ?it/s]
231
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 84.11it/s]
232
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 12.98it/s]
233
+ propagate_in_video: 0it [00:00, ?it/s]
234
+ frame loading (image folder) [rank=0]: 100%|██████████| 121/121 [00:01<00:00, 80.99it/s]
235
+ propagate_in_video: 100%|██████████| 121/121 [00:09<00:00, 13.26it/s]
236
+ propagate_in_video: 0it [00:00, ?it/s]
237
+ [CReflow rollout] done: 20 samples, rewards = [1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 1.00, 0.00, 0.00]
238
+ Rollout done: 20 samples, good=18, bad=2, avg_reward=0.900
239
+ 1b67584bff87: mean=1.00, good=2/2
240
+ 1b70247c6412: mean=0.00, good=0/2
241
+ 41855438a8e0: mean=1.00, good=2/2
242
+ 4568a9603f3b: mean=1.00, good=2/2
243
+ 6b20973f10ef: mean=1.00, good=2/2
244
+ 7ebc8b336779: mean=1.00, good=2/2
245
+ 8077c679fb62: mean=1.00, good=2/2
246
+ a67b02f4e7ce: mean=1.00, good=2/2
247
+ cff33035b28a: mean=1.00, good=2/2
248
+ f4f8079f6637: mean=1.00, good=2/2
249
+ [Phase 2] Library update ...
250
+ [Library] rank=0 has 18 good samples
251
+ [Library] after update: SuccessLibrary(total=127, coverage=10/10, max_per_condition=64)
252
+ [Phase 3] Re-pairing ...
253
+ [Re-pair] train_set: 20 samples (anchor=18, correction=2, skipped=0)
254
+ [Phase 4] Training (100 epochs, 20 samples) ...
255
+ [DEBUG] v_pred requires_grad=False, grad_fn=False
256
+ Traceback (most recent call last):
257
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1039, in <module>
258
+ main()
259
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 939, in main
260
+ train_metrics = train_one_round(
261
+ ^^^^^^^^^^^^^^^^
262
+ File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 579, in train_one_round
263
+ loss.backward()
264
+ File "/usr/local/lib/python3.11/dist-packages/torch/_tensor.py", line 625, in backward
265
+ torch.autograd.backward(
266
+ File "/usr/local/lib/python3.11/dist-packages/torch/autograd/__init__.py", line 354, in backward
267
+ _engine_run_backward(
268
+ File "/usr/local/lib/python3.11/dist-packages/torch/autograd/graph.py", line 841, in _engine_run_backward
269
+ return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
270
+ ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
271
+ RuntimeError: element 0 of tensors does not require grad and does not have a grad_fn
272
+ [rank0]: Traceback (most recent call last):
273
+ [rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 1039, in <module>
274
+ [rank0]: main()
275
+ [rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 939, in main
276
+ [rank0]: train_metrics = train_one_round(
277
+ [rank0]: ^^^^^^^^^^^^^^^^
278
+ [rank0]: File "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py", line 579, in train_one_round
279
+ [rank0]: loss.backward()
280
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/_tensor.py", line 625, in backward
281
+ [rank0]: torch.autograd.backward(
282
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/autograd/__init__.py", line 354, in backward
283
+ [rank0]: _engine_run_backward(
284
+ [rank0]: File "/usr/local/lib/python3.11/dist-packages/torch/autograd/graph.py", line 841, in _engine_run_backward
285
+ [rank0]: return Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
286
+ [rank0]: ^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^
287
+ [rank0]: RuntimeError: element 0 of tensors does not require grad and does not have a grad_fn
wandb/run-20260408_014250-d3p1n5w7/files/wandb-metadata.json ADDED
@@ -0,0 +1,144 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-5.15.120.bsk.6-amd64-x86_64-with-glibc2.36",
3
+ "python": "3.11.2",
4
+ "startedAt": "2026-04-08T01:42:50.836955Z",
5
+ "args": [
6
+ "--task",
7
+ "ti2v-5B",
8
+ "--size",
9
+ "640*736",
10
+ "--frame_num",
11
+ "121",
12
+ "--ckpt_dir",
13
+ "ckpts/Wan2.2-TI2V-5B",
14
+ "--pt_dir",
15
+ "ckpts/vidar_ckpt/merged_vidar_lora.pt",
16
+ "--dataset_json",
17
+ "data/rl_train/robotwin_stack_blocks_three.json",
18
+ "--output_dir",
19
+ "data/outputs/creflow_stack_blocks_three",
20
+ "--num_ode_steps",
21
+ "20",
22
+ "--sample_shift",
23
+ "5.0",
24
+ "--sample_guide_scale",
25
+ "5.0",
26
+ "--K",
27
+ "16",
28
+ "--num_rounds",
29
+ "10",
30
+ "--epochs_per_round",
31
+ "100",
32
+ "--max_per_condition",
33
+ "64",
34
+ "--effective_batch_size",
35
+ "16",
36
+ "--w_bad",
37
+ "1.0",
38
+ "--lambda_gripper",
39
+ "1.0",
40
+ "--seed",
41
+ "42",
42
+ "--reward_config",
43
+ "blocks_stack_v2",
44
+ "--convert_model_dtype",
45
+ "--offload_model",
46
+ "false",
47
+ "--learning_rate",
48
+ "1e-5",
49
+ "--weight_decay",
50
+ "0.01",
51
+ "--max_grad_norm",
52
+ "2.0",
53
+ "--checkpointing_steps",
54
+ "10",
55
+ "--lora_rank",
56
+ "64",
57
+ "--lora_alpha",
58
+ "64",
59
+ "--wandb_project",
60
+ "Corrective-Reflow",
61
+ "--wandb_run_name",
62
+ "creflow_stack_blocks_three"
63
+ ],
64
+ "program": "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py",
65
+ "codePath": "fastvideo/train_creflow_wan_2_2_ti2v.py",
66
+ "git": {
67
+ "remote": "git@github.com:VincentNi0107/EmbodiedVideoRL.git",
68
+ "commit": "a8f5b170f0ef7d865af7e8f32c4aebb46efec2c1"
69
+ },
70
+ "email": "1163051845@qq.com",
71
+ "root": "data/outputs/creflow_stack_blocks_three",
72
+ "host": "n124-104-180",
73
+ "username": "tiger",
74
+ "executable": "/usr/bin/python",
75
+ "codePathLocal": "fastvideo/train_creflow_wan_2_2_ti2v.py",
76
+ "cpu_count": 104,
77
+ "cpu_count_logical": 208,
78
+ "gpu": "NVIDIA H100 80GB HBM3",
79
+ "gpu_count": 8,
80
+ "disk": {
81
+ "/": {
82
+ "total": "1055740600320",
83
+ "used": "482529443840"
84
+ }
85
+ },
86
+ "memory": {
87
+ "total": "1977660338176"
88
+ },
89
+ "cpu": {
90
+ "count": 104,
91
+ "countLogical": 208
92
+ },
93
+ "gpu_nvidia": [
94
+ {
95
+ "name": "NVIDIA H100 80GB HBM3",
96
+ "memoryTotal": "85520809984",
97
+ "cudaCores": 16896,
98
+ "architecture": "Hopper"
99
+ },
100
+ {
101
+ "name": "NVIDIA H100 80GB HBM3",
102
+ "memoryTotal": "85520809984",
103
+ "cudaCores": 16896,
104
+ "architecture": "Hopper"
105
+ },
106
+ {
107
+ "name": "NVIDIA H100 80GB HBM3",
108
+ "memoryTotal": "85520809984",
109
+ "cudaCores": 16896,
110
+ "architecture": "Hopper"
111
+ },
112
+ {
113
+ "name": "NVIDIA H100 80GB HBM3",
114
+ "memoryTotal": "85520809984",
115
+ "cudaCores": 16896,
116
+ "architecture": "Hopper"
117
+ },
118
+ {
119
+ "name": "NVIDIA H100 80GB HBM3",
120
+ "memoryTotal": "85520809984",
121
+ "cudaCores": 16896,
122
+ "architecture": "Hopper"
123
+ },
124
+ {
125
+ "name": "NVIDIA H100 80GB HBM3",
126
+ "memoryTotal": "85520809984",
127
+ "cudaCores": 16896,
128
+ "architecture": "Hopper"
129
+ },
130
+ {
131
+ "name": "NVIDIA H100 80GB HBM3",
132
+ "memoryTotal": "85520809984",
133
+ "cudaCores": 16896,
134
+ "architecture": "Hopper"
135
+ },
136
+ {
137
+ "name": "NVIDIA H100 80GB HBM3",
138
+ "memoryTotal": "85520809984",
139
+ "cudaCores": 16896,
140
+ "architecture": "Hopper"
141
+ }
142
+ ],
143
+ "cudaVersion": "12.9"
144
+ }
wandb/run-20260408_014250-d3p1n5w7/files/wandb-summary.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"_wandb":{"runtime":1586}}
wandb/run-20260408_014250-d3p1n5w7/logs/debug-internal.log ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-04-08T01:42:50.846912727Z","level":"INFO","msg":"using version","core version":"0.18.5"}
2
+ {"time":"2026-04-08T01:42:50.846924973Z","level":"INFO","msg":"created symlink","path":"data/outputs/creflow_stack_blocks_three/wandb/run-20260408_014250-d3p1n5w7/logs/debug-core.log"}
3
+ {"time":"2026-04-08T01:42:51.065325741Z","level":"INFO","msg":"created new stream","id":"d3p1n5w7"}
4
+ {"time":"2026-04-08T01:42:51.06552384Z","level":"INFO","msg":"stream: started","id":"d3p1n5w7"}
5
+ {"time":"2026-04-08T01:42:51.065611464Z","level":"INFO","msg":"sender: started","stream_id":"d3p1n5w7"}
6
+ {"time":"2026-04-08T01:42:51.06558469Z","level":"INFO","msg":"handler: started","stream_id":{"value":"d3p1n5w7"}}
7
+ {"time":"2026-04-08T01:42:51.065564877Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"d3p1n5w7"}}
8
+ {"time":"2026-04-08T01:42:51.47008492Z","level":"INFO","msg":"Starting system monitor"}
9
+ {"time":"2026-04-08T02:00:36.824476956Z","level":"INFO","msg":"api: retrying error","error":"Post \"https://api.wandb.ai/graphql\": net/http: request canceled (Client.Timeout exceeded while awaiting headers)"}
10
+ {"time":"2026-04-08T02:01:09.269479394Z","level":"INFO","msg":"api: retrying error","error":"Post \"https://api.wandb.ai/graphql\": context deadline exceeded"}
11
+ {"time":"2026-04-08T02:09:17.196524875Z","level":"INFO","msg":"stream: closing","id":"d3p1n5w7"}
12
+ {"time":"2026-04-08T02:09:17.196574831Z","level":"INFO","msg":"Stopping system monitor"}
13
+ {"time":"2026-04-08T02:09:17.201147257Z","level":"INFO","msg":"Stopped system monitor"}
14
+ {"time":"2026-04-08T02:09:18.01646151Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
15
+ {"time":"2026-04-08T02:09:18.177322397Z","level":"INFO","msg":"handler: closed","stream_id":{"value":"d3p1n5w7"}}
16
+ {"time":"2026-04-08T02:09:18.177361768Z","level":"INFO","msg":"sender: closed","stream_id":"d3p1n5w7"}
17
+ {"time":"2026-04-08T02:09:18.177361574Z","level":"INFO","msg":"writer: Close: closed","stream_id":{"value":"d3p1n5w7"}}
18
+ {"time":"2026-04-08T02:09:18.183256398Z","level":"INFO","msg":"stream: closed","id":"d3p1n5w7"}
wandb/run-20260408_014250-d3p1n5w7/logs/debug.log ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 2026-04-08 01:42:50,803 INFO MainThread:198734 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5
2
+ 2026-04-08 01:42:50,803 INFO MainThread:198734 [wandb_setup.py:_flush():79] Configure stats pid to 198734
3
+ 2026-04-08 01:42:50,803 INFO MainThread:198734 [wandb_setup.py:_flush():79] Loading settings from /home/tiger/.config/wandb/settings
4
+ 2026-04-08 01:42:50,803 INFO MainThread:198734 [wandb_setup.py:_flush():79] Loading settings from /mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/wandb/settings
5
+ 2026-04-08 01:42:50,803 INFO MainThread:198734 [wandb_setup.py:_flush():79] Loading settings from environment variables: {'disabled': 'false', 'project': 'Corrective-Reflow', 'api_key': '***REDACTED***', 'mode': 'online'}
6
+ 2026-04-08 01:42:50,803 INFO MainThread:198734 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': 'online', '_disable_service': None}
7
+ 2026-04-08 01:42:50,803 INFO MainThread:198734 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'fastvideo/train_creflow_wan_2_2_ti2v.py', 'program_abspath': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py', 'program': '/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py'}
8
+ 2026-04-08 01:42:50,803 INFO MainThread:198734 [wandb_setup.py:_flush():79] Applying login settings: {}
9
+ 2026-04-08 01:42:50,805 INFO MainThread:198734 [wandb_init.py:_log_setup():534] Logging user logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_014250-d3p1n5w7/logs/debug.log
10
+ 2026-04-08 01:42:50,807 INFO MainThread:198734 [wandb_init.py:_log_setup():535] Logging internal logs to data/outputs/creflow_stack_blocks_three/wandb/run-20260408_014250-d3p1n5w7/logs/debug-internal.log
11
+ 2026-04-08 01:42:50,807 INFO MainThread:198734 [wandb_init.py:init():621] calling init triggers
12
+ 2026-04-08 01:42:50,807 INFO MainThread:198734 [wandb_init.py:init():628] wandb.init called with sweep_config: {}
13
+ config: {'vidar_root': '', 'task': 'ti2v-5B', 'size': '640*736', 'frame_num': 121, 'ckpt_dir': 'ckpts/Wan2.2-TI2V-5B', 'pt_dir': 'ckpts/vidar_ckpt/merged_vidar_lora.pt', 'convert_model_dtype': True, 'offload_model': False, 'dataset_json': 'data/rl_train/robotwin_stack_blocks_three.json', 'max_samples': -1, 'output_dir': 'data/outputs/creflow_stack_blocks_three', 'num_ode_steps': 20, 'sample_shift': 5.0, 'sample_guide_scale': 5.0, 'neg_prompt': '色调艳丽,过曝,静态,细节模糊不清,字幕,风格,作品,画作,画面,静止,整体发灰,最差质量,低质量,JPEG压缩残留,丑陋的,残缺的,多余的手指,画得不好的手部,画得不好的脸部,畸形的,毁容的,形态畸形的肢体,手指融合,静止不动的画面,杂乱的背景,三条腿,背景人很多,倒着走', 'seed': 42, 'K': 16, 'num_rounds': 10, 'epochs_per_round': 100, 'effective_batch_size': 16, 'w_bad': 1.0, 'max_per_condition': 64, 'lambda_gripper': 1.0, 'reset_optimizer_per_round': False, 'num_train_timesteps': 1000, 'learning_rate': 1e-05, 'weight_decay': 0.01, 'max_grad_norm': 2.0, 'checkpointing_steps': 10, 'gradient_checkpointing': True, 'use_8bit_adam': True, 'lora_rank': 64, 'lora_alpha': 64, 'lora_target_modules': None, 'resume_from_lora_checkpoint': None, 'reward_config': 'blocks_stack_v2', 'reward_backend': 'blocks_stack_v2', 'skip_reward_debug_video': True, 'wandb_project': 'Corrective-Reflow', 'wandb_run_name': 'creflow_stack_blocks_three', 'hallucination_crop_top_ratio': 0.6667, 'bsv2_prompts': ['red block', 'green block', 'blue block'], 'bsv2_pick_thr': 0.05, 'bsv2_place_thr': 0.05, 'bsv2_vertical_sep_thr': 0.03, 'bsv2_x_align_thr': 0.025, 'bsv2_check_window_frac': 0.2, 'bsv2_dup_max_frames': 0, 'bsv2_expected_grip_changes': 6, 'bsv2_mj_lo': 0.6, 'bsv2_mj_hi': 1.3, 'bsv2_idm_ckpt_path': 'ckpts/vidar_ckpt/idm.pt'}
14
+ 2026-04-08 01:42:50,807 INFO MainThread:198734 [wandb_init.py:init():671] starting backend
15
+ 2026-04-08 01:42:50,807 INFO MainThread:198734 [wandb_init.py:init():675] sending inform_init request
16
+ 2026-04-08 01:42:50,834 INFO MainThread:198734 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
17
+ 2026-04-08 01:42:50,835 INFO MainThread:198734 [wandb_init.py:init():688] backend started and connected
18
+ 2026-04-08 01:42:50,889 INFO MainThread:198734 [wandb_init.py:init():783] updated telemetry
19
+ 2026-04-08 01:42:51,055 INFO MainThread:198734 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout
20
+ 2026-04-08 01:42:51,420 INFO MainThread:198734 [wandb_init.py:init():867] starting run threads in backend
21
+ 2026-04-08 01:42:51,784 INFO MainThread:198734 [wandb_run.py:_console_start():2463] atexit reg
22
+ 2026-04-08 01:42:51,785 INFO MainThread:198734 [wandb_run.py:_redirect():2311] redirect: wrap_raw
23
+ 2026-04-08 01:42:51,785 INFO MainThread:198734 [wandb_run.py:_redirect():2376] Wrapping output streams.
24
+ 2026-04-08 01:42:51,785 INFO MainThread:198734 [wandb_run.py:_redirect():2401] Redirects installed.
25
+ 2026-04-08 01:42:51,790 INFO MainThread:198734 [wandb_init.py:init():911] run started, returning control to user process
26
+ 2026-04-08 02:09:17,196 WARNING MsgRouterThr:198734 [router.py:message_loop():77] message_loop has been closed
wandb/run-20260408_025012-pf74wm1f/logs/debug-internal.log ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {"time":"2026-04-08T02:50:12.360136031Z","level":"INFO","msg":"using version","core version":"0.18.5"}
2
+ {"time":"2026-04-08T02:50:12.360147Z","level":"INFO","msg":"created symlink","path":"data/outputs/creflow_stack_blocks_three/wandb/run-20260408_025012-pf74wm1f/logs/debug-core.log"}
3
+ {"time":"2026-04-08T02:50:12.578898735Z","level":"INFO","msg":"created new stream","id":"pf74wm1f"}
4
+ {"time":"2026-04-08T02:50:12.579095021Z","level":"INFO","msg":"stream: started","id":"pf74wm1f"}
5
+ {"time":"2026-04-08T02:50:12.5791654Z","level":"INFO","msg":"sender: started","stream_id":"pf74wm1f"}
6
+ {"time":"2026-04-08T02:50:12.579140373Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"pf74wm1f"}}
7
+ {"time":"2026-04-08T02:50:12.579178443Z","level":"INFO","msg":"handler: started","stream_id":{"value":"pf74wm1f"}}
8
+ {"time":"2026-04-08T02:50:12.901012044Z","level":"INFO","msg":"Starting system monitor"}
wandb/run-20260408_164919-x63spn6n/files/wandb-metadata.json ADDED
@@ -0,0 +1,144 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-5.15.120.bsk.6-amd64-x86_64-with-glibc2.36",
3
+ "python": "3.11.2",
4
+ "startedAt": "2026-04-08T16:49:19.417199Z",
5
+ "args": [
6
+ "--task",
7
+ "ti2v-5B",
8
+ "--size",
9
+ "640*736",
10
+ "--frame_num",
11
+ "121",
12
+ "--ckpt_dir",
13
+ "ckpts/Wan2.2-TI2V-5B",
14
+ "--pt_dir",
15
+ "ckpts/vidar_ckpt/merged_vidar_lora.pt",
16
+ "--dataset_json",
17
+ "data/rl_train/robotwin_stack_blocks_three.json",
18
+ "--output_dir",
19
+ "data/outputs/creflow_stack_blocks_three",
20
+ "--num_ode_steps",
21
+ "20",
22
+ "--sample_shift",
23
+ "5.0",
24
+ "--sample_guide_scale",
25
+ "5.0",
26
+ "--K",
27
+ "16",
28
+ "--num_rounds",
29
+ "10",
30
+ "--epochs_per_round",
31
+ "100",
32
+ "--max_per_condition",
33
+ "64",
34
+ "--effective_batch_size",
35
+ "16",
36
+ "--w_bad",
37
+ "1.0",
38
+ "--lambda_gripper",
39
+ "1.0",
40
+ "--seed",
41
+ "42",
42
+ "--reward_config",
43
+ "blocks_stack_v2",
44
+ "--convert_model_dtype",
45
+ "--offload_model",
46
+ "false",
47
+ "--learning_rate",
48
+ "1e-5",
49
+ "--weight_decay",
50
+ "0.01",
51
+ "--max_grad_norm",
52
+ "2.0",
53
+ "--checkpointing_steps",
54
+ "10",
55
+ "--lora_rank",
56
+ "64",
57
+ "--lora_alpha",
58
+ "64",
59
+ "--wandb_project",
60
+ "Corrective-Reflow",
61
+ "--wandb_run_name",
62
+ "creflow_stack_blocks_three"
63
+ ],
64
+ "program": "/mnt/bn/themis/yijiangli/project/EmbodiedVideoRL/fastvideo/train_creflow_wan_2_2_ti2v.py",
65
+ "codePath": "fastvideo/train_creflow_wan_2_2_ti2v.py",
66
+ "git": {
67
+ "remote": "git@github.com:VincentNi0107/EmbodiedVideoRL.git",
68
+ "commit": "d5e0e14ab1e94676078e078720b47c2281e8446e"
69
+ },
70
+ "email": "1163051845@qq.com",
71
+ "root": "data/outputs/creflow_stack_blocks_three",
72
+ "host": "n124-106-114",
73
+ "username": "tiger",
74
+ "executable": "/usr/bin/python",
75
+ "codePathLocal": "fastvideo/train_creflow_wan_2_2_ti2v.py",
76
+ "cpu_count": 104,
77
+ "cpu_count_logical": 208,
78
+ "gpu": "NVIDIA H100 80GB HBM3",
79
+ "gpu_count": 8,
80
+ "disk": {
81
+ "/": {
82
+ "total": "1055740600320",
83
+ "used": "290213711872"
84
+ }
85
+ },
86
+ "memory": {
87
+ "total": "1977660334080"
88
+ },
89
+ "cpu": {
90
+ "count": 104,
91
+ "countLogical": 208
92
+ },
93
+ "gpu_nvidia": [
94
+ {
95
+ "name": "NVIDIA H100 80GB HBM3",
96
+ "memoryTotal": "85520809984",
97
+ "cudaCores": 16896,
98
+ "architecture": "Hopper"
99
+ },
100
+ {
101
+ "name": "NVIDIA H100 80GB HBM3",
102
+ "memoryTotal": "85520809984",
103
+ "cudaCores": 16896,
104
+ "architecture": "Hopper"
105
+ },
106
+ {
107
+ "name": "NVIDIA H100 80GB HBM3",
108
+ "memoryTotal": "85520809984",
109
+ "cudaCores": 16896,
110
+ "architecture": "Hopper"
111
+ },
112
+ {
113
+ "name": "NVIDIA H100 80GB HBM3",
114
+ "memoryTotal": "85520809984",
115
+ "cudaCores": 16896,
116
+ "architecture": "Hopper"
117
+ },
118
+ {
119
+ "name": "NVIDIA H100 80GB HBM3",
120
+ "memoryTotal": "85520809984",
121
+ "cudaCores": 16896,
122
+ "architecture": "Hopper"
123
+ },
124
+ {
125
+ "name": "NVIDIA H100 80GB HBM3",
126
+ "memoryTotal": "85520809984",
127
+ "cudaCores": 16896,
128
+ "architecture": "Hopper"
129
+ },
130
+ {
131
+ "name": "NVIDIA H100 80GB HBM3",
132
+ "memoryTotal": "85520809984",
133
+ "cudaCores": 16896,
134
+ "architecture": "Hopper"
135
+ },
136
+ {
137
+ "name": "NVIDIA H100 80GB HBM3",
138
+ "memoryTotal": "85520809984",
139
+ "cudaCores": 16896,
140
+ "architecture": "Hopper"
141
+ }
142
+ ],
143
+ "cudaVersion": "12.9"
144
+ }