sync run artifacts: vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/best.pt +3 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/git-info.txt +2 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_0_log.err +45 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_0_log.out +0 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_10_log.err +9 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_10_log.out +288 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_11_log.err +9 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_11_log.out +288 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_12_log.err +18 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_12_log.out +292 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_13_log.err +9 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_13_log.out +288 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_14_log.err +9 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_14_log.out +288 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_15_log.err +9 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_15_log.out +288 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_1_log.err +15 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_1_log.out +288 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_2_log.err +15 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_2_log.out +288 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_3_log.err +15 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_3_log.out +288 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_4_log.err +18 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_4_log.out +292 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_5_log.err +9 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_5_log.out +288 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_6_log.err +9 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_6_log.out +288 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_7_log.err +9 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_7_log.out +288 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_8_log.err +18 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_8_log.out +292 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_9_log.err +9 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_9_log.out +288 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/latest.pt +3 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r0.csv +0 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r1.csv +0 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r10.csv +0 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r11.csv +0 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r12.csv +0 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r13.csv +0 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r14.csv +0 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r15.csv +0 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r2.csv +0 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r3.csv +0 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r4.csv +0 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r5.csv +0 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r6.csv +0 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r7.csv +0 -0
- vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r8.csv +0 -0
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/best.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:302d99bc96a1694bb6cfe20dfe8938acb2b195884b0f149a6c7bcd3b709141e5
|
| 3 |
+
size 7397286257
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/git-info.txt
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
branch: main
|
| 2 |
+
commit: 857f73bfc0cf1648567212e23a081f468aab48c2
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_0_log.err
ADDED
|
@@ -0,0 +1,45 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/timm/models/layers/__init__.py:49: FutureWarning: Importing from timm.models.layers is deprecated, please import via timm.layers
|
| 2 |
+
warnings.warn(f"Importing from {__name__} is deprecated, please import via timm.layers", FutureWarning)
|
| 3 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
|
| 4 |
+
return func(*args, **kwargs)
|
| 5 |
+
[rank0]:[W514 09:36:45.928252362 ProcessGroupNCCL.cpp:5138] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 6 |
+
wandb: Currently logged in as: dgcnz (uvjepa) to https://api.wandb.ai. Use `wandb login --relogin` to force relogin
|
| 7 |
+
wandb: setting up run rxtoljxd
|
| 8 |
+
wandb: Tracking run with wandb version 0.23.1
|
| 9 |
+
wandb: Run data is saved locally in /scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/wandb/run-20260514_093658-rxtoljxd
|
| 10 |
+
wandb: Run `wandb offline` to turn off syncing.
|
| 11 |
+
wandb: Syncing run c003_vitl_k16_simple_cross_factorized_stable
|
| 12 |
+
wandb: ⭐️ View project at https://wandb.ai/uvjepa/vjepa_ujepaside
|
| 13 |
+
wandb: 🚀 View run at https://wandb.ai/uvjepa/vjepa_ujepaside/runs/rxtoljxd
|
| 14 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/utils.py:796: FutureWarning: `torch.cuda.amp.GradScaler(args...)` is deprecated. Please use `torch.amp.GradScaler('cuda', args...)` instead.
|
| 15 |
+
scaler = torch.cuda.amp.GradScaler() if mixed_precision else None
|
| 16 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 17 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 18 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
|
| 19 |
+
return func(*args, **kwargs)
|
| 20 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
|
| 21 |
+
return func(*args, **kwargs)
|
| 22 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 23 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 24 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
|
| 25 |
+
return func(*args, **kwargs)
|
| 26 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 27 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 28 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/torch/distributed/c10d_logger.py:83: UserWarning: barrier(): using the device under current context. You can specify `device_id` in `init_process_group` to mute this warning.
|
| 29 |
+
return func(*args, **kwargs)
|
| 30 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 31 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 32 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 33 |
+
[2026-05-14T13:38:23.807] error: *** JOB 22743106 ON gcn75 CANCELLED AT 2026-05-14T13:38:23 DUE to SIGNAL Terminated ***
|
| 34 |
+
srun: Job step aborted: Waiting up to 32 seconds for job step to finish.
|
| 35 |
+
[2026-05-14T13:38:23.808] error: *** STEP 22743106.0 ON gcn75 CANCELLED AT 2026-05-14T13:38:23 DUE to SIGNAL Terminated ***
|
| 36 |
+
submitit WARNING (2026-05-14 13:38:23,813) - Bypassing signal SIGTERM
|
| 37 |
+
[2026-05-14T13:38:57.539] error: namespace_p_join: open failed for /slurm/22743106/.ns: No such file or directory
|
| 38 |
+
[2026-05-14T13:38:57.540] error: namespace_g_join(JobId=22743106 SLUID=s8FXKRJP5N0000): No such file or directory
|
| 39 |
+
[2026-05-14T13:38:59.198] error: Failed to send MESSAGE_TASK_EXIT: Connection refused
|
| 40 |
+
[2026-05-14T13:39:03.204] error: namespace_p_join: open failed for /slurm/22743106/.ns: No such file or directory
|
| 41 |
+
[2026-05-14T13:39:03.205] error: namespace_g_join(JobId=22743106 SLUID=s8FXKRJP5N0000): No such file or directory
|
| 42 |
+
[2026-05-14T13:39:03.345] error: namespace_p_join: open failed for /slurm/22743106/.ns: No such file or directory
|
| 43 |
+
[2026-05-14T13:39:03.346] error: namespace_g_join(JobId=22743106 SLUID=s8FXKRJP5N0000): No such file or directory
|
| 44 |
+
[2026-05-14T13:39:03.385] error: namespace_p_join: open failed for /slurm/22743106/.ns: No such file or directory
|
| 45 |
+
[2026-05-14T13:39:03.385] error: namespace_g_join(JobId=22743106 SLUID=s8FXKRJP5N0000): No such file or directory
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_0_log.out
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_10_log.err
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/timm/models/layers/__init__.py:49: FutureWarning: Importing from timm.models.layers is deprecated, please import via timm.layers
|
| 2 |
+
warnings.warn(f"Importing from {__name__} is deprecated, please import via timm.layers", FutureWarning)
|
| 3 |
+
[rank10]:[W514 09:35:52.955169794 ProcessGroupNCCL.cpp:5138] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 4 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/utils.py:796: FutureWarning: `torch.cuda.amp.GradScaler(args...)` is deprecated. Please use `torch.amp.GradScaler('cuda', args...)` instead.
|
| 5 |
+
scaler = torch.cuda.amp.GradScaler() if mixed_precision else None
|
| 6 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 7 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 8 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 9 |
+
submitit WARNING (2026-05-14 13:38:23,852) - Bypassing signal SIGTERM
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_10_log.out
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
submitit INFO (2026-05-14 09:28:14,530) - Starting with JobEnvironment(job_id=22743106, hostname=gcn81.local.snellius.surf.nl, local_rank=2(4), node=2(4), global_rank=10(16))
|
| 2 |
+
submitit INFO (2026-05-14 09:28:14,530) - Loading pickle: /scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_submitted.pkl
|
| 3 |
+
INFO:root:loaded pretrain params...
|
| 4 |
+
{ 'app': 'vjepa',
|
| 5 |
+
'cpus_per_task': 16,
|
| 6 |
+
'data': { 'batch_size': 64,
|
| 7 |
+
'crop_size': 224,
|
| 8 |
+
'dataset_fpcs': [16, 16],
|
| 9 |
+
'dataset_type': 'VideoDataset',
|
| 10 |
+
'datasets': [ '/scratch-shared/dcanez/data/kinetics/k400/train.csv',
|
| 11 |
+
'/scratch-shared/dcanez/data/ssv2/train.csv'],
|
| 12 |
+
'datasets_weights': [0.65, 0.35],
|
| 13 |
+
'fps': 4,
|
| 14 |
+
'num_workers': 10,
|
| 15 |
+
'patch_size': 14,
|
| 16 |
+
'persistent_workers': True,
|
| 17 |
+
'pin_mem': True,
|
| 18 |
+
'stage': [ { 'dest': 'kinetics_240',
|
| 19 |
+
'format': 'targz_parts',
|
| 20 |
+
'src': '/scratch-shared/dcanez/data/kinetics/k400/tars_240/'},
|
| 21 |
+
{ 'dest': 'ssv2',
|
| 22 |
+
'format': 'multipart_tar',
|
| 23 |
+
'src': '/scratch-nvme/ml-datasets/something-something-v2/'}],
|
| 24 |
+
'tubelet_size': 1},
|
| 25 |
+
'data_aug': { 'auto_augment': False,
|
| 26 |
+
'motion_shift': False,
|
| 27 |
+
'random_resize_aspect_ratio': [0.75, 1.35],
|
| 28 |
+
'random_resize_scale': [0.3, 1.0],
|
| 29 |
+
'reprob': 0.0},
|
| 30 |
+
'folder': '/scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable',
|
| 31 |
+
'loss': {'loss_exp': 1.0},
|
| 32 |
+
'mask': [ { 'aspect_ratio': [0.75, 1.5],
|
| 33 |
+
'full_complement': False,
|
| 34 |
+
'max_keep': None,
|
| 35 |
+
'max_temporal_keep': 1.0,
|
| 36 |
+
'num_blocks': 8,
|
| 37 |
+
'spatial_scale': [0.15, 0.15],
|
| 38 |
+
'temporal_scale': [1.0, 1.0]},
|
| 39 |
+
{ 'aspect_ratio': [0.75, 1.5],
|
| 40 |
+
'full_complement': False,
|
| 41 |
+
'max_keep': None,
|
| 42 |
+
'max_temporal_keep': 1.0,
|
| 43 |
+
'num_blocks': 2,
|
| 44 |
+
'spatial_scale': [0.7, 0.7],
|
| 45 |
+
'temporal_scale': [1.0, 1.0]}],
|
| 46 |
+
'mem_per_gpu': '180G',
|
| 47 |
+
'meta': { 'dtype': 'bfloat16',
|
| 48 |
+
'knn_eval_epoch0': False,
|
| 49 |
+
'knn_eval_freq': 5,
|
| 50 |
+
'knn_eval_presets': [ { 'config': { 'batch_size': 64,
|
| 51 |
+
'dataset_train': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/train.csv',
|
| 52 |
+
'dataset_val': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/val.csv',
|
| 53 |
+
'eval_videos_per_class': 25,
|
| 54 |
+
'num_workers': 8,
|
| 55 |
+
'pool_type': 'slot_temporal_concat',
|
| 56 |
+
'train_videos_per_class': 100},
|
| 57 |
+
'preset': 'ucf101'},
|
| 58 |
+
{ 'config': { 'batch_size': 64,
|
| 59 |
+
'dataset_train': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/train_coarse10.csv',
|
| 60 |
+
'dataset_val': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/val_coarse10.csv',
|
| 61 |
+
'eval_videos_per_class': 100,
|
| 62 |
+
'linear_probe': True,
|
| 63 |
+
'num_workers': 8,
|
| 64 |
+
'pool_type': 'slot_temporal_concat',
|
| 65 |
+
'train_videos_per_class': 500},
|
| 66 |
+
'preset': 'ssv2_coarse10'}],
|
| 67 |
+
'load_checkpoint': True,
|
| 68 |
+
'read_checkpoint': None,
|
| 69 |
+
'save_every_freq': 5,
|
| 70 |
+
'seed': 239,
|
| 71 |
+
'use_sdpa': True,
|
| 72 |
+
'use_wandb': True,
|
| 73 |
+
'wandb_project': 'vjepa_ujepaside'},
|
| 74 |
+
'metrics': {'sigreg': {}, 'std': {}},
|
| 75 |
+
'model': { 'model_name': 'ujepaside_large_patch14_capi_lvd1689m',
|
| 76 |
+
'pred_depth': 6,
|
| 77 |
+
'pred_embed_dim': 384,
|
| 78 |
+
'pred_num_heads': 12,
|
| 79 |
+
'predictor': 'v2_cross',
|
| 80 |
+
'st_causal': False,
|
| 81 |
+
'st_drop_path': 0.2,
|
| 82 |
+
'st_flex_enable': False,
|
| 83 |
+
'st_layer_scale_init': 1e-05,
|
| 84 |
+
'st_num_slots': 16,
|
| 85 |
+
'st_side_block_type': 'factorized',
|
| 86 |
+
'st_slots_causal_within_frame': False,
|
| 87 |
+
'target_kind': 'frozen_2d',
|
| 88 |
+
'target_type': 'vit_large_patch14_capi.lvd1689m',
|
| 89 |
+
'temporal_spacing': 1.0,
|
| 90 |
+
'uniform_power': True,
|
| 91 |
+
'use_activation_checkpointing': True,
|
| 92 |
+
'use_mask_tokens': True,
|
| 93 |
+
'use_rope': True,
|
| 94 |
+
'use_sdpa': True,
|
| 95 |
+
'zero_init_mask_tokens': True},
|
| 96 |
+
'nodes': 4,
|
| 97 |
+
'optimization': { 'clip_grad': 3.0,
|
| 98 |
+
'ema': [0.99925, 0.99925],
|
| 99 |
+
'epochs': 100,
|
| 100 |
+
'final_lr': 0.0001,
|
| 101 |
+
'final_weight_decay': 0.04,
|
| 102 |
+
'ipe': 300,
|
| 103 |
+
'ipe_scale': 1.0,
|
| 104 |
+
'lr': 0.0005,
|
| 105 |
+
'start_lr': 0.0001,
|
| 106 |
+
'warmup': 10,
|
| 107 |
+
'weight_decay': 0.04},
|
| 108 |
+
'tasks_per_node': 4}
|
| 109 |
+
INFO:root:Running pre-training of app: vjepa
|
| 110 |
+
[INFO ][2026-05-14 09:32:44][app.vjepa.train ][main ] which_dtype='bfloat16'
|
| 111 |
+
[INFO ][2026-05-14 09:32:44][app.vjepa.train ][main ] Disabling persistent_workers (incompatible with KNN eval)
|
| 112 |
+
[INFO ][2026-05-14 09:32:44][app.vjepa.train ][main ] NCCL_SOCKET_IFNAME=eno
|
| 113 |
+
[INFO ][2026-05-14 09:32:47][app.vjepa.train ][main ] Initialized (rank/world-size) 10/16, tasks_per_node=4
|
| 114 |
+
[INFO ][2026-05-14 09:32:47][root ][stage_datasets ] [local_rank 2/4] Staging kinetics_240 (targz_parts)
|
| 115 |
+
[INFO ][2026-05-14 09:32:47][root ][_stage_targz_parts ] [rank 2] Extracting 26/103 tar.gz parts to /scratch-node/dcanez.22743106/kinetics_240
|
| 116 |
+
[INFO ][2026-05-14 09:33:01][root ][_stage_targz_parts ] [local_rank 2] Extracted 2/26 parts
|
| 117 |
+
[INFO ][2026-05-14 09:33:15][root ][_stage_targz_parts ] [local_rank 2] Extracted 4/26 parts
|
| 118 |
+
[INFO ][2026-05-14 09:33:29][root ][_stage_targz_parts ] [local_rank 2] Extracted 6/26 parts
|
| 119 |
+
[INFO ][2026-05-14 09:33:44][root ][_stage_targz_parts ] [local_rank 2] Extracted 8/26 parts
|
| 120 |
+
[INFO ][2026-05-14 09:33:58][root ][_stage_targz_parts ] [local_rank 2] Extracted 10/26 parts
|
| 121 |
+
[INFO ][2026-05-14 09:34:12][root ][_stage_targz_parts ] [local_rank 2] Extracted 12/26 parts
|
| 122 |
+
[INFO ][2026-05-14 09:34:25][root ][_stage_targz_parts ] [local_rank 2] Extracted 14/26 parts
|
| 123 |
+
[INFO ][2026-05-14 09:34:40][root ][_stage_targz_parts ] [local_rank 2] Extracted 16/26 parts
|
| 124 |
+
[INFO ][2026-05-14 09:34:54][root ][_stage_targz_parts ] [local_rank 2] Extracted 18/26 parts
|
| 125 |
+
[INFO ][2026-05-14 09:35:09][root ][_stage_targz_parts ] [local_rank 2] Extracted 20/26 parts
|
| 126 |
+
[INFO ][2026-05-14 09:35:23][root ][_stage_targz_parts ] [local_rank 2] Extracted 22/26 parts
|
| 127 |
+
[INFO ][2026-05-14 09:35:37][root ][_stage_targz_parts ] [local_rank 2] Extracted 24/26 parts
|
| 128 |
+
[INFO ][2026-05-14 09:35:52][root ][_stage_targz_parts ] [local_rank 2] Extracted 26/26 parts
|
| 129 |
+
[INFO ][2026-05-14 09:35:52][root ][stage_datasets ] [local_rank 2/4] Staging ssv2 (multipart_tar)
|
| 130 |
+
[INFO ][2026-05-14 09:36:49][app.vjepa.train ][main ] Staged datasets: ['/scratch-node/dcanez.22743106/kinetics_240/train.csv', '/scratch-node/dcanez.22743106/ssv2/train.csv']
|
| 131 |
+
[INFO ][2026-05-14 09:43:14][experiments.stmodels.vision_transformers_v3][vit_large_patch14_capi_lvd1689m] Loading pretrained weights for vit_large_patch14_capi_lvd1689m from https://dl.fbaipublicfiles.com/capi/capi_vitl14_lvd.pth
|
| 132 |
+
[INFO ][2026-05-14 09:43:28][root ][_build_frozen_2d_target ] Loaded frozen_2d target from timm: vit_large_patch14_capi.lvd1689m
|
| 133 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] ViTMultiSeqWrapper(
|
| 134 |
+
(backbone): UJEPAside(
|
| 135 |
+
(patch_embed): PatchEmbed3D(
|
| 136 |
+
(proj): Conv3d(3, 1024, kernel_size=(1, 14, 14), stride=(1, 14, 14))
|
| 137 |
+
)
|
| 138 |
+
(rope): CAPI2DRoPE()
|
| 139 |
+
(blocks): ModuleList(
|
| 140 |
+
(0-23): 24 x Block(
|
| 141 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 142 |
+
(rope_impl): CAPI2DRoPE()
|
| 143 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 144 |
+
(drop_path1): Identity()
|
| 145 |
+
(drop_path2): Identity()
|
| 146 |
+
(attn): Attention(
|
| 147 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 148 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 149 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 150 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 151 |
+
(rope_impl): CAPI2DRoPE()
|
| 152 |
+
)
|
| 153 |
+
(mlp): MLP(
|
| 154 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 155 |
+
(act): GELU(approximate='none')
|
| 156 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 157 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 158 |
+
)
|
| 159 |
+
)
|
| 160 |
+
)
|
| 161 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=False)
|
| 162 |
+
(st_blocks): ModuleList(
|
| 163 |
+
(0-23): 24 x FactorizedSlotSideBlock(
|
| 164 |
+
(cross_attn): EfficientResidual(
|
| 165 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 166 |
+
(fn): Attention(
|
| 167 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 168 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 169 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 170 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 171 |
+
(rope_impl): CAPI3DRoPE()
|
| 172 |
+
)
|
| 173 |
+
)
|
| 174 |
+
(self_attn): EfficientResidual(
|
| 175 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 176 |
+
(fn): Attention(
|
| 177 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 178 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 179 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 180 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 181 |
+
(rope_impl): CAPI3DRoPE()
|
| 182 |
+
)
|
| 183 |
+
)
|
| 184 |
+
(mlp): EfficientResidual(
|
| 185 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 186 |
+
(fn): MLP(
|
| 187 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 188 |
+
(act): GELU(approximate='none')
|
| 189 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 190 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 191 |
+
)
|
| 192 |
+
)
|
| 193 |
+
)
|
| 194 |
+
)
|
| 195 |
+
(st_rope): CAPI3DRoPE()
|
| 196 |
+
)
|
| 197 |
+
)
|
| 198 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] UJEPAsidePredictorMultiSeqWrapper(
|
| 199 |
+
(backbone): PredictorV2(
|
| 200 |
+
(predictor_embed): Linear(in_features=1024, out_features=384, bias=True)
|
| 201 |
+
(mask_tokens): ParameterList(
|
| 202 |
+
(0): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 203 |
+
(1): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 204 |
+
(2): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 205 |
+
(3): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 206 |
+
)
|
| 207 |
+
(predictor_blocks): ModuleList(
|
| 208 |
+
(0-5): 6 x Block(
|
| 209 |
+
(residual1): EfficientResidual(
|
| 210 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 211 |
+
(fn): Attention(
|
| 212 |
+
(q_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 213 |
+
(k_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 214 |
+
(v_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 215 |
+
(proj): Linear(in_features=384, out_features=384, bias=False)
|
| 216 |
+
(rope): Rope()
|
| 217 |
+
)
|
| 218 |
+
)
|
| 219 |
+
(residual2): EfficientResidual(
|
| 220 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 221 |
+
(fn): MLP(
|
| 222 |
+
(fc1): Linear(in_features=384, out_features=1536, bias=False)
|
| 223 |
+
(act): GELU(approximate='none')
|
| 224 |
+
(fc2): Linear(in_features=1536, out_features=384, bias=False)
|
| 225 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 226 |
+
)
|
| 227 |
+
)
|
| 228 |
+
)
|
| 229 |
+
)
|
| 230 |
+
(predictor_norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 231 |
+
(predictor_proj): Linear(in_features=384, out_features=1024, bias=True)
|
| 232 |
+
)
|
| 233 |
+
)
|
| 234 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] MultiSeqWrapper(
|
| 235 |
+
(backbone): Frozen2DTargetWrapper(
|
| 236 |
+
(backbone): Eva(
|
| 237 |
+
(patch_embed): PatchEmbed(
|
| 238 |
+
(proj): Conv2d(3, 1024, kernel_size=(14, 14), stride=(14, 14))
|
| 239 |
+
(norm): Identity()
|
| 240 |
+
)
|
| 241 |
+
(pos_drop): Dropout(p=0.0, inplace=False)
|
| 242 |
+
(norm_pre): Identity()
|
| 243 |
+
(blocks): ModuleList(
|
| 244 |
+
(0-23): 24 x EvaBlock(
|
| 245 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 246 |
+
(attn): EvaAttention(
|
| 247 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 248 |
+
(q_norm): Identity()
|
| 249 |
+
(k_norm): Identity()
|
| 250 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 251 |
+
(norm): Identity()
|
| 252 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=True)
|
| 253 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 254 |
+
)
|
| 255 |
+
(drop_path1): Identity()
|
| 256 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 257 |
+
(mlp): Mlp(
|
| 258 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=True)
|
| 259 |
+
(act): GELU(approximate='none')
|
| 260 |
+
(drop1): Dropout(p=0.0, inplace=False)
|
| 261 |
+
(norm): Identity()
|
| 262 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=True)
|
| 263 |
+
(drop2): Dropout(p=0.0, inplace=False)
|
| 264 |
+
)
|
| 265 |
+
(drop_path2): Identity()
|
| 266 |
+
)
|
| 267 |
+
)
|
| 268 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 269 |
+
(fc_norm): Identity()
|
| 270 |
+
(head_drop): Dropout(p=0.0, inplace=False)
|
| 271 |
+
(head): Identity()
|
| 272 |
+
(rope): _CapiPatchRoPE()
|
| 273 |
+
)
|
| 274 |
+
)
|
| 275 |
+
)
|
| 276 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] Encoder number of parameters: 403136512
|
| 277 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] Predictor number of parameters: 11416192
|
| 278 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] Target encoder number of parameters: 0
|
| 279 |
+
[INFO ][2026-05-14 09:43:41][root ][make_videodataset ] VideoDataset dataset created
|
| 280 |
+
[INFO ][2026-05-14 09:43:41][WeightedSampler ][__init__ ] Using DistributedWeightedSampler with rank 10 / 16
|
| 281 |
+
[INFO ][2026-05-14 09:43:41][root ][make_videodataset ] VideoDataset unsupervised data loader created
|
| 282 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] iterations per epoch/dataset length: 300/399
|
| 283 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Wrapping models in DDP (rank 10)...
|
| 284 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Initializing loader...
|
| 285 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 286 |
+
submitit WARNING (2026-05-14 13:38:23,852) - Bypassing signal SIGTERM
|
| 287 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGTERM
|
| 288 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGCONT
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_11_log.err
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/timm/models/layers/__init__.py:49: FutureWarning: Importing from timm.models.layers is deprecated, please import via timm.layers
|
| 2 |
+
warnings.warn(f"Importing from {__name__} is deprecated, please import via timm.layers", FutureWarning)
|
| 3 |
+
[rank11]:[W514 09:35:46.057936025 ProcessGroupNCCL.cpp:5138] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 4 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/utils.py:796: FutureWarning: `torch.cuda.amp.GradScaler(args...)` is deprecated. Please use `torch.amp.GradScaler('cuda', args...)` instead.
|
| 5 |
+
scaler = torch.cuda.amp.GradScaler() if mixed_precision else None
|
| 6 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 7 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 8 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 9 |
+
submitit WARNING (2026-05-14 13:38:23,846) - Bypassing signal SIGTERM
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_11_log.out
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
submitit INFO (2026-05-14 09:28:14,530) - Starting with JobEnvironment(job_id=22743106, hostname=gcn81.local.snellius.surf.nl, local_rank=3(4), node=2(4), global_rank=11(16))
|
| 2 |
+
submitit INFO (2026-05-14 09:28:14,530) - Loading pickle: /scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_submitted.pkl
|
| 3 |
+
INFO:root:loaded pretrain params...
|
| 4 |
+
{ 'app': 'vjepa',
|
| 5 |
+
'cpus_per_task': 16,
|
| 6 |
+
'data': { 'batch_size': 64,
|
| 7 |
+
'crop_size': 224,
|
| 8 |
+
'dataset_fpcs': [16, 16],
|
| 9 |
+
'dataset_type': 'VideoDataset',
|
| 10 |
+
'datasets': [ '/scratch-shared/dcanez/data/kinetics/k400/train.csv',
|
| 11 |
+
'/scratch-shared/dcanez/data/ssv2/train.csv'],
|
| 12 |
+
'datasets_weights': [0.65, 0.35],
|
| 13 |
+
'fps': 4,
|
| 14 |
+
'num_workers': 10,
|
| 15 |
+
'patch_size': 14,
|
| 16 |
+
'persistent_workers': True,
|
| 17 |
+
'pin_mem': True,
|
| 18 |
+
'stage': [ { 'dest': 'kinetics_240',
|
| 19 |
+
'format': 'targz_parts',
|
| 20 |
+
'src': '/scratch-shared/dcanez/data/kinetics/k400/tars_240/'},
|
| 21 |
+
{ 'dest': 'ssv2',
|
| 22 |
+
'format': 'multipart_tar',
|
| 23 |
+
'src': '/scratch-nvme/ml-datasets/something-something-v2/'}],
|
| 24 |
+
'tubelet_size': 1},
|
| 25 |
+
'data_aug': { 'auto_augment': False,
|
| 26 |
+
'motion_shift': False,
|
| 27 |
+
'random_resize_aspect_ratio': [0.75, 1.35],
|
| 28 |
+
'random_resize_scale': [0.3, 1.0],
|
| 29 |
+
'reprob': 0.0},
|
| 30 |
+
'folder': '/scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable',
|
| 31 |
+
'loss': {'loss_exp': 1.0},
|
| 32 |
+
'mask': [ { 'aspect_ratio': [0.75, 1.5],
|
| 33 |
+
'full_complement': False,
|
| 34 |
+
'max_keep': None,
|
| 35 |
+
'max_temporal_keep': 1.0,
|
| 36 |
+
'num_blocks': 8,
|
| 37 |
+
'spatial_scale': [0.15, 0.15],
|
| 38 |
+
'temporal_scale': [1.0, 1.0]},
|
| 39 |
+
{ 'aspect_ratio': [0.75, 1.5],
|
| 40 |
+
'full_complement': False,
|
| 41 |
+
'max_keep': None,
|
| 42 |
+
'max_temporal_keep': 1.0,
|
| 43 |
+
'num_blocks': 2,
|
| 44 |
+
'spatial_scale': [0.7, 0.7],
|
| 45 |
+
'temporal_scale': [1.0, 1.0]}],
|
| 46 |
+
'mem_per_gpu': '180G',
|
| 47 |
+
'meta': { 'dtype': 'bfloat16',
|
| 48 |
+
'knn_eval_epoch0': False,
|
| 49 |
+
'knn_eval_freq': 5,
|
| 50 |
+
'knn_eval_presets': [ { 'config': { 'batch_size': 64,
|
| 51 |
+
'dataset_train': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/train.csv',
|
| 52 |
+
'dataset_val': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/val.csv',
|
| 53 |
+
'eval_videos_per_class': 25,
|
| 54 |
+
'num_workers': 8,
|
| 55 |
+
'pool_type': 'slot_temporal_concat',
|
| 56 |
+
'train_videos_per_class': 100},
|
| 57 |
+
'preset': 'ucf101'},
|
| 58 |
+
{ 'config': { 'batch_size': 64,
|
| 59 |
+
'dataset_train': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/train_coarse10.csv',
|
| 60 |
+
'dataset_val': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/val_coarse10.csv',
|
| 61 |
+
'eval_videos_per_class': 100,
|
| 62 |
+
'linear_probe': True,
|
| 63 |
+
'num_workers': 8,
|
| 64 |
+
'pool_type': 'slot_temporal_concat',
|
| 65 |
+
'train_videos_per_class': 500},
|
| 66 |
+
'preset': 'ssv2_coarse10'}],
|
| 67 |
+
'load_checkpoint': True,
|
| 68 |
+
'read_checkpoint': None,
|
| 69 |
+
'save_every_freq': 5,
|
| 70 |
+
'seed': 239,
|
| 71 |
+
'use_sdpa': True,
|
| 72 |
+
'use_wandb': True,
|
| 73 |
+
'wandb_project': 'vjepa_ujepaside'},
|
| 74 |
+
'metrics': {'sigreg': {}, 'std': {}},
|
| 75 |
+
'model': { 'model_name': 'ujepaside_large_patch14_capi_lvd1689m',
|
| 76 |
+
'pred_depth': 6,
|
| 77 |
+
'pred_embed_dim': 384,
|
| 78 |
+
'pred_num_heads': 12,
|
| 79 |
+
'predictor': 'v2_cross',
|
| 80 |
+
'st_causal': False,
|
| 81 |
+
'st_drop_path': 0.2,
|
| 82 |
+
'st_flex_enable': False,
|
| 83 |
+
'st_layer_scale_init': 1e-05,
|
| 84 |
+
'st_num_slots': 16,
|
| 85 |
+
'st_side_block_type': 'factorized',
|
| 86 |
+
'st_slots_causal_within_frame': False,
|
| 87 |
+
'target_kind': 'frozen_2d',
|
| 88 |
+
'target_type': 'vit_large_patch14_capi.lvd1689m',
|
| 89 |
+
'temporal_spacing': 1.0,
|
| 90 |
+
'uniform_power': True,
|
| 91 |
+
'use_activation_checkpointing': True,
|
| 92 |
+
'use_mask_tokens': True,
|
| 93 |
+
'use_rope': True,
|
| 94 |
+
'use_sdpa': True,
|
| 95 |
+
'zero_init_mask_tokens': True},
|
| 96 |
+
'nodes': 4,
|
| 97 |
+
'optimization': { 'clip_grad': 3.0,
|
| 98 |
+
'ema': [0.99925, 0.99925],
|
| 99 |
+
'epochs': 100,
|
| 100 |
+
'final_lr': 0.0001,
|
| 101 |
+
'final_weight_decay': 0.04,
|
| 102 |
+
'ipe': 300,
|
| 103 |
+
'ipe_scale': 1.0,
|
| 104 |
+
'lr': 0.0005,
|
| 105 |
+
'start_lr': 0.0001,
|
| 106 |
+
'warmup': 10,
|
| 107 |
+
'weight_decay': 0.04},
|
| 108 |
+
'tasks_per_node': 4}
|
| 109 |
+
INFO:root:Running pre-training of app: vjepa
|
| 110 |
+
[INFO ][2026-05-14 09:32:44][app.vjepa.train ][main ] which_dtype='bfloat16'
|
| 111 |
+
[INFO ][2026-05-14 09:32:44][app.vjepa.train ][main ] Disabling persistent_workers (incompatible with KNN eval)
|
| 112 |
+
[INFO ][2026-05-14 09:32:44][app.vjepa.train ][main ] NCCL_SOCKET_IFNAME=eno
|
| 113 |
+
[INFO ][2026-05-14 09:32:46][app.vjepa.train ][main ] Initialized (rank/world-size) 11/16, tasks_per_node=4
|
| 114 |
+
[INFO ][2026-05-14 09:32:46][root ][stage_datasets ] [local_rank 3/4] Staging kinetics_240 (targz_parts)
|
| 115 |
+
[INFO ][2026-05-14 09:32:46][root ][_stage_targz_parts ] [rank 3] Extracting 25/103 tar.gz parts to /scratch-node/dcanez.22743106/kinetics_240
|
| 116 |
+
[INFO ][2026-05-14 09:33:00][root ][_stage_targz_parts ] [local_rank 3] Extracted 2/25 parts
|
| 117 |
+
[INFO ][2026-05-14 09:33:14][root ][_stage_targz_parts ] [local_rank 3] Extracted 4/25 parts
|
| 118 |
+
[INFO ][2026-05-14 09:33:28][root ][_stage_targz_parts ] [local_rank 3] Extracted 6/25 parts
|
| 119 |
+
[INFO ][2026-05-14 09:33:43][root ][_stage_targz_parts ] [local_rank 3] Extracted 8/25 parts
|
| 120 |
+
[INFO ][2026-05-14 09:33:57][root ][_stage_targz_parts ] [local_rank 3] Extracted 10/25 parts
|
| 121 |
+
[INFO ][2026-05-14 09:34:11][root ][_stage_targz_parts ] [local_rank 3] Extracted 12/25 parts
|
| 122 |
+
[INFO ][2026-05-14 09:34:25][root ][_stage_targz_parts ] [local_rank 3] Extracted 14/25 parts
|
| 123 |
+
[INFO ][2026-05-14 09:34:40][root ][_stage_targz_parts ] [local_rank 3] Extracted 16/25 parts
|
| 124 |
+
[INFO ][2026-05-14 09:34:54][root ][_stage_targz_parts ] [local_rank 3] Extracted 18/25 parts
|
| 125 |
+
[INFO ][2026-05-14 09:35:08][root ][_stage_targz_parts ] [local_rank 3] Extracted 20/25 parts
|
| 126 |
+
[INFO ][2026-05-14 09:35:23][root ][_stage_targz_parts ] [local_rank 3] Extracted 22/25 parts
|
| 127 |
+
[INFO ][2026-05-14 09:35:37][root ][_stage_targz_parts ] [local_rank 3] Extracted 24/25 parts
|
| 128 |
+
[INFO ][2026-05-14 09:35:45][root ][_stage_targz_parts ] [local_rank 3] Extracted 25/25 parts
|
| 129 |
+
[INFO ][2026-05-14 09:35:45][root ][stage_datasets ] [local_rank 3/4] Staging ssv2 (multipart_tar)
|
| 130 |
+
[INFO ][2026-05-14 09:36:49][app.vjepa.train ][main ] Staged datasets: ['/scratch-node/dcanez.22743106/kinetics_240/train.csv', '/scratch-node/dcanez.22743106/ssv2/train.csv']
|
| 131 |
+
[INFO ][2026-05-14 09:43:14][experiments.stmodels.vision_transformers_v3][vit_large_patch14_capi_lvd1689m] Loading pretrained weights for vit_large_patch14_capi_lvd1689m from https://dl.fbaipublicfiles.com/capi/capi_vitl14_lvd.pth
|
| 132 |
+
[INFO ][2026-05-14 09:43:28][root ][_build_frozen_2d_target ] Loaded frozen_2d target from timm: vit_large_patch14_capi.lvd1689m
|
| 133 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] ViTMultiSeqWrapper(
|
| 134 |
+
(backbone): UJEPAside(
|
| 135 |
+
(patch_embed): PatchEmbed3D(
|
| 136 |
+
(proj): Conv3d(3, 1024, kernel_size=(1, 14, 14), stride=(1, 14, 14))
|
| 137 |
+
)
|
| 138 |
+
(rope): CAPI2DRoPE()
|
| 139 |
+
(blocks): ModuleList(
|
| 140 |
+
(0-23): 24 x Block(
|
| 141 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 142 |
+
(rope_impl): CAPI2DRoPE()
|
| 143 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 144 |
+
(drop_path1): Identity()
|
| 145 |
+
(drop_path2): Identity()
|
| 146 |
+
(attn): Attention(
|
| 147 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 148 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 149 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 150 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 151 |
+
(rope_impl): CAPI2DRoPE()
|
| 152 |
+
)
|
| 153 |
+
(mlp): MLP(
|
| 154 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 155 |
+
(act): GELU(approximate='none')
|
| 156 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 157 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 158 |
+
)
|
| 159 |
+
)
|
| 160 |
+
)
|
| 161 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=False)
|
| 162 |
+
(st_blocks): ModuleList(
|
| 163 |
+
(0-23): 24 x FactorizedSlotSideBlock(
|
| 164 |
+
(cross_attn): EfficientResidual(
|
| 165 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 166 |
+
(fn): Attention(
|
| 167 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 168 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 169 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 170 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 171 |
+
(rope_impl): CAPI3DRoPE()
|
| 172 |
+
)
|
| 173 |
+
)
|
| 174 |
+
(self_attn): EfficientResidual(
|
| 175 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 176 |
+
(fn): Attention(
|
| 177 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 178 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 179 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 180 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 181 |
+
(rope_impl): CAPI3DRoPE()
|
| 182 |
+
)
|
| 183 |
+
)
|
| 184 |
+
(mlp): EfficientResidual(
|
| 185 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 186 |
+
(fn): MLP(
|
| 187 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 188 |
+
(act): GELU(approximate='none')
|
| 189 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 190 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 191 |
+
)
|
| 192 |
+
)
|
| 193 |
+
)
|
| 194 |
+
)
|
| 195 |
+
(st_rope): CAPI3DRoPE()
|
| 196 |
+
)
|
| 197 |
+
)
|
| 198 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] UJEPAsidePredictorMultiSeqWrapper(
|
| 199 |
+
(backbone): PredictorV2(
|
| 200 |
+
(predictor_embed): Linear(in_features=1024, out_features=384, bias=True)
|
| 201 |
+
(mask_tokens): ParameterList(
|
| 202 |
+
(0): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 203 |
+
(1): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 204 |
+
(2): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 205 |
+
(3): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 206 |
+
)
|
| 207 |
+
(predictor_blocks): ModuleList(
|
| 208 |
+
(0-5): 6 x Block(
|
| 209 |
+
(residual1): EfficientResidual(
|
| 210 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 211 |
+
(fn): Attention(
|
| 212 |
+
(q_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 213 |
+
(k_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 214 |
+
(v_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 215 |
+
(proj): Linear(in_features=384, out_features=384, bias=False)
|
| 216 |
+
(rope): Rope()
|
| 217 |
+
)
|
| 218 |
+
)
|
| 219 |
+
(residual2): EfficientResidual(
|
| 220 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 221 |
+
(fn): MLP(
|
| 222 |
+
(fc1): Linear(in_features=384, out_features=1536, bias=False)
|
| 223 |
+
(act): GELU(approximate='none')
|
| 224 |
+
(fc2): Linear(in_features=1536, out_features=384, bias=False)
|
| 225 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 226 |
+
)
|
| 227 |
+
)
|
| 228 |
+
)
|
| 229 |
+
)
|
| 230 |
+
(predictor_norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 231 |
+
(predictor_proj): Linear(in_features=384, out_features=1024, bias=True)
|
| 232 |
+
)
|
| 233 |
+
)
|
| 234 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] MultiSeqWrapper(
|
| 235 |
+
(backbone): Frozen2DTargetWrapper(
|
| 236 |
+
(backbone): Eva(
|
| 237 |
+
(patch_embed): PatchEmbed(
|
| 238 |
+
(proj): Conv2d(3, 1024, kernel_size=(14, 14), stride=(14, 14))
|
| 239 |
+
(norm): Identity()
|
| 240 |
+
)
|
| 241 |
+
(pos_drop): Dropout(p=0.0, inplace=False)
|
| 242 |
+
(norm_pre): Identity()
|
| 243 |
+
(blocks): ModuleList(
|
| 244 |
+
(0-23): 24 x EvaBlock(
|
| 245 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 246 |
+
(attn): EvaAttention(
|
| 247 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 248 |
+
(q_norm): Identity()
|
| 249 |
+
(k_norm): Identity()
|
| 250 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 251 |
+
(norm): Identity()
|
| 252 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=True)
|
| 253 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 254 |
+
)
|
| 255 |
+
(drop_path1): Identity()
|
| 256 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 257 |
+
(mlp): Mlp(
|
| 258 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=True)
|
| 259 |
+
(act): GELU(approximate='none')
|
| 260 |
+
(drop1): Dropout(p=0.0, inplace=False)
|
| 261 |
+
(norm): Identity()
|
| 262 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=True)
|
| 263 |
+
(drop2): Dropout(p=0.0, inplace=False)
|
| 264 |
+
)
|
| 265 |
+
(drop_path2): Identity()
|
| 266 |
+
)
|
| 267 |
+
)
|
| 268 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 269 |
+
(fc_norm): Identity()
|
| 270 |
+
(head_drop): Dropout(p=0.0, inplace=False)
|
| 271 |
+
(head): Identity()
|
| 272 |
+
(rope): _CapiPatchRoPE()
|
| 273 |
+
)
|
| 274 |
+
)
|
| 275 |
+
)
|
| 276 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] Encoder number of parameters: 403136512
|
| 277 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] Predictor number of parameters: 11416192
|
| 278 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] Target encoder number of parameters: 0
|
| 279 |
+
[INFO ][2026-05-14 09:43:41][root ][make_videodataset ] VideoDataset dataset created
|
| 280 |
+
[INFO ][2026-05-14 09:43:41][WeightedSampler ][__init__ ] Using DistributedWeightedSampler with rank 11 / 16
|
| 281 |
+
[INFO ][2026-05-14 09:43:41][root ][make_videodataset ] VideoDataset unsupervised data loader created
|
| 282 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] iterations per epoch/dataset length: 300/399
|
| 283 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Wrapping models in DDP (rank 11)...
|
| 284 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Initializing loader...
|
| 285 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 286 |
+
submitit WARNING (2026-05-14 13:38:23,846) - Bypassing signal SIGTERM
|
| 287 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGTERM
|
| 288 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGCONT
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_12_log.err
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/timm/models/layers/__init__.py:49: FutureWarning: Importing from timm.models.layers is deprecated, please import via timm.layers
|
| 2 |
+
warnings.warn(f"Importing from {__name__} is deprecated, please import via timm.layers", FutureWarning)
|
| 3 |
+
[rank12]:[W514 09:36:45.287417238 ProcessGroupNCCL.cpp:5138] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 4 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/utils.py:796: FutureWarning: `torch.cuda.amp.GradScaler(args...)` is deprecated. Please use `torch.amp.GradScaler('cuda', args...)` instead.
|
| 5 |
+
scaler = torch.cuda.amp.GradScaler() if mixed_precision else None
|
| 6 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 7 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 8 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 9 |
+
submitit WARNING (2026-05-14 13:38:23,838) - Bypassing signal SIGTERM
|
| 10 |
+
[2026-05-14T13:38:56.223] error: namespace_p_join: open failed for /slurm/22743106/.ns: No such file or directory
|
| 11 |
+
[2026-05-14T13:38:56.223] error: namespace_g_join(JobId=22743106 SLUID=s8FXKRJP5N0000): No such file or directory
|
| 12 |
+
[2026-05-14T13:38:57.868] error: Failed to send MESSAGE_TASK_EXIT: Connection refused
|
| 13 |
+
[2026-05-14T13:38:59.637] error: namespace_p_join: open failed for /slurm/22743106/.ns: No such file or directory
|
| 14 |
+
[2026-05-14T13:38:59.637] error: namespace_g_join(JobId=22743106 SLUID=s8FXKRJP5N0000): No such file or directory
|
| 15 |
+
[2026-05-14T13:38:59.776] error: namespace_p_join: open failed for /slurm/22743106/.ns: No such file or directory
|
| 16 |
+
[2026-05-14T13:38:59.776] error: namespace_g_join(JobId=22743106 SLUID=s8FXKRJP5N0000): No such file or directory
|
| 17 |
+
[2026-05-14T13:38:59.815] error: namespace_p_join: open failed for /slurm/22743106/.ns: No such file or directory
|
| 18 |
+
[2026-05-14T13:38:59.815] error: namespace_g_join(JobId=22743106 SLUID=s8FXKRJP5N0000): No such file or directory
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_12_log.out
ADDED
|
@@ -0,0 +1,292 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
submitit INFO (2026-05-14 09:28:14,530) - Starting with JobEnvironment(job_id=22743106, hostname=gcn83.local.snellius.surf.nl, local_rank=0(4), node=3(4), global_rank=12(16))
|
| 2 |
+
submitit INFO (2026-05-14 09:28:14,530) - Loading pickle: /scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_submitted.pkl
|
| 3 |
+
INFO:root:loaded pretrain params...
|
| 4 |
+
{ 'app': 'vjepa',
|
| 5 |
+
'cpus_per_task': 16,
|
| 6 |
+
'data': { 'batch_size': 64,
|
| 7 |
+
'crop_size': 224,
|
| 8 |
+
'dataset_fpcs': [16, 16],
|
| 9 |
+
'dataset_type': 'VideoDataset',
|
| 10 |
+
'datasets': [ '/scratch-shared/dcanez/data/kinetics/k400/train.csv',
|
| 11 |
+
'/scratch-shared/dcanez/data/ssv2/train.csv'],
|
| 12 |
+
'datasets_weights': [0.65, 0.35],
|
| 13 |
+
'fps': 4,
|
| 14 |
+
'num_workers': 10,
|
| 15 |
+
'patch_size': 14,
|
| 16 |
+
'persistent_workers': True,
|
| 17 |
+
'pin_mem': True,
|
| 18 |
+
'stage': [ { 'dest': 'kinetics_240',
|
| 19 |
+
'format': 'targz_parts',
|
| 20 |
+
'src': '/scratch-shared/dcanez/data/kinetics/k400/tars_240/'},
|
| 21 |
+
{ 'dest': 'ssv2',
|
| 22 |
+
'format': 'multipart_tar',
|
| 23 |
+
'src': '/scratch-nvme/ml-datasets/something-something-v2/'}],
|
| 24 |
+
'tubelet_size': 1},
|
| 25 |
+
'data_aug': { 'auto_augment': False,
|
| 26 |
+
'motion_shift': False,
|
| 27 |
+
'random_resize_aspect_ratio': [0.75, 1.35],
|
| 28 |
+
'random_resize_scale': [0.3, 1.0],
|
| 29 |
+
'reprob': 0.0},
|
| 30 |
+
'folder': '/scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable',
|
| 31 |
+
'loss': {'loss_exp': 1.0},
|
| 32 |
+
'mask': [ { 'aspect_ratio': [0.75, 1.5],
|
| 33 |
+
'full_complement': False,
|
| 34 |
+
'max_keep': None,
|
| 35 |
+
'max_temporal_keep': 1.0,
|
| 36 |
+
'num_blocks': 8,
|
| 37 |
+
'spatial_scale': [0.15, 0.15],
|
| 38 |
+
'temporal_scale': [1.0, 1.0]},
|
| 39 |
+
{ 'aspect_ratio': [0.75, 1.5],
|
| 40 |
+
'full_complement': False,
|
| 41 |
+
'max_keep': None,
|
| 42 |
+
'max_temporal_keep': 1.0,
|
| 43 |
+
'num_blocks': 2,
|
| 44 |
+
'spatial_scale': [0.7, 0.7],
|
| 45 |
+
'temporal_scale': [1.0, 1.0]}],
|
| 46 |
+
'mem_per_gpu': '180G',
|
| 47 |
+
'meta': { 'dtype': 'bfloat16',
|
| 48 |
+
'knn_eval_epoch0': False,
|
| 49 |
+
'knn_eval_freq': 5,
|
| 50 |
+
'knn_eval_presets': [ { 'config': { 'batch_size': 64,
|
| 51 |
+
'dataset_train': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/train.csv',
|
| 52 |
+
'dataset_val': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/val.csv',
|
| 53 |
+
'eval_videos_per_class': 25,
|
| 54 |
+
'num_workers': 8,
|
| 55 |
+
'pool_type': 'slot_temporal_concat',
|
| 56 |
+
'train_videos_per_class': 100},
|
| 57 |
+
'preset': 'ucf101'},
|
| 58 |
+
{ 'config': { 'batch_size': 64,
|
| 59 |
+
'dataset_train': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/train_coarse10.csv',
|
| 60 |
+
'dataset_val': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/val_coarse10.csv',
|
| 61 |
+
'eval_videos_per_class': 100,
|
| 62 |
+
'linear_probe': True,
|
| 63 |
+
'num_workers': 8,
|
| 64 |
+
'pool_type': 'slot_temporal_concat',
|
| 65 |
+
'train_videos_per_class': 500},
|
| 66 |
+
'preset': 'ssv2_coarse10'}],
|
| 67 |
+
'load_checkpoint': True,
|
| 68 |
+
'read_checkpoint': None,
|
| 69 |
+
'save_every_freq': 5,
|
| 70 |
+
'seed': 239,
|
| 71 |
+
'use_sdpa': True,
|
| 72 |
+
'use_wandb': True,
|
| 73 |
+
'wandb_project': 'vjepa_ujepaside'},
|
| 74 |
+
'metrics': {'sigreg': {}, 'std': {}},
|
| 75 |
+
'model': { 'model_name': 'ujepaside_large_patch14_capi_lvd1689m',
|
| 76 |
+
'pred_depth': 6,
|
| 77 |
+
'pred_embed_dim': 384,
|
| 78 |
+
'pred_num_heads': 12,
|
| 79 |
+
'predictor': 'v2_cross',
|
| 80 |
+
'st_causal': False,
|
| 81 |
+
'st_drop_path': 0.2,
|
| 82 |
+
'st_flex_enable': False,
|
| 83 |
+
'st_layer_scale_init': 1e-05,
|
| 84 |
+
'st_num_slots': 16,
|
| 85 |
+
'st_side_block_type': 'factorized',
|
| 86 |
+
'st_slots_causal_within_frame': False,
|
| 87 |
+
'target_kind': 'frozen_2d',
|
| 88 |
+
'target_type': 'vit_large_patch14_capi.lvd1689m',
|
| 89 |
+
'temporal_spacing': 1.0,
|
| 90 |
+
'uniform_power': True,
|
| 91 |
+
'use_activation_checkpointing': True,
|
| 92 |
+
'use_mask_tokens': True,
|
| 93 |
+
'use_rope': True,
|
| 94 |
+
'use_sdpa': True,
|
| 95 |
+
'zero_init_mask_tokens': True},
|
| 96 |
+
'nodes': 4,
|
| 97 |
+
'optimization': { 'clip_grad': 3.0,
|
| 98 |
+
'ema': [0.99925, 0.99925],
|
| 99 |
+
'epochs': 100,
|
| 100 |
+
'final_lr': 0.0001,
|
| 101 |
+
'final_weight_decay': 0.04,
|
| 102 |
+
'ipe': 300,
|
| 103 |
+
'ipe_scale': 1.0,
|
| 104 |
+
'lr': 0.0005,
|
| 105 |
+
'start_lr': 0.0001,
|
| 106 |
+
'warmup': 10,
|
| 107 |
+
'weight_decay': 0.04},
|
| 108 |
+
'tasks_per_node': 4}
|
| 109 |
+
INFO:root:Running pre-training of app: vjepa
|
| 110 |
+
[INFO ][2026-05-14 09:32:42][app.vjepa.train ][main ] which_dtype='bfloat16'
|
| 111 |
+
[INFO ][2026-05-14 09:32:42][app.vjepa.train ][main ] Disabling persistent_workers (incompatible with KNN eval)
|
| 112 |
+
[INFO ][2026-05-14 09:32:42][app.vjepa.train ][main ] NCCL_SOCKET_IFNAME=eno
|
| 113 |
+
[INFO ][2026-05-14 09:32:44][app.vjepa.train ][main ] Initialized (rank/world-size) 12/16, tasks_per_node=4
|
| 114 |
+
[INFO ][2026-05-14 09:32:44][root ][stage_datasets ] [local_rank 0/4] Staging kinetics_240 (targz_parts)
|
| 115 |
+
[INFO ][2026-05-14 09:32:44][root ][_stage_targz_parts ] [rank 0] Extracting 26/103 tar.gz parts to /scratch-node/dcanez.22743106/kinetics_240
|
| 116 |
+
[INFO ][2026-05-14 09:32:59][root ][_stage_targz_parts ] [local_rank 0] Extracted 2/26 parts
|
| 117 |
+
[INFO ][2026-05-14 09:33:13][root ][_stage_targz_parts ] [local_rank 0] Extracted 4/26 parts
|
| 118 |
+
[INFO ][2026-05-14 09:33:27][root ][_stage_targz_parts ] [local_rank 0] Extracted 6/26 parts
|
| 119 |
+
[INFO ][2026-05-14 09:33:41][root ][_stage_targz_parts ] [local_rank 0] Extracted 8/26 parts
|
| 120 |
+
[INFO ][2026-05-14 09:33:55][root ][_stage_targz_parts ] [local_rank 0] Extracted 10/26 parts
|
| 121 |
+
[INFO ][2026-05-14 09:34:10][root ][_stage_targz_parts ] [local_rank 0] Extracted 12/26 parts
|
| 122 |
+
[INFO ][2026-05-14 09:34:24][root ][_stage_targz_parts ] [local_rank 0] Extracted 14/26 parts
|
| 123 |
+
[INFO ][2026-05-14 09:34:38][root ][_stage_targz_parts ] [local_rank 0] Extracted 16/26 parts
|
| 124 |
+
[INFO ][2026-05-14 09:34:52][root ][_stage_targz_parts ] [local_rank 0] Extracted 18/26 parts
|
| 125 |
+
[INFO ][2026-05-14 09:35:07][root ][_stage_targz_parts ] [local_rank 0] Extracted 20/26 parts
|
| 126 |
+
[INFO ][2026-05-14 09:35:21][root ][_stage_targz_parts ] [local_rank 0] Extracted 22/26 parts
|
| 127 |
+
[INFO ][2026-05-14 09:35:36][root ][_stage_targz_parts ] [local_rank 0] Extracted 24/26 parts
|
| 128 |
+
[INFO ][2026-05-14 09:35:50][root ][_stage_targz_parts ] [local_rank 0] Extracted 26/26 parts
|
| 129 |
+
[INFO ][2026-05-14 09:35:50][root ][stage_datasets ] [local_rank 0/4] Staging ssv2 (multipart_tar)
|
| 130 |
+
[INFO ][2026-05-14 09:35:50][root ][_stage_multipart_tar ] [rank 0] Extracting multipart tar (2 files) to /scratch-node/dcanez.22743106/ssv2
|
| 131 |
+
[INFO ][2026-05-14 09:36:48][root ][stage_datasets ] Data staging completed in 243.6s (4.1min)
|
| 132 |
+
[INFO ][2026-05-14 09:36:49][root ][_rewrite_csv ] Wrote local CSV: /scratch-node/dcanez.22743106/kinetics_240/train.csv (239789 entries)
|
| 133 |
+
[INFO ][2026-05-14 09:36:49][root ][_rewrite_csv ] Wrote local CSV: /scratch-node/dcanez.22743106/ssv2/train.csv (168913 entries)
|
| 134 |
+
[INFO ][2026-05-14 09:36:49][app.vjepa.train ][main ] Staged datasets: ['/scratch-node/dcanez.22743106/kinetics_240/train.csv', '/scratch-node/dcanez.22743106/ssv2/train.csv']
|
| 135 |
+
[INFO ][2026-05-14 09:43:13][experiments.stmodels.vision_transformers_v3][vit_large_patch14_capi_lvd1689m] Loading pretrained weights for vit_large_patch14_capi_lvd1689m from https://dl.fbaipublicfiles.com/capi/capi_vitl14_lvd.pth
|
| 136 |
+
[INFO ][2026-05-14 09:43:23][root ][_build_frozen_2d_target ] Loaded frozen_2d target from timm: vit_large_patch14_capi.lvd1689m
|
| 137 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] ViTMultiSeqWrapper(
|
| 138 |
+
(backbone): UJEPAside(
|
| 139 |
+
(patch_embed): PatchEmbed3D(
|
| 140 |
+
(proj): Conv3d(3, 1024, kernel_size=(1, 14, 14), stride=(1, 14, 14))
|
| 141 |
+
)
|
| 142 |
+
(rope): CAPI2DRoPE()
|
| 143 |
+
(blocks): ModuleList(
|
| 144 |
+
(0-23): 24 x Block(
|
| 145 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 146 |
+
(rope_impl): CAPI2DRoPE()
|
| 147 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 148 |
+
(drop_path1): Identity()
|
| 149 |
+
(drop_path2): Identity()
|
| 150 |
+
(attn): Attention(
|
| 151 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 152 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 153 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 154 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 155 |
+
(rope_impl): CAPI2DRoPE()
|
| 156 |
+
)
|
| 157 |
+
(mlp): MLP(
|
| 158 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 159 |
+
(act): GELU(approximate='none')
|
| 160 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 161 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 162 |
+
)
|
| 163 |
+
)
|
| 164 |
+
)
|
| 165 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=False)
|
| 166 |
+
(st_blocks): ModuleList(
|
| 167 |
+
(0-23): 24 x FactorizedSlotSideBlock(
|
| 168 |
+
(cross_attn): EfficientResidual(
|
| 169 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 170 |
+
(fn): Attention(
|
| 171 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 172 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 173 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 174 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 175 |
+
(rope_impl): CAPI3DRoPE()
|
| 176 |
+
)
|
| 177 |
+
)
|
| 178 |
+
(self_attn): EfficientResidual(
|
| 179 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 180 |
+
(fn): Attention(
|
| 181 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 182 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 183 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 184 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 185 |
+
(rope_impl): CAPI3DRoPE()
|
| 186 |
+
)
|
| 187 |
+
)
|
| 188 |
+
(mlp): EfficientResidual(
|
| 189 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 190 |
+
(fn): MLP(
|
| 191 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 192 |
+
(act): GELU(approximate='none')
|
| 193 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 194 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 195 |
+
)
|
| 196 |
+
)
|
| 197 |
+
)
|
| 198 |
+
)
|
| 199 |
+
(st_rope): CAPI3DRoPE()
|
| 200 |
+
)
|
| 201 |
+
)
|
| 202 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] UJEPAsidePredictorMultiSeqWrapper(
|
| 203 |
+
(backbone): PredictorV2(
|
| 204 |
+
(predictor_embed): Linear(in_features=1024, out_features=384, bias=True)
|
| 205 |
+
(mask_tokens): ParameterList(
|
| 206 |
+
(0): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 207 |
+
(1): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 208 |
+
(2): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 209 |
+
(3): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 210 |
+
)
|
| 211 |
+
(predictor_blocks): ModuleList(
|
| 212 |
+
(0-5): 6 x Block(
|
| 213 |
+
(residual1): EfficientResidual(
|
| 214 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 215 |
+
(fn): Attention(
|
| 216 |
+
(q_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 217 |
+
(k_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 218 |
+
(v_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 219 |
+
(proj): Linear(in_features=384, out_features=384, bias=False)
|
| 220 |
+
(rope): Rope()
|
| 221 |
+
)
|
| 222 |
+
)
|
| 223 |
+
(residual2): EfficientResidual(
|
| 224 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 225 |
+
(fn): MLP(
|
| 226 |
+
(fc1): Linear(in_features=384, out_features=1536, bias=False)
|
| 227 |
+
(act): GELU(approximate='none')
|
| 228 |
+
(fc2): Linear(in_features=1536, out_features=384, bias=False)
|
| 229 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 230 |
+
)
|
| 231 |
+
)
|
| 232 |
+
)
|
| 233 |
+
)
|
| 234 |
+
(predictor_norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 235 |
+
(predictor_proj): Linear(in_features=384, out_features=1024, bias=True)
|
| 236 |
+
)
|
| 237 |
+
)
|
| 238 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] MultiSeqWrapper(
|
| 239 |
+
(backbone): Frozen2DTargetWrapper(
|
| 240 |
+
(backbone): Eva(
|
| 241 |
+
(patch_embed): PatchEmbed(
|
| 242 |
+
(proj): Conv2d(3, 1024, kernel_size=(14, 14), stride=(14, 14))
|
| 243 |
+
(norm): Identity()
|
| 244 |
+
)
|
| 245 |
+
(pos_drop): Dropout(p=0.0, inplace=False)
|
| 246 |
+
(norm_pre): Identity()
|
| 247 |
+
(blocks): ModuleList(
|
| 248 |
+
(0-23): 24 x EvaBlock(
|
| 249 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 250 |
+
(attn): EvaAttention(
|
| 251 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 252 |
+
(q_norm): Identity()
|
| 253 |
+
(k_norm): Identity()
|
| 254 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 255 |
+
(norm): Identity()
|
| 256 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=True)
|
| 257 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 258 |
+
)
|
| 259 |
+
(drop_path1): Identity()
|
| 260 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 261 |
+
(mlp): Mlp(
|
| 262 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=True)
|
| 263 |
+
(act): GELU(approximate='none')
|
| 264 |
+
(drop1): Dropout(p=0.0, inplace=False)
|
| 265 |
+
(norm): Identity()
|
| 266 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=True)
|
| 267 |
+
(drop2): Dropout(p=0.0, inplace=False)
|
| 268 |
+
)
|
| 269 |
+
(drop_path2): Identity()
|
| 270 |
+
)
|
| 271 |
+
)
|
| 272 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 273 |
+
(fc_norm): Identity()
|
| 274 |
+
(head_drop): Dropout(p=0.0, inplace=False)
|
| 275 |
+
(head): Identity()
|
| 276 |
+
(rope): _CapiPatchRoPE()
|
| 277 |
+
)
|
| 278 |
+
)
|
| 279 |
+
)
|
| 280 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] Encoder number of parameters: 403136512
|
| 281 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] Predictor number of parameters: 11416192
|
| 282 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] Target encoder number of parameters: 0
|
| 283 |
+
[INFO ][2026-05-14 09:43:36][root ][make_videodataset ] VideoDataset dataset created
|
| 284 |
+
[INFO ][2026-05-14 09:43:36][WeightedSampler ][__init__ ] Using DistributedWeightedSampler with rank 12 / 16
|
| 285 |
+
[INFO ][2026-05-14 09:43:36][root ][make_videodataset ] VideoDataset unsupervised data loader created
|
| 286 |
+
[INFO ][2026-05-14 09:43:36][app.vjepa.train ][main ] iterations per epoch/dataset length: 300/399
|
| 287 |
+
[INFO ][2026-05-14 09:43:36][app.vjepa.train ][main ] Wrapping models in DDP (rank 12)...
|
| 288 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Initializing loader...
|
| 289 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 290 |
+
submitit WARNING (2026-05-14 13:38:23,838) - Bypassing signal SIGTERM
|
| 291 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGTERM
|
| 292 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGCONT
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_13_log.err
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/timm/models/layers/__init__.py:49: FutureWarning: Importing from timm.models.layers is deprecated, please import via timm.layers
|
| 2 |
+
warnings.warn(f"Importing from {__name__} is deprecated, please import via timm.layers", FutureWarning)
|
| 3 |
+
[rank13]:[W514 09:35:49.225966127 ProcessGroupNCCL.cpp:5138] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 4 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/utils.py:796: FutureWarning: `torch.cuda.amp.GradScaler(args...)` is deprecated. Please use `torch.amp.GradScaler('cuda', args...)` instead.
|
| 5 |
+
scaler = torch.cuda.amp.GradScaler() if mixed_precision else None
|
| 6 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 7 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 8 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 9 |
+
submitit WARNING (2026-05-14 13:38:23,823) - Bypassing signal SIGTERM
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_13_log.out
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
submitit INFO (2026-05-14 09:28:14,530) - Starting with JobEnvironment(job_id=22743106, hostname=gcn83.local.snellius.surf.nl, local_rank=1(4), node=3(4), global_rank=13(16))
|
| 2 |
+
submitit INFO (2026-05-14 09:28:14,530) - Loading pickle: /scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_submitted.pkl
|
| 3 |
+
INFO:root:loaded pretrain params...
|
| 4 |
+
{ 'app': 'vjepa',
|
| 5 |
+
'cpus_per_task': 16,
|
| 6 |
+
'data': { 'batch_size': 64,
|
| 7 |
+
'crop_size': 224,
|
| 8 |
+
'dataset_fpcs': [16, 16],
|
| 9 |
+
'dataset_type': 'VideoDataset',
|
| 10 |
+
'datasets': [ '/scratch-shared/dcanez/data/kinetics/k400/train.csv',
|
| 11 |
+
'/scratch-shared/dcanez/data/ssv2/train.csv'],
|
| 12 |
+
'datasets_weights': [0.65, 0.35],
|
| 13 |
+
'fps': 4,
|
| 14 |
+
'num_workers': 10,
|
| 15 |
+
'patch_size': 14,
|
| 16 |
+
'persistent_workers': True,
|
| 17 |
+
'pin_mem': True,
|
| 18 |
+
'stage': [ { 'dest': 'kinetics_240',
|
| 19 |
+
'format': 'targz_parts',
|
| 20 |
+
'src': '/scratch-shared/dcanez/data/kinetics/k400/tars_240/'},
|
| 21 |
+
{ 'dest': 'ssv2',
|
| 22 |
+
'format': 'multipart_tar',
|
| 23 |
+
'src': '/scratch-nvme/ml-datasets/something-something-v2/'}],
|
| 24 |
+
'tubelet_size': 1},
|
| 25 |
+
'data_aug': { 'auto_augment': False,
|
| 26 |
+
'motion_shift': False,
|
| 27 |
+
'random_resize_aspect_ratio': [0.75, 1.35],
|
| 28 |
+
'random_resize_scale': [0.3, 1.0],
|
| 29 |
+
'reprob': 0.0},
|
| 30 |
+
'folder': '/scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable',
|
| 31 |
+
'loss': {'loss_exp': 1.0},
|
| 32 |
+
'mask': [ { 'aspect_ratio': [0.75, 1.5],
|
| 33 |
+
'full_complement': False,
|
| 34 |
+
'max_keep': None,
|
| 35 |
+
'max_temporal_keep': 1.0,
|
| 36 |
+
'num_blocks': 8,
|
| 37 |
+
'spatial_scale': [0.15, 0.15],
|
| 38 |
+
'temporal_scale': [1.0, 1.0]},
|
| 39 |
+
{ 'aspect_ratio': [0.75, 1.5],
|
| 40 |
+
'full_complement': False,
|
| 41 |
+
'max_keep': None,
|
| 42 |
+
'max_temporal_keep': 1.0,
|
| 43 |
+
'num_blocks': 2,
|
| 44 |
+
'spatial_scale': [0.7, 0.7],
|
| 45 |
+
'temporal_scale': [1.0, 1.0]}],
|
| 46 |
+
'mem_per_gpu': '180G',
|
| 47 |
+
'meta': { 'dtype': 'bfloat16',
|
| 48 |
+
'knn_eval_epoch0': False,
|
| 49 |
+
'knn_eval_freq': 5,
|
| 50 |
+
'knn_eval_presets': [ { 'config': { 'batch_size': 64,
|
| 51 |
+
'dataset_train': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/train.csv',
|
| 52 |
+
'dataset_val': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/val.csv',
|
| 53 |
+
'eval_videos_per_class': 25,
|
| 54 |
+
'num_workers': 8,
|
| 55 |
+
'pool_type': 'slot_temporal_concat',
|
| 56 |
+
'train_videos_per_class': 100},
|
| 57 |
+
'preset': 'ucf101'},
|
| 58 |
+
{ 'config': { 'batch_size': 64,
|
| 59 |
+
'dataset_train': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/train_coarse10.csv',
|
| 60 |
+
'dataset_val': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/val_coarse10.csv',
|
| 61 |
+
'eval_videos_per_class': 100,
|
| 62 |
+
'linear_probe': True,
|
| 63 |
+
'num_workers': 8,
|
| 64 |
+
'pool_type': 'slot_temporal_concat',
|
| 65 |
+
'train_videos_per_class': 500},
|
| 66 |
+
'preset': 'ssv2_coarse10'}],
|
| 67 |
+
'load_checkpoint': True,
|
| 68 |
+
'read_checkpoint': None,
|
| 69 |
+
'save_every_freq': 5,
|
| 70 |
+
'seed': 239,
|
| 71 |
+
'use_sdpa': True,
|
| 72 |
+
'use_wandb': True,
|
| 73 |
+
'wandb_project': 'vjepa_ujepaside'},
|
| 74 |
+
'metrics': {'sigreg': {}, 'std': {}},
|
| 75 |
+
'model': { 'model_name': 'ujepaside_large_patch14_capi_lvd1689m',
|
| 76 |
+
'pred_depth': 6,
|
| 77 |
+
'pred_embed_dim': 384,
|
| 78 |
+
'pred_num_heads': 12,
|
| 79 |
+
'predictor': 'v2_cross',
|
| 80 |
+
'st_causal': False,
|
| 81 |
+
'st_drop_path': 0.2,
|
| 82 |
+
'st_flex_enable': False,
|
| 83 |
+
'st_layer_scale_init': 1e-05,
|
| 84 |
+
'st_num_slots': 16,
|
| 85 |
+
'st_side_block_type': 'factorized',
|
| 86 |
+
'st_slots_causal_within_frame': False,
|
| 87 |
+
'target_kind': 'frozen_2d',
|
| 88 |
+
'target_type': 'vit_large_patch14_capi.lvd1689m',
|
| 89 |
+
'temporal_spacing': 1.0,
|
| 90 |
+
'uniform_power': True,
|
| 91 |
+
'use_activation_checkpointing': True,
|
| 92 |
+
'use_mask_tokens': True,
|
| 93 |
+
'use_rope': True,
|
| 94 |
+
'use_sdpa': True,
|
| 95 |
+
'zero_init_mask_tokens': True},
|
| 96 |
+
'nodes': 4,
|
| 97 |
+
'optimization': { 'clip_grad': 3.0,
|
| 98 |
+
'ema': [0.99925, 0.99925],
|
| 99 |
+
'epochs': 100,
|
| 100 |
+
'final_lr': 0.0001,
|
| 101 |
+
'final_weight_decay': 0.04,
|
| 102 |
+
'ipe': 300,
|
| 103 |
+
'ipe_scale': 1.0,
|
| 104 |
+
'lr': 0.0005,
|
| 105 |
+
'start_lr': 0.0001,
|
| 106 |
+
'warmup': 10,
|
| 107 |
+
'weight_decay': 0.04},
|
| 108 |
+
'tasks_per_node': 4}
|
| 109 |
+
INFO:root:Running pre-training of app: vjepa
|
| 110 |
+
[INFO ][2026-05-14 09:32:42][app.vjepa.train ][main ] which_dtype='bfloat16'
|
| 111 |
+
[INFO ][2026-05-14 09:32:42][app.vjepa.train ][main ] Disabling persistent_workers (incompatible with KNN eval)
|
| 112 |
+
[INFO ][2026-05-14 09:32:42][app.vjepa.train ][main ] NCCL_SOCKET_IFNAME=eno
|
| 113 |
+
[INFO ][2026-05-14 09:32:44][app.vjepa.train ][main ] Initialized (rank/world-size) 13/16, tasks_per_node=4
|
| 114 |
+
[INFO ][2026-05-14 09:32:44][root ][stage_datasets ] [local_rank 1/4] Staging kinetics_240 (targz_parts)
|
| 115 |
+
[INFO ][2026-05-14 09:32:44][root ][_stage_targz_parts ] [rank 1] Extracting 26/103 tar.gz parts to /scratch-node/dcanez.22743106/kinetics_240
|
| 116 |
+
[INFO ][2026-05-14 09:32:57][root ][_stage_targz_parts ] [local_rank 1] Extracted 2/26 parts
|
| 117 |
+
[INFO ][2026-05-14 09:33:11][root ][_stage_targz_parts ] [local_rank 1] Extracted 4/26 parts
|
| 118 |
+
[INFO ][2026-05-14 09:33:25][root ][_stage_targz_parts ] [local_rank 1] Extracted 6/26 parts
|
| 119 |
+
[INFO ][2026-05-14 09:33:39][root ][_stage_targz_parts ] [local_rank 1] Extracted 8/26 parts
|
| 120 |
+
[INFO ][2026-05-14 09:33:53][root ][_stage_targz_parts ] [local_rank 1] Extracted 10/26 parts
|
| 121 |
+
[INFO ][2026-05-14 09:34:08][root ][_stage_targz_parts ] [local_rank 1] Extracted 12/26 parts
|
| 122 |
+
[INFO ][2026-05-14 09:34:22][root ][_stage_targz_parts ] [local_rank 1] Extracted 14/26 parts
|
| 123 |
+
[INFO ][2026-05-14 09:34:36][root ][_stage_targz_parts ] [local_rank 1] Extracted 16/26 parts
|
| 124 |
+
[INFO ][2026-05-14 09:34:51][root ][_stage_targz_parts ] [local_rank 1] Extracted 18/26 parts
|
| 125 |
+
[INFO ][2026-05-14 09:35:06][root ][_stage_targz_parts ] [local_rank 1] Extracted 20/26 parts
|
| 126 |
+
[INFO ][2026-05-14 09:35:20][root ][_stage_targz_parts ] [local_rank 1] Extracted 22/26 parts
|
| 127 |
+
[INFO ][2026-05-14 09:35:34][root ][_stage_targz_parts ] [local_rank 1] Extracted 24/26 parts
|
| 128 |
+
[INFO ][2026-05-14 09:35:49][root ][_stage_targz_parts ] [local_rank 1] Extracted 26/26 parts
|
| 129 |
+
[INFO ][2026-05-14 09:35:49][root ][stage_datasets ] [local_rank 1/4] Staging ssv2 (multipart_tar)
|
| 130 |
+
[INFO ][2026-05-14 09:36:49][app.vjepa.train ][main ] Staged datasets: ['/scratch-node/dcanez.22743106/kinetics_240/train.csv', '/scratch-node/dcanez.22743106/ssv2/train.csv']
|
| 131 |
+
[INFO ][2026-05-14 09:43:14][experiments.stmodels.vision_transformers_v3][vit_large_patch14_capi_lvd1689m] Loading pretrained weights for vit_large_patch14_capi_lvd1689m from https://dl.fbaipublicfiles.com/capi/capi_vitl14_lvd.pth
|
| 132 |
+
[INFO ][2026-05-14 09:43:23][root ][_build_frozen_2d_target ] Loaded frozen_2d target from timm: vit_large_patch14_capi.lvd1689m
|
| 133 |
+
[INFO ][2026-05-14 09:43:23][root ][init_video_model ] ViTMultiSeqWrapper(
|
| 134 |
+
(backbone): UJEPAside(
|
| 135 |
+
(patch_embed): PatchEmbed3D(
|
| 136 |
+
(proj): Conv3d(3, 1024, kernel_size=(1, 14, 14), stride=(1, 14, 14))
|
| 137 |
+
)
|
| 138 |
+
(rope): CAPI2DRoPE()
|
| 139 |
+
(blocks): ModuleList(
|
| 140 |
+
(0-23): 24 x Block(
|
| 141 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 142 |
+
(rope_impl): CAPI2DRoPE()
|
| 143 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 144 |
+
(drop_path1): Identity()
|
| 145 |
+
(drop_path2): Identity()
|
| 146 |
+
(attn): Attention(
|
| 147 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 148 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 149 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 150 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 151 |
+
(rope_impl): CAPI2DRoPE()
|
| 152 |
+
)
|
| 153 |
+
(mlp): MLP(
|
| 154 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 155 |
+
(act): GELU(approximate='none')
|
| 156 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 157 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 158 |
+
)
|
| 159 |
+
)
|
| 160 |
+
)
|
| 161 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=False)
|
| 162 |
+
(st_blocks): ModuleList(
|
| 163 |
+
(0-23): 24 x FactorizedSlotSideBlock(
|
| 164 |
+
(cross_attn): EfficientResidual(
|
| 165 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 166 |
+
(fn): Attention(
|
| 167 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 168 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 169 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 170 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 171 |
+
(rope_impl): CAPI3DRoPE()
|
| 172 |
+
)
|
| 173 |
+
)
|
| 174 |
+
(self_attn): EfficientResidual(
|
| 175 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 176 |
+
(fn): Attention(
|
| 177 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 178 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 179 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 180 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 181 |
+
(rope_impl): CAPI3DRoPE()
|
| 182 |
+
)
|
| 183 |
+
)
|
| 184 |
+
(mlp): EfficientResidual(
|
| 185 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 186 |
+
(fn): MLP(
|
| 187 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 188 |
+
(act): GELU(approximate='none')
|
| 189 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 190 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 191 |
+
)
|
| 192 |
+
)
|
| 193 |
+
)
|
| 194 |
+
)
|
| 195 |
+
(st_rope): CAPI3DRoPE()
|
| 196 |
+
)
|
| 197 |
+
)
|
| 198 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] UJEPAsidePredictorMultiSeqWrapper(
|
| 199 |
+
(backbone): PredictorV2(
|
| 200 |
+
(predictor_embed): Linear(in_features=1024, out_features=384, bias=True)
|
| 201 |
+
(mask_tokens): ParameterList(
|
| 202 |
+
(0): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 203 |
+
(1): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 204 |
+
(2): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 205 |
+
(3): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 206 |
+
)
|
| 207 |
+
(predictor_blocks): ModuleList(
|
| 208 |
+
(0-5): 6 x Block(
|
| 209 |
+
(residual1): EfficientResidual(
|
| 210 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 211 |
+
(fn): Attention(
|
| 212 |
+
(q_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 213 |
+
(k_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 214 |
+
(v_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 215 |
+
(proj): Linear(in_features=384, out_features=384, bias=False)
|
| 216 |
+
(rope): Rope()
|
| 217 |
+
)
|
| 218 |
+
)
|
| 219 |
+
(residual2): EfficientResidual(
|
| 220 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 221 |
+
(fn): MLP(
|
| 222 |
+
(fc1): Linear(in_features=384, out_features=1536, bias=False)
|
| 223 |
+
(act): GELU(approximate='none')
|
| 224 |
+
(fc2): Linear(in_features=1536, out_features=384, bias=False)
|
| 225 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 226 |
+
)
|
| 227 |
+
)
|
| 228 |
+
)
|
| 229 |
+
)
|
| 230 |
+
(predictor_norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 231 |
+
(predictor_proj): Linear(in_features=384, out_features=1024, bias=True)
|
| 232 |
+
)
|
| 233 |
+
)
|
| 234 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] MultiSeqWrapper(
|
| 235 |
+
(backbone): Frozen2DTargetWrapper(
|
| 236 |
+
(backbone): Eva(
|
| 237 |
+
(patch_embed): PatchEmbed(
|
| 238 |
+
(proj): Conv2d(3, 1024, kernel_size=(14, 14), stride=(14, 14))
|
| 239 |
+
(norm): Identity()
|
| 240 |
+
)
|
| 241 |
+
(pos_drop): Dropout(p=0.0, inplace=False)
|
| 242 |
+
(norm_pre): Identity()
|
| 243 |
+
(blocks): ModuleList(
|
| 244 |
+
(0-23): 24 x EvaBlock(
|
| 245 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 246 |
+
(attn): EvaAttention(
|
| 247 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 248 |
+
(q_norm): Identity()
|
| 249 |
+
(k_norm): Identity()
|
| 250 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 251 |
+
(norm): Identity()
|
| 252 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=True)
|
| 253 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 254 |
+
)
|
| 255 |
+
(drop_path1): Identity()
|
| 256 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 257 |
+
(mlp): Mlp(
|
| 258 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=True)
|
| 259 |
+
(act): GELU(approximate='none')
|
| 260 |
+
(drop1): Dropout(p=0.0, inplace=False)
|
| 261 |
+
(norm): Identity()
|
| 262 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=True)
|
| 263 |
+
(drop2): Dropout(p=0.0, inplace=False)
|
| 264 |
+
)
|
| 265 |
+
(drop_path2): Identity()
|
| 266 |
+
)
|
| 267 |
+
)
|
| 268 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 269 |
+
(fc_norm): Identity()
|
| 270 |
+
(head_drop): Dropout(p=0.0, inplace=False)
|
| 271 |
+
(head): Identity()
|
| 272 |
+
(rope): _CapiPatchRoPE()
|
| 273 |
+
)
|
| 274 |
+
)
|
| 275 |
+
)
|
| 276 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] Encoder number of parameters: 403136512
|
| 277 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] Predictor number of parameters: 11416192
|
| 278 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] Target encoder number of parameters: 0
|
| 279 |
+
[INFO ][2026-05-14 09:43:36][root ][make_videodataset ] VideoDataset dataset created
|
| 280 |
+
[INFO ][2026-05-14 09:43:36][WeightedSampler ][__init__ ] Using DistributedWeightedSampler with rank 13 / 16
|
| 281 |
+
[INFO ][2026-05-14 09:43:36][root ][make_videodataset ] VideoDataset unsupervised data loader created
|
| 282 |
+
[INFO ][2026-05-14 09:43:36][app.vjepa.train ][main ] iterations per epoch/dataset length: 300/399
|
| 283 |
+
[INFO ][2026-05-14 09:43:36][app.vjepa.train ][main ] Wrapping models in DDP (rank 13)...
|
| 284 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Initializing loader...
|
| 285 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 286 |
+
submitit WARNING (2026-05-14 13:38:23,823) - Bypassing signal SIGTERM
|
| 287 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGTERM
|
| 288 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGCONT
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_14_log.err
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/timm/models/layers/__init__.py:49: FutureWarning: Importing from timm.models.layers is deprecated, please import via timm.layers
|
| 2 |
+
warnings.warn(f"Importing from {__name__} is deprecated, please import via timm.layers", FutureWarning)
|
| 3 |
+
[rank14]:[W514 09:35:52.647814712 ProcessGroupNCCL.cpp:5138] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 4 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/utils.py:796: FutureWarning: `torch.cuda.amp.GradScaler(args...)` is deprecated. Please use `torch.amp.GradScaler('cuda', args...)` instead.
|
| 5 |
+
scaler = torch.cuda.amp.GradScaler() if mixed_precision else None
|
| 6 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 7 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 8 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 9 |
+
submitit WARNING (2026-05-14 13:38:23,826) - Bypassing signal SIGTERM
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_14_log.out
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
submitit INFO (2026-05-14 09:28:14,530) - Starting with JobEnvironment(job_id=22743106, hostname=gcn83.local.snellius.surf.nl, local_rank=2(4), node=3(4), global_rank=14(16))
|
| 2 |
+
submitit INFO (2026-05-14 09:28:14,530) - Loading pickle: /scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_submitted.pkl
|
| 3 |
+
INFO:root:loaded pretrain params...
|
| 4 |
+
{ 'app': 'vjepa',
|
| 5 |
+
'cpus_per_task': 16,
|
| 6 |
+
'data': { 'batch_size': 64,
|
| 7 |
+
'crop_size': 224,
|
| 8 |
+
'dataset_fpcs': [16, 16],
|
| 9 |
+
'dataset_type': 'VideoDataset',
|
| 10 |
+
'datasets': [ '/scratch-shared/dcanez/data/kinetics/k400/train.csv',
|
| 11 |
+
'/scratch-shared/dcanez/data/ssv2/train.csv'],
|
| 12 |
+
'datasets_weights': [0.65, 0.35],
|
| 13 |
+
'fps': 4,
|
| 14 |
+
'num_workers': 10,
|
| 15 |
+
'patch_size': 14,
|
| 16 |
+
'persistent_workers': True,
|
| 17 |
+
'pin_mem': True,
|
| 18 |
+
'stage': [ { 'dest': 'kinetics_240',
|
| 19 |
+
'format': 'targz_parts',
|
| 20 |
+
'src': '/scratch-shared/dcanez/data/kinetics/k400/tars_240/'},
|
| 21 |
+
{ 'dest': 'ssv2',
|
| 22 |
+
'format': 'multipart_tar',
|
| 23 |
+
'src': '/scratch-nvme/ml-datasets/something-something-v2/'}],
|
| 24 |
+
'tubelet_size': 1},
|
| 25 |
+
'data_aug': { 'auto_augment': False,
|
| 26 |
+
'motion_shift': False,
|
| 27 |
+
'random_resize_aspect_ratio': [0.75, 1.35],
|
| 28 |
+
'random_resize_scale': [0.3, 1.0],
|
| 29 |
+
'reprob': 0.0},
|
| 30 |
+
'folder': '/scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable',
|
| 31 |
+
'loss': {'loss_exp': 1.0},
|
| 32 |
+
'mask': [ { 'aspect_ratio': [0.75, 1.5],
|
| 33 |
+
'full_complement': False,
|
| 34 |
+
'max_keep': None,
|
| 35 |
+
'max_temporal_keep': 1.0,
|
| 36 |
+
'num_blocks': 8,
|
| 37 |
+
'spatial_scale': [0.15, 0.15],
|
| 38 |
+
'temporal_scale': [1.0, 1.0]},
|
| 39 |
+
{ 'aspect_ratio': [0.75, 1.5],
|
| 40 |
+
'full_complement': False,
|
| 41 |
+
'max_keep': None,
|
| 42 |
+
'max_temporal_keep': 1.0,
|
| 43 |
+
'num_blocks': 2,
|
| 44 |
+
'spatial_scale': [0.7, 0.7],
|
| 45 |
+
'temporal_scale': [1.0, 1.0]}],
|
| 46 |
+
'mem_per_gpu': '180G',
|
| 47 |
+
'meta': { 'dtype': 'bfloat16',
|
| 48 |
+
'knn_eval_epoch0': False,
|
| 49 |
+
'knn_eval_freq': 5,
|
| 50 |
+
'knn_eval_presets': [ { 'config': { 'batch_size': 64,
|
| 51 |
+
'dataset_train': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/train.csv',
|
| 52 |
+
'dataset_val': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/val.csv',
|
| 53 |
+
'eval_videos_per_class': 25,
|
| 54 |
+
'num_workers': 8,
|
| 55 |
+
'pool_type': 'slot_temporal_concat',
|
| 56 |
+
'train_videos_per_class': 100},
|
| 57 |
+
'preset': 'ucf101'},
|
| 58 |
+
{ 'config': { 'batch_size': 64,
|
| 59 |
+
'dataset_train': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/train_coarse10.csv',
|
| 60 |
+
'dataset_val': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/val_coarse10.csv',
|
| 61 |
+
'eval_videos_per_class': 100,
|
| 62 |
+
'linear_probe': True,
|
| 63 |
+
'num_workers': 8,
|
| 64 |
+
'pool_type': 'slot_temporal_concat',
|
| 65 |
+
'train_videos_per_class': 500},
|
| 66 |
+
'preset': 'ssv2_coarse10'}],
|
| 67 |
+
'load_checkpoint': True,
|
| 68 |
+
'read_checkpoint': None,
|
| 69 |
+
'save_every_freq': 5,
|
| 70 |
+
'seed': 239,
|
| 71 |
+
'use_sdpa': True,
|
| 72 |
+
'use_wandb': True,
|
| 73 |
+
'wandb_project': 'vjepa_ujepaside'},
|
| 74 |
+
'metrics': {'sigreg': {}, 'std': {}},
|
| 75 |
+
'model': { 'model_name': 'ujepaside_large_patch14_capi_lvd1689m',
|
| 76 |
+
'pred_depth': 6,
|
| 77 |
+
'pred_embed_dim': 384,
|
| 78 |
+
'pred_num_heads': 12,
|
| 79 |
+
'predictor': 'v2_cross',
|
| 80 |
+
'st_causal': False,
|
| 81 |
+
'st_drop_path': 0.2,
|
| 82 |
+
'st_flex_enable': False,
|
| 83 |
+
'st_layer_scale_init': 1e-05,
|
| 84 |
+
'st_num_slots': 16,
|
| 85 |
+
'st_side_block_type': 'factorized',
|
| 86 |
+
'st_slots_causal_within_frame': False,
|
| 87 |
+
'target_kind': 'frozen_2d',
|
| 88 |
+
'target_type': 'vit_large_patch14_capi.lvd1689m',
|
| 89 |
+
'temporal_spacing': 1.0,
|
| 90 |
+
'uniform_power': True,
|
| 91 |
+
'use_activation_checkpointing': True,
|
| 92 |
+
'use_mask_tokens': True,
|
| 93 |
+
'use_rope': True,
|
| 94 |
+
'use_sdpa': True,
|
| 95 |
+
'zero_init_mask_tokens': True},
|
| 96 |
+
'nodes': 4,
|
| 97 |
+
'optimization': { 'clip_grad': 3.0,
|
| 98 |
+
'ema': [0.99925, 0.99925],
|
| 99 |
+
'epochs': 100,
|
| 100 |
+
'final_lr': 0.0001,
|
| 101 |
+
'final_weight_decay': 0.04,
|
| 102 |
+
'ipe': 300,
|
| 103 |
+
'ipe_scale': 1.0,
|
| 104 |
+
'lr': 0.0005,
|
| 105 |
+
'start_lr': 0.0001,
|
| 106 |
+
'warmup': 10,
|
| 107 |
+
'weight_decay': 0.04},
|
| 108 |
+
'tasks_per_node': 4}
|
| 109 |
+
INFO:root:Running pre-training of app: vjepa
|
| 110 |
+
[INFO ][2026-05-14 09:32:42][app.vjepa.train ][main ] which_dtype='bfloat16'
|
| 111 |
+
[INFO ][2026-05-14 09:32:42][app.vjepa.train ][main ] Disabling persistent_workers (incompatible with KNN eval)
|
| 112 |
+
[INFO ][2026-05-14 09:32:42][app.vjepa.train ][main ] NCCL_SOCKET_IFNAME=eno
|
| 113 |
+
[INFO ][2026-05-14 09:32:46][app.vjepa.train ][main ] Initialized (rank/world-size) 14/16, tasks_per_node=4
|
| 114 |
+
[INFO ][2026-05-14 09:32:46][root ][stage_datasets ] [local_rank 2/4] Staging kinetics_240 (targz_parts)
|
| 115 |
+
[INFO ][2026-05-14 09:32:46][root ][_stage_targz_parts ] [rank 2] Extracting 26/103 tar.gz parts to /scratch-node/dcanez.22743106/kinetics_240
|
| 116 |
+
[INFO ][2026-05-14 09:33:00][root ][_stage_targz_parts ] [local_rank 2] Extracted 2/26 parts
|
| 117 |
+
[INFO ][2026-05-14 09:33:14][root ][_stage_targz_parts ] [local_rank 2] Extracted 4/26 parts
|
| 118 |
+
[INFO ][2026-05-14 09:33:28][root ][_stage_targz_parts ] [local_rank 2] Extracted 6/26 parts
|
| 119 |
+
[INFO ][2026-05-14 09:33:43][root ][_stage_targz_parts ] [local_rank 2] Extracted 8/26 parts
|
| 120 |
+
[INFO ][2026-05-14 09:33:57][root ][_stage_targz_parts ] [local_rank 2] Extracted 10/26 parts
|
| 121 |
+
[INFO ][2026-05-14 09:34:11][root ][_stage_targz_parts ] [local_rank 2] Extracted 12/26 parts
|
| 122 |
+
[INFO ][2026-05-14 09:34:25][root ][_stage_targz_parts ] [local_rank 2] Extracted 14/26 parts
|
| 123 |
+
[INFO ][2026-05-14 09:34:40][root ][_stage_targz_parts ] [local_rank 2] Extracted 16/26 parts
|
| 124 |
+
[INFO ][2026-05-14 09:34:54][root ][_stage_targz_parts ] [local_rank 2] Extracted 18/26 parts
|
| 125 |
+
[INFO ][2026-05-14 09:35:08][root ][_stage_targz_parts ] [local_rank 2] Extracted 20/26 parts
|
| 126 |
+
[INFO ][2026-05-14 09:35:23][root ][_stage_targz_parts ] [local_rank 2] Extracted 22/26 parts
|
| 127 |
+
[INFO ][2026-05-14 09:35:37][root ][_stage_targz_parts ] [local_rank 2] Extracted 24/26 parts
|
| 128 |
+
[INFO ][2026-05-14 09:35:52][root ][_stage_targz_parts ] [local_rank 2] Extracted 26/26 parts
|
| 129 |
+
[INFO ][2026-05-14 09:35:52][root ][stage_datasets ] [local_rank 2/4] Staging ssv2 (multipart_tar)
|
| 130 |
+
[INFO ][2026-05-14 09:36:49][app.vjepa.train ][main ] Staged datasets: ['/scratch-node/dcanez.22743106/kinetics_240/train.csv', '/scratch-node/dcanez.22743106/ssv2/train.csv']
|
| 131 |
+
[INFO ][2026-05-14 09:43:14][experiments.stmodels.vision_transformers_v3][vit_large_patch14_capi_lvd1689m] Loading pretrained weights for vit_large_patch14_capi_lvd1689m from https://dl.fbaipublicfiles.com/capi/capi_vitl14_lvd.pth
|
| 132 |
+
[INFO ][2026-05-14 09:43:23][root ][_build_frozen_2d_target ] Loaded frozen_2d target from timm: vit_large_patch14_capi.lvd1689m
|
| 133 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] ViTMultiSeqWrapper(
|
| 134 |
+
(backbone): UJEPAside(
|
| 135 |
+
(patch_embed): PatchEmbed3D(
|
| 136 |
+
(proj): Conv3d(3, 1024, kernel_size=(1, 14, 14), stride=(1, 14, 14))
|
| 137 |
+
)
|
| 138 |
+
(rope): CAPI2DRoPE()
|
| 139 |
+
(blocks): ModuleList(
|
| 140 |
+
(0-23): 24 x Block(
|
| 141 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 142 |
+
(rope_impl): CAPI2DRoPE()
|
| 143 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 144 |
+
(drop_path1): Identity()
|
| 145 |
+
(drop_path2): Identity()
|
| 146 |
+
(attn): Attention(
|
| 147 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 148 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 149 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 150 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 151 |
+
(rope_impl): CAPI2DRoPE()
|
| 152 |
+
)
|
| 153 |
+
(mlp): MLP(
|
| 154 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 155 |
+
(act): GELU(approximate='none')
|
| 156 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 157 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 158 |
+
)
|
| 159 |
+
)
|
| 160 |
+
)
|
| 161 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=False)
|
| 162 |
+
(st_blocks): ModuleList(
|
| 163 |
+
(0-23): 24 x FactorizedSlotSideBlock(
|
| 164 |
+
(cross_attn): EfficientResidual(
|
| 165 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 166 |
+
(fn): Attention(
|
| 167 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 168 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 169 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 170 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 171 |
+
(rope_impl): CAPI3DRoPE()
|
| 172 |
+
)
|
| 173 |
+
)
|
| 174 |
+
(self_attn): EfficientResidual(
|
| 175 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 176 |
+
(fn): Attention(
|
| 177 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 178 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 179 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 180 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 181 |
+
(rope_impl): CAPI3DRoPE()
|
| 182 |
+
)
|
| 183 |
+
)
|
| 184 |
+
(mlp): EfficientResidual(
|
| 185 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 186 |
+
(fn): MLP(
|
| 187 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 188 |
+
(act): GELU(approximate='none')
|
| 189 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 190 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 191 |
+
)
|
| 192 |
+
)
|
| 193 |
+
)
|
| 194 |
+
)
|
| 195 |
+
(st_rope): CAPI3DRoPE()
|
| 196 |
+
)
|
| 197 |
+
)
|
| 198 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] UJEPAsidePredictorMultiSeqWrapper(
|
| 199 |
+
(backbone): PredictorV2(
|
| 200 |
+
(predictor_embed): Linear(in_features=1024, out_features=384, bias=True)
|
| 201 |
+
(mask_tokens): ParameterList(
|
| 202 |
+
(0): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 203 |
+
(1): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 204 |
+
(2): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 205 |
+
(3): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 206 |
+
)
|
| 207 |
+
(predictor_blocks): ModuleList(
|
| 208 |
+
(0-5): 6 x Block(
|
| 209 |
+
(residual1): EfficientResidual(
|
| 210 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 211 |
+
(fn): Attention(
|
| 212 |
+
(q_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 213 |
+
(k_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 214 |
+
(v_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 215 |
+
(proj): Linear(in_features=384, out_features=384, bias=False)
|
| 216 |
+
(rope): Rope()
|
| 217 |
+
)
|
| 218 |
+
)
|
| 219 |
+
(residual2): EfficientResidual(
|
| 220 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 221 |
+
(fn): MLP(
|
| 222 |
+
(fc1): Linear(in_features=384, out_features=1536, bias=False)
|
| 223 |
+
(act): GELU(approximate='none')
|
| 224 |
+
(fc2): Linear(in_features=1536, out_features=384, bias=False)
|
| 225 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 226 |
+
)
|
| 227 |
+
)
|
| 228 |
+
)
|
| 229 |
+
)
|
| 230 |
+
(predictor_norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 231 |
+
(predictor_proj): Linear(in_features=384, out_features=1024, bias=True)
|
| 232 |
+
)
|
| 233 |
+
)
|
| 234 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] MultiSeqWrapper(
|
| 235 |
+
(backbone): Frozen2DTargetWrapper(
|
| 236 |
+
(backbone): Eva(
|
| 237 |
+
(patch_embed): PatchEmbed(
|
| 238 |
+
(proj): Conv2d(3, 1024, kernel_size=(14, 14), stride=(14, 14))
|
| 239 |
+
(norm): Identity()
|
| 240 |
+
)
|
| 241 |
+
(pos_drop): Dropout(p=0.0, inplace=False)
|
| 242 |
+
(norm_pre): Identity()
|
| 243 |
+
(blocks): ModuleList(
|
| 244 |
+
(0-23): 24 x EvaBlock(
|
| 245 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 246 |
+
(attn): EvaAttention(
|
| 247 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 248 |
+
(q_norm): Identity()
|
| 249 |
+
(k_norm): Identity()
|
| 250 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 251 |
+
(norm): Identity()
|
| 252 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=True)
|
| 253 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 254 |
+
)
|
| 255 |
+
(drop_path1): Identity()
|
| 256 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 257 |
+
(mlp): Mlp(
|
| 258 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=True)
|
| 259 |
+
(act): GELU(approximate='none')
|
| 260 |
+
(drop1): Dropout(p=0.0, inplace=False)
|
| 261 |
+
(norm): Identity()
|
| 262 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=True)
|
| 263 |
+
(drop2): Dropout(p=0.0, inplace=False)
|
| 264 |
+
)
|
| 265 |
+
(drop_path2): Identity()
|
| 266 |
+
)
|
| 267 |
+
)
|
| 268 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 269 |
+
(fc_norm): Identity()
|
| 270 |
+
(head_drop): Dropout(p=0.0, inplace=False)
|
| 271 |
+
(head): Identity()
|
| 272 |
+
(rope): _CapiPatchRoPE()
|
| 273 |
+
)
|
| 274 |
+
)
|
| 275 |
+
)
|
| 276 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] Encoder number of parameters: 403136512
|
| 277 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] Predictor number of parameters: 11416192
|
| 278 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] Target encoder number of parameters: 0
|
| 279 |
+
[INFO ][2026-05-14 09:43:36][root ][make_videodataset ] VideoDataset dataset created
|
| 280 |
+
[INFO ][2026-05-14 09:43:36][WeightedSampler ][__init__ ] Using DistributedWeightedSampler with rank 14 / 16
|
| 281 |
+
[INFO ][2026-05-14 09:43:36][root ][make_videodataset ] VideoDataset unsupervised data loader created
|
| 282 |
+
[INFO ][2026-05-14 09:43:36][app.vjepa.train ][main ] iterations per epoch/dataset length: 300/399
|
| 283 |
+
[INFO ][2026-05-14 09:43:36][app.vjepa.train ][main ] Wrapping models in DDP (rank 14)...
|
| 284 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Initializing loader...
|
| 285 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 286 |
+
submitit WARNING (2026-05-14 13:38:23,826) - Bypassing signal SIGTERM
|
| 287 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGTERM
|
| 288 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGCONT
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_15_log.err
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/timm/models/layers/__init__.py:49: FutureWarning: Importing from timm.models.layers is deprecated, please import via timm.layers
|
| 2 |
+
warnings.warn(f"Importing from {__name__} is deprecated, please import via timm.layers", FutureWarning)
|
| 3 |
+
[rank15]:[W514 09:35:46.486928483 ProcessGroupNCCL.cpp:5138] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 4 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/utils.py:796: FutureWarning: `torch.cuda.amp.GradScaler(args...)` is deprecated. Please use `torch.amp.GradScaler('cuda', args...)` instead.
|
| 5 |
+
scaler = torch.cuda.amp.GradScaler() if mixed_precision else None
|
| 6 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 7 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 8 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 9 |
+
submitit WARNING (2026-05-14 13:38:23,831) - Bypassing signal SIGTERM
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_15_log.out
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
submitit INFO (2026-05-14 09:28:14,530) - Starting with JobEnvironment(job_id=22743106, hostname=gcn83.local.snellius.surf.nl, local_rank=3(4), node=3(4), global_rank=15(16))
|
| 2 |
+
submitit INFO (2026-05-14 09:28:14,530) - Loading pickle: /scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_submitted.pkl
|
| 3 |
+
INFO:root:loaded pretrain params...
|
| 4 |
+
{ 'app': 'vjepa',
|
| 5 |
+
'cpus_per_task': 16,
|
| 6 |
+
'data': { 'batch_size': 64,
|
| 7 |
+
'crop_size': 224,
|
| 8 |
+
'dataset_fpcs': [16, 16],
|
| 9 |
+
'dataset_type': 'VideoDataset',
|
| 10 |
+
'datasets': [ '/scratch-shared/dcanez/data/kinetics/k400/train.csv',
|
| 11 |
+
'/scratch-shared/dcanez/data/ssv2/train.csv'],
|
| 12 |
+
'datasets_weights': [0.65, 0.35],
|
| 13 |
+
'fps': 4,
|
| 14 |
+
'num_workers': 10,
|
| 15 |
+
'patch_size': 14,
|
| 16 |
+
'persistent_workers': True,
|
| 17 |
+
'pin_mem': True,
|
| 18 |
+
'stage': [ { 'dest': 'kinetics_240',
|
| 19 |
+
'format': 'targz_parts',
|
| 20 |
+
'src': '/scratch-shared/dcanez/data/kinetics/k400/tars_240/'},
|
| 21 |
+
{ 'dest': 'ssv2',
|
| 22 |
+
'format': 'multipart_tar',
|
| 23 |
+
'src': '/scratch-nvme/ml-datasets/something-something-v2/'}],
|
| 24 |
+
'tubelet_size': 1},
|
| 25 |
+
'data_aug': { 'auto_augment': False,
|
| 26 |
+
'motion_shift': False,
|
| 27 |
+
'random_resize_aspect_ratio': [0.75, 1.35],
|
| 28 |
+
'random_resize_scale': [0.3, 1.0],
|
| 29 |
+
'reprob': 0.0},
|
| 30 |
+
'folder': '/scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable',
|
| 31 |
+
'loss': {'loss_exp': 1.0},
|
| 32 |
+
'mask': [ { 'aspect_ratio': [0.75, 1.5],
|
| 33 |
+
'full_complement': False,
|
| 34 |
+
'max_keep': None,
|
| 35 |
+
'max_temporal_keep': 1.0,
|
| 36 |
+
'num_blocks': 8,
|
| 37 |
+
'spatial_scale': [0.15, 0.15],
|
| 38 |
+
'temporal_scale': [1.0, 1.0]},
|
| 39 |
+
{ 'aspect_ratio': [0.75, 1.5],
|
| 40 |
+
'full_complement': False,
|
| 41 |
+
'max_keep': None,
|
| 42 |
+
'max_temporal_keep': 1.0,
|
| 43 |
+
'num_blocks': 2,
|
| 44 |
+
'spatial_scale': [0.7, 0.7],
|
| 45 |
+
'temporal_scale': [1.0, 1.0]}],
|
| 46 |
+
'mem_per_gpu': '180G',
|
| 47 |
+
'meta': { 'dtype': 'bfloat16',
|
| 48 |
+
'knn_eval_epoch0': False,
|
| 49 |
+
'knn_eval_freq': 5,
|
| 50 |
+
'knn_eval_presets': [ { 'config': { 'batch_size': 64,
|
| 51 |
+
'dataset_train': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/train.csv',
|
| 52 |
+
'dataset_val': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/val.csv',
|
| 53 |
+
'eval_videos_per_class': 25,
|
| 54 |
+
'num_workers': 8,
|
| 55 |
+
'pool_type': 'slot_temporal_concat',
|
| 56 |
+
'train_videos_per_class': 100},
|
| 57 |
+
'preset': 'ucf101'},
|
| 58 |
+
{ 'config': { 'batch_size': 64,
|
| 59 |
+
'dataset_train': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/train_coarse10.csv',
|
| 60 |
+
'dataset_val': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/val_coarse10.csv',
|
| 61 |
+
'eval_videos_per_class': 100,
|
| 62 |
+
'linear_probe': True,
|
| 63 |
+
'num_workers': 8,
|
| 64 |
+
'pool_type': 'slot_temporal_concat',
|
| 65 |
+
'train_videos_per_class': 500},
|
| 66 |
+
'preset': 'ssv2_coarse10'}],
|
| 67 |
+
'load_checkpoint': True,
|
| 68 |
+
'read_checkpoint': None,
|
| 69 |
+
'save_every_freq': 5,
|
| 70 |
+
'seed': 239,
|
| 71 |
+
'use_sdpa': True,
|
| 72 |
+
'use_wandb': True,
|
| 73 |
+
'wandb_project': 'vjepa_ujepaside'},
|
| 74 |
+
'metrics': {'sigreg': {}, 'std': {}},
|
| 75 |
+
'model': { 'model_name': 'ujepaside_large_patch14_capi_lvd1689m',
|
| 76 |
+
'pred_depth': 6,
|
| 77 |
+
'pred_embed_dim': 384,
|
| 78 |
+
'pred_num_heads': 12,
|
| 79 |
+
'predictor': 'v2_cross',
|
| 80 |
+
'st_causal': False,
|
| 81 |
+
'st_drop_path': 0.2,
|
| 82 |
+
'st_flex_enable': False,
|
| 83 |
+
'st_layer_scale_init': 1e-05,
|
| 84 |
+
'st_num_slots': 16,
|
| 85 |
+
'st_side_block_type': 'factorized',
|
| 86 |
+
'st_slots_causal_within_frame': False,
|
| 87 |
+
'target_kind': 'frozen_2d',
|
| 88 |
+
'target_type': 'vit_large_patch14_capi.lvd1689m',
|
| 89 |
+
'temporal_spacing': 1.0,
|
| 90 |
+
'uniform_power': True,
|
| 91 |
+
'use_activation_checkpointing': True,
|
| 92 |
+
'use_mask_tokens': True,
|
| 93 |
+
'use_rope': True,
|
| 94 |
+
'use_sdpa': True,
|
| 95 |
+
'zero_init_mask_tokens': True},
|
| 96 |
+
'nodes': 4,
|
| 97 |
+
'optimization': { 'clip_grad': 3.0,
|
| 98 |
+
'ema': [0.99925, 0.99925],
|
| 99 |
+
'epochs': 100,
|
| 100 |
+
'final_lr': 0.0001,
|
| 101 |
+
'final_weight_decay': 0.04,
|
| 102 |
+
'ipe': 300,
|
| 103 |
+
'ipe_scale': 1.0,
|
| 104 |
+
'lr': 0.0005,
|
| 105 |
+
'start_lr': 0.0001,
|
| 106 |
+
'warmup': 10,
|
| 107 |
+
'weight_decay': 0.04},
|
| 108 |
+
'tasks_per_node': 4}
|
| 109 |
+
INFO:root:Running pre-training of app: vjepa
|
| 110 |
+
[INFO ][2026-05-14 09:32:42][app.vjepa.train ][main ] which_dtype='bfloat16'
|
| 111 |
+
[INFO ][2026-05-14 09:32:42][app.vjepa.train ][main ] Disabling persistent_workers (incompatible with KNN eval)
|
| 112 |
+
[INFO ][2026-05-14 09:32:42][app.vjepa.train ][main ] NCCL_SOCKET_IFNAME=eno
|
| 113 |
+
[INFO ][2026-05-14 09:32:45][app.vjepa.train ][main ] Initialized (rank/world-size) 15/16, tasks_per_node=4
|
| 114 |
+
[INFO ][2026-05-14 09:32:45][root ][stage_datasets ] [local_rank 3/4] Staging kinetics_240 (targz_parts)
|
| 115 |
+
[INFO ][2026-05-14 09:32:45][root ][_stage_targz_parts ] [rank 3] Extracting 25/103 tar.gz parts to /scratch-node/dcanez.22743106/kinetics_240
|
| 116 |
+
[INFO ][2026-05-14 09:32:59][root ][_stage_targz_parts ] [local_rank 3] Extracted 2/25 parts
|
| 117 |
+
[INFO ][2026-05-14 09:33:14][root ][_stage_targz_parts ] [local_rank 3] Extracted 4/25 parts
|
| 118 |
+
[INFO ][2026-05-14 09:33:28][root ][_stage_targz_parts ] [local_rank 3] Extracted 6/25 parts
|
| 119 |
+
[INFO ][2026-05-14 09:33:42][root ][_stage_targz_parts ] [local_rank 3] Extracted 8/25 parts
|
| 120 |
+
[INFO ][2026-05-14 09:33:57][root ][_stage_targz_parts ] [local_rank 3] Extracted 10/25 parts
|
| 121 |
+
[INFO ][2026-05-14 09:34:11][root ][_stage_targz_parts ] [local_rank 3] Extracted 12/25 parts
|
| 122 |
+
[INFO ][2026-05-14 09:34:26][root ][_stage_targz_parts ] [local_rank 3] Extracted 14/25 parts
|
| 123 |
+
[INFO ][2026-05-14 09:34:40][root ][_stage_targz_parts ] [local_rank 3] Extracted 16/25 parts
|
| 124 |
+
[INFO ][2026-05-14 09:34:54][root ][_stage_targz_parts ] [local_rank 3] Extracted 18/25 parts
|
| 125 |
+
[INFO ][2026-05-14 09:35:09][root ][_stage_targz_parts ] [local_rank 3] Extracted 20/25 parts
|
| 126 |
+
[INFO ][2026-05-14 09:35:23][root ][_stage_targz_parts ] [local_rank 3] Extracted 22/25 parts
|
| 127 |
+
[INFO ][2026-05-14 09:35:37][root ][_stage_targz_parts ] [local_rank 3] Extracted 24/25 parts
|
| 128 |
+
[INFO ][2026-05-14 09:35:45][root ][_stage_targz_parts ] [local_rank 3] Extracted 25/25 parts
|
| 129 |
+
[INFO ][2026-05-14 09:35:45][root ][stage_datasets ] [local_rank 3/4] Staging ssv2 (multipart_tar)
|
| 130 |
+
[INFO ][2026-05-14 09:36:49][app.vjepa.train ][main ] Staged datasets: ['/scratch-node/dcanez.22743106/kinetics_240/train.csv', '/scratch-node/dcanez.22743106/ssv2/train.csv']
|
| 131 |
+
[INFO ][2026-05-14 09:43:14][experiments.stmodels.vision_transformers_v3][vit_large_patch14_capi_lvd1689m] Loading pretrained weights for vit_large_patch14_capi_lvd1689m from https://dl.fbaipublicfiles.com/capi/capi_vitl14_lvd.pth
|
| 132 |
+
[INFO ][2026-05-14 09:43:23][root ][_build_frozen_2d_target ] Loaded frozen_2d target from timm: vit_large_patch14_capi.lvd1689m
|
| 133 |
+
[INFO ][2026-05-14 09:43:23][root ][init_video_model ] ViTMultiSeqWrapper(
|
| 134 |
+
(backbone): UJEPAside(
|
| 135 |
+
(patch_embed): PatchEmbed3D(
|
| 136 |
+
(proj): Conv3d(3, 1024, kernel_size=(1, 14, 14), stride=(1, 14, 14))
|
| 137 |
+
)
|
| 138 |
+
(rope): CAPI2DRoPE()
|
| 139 |
+
(blocks): ModuleList(
|
| 140 |
+
(0-23): 24 x Block(
|
| 141 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 142 |
+
(rope_impl): CAPI2DRoPE()
|
| 143 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 144 |
+
(drop_path1): Identity()
|
| 145 |
+
(drop_path2): Identity()
|
| 146 |
+
(attn): Attention(
|
| 147 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 148 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 149 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 150 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 151 |
+
(rope_impl): CAPI2DRoPE()
|
| 152 |
+
)
|
| 153 |
+
(mlp): MLP(
|
| 154 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 155 |
+
(act): GELU(approximate='none')
|
| 156 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 157 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 158 |
+
)
|
| 159 |
+
)
|
| 160 |
+
)
|
| 161 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=False)
|
| 162 |
+
(st_blocks): ModuleList(
|
| 163 |
+
(0-23): 24 x FactorizedSlotSideBlock(
|
| 164 |
+
(cross_attn): EfficientResidual(
|
| 165 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 166 |
+
(fn): Attention(
|
| 167 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 168 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 169 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 170 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 171 |
+
(rope_impl): CAPI3DRoPE()
|
| 172 |
+
)
|
| 173 |
+
)
|
| 174 |
+
(self_attn): EfficientResidual(
|
| 175 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 176 |
+
(fn): Attention(
|
| 177 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 178 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 179 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 180 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 181 |
+
(rope_impl): CAPI3DRoPE()
|
| 182 |
+
)
|
| 183 |
+
)
|
| 184 |
+
(mlp): EfficientResidual(
|
| 185 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 186 |
+
(fn): MLP(
|
| 187 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 188 |
+
(act): GELU(approximate='none')
|
| 189 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 190 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 191 |
+
)
|
| 192 |
+
)
|
| 193 |
+
)
|
| 194 |
+
)
|
| 195 |
+
(st_rope): CAPI3DRoPE()
|
| 196 |
+
)
|
| 197 |
+
)
|
| 198 |
+
[INFO ][2026-05-14 09:43:23][root ][init_video_model ] UJEPAsidePredictorMultiSeqWrapper(
|
| 199 |
+
(backbone): PredictorV2(
|
| 200 |
+
(predictor_embed): Linear(in_features=1024, out_features=384, bias=True)
|
| 201 |
+
(mask_tokens): ParameterList(
|
| 202 |
+
(0): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 203 |
+
(1): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 204 |
+
(2): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 205 |
+
(3): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 206 |
+
)
|
| 207 |
+
(predictor_blocks): ModuleList(
|
| 208 |
+
(0-5): 6 x Block(
|
| 209 |
+
(residual1): EfficientResidual(
|
| 210 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 211 |
+
(fn): Attention(
|
| 212 |
+
(q_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 213 |
+
(k_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 214 |
+
(v_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 215 |
+
(proj): Linear(in_features=384, out_features=384, bias=False)
|
| 216 |
+
(rope): Rope()
|
| 217 |
+
)
|
| 218 |
+
)
|
| 219 |
+
(residual2): EfficientResidual(
|
| 220 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 221 |
+
(fn): MLP(
|
| 222 |
+
(fc1): Linear(in_features=384, out_features=1536, bias=False)
|
| 223 |
+
(act): GELU(approximate='none')
|
| 224 |
+
(fc2): Linear(in_features=1536, out_features=384, bias=False)
|
| 225 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 226 |
+
)
|
| 227 |
+
)
|
| 228 |
+
)
|
| 229 |
+
)
|
| 230 |
+
(predictor_norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 231 |
+
(predictor_proj): Linear(in_features=384, out_features=1024, bias=True)
|
| 232 |
+
)
|
| 233 |
+
)
|
| 234 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] MultiSeqWrapper(
|
| 235 |
+
(backbone): Frozen2DTargetWrapper(
|
| 236 |
+
(backbone): Eva(
|
| 237 |
+
(patch_embed): PatchEmbed(
|
| 238 |
+
(proj): Conv2d(3, 1024, kernel_size=(14, 14), stride=(14, 14))
|
| 239 |
+
(norm): Identity()
|
| 240 |
+
)
|
| 241 |
+
(pos_drop): Dropout(p=0.0, inplace=False)
|
| 242 |
+
(norm_pre): Identity()
|
| 243 |
+
(blocks): ModuleList(
|
| 244 |
+
(0-23): 24 x EvaBlock(
|
| 245 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 246 |
+
(attn): EvaAttention(
|
| 247 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 248 |
+
(q_norm): Identity()
|
| 249 |
+
(k_norm): Identity()
|
| 250 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 251 |
+
(norm): Identity()
|
| 252 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=True)
|
| 253 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 254 |
+
)
|
| 255 |
+
(drop_path1): Identity()
|
| 256 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 257 |
+
(mlp): Mlp(
|
| 258 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=True)
|
| 259 |
+
(act): GELU(approximate='none')
|
| 260 |
+
(drop1): Dropout(p=0.0, inplace=False)
|
| 261 |
+
(norm): Identity()
|
| 262 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=True)
|
| 263 |
+
(drop2): Dropout(p=0.0, inplace=False)
|
| 264 |
+
)
|
| 265 |
+
(drop_path2): Identity()
|
| 266 |
+
)
|
| 267 |
+
)
|
| 268 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 269 |
+
(fc_norm): Identity()
|
| 270 |
+
(head_drop): Dropout(p=0.0, inplace=False)
|
| 271 |
+
(head): Identity()
|
| 272 |
+
(rope): _CapiPatchRoPE()
|
| 273 |
+
)
|
| 274 |
+
)
|
| 275 |
+
)
|
| 276 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] Encoder number of parameters: 403136512
|
| 277 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] Predictor number of parameters: 11416192
|
| 278 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] Target encoder number of parameters: 0
|
| 279 |
+
[INFO ][2026-05-14 09:43:36][root ][make_videodataset ] VideoDataset dataset created
|
| 280 |
+
[INFO ][2026-05-14 09:43:36][WeightedSampler ][__init__ ] Using DistributedWeightedSampler with rank 15 / 16
|
| 281 |
+
[INFO ][2026-05-14 09:43:36][root ][make_videodataset ] VideoDataset unsupervised data loader created
|
| 282 |
+
[INFO ][2026-05-14 09:43:36][app.vjepa.train ][main ] iterations per epoch/dataset length: 300/399
|
| 283 |
+
[INFO ][2026-05-14 09:43:36][app.vjepa.train ][main ] Wrapping models in DDP (rank 15)...
|
| 284 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Initializing loader...
|
| 285 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 286 |
+
submitit WARNING (2026-05-14 13:38:23,831) - Bypassing signal SIGTERM
|
| 287 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGTERM
|
| 288 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGCONT
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_1_log.err
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/timm/models/layers/__init__.py:49: FutureWarning: Importing from timm.models.layers is deprecated, please import via timm.layers
|
| 2 |
+
warnings.warn(f"Importing from {__name__} is deprecated, please import via timm.layers", FutureWarning)
|
| 3 |
+
[rank1]:[W514 09:35:35.129726343 ProcessGroupNCCL.cpp:5138] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 4 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/utils.py:796: FutureWarning: `torch.cuda.amp.GradScaler(args...)` is deprecated. Please use `torch.amp.GradScaler('cuda', args...)` instead.
|
| 5 |
+
scaler = torch.cuda.amp.GradScaler() if mixed_precision else None
|
| 6 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 7 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 8 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 9 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 10 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 11 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 12 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 13 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 14 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 15 |
+
submitit WARNING (2026-05-14 13:38:23,907) - Bypassing signal SIGTERM
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_1_log.out
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
submitit INFO (2026-05-14 09:28:14,530) - Starting with JobEnvironment(job_id=22743106, hostname=gcn75.local.snellius.surf.nl, local_rank=1(4), node=0(4), global_rank=1(16))
|
| 2 |
+
submitit INFO (2026-05-14 09:28:14,530) - Loading pickle: /scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_submitted.pkl
|
| 3 |
+
INFO:root:loaded pretrain params...
|
| 4 |
+
{ 'app': 'vjepa',
|
| 5 |
+
'cpus_per_task': 16,
|
| 6 |
+
'data': { 'batch_size': 64,
|
| 7 |
+
'crop_size': 224,
|
| 8 |
+
'dataset_fpcs': [16, 16],
|
| 9 |
+
'dataset_type': 'VideoDataset',
|
| 10 |
+
'datasets': [ '/scratch-shared/dcanez/data/kinetics/k400/train.csv',
|
| 11 |
+
'/scratch-shared/dcanez/data/ssv2/train.csv'],
|
| 12 |
+
'datasets_weights': [0.65, 0.35],
|
| 13 |
+
'fps': 4,
|
| 14 |
+
'num_workers': 10,
|
| 15 |
+
'patch_size': 14,
|
| 16 |
+
'persistent_workers': True,
|
| 17 |
+
'pin_mem': True,
|
| 18 |
+
'stage': [ { 'dest': 'kinetics_240',
|
| 19 |
+
'format': 'targz_parts',
|
| 20 |
+
'src': '/scratch-shared/dcanez/data/kinetics/k400/tars_240/'},
|
| 21 |
+
{ 'dest': 'ssv2',
|
| 22 |
+
'format': 'multipart_tar',
|
| 23 |
+
'src': '/scratch-nvme/ml-datasets/something-something-v2/'}],
|
| 24 |
+
'tubelet_size': 1},
|
| 25 |
+
'data_aug': { 'auto_augment': False,
|
| 26 |
+
'motion_shift': False,
|
| 27 |
+
'random_resize_aspect_ratio': [0.75, 1.35],
|
| 28 |
+
'random_resize_scale': [0.3, 1.0],
|
| 29 |
+
'reprob': 0.0},
|
| 30 |
+
'folder': '/scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable',
|
| 31 |
+
'loss': {'loss_exp': 1.0},
|
| 32 |
+
'mask': [ { 'aspect_ratio': [0.75, 1.5],
|
| 33 |
+
'full_complement': False,
|
| 34 |
+
'max_keep': None,
|
| 35 |
+
'max_temporal_keep': 1.0,
|
| 36 |
+
'num_blocks': 8,
|
| 37 |
+
'spatial_scale': [0.15, 0.15],
|
| 38 |
+
'temporal_scale': [1.0, 1.0]},
|
| 39 |
+
{ 'aspect_ratio': [0.75, 1.5],
|
| 40 |
+
'full_complement': False,
|
| 41 |
+
'max_keep': None,
|
| 42 |
+
'max_temporal_keep': 1.0,
|
| 43 |
+
'num_blocks': 2,
|
| 44 |
+
'spatial_scale': [0.7, 0.7],
|
| 45 |
+
'temporal_scale': [1.0, 1.0]}],
|
| 46 |
+
'mem_per_gpu': '180G',
|
| 47 |
+
'meta': { 'dtype': 'bfloat16',
|
| 48 |
+
'knn_eval_epoch0': False,
|
| 49 |
+
'knn_eval_freq': 5,
|
| 50 |
+
'knn_eval_presets': [ { 'config': { 'batch_size': 64,
|
| 51 |
+
'dataset_train': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/train.csv',
|
| 52 |
+
'dataset_val': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/val.csv',
|
| 53 |
+
'eval_videos_per_class': 25,
|
| 54 |
+
'num_workers': 8,
|
| 55 |
+
'pool_type': 'slot_temporal_concat',
|
| 56 |
+
'train_videos_per_class': 100},
|
| 57 |
+
'preset': 'ucf101'},
|
| 58 |
+
{ 'config': { 'batch_size': 64,
|
| 59 |
+
'dataset_train': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/train_coarse10.csv',
|
| 60 |
+
'dataset_val': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/val_coarse10.csv',
|
| 61 |
+
'eval_videos_per_class': 100,
|
| 62 |
+
'linear_probe': True,
|
| 63 |
+
'num_workers': 8,
|
| 64 |
+
'pool_type': 'slot_temporal_concat',
|
| 65 |
+
'train_videos_per_class': 500},
|
| 66 |
+
'preset': 'ssv2_coarse10'}],
|
| 67 |
+
'load_checkpoint': True,
|
| 68 |
+
'read_checkpoint': None,
|
| 69 |
+
'save_every_freq': 5,
|
| 70 |
+
'seed': 239,
|
| 71 |
+
'use_sdpa': True,
|
| 72 |
+
'use_wandb': True,
|
| 73 |
+
'wandb_project': 'vjepa_ujepaside'},
|
| 74 |
+
'metrics': {'sigreg': {}, 'std': {}},
|
| 75 |
+
'model': { 'model_name': 'ujepaside_large_patch14_capi_lvd1689m',
|
| 76 |
+
'pred_depth': 6,
|
| 77 |
+
'pred_embed_dim': 384,
|
| 78 |
+
'pred_num_heads': 12,
|
| 79 |
+
'predictor': 'v2_cross',
|
| 80 |
+
'st_causal': False,
|
| 81 |
+
'st_drop_path': 0.2,
|
| 82 |
+
'st_flex_enable': False,
|
| 83 |
+
'st_layer_scale_init': 1e-05,
|
| 84 |
+
'st_num_slots': 16,
|
| 85 |
+
'st_side_block_type': 'factorized',
|
| 86 |
+
'st_slots_causal_within_frame': False,
|
| 87 |
+
'target_kind': 'frozen_2d',
|
| 88 |
+
'target_type': 'vit_large_patch14_capi.lvd1689m',
|
| 89 |
+
'temporal_spacing': 1.0,
|
| 90 |
+
'uniform_power': True,
|
| 91 |
+
'use_activation_checkpointing': True,
|
| 92 |
+
'use_mask_tokens': True,
|
| 93 |
+
'use_rope': True,
|
| 94 |
+
'use_sdpa': True,
|
| 95 |
+
'zero_init_mask_tokens': True},
|
| 96 |
+
'nodes': 4,
|
| 97 |
+
'optimization': { 'clip_grad': 3.0,
|
| 98 |
+
'ema': [0.99925, 0.99925],
|
| 99 |
+
'epochs': 100,
|
| 100 |
+
'final_lr': 0.0001,
|
| 101 |
+
'final_weight_decay': 0.04,
|
| 102 |
+
'ipe': 300,
|
| 103 |
+
'ipe_scale': 1.0,
|
| 104 |
+
'lr': 0.0005,
|
| 105 |
+
'start_lr': 0.0001,
|
| 106 |
+
'warmup': 10,
|
| 107 |
+
'weight_decay': 0.04},
|
| 108 |
+
'tasks_per_node': 4}
|
| 109 |
+
INFO:root:Running pre-training of app: vjepa
|
| 110 |
+
[INFO ][2026-05-14 09:32:27][app.vjepa.train ][main ] which_dtype='bfloat16'
|
| 111 |
+
[INFO ][2026-05-14 09:32:27][app.vjepa.train ][main ] Disabling persistent_workers (incompatible with KNN eval)
|
| 112 |
+
[INFO ][2026-05-14 09:32:27][app.vjepa.train ][main ] NCCL_SOCKET_IFNAME=eno
|
| 113 |
+
[INFO ][2026-05-14 09:32:30][app.vjepa.train ][main ] Initialized (rank/world-size) 1/16, tasks_per_node=4
|
| 114 |
+
[INFO ][2026-05-14 09:32:30][root ][stage_datasets ] [local_rank 1/4] Staging kinetics_240 (targz_parts)
|
| 115 |
+
[INFO ][2026-05-14 09:32:30][root ][_stage_targz_parts ] [rank 1] Extracting 26/103 tar.gz parts to /scratch-node/dcanez.22743106/kinetics_240
|
| 116 |
+
[INFO ][2026-05-14 09:32:43][root ][_stage_targz_parts ] [local_rank 1] Extracted 2/26 parts
|
| 117 |
+
[INFO ][2026-05-14 09:32:56][root ][_stage_targz_parts ] [local_rank 1] Extracted 4/26 parts
|
| 118 |
+
[INFO ][2026-05-14 09:33:11][root ][_stage_targz_parts ] [local_rank 1] Extracted 6/26 parts
|
| 119 |
+
[INFO ][2026-05-14 09:33:25][root ][_stage_targz_parts ] [local_rank 1] Extracted 8/26 parts
|
| 120 |
+
[INFO ][2026-05-14 09:33:39][root ][_stage_targz_parts ] [local_rank 1] Extracted 10/26 parts
|
| 121 |
+
[INFO ][2026-05-14 09:33:54][root ][_stage_targz_parts ] [local_rank 1] Extracted 12/26 parts
|
| 122 |
+
[INFO ][2026-05-14 09:34:07][root ][_stage_targz_parts ] [local_rank 1] Extracted 14/26 parts
|
| 123 |
+
[INFO ][2026-05-14 09:34:22][root ][_stage_targz_parts ] [local_rank 1] Extracted 16/26 parts
|
| 124 |
+
[INFO ][2026-05-14 09:34:36][root ][_stage_targz_parts ] [local_rank 1] Extracted 18/26 parts
|
| 125 |
+
[INFO ][2026-05-14 09:34:51][root ][_stage_targz_parts ] [local_rank 1] Extracted 20/26 parts
|
| 126 |
+
[INFO ][2026-05-14 09:35:06][root ][_stage_targz_parts ] [local_rank 1] Extracted 22/26 parts
|
| 127 |
+
[INFO ][2026-05-14 09:35:20][root ][_stage_targz_parts ] [local_rank 1] Extracted 24/26 parts
|
| 128 |
+
[INFO ][2026-05-14 09:35:35][root ][_stage_targz_parts ] [local_rank 1] Extracted 26/26 parts
|
| 129 |
+
[INFO ][2026-05-14 09:35:35][root ][stage_datasets ] [local_rank 1/4] Staging ssv2 (multipart_tar)
|
| 130 |
+
[INFO ][2026-05-14 09:36:49][app.vjepa.train ][main ] Staged datasets: ['/scratch-node/dcanez.22743106/kinetics_240/train.csv', '/scratch-node/dcanez.22743106/ssv2/train.csv']
|
| 131 |
+
[INFO ][2026-05-14 09:43:14][experiments.stmodels.vision_transformers_v3][vit_large_patch14_capi_lvd1689m] Loading pretrained weights for vit_large_patch14_capi_lvd1689m from https://dl.fbaipublicfiles.com/capi/capi_vitl14_lvd.pth
|
| 132 |
+
[INFO ][2026-05-14 09:43:23][root ][_build_frozen_2d_target ] Loaded frozen_2d target from timm: vit_large_patch14_capi.lvd1689m
|
| 133 |
+
[INFO ][2026-05-14 09:43:23][root ][init_video_model ] ViTMultiSeqWrapper(
|
| 134 |
+
(backbone): UJEPAside(
|
| 135 |
+
(patch_embed): PatchEmbed3D(
|
| 136 |
+
(proj): Conv3d(3, 1024, kernel_size=(1, 14, 14), stride=(1, 14, 14))
|
| 137 |
+
)
|
| 138 |
+
(rope): CAPI2DRoPE()
|
| 139 |
+
(blocks): ModuleList(
|
| 140 |
+
(0-23): 24 x Block(
|
| 141 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 142 |
+
(rope_impl): CAPI2DRoPE()
|
| 143 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 144 |
+
(drop_path1): Identity()
|
| 145 |
+
(drop_path2): Identity()
|
| 146 |
+
(attn): Attention(
|
| 147 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 148 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 149 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 150 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 151 |
+
(rope_impl): CAPI2DRoPE()
|
| 152 |
+
)
|
| 153 |
+
(mlp): MLP(
|
| 154 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 155 |
+
(act): GELU(approximate='none')
|
| 156 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 157 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 158 |
+
)
|
| 159 |
+
)
|
| 160 |
+
)
|
| 161 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=False)
|
| 162 |
+
(st_blocks): ModuleList(
|
| 163 |
+
(0-23): 24 x FactorizedSlotSideBlock(
|
| 164 |
+
(cross_attn): EfficientResidual(
|
| 165 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 166 |
+
(fn): Attention(
|
| 167 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 168 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 169 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 170 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 171 |
+
(rope_impl): CAPI3DRoPE()
|
| 172 |
+
)
|
| 173 |
+
)
|
| 174 |
+
(self_attn): EfficientResidual(
|
| 175 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 176 |
+
(fn): Attention(
|
| 177 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 178 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 179 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 180 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 181 |
+
(rope_impl): CAPI3DRoPE()
|
| 182 |
+
)
|
| 183 |
+
)
|
| 184 |
+
(mlp): EfficientResidual(
|
| 185 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 186 |
+
(fn): MLP(
|
| 187 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 188 |
+
(act): GELU(approximate='none')
|
| 189 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 190 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 191 |
+
)
|
| 192 |
+
)
|
| 193 |
+
)
|
| 194 |
+
)
|
| 195 |
+
(st_rope): CAPI3DRoPE()
|
| 196 |
+
)
|
| 197 |
+
)
|
| 198 |
+
[INFO ][2026-05-14 09:43:23][root ][init_video_model ] UJEPAsidePredictorMultiSeqWrapper(
|
| 199 |
+
(backbone): PredictorV2(
|
| 200 |
+
(predictor_embed): Linear(in_features=1024, out_features=384, bias=True)
|
| 201 |
+
(mask_tokens): ParameterList(
|
| 202 |
+
(0): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 203 |
+
(1): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 204 |
+
(2): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 205 |
+
(3): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 206 |
+
)
|
| 207 |
+
(predictor_blocks): ModuleList(
|
| 208 |
+
(0-5): 6 x Block(
|
| 209 |
+
(residual1): EfficientResidual(
|
| 210 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 211 |
+
(fn): Attention(
|
| 212 |
+
(q_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 213 |
+
(k_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 214 |
+
(v_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 215 |
+
(proj): Linear(in_features=384, out_features=384, bias=False)
|
| 216 |
+
(rope): Rope()
|
| 217 |
+
)
|
| 218 |
+
)
|
| 219 |
+
(residual2): EfficientResidual(
|
| 220 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 221 |
+
(fn): MLP(
|
| 222 |
+
(fc1): Linear(in_features=384, out_features=1536, bias=False)
|
| 223 |
+
(act): GELU(approximate='none')
|
| 224 |
+
(fc2): Linear(in_features=1536, out_features=384, bias=False)
|
| 225 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 226 |
+
)
|
| 227 |
+
)
|
| 228 |
+
)
|
| 229 |
+
)
|
| 230 |
+
(predictor_norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 231 |
+
(predictor_proj): Linear(in_features=384, out_features=1024, bias=True)
|
| 232 |
+
)
|
| 233 |
+
)
|
| 234 |
+
[INFO ][2026-05-14 09:43:23][root ][init_video_model ] MultiSeqWrapper(
|
| 235 |
+
(backbone): Frozen2DTargetWrapper(
|
| 236 |
+
(backbone): Eva(
|
| 237 |
+
(patch_embed): PatchEmbed(
|
| 238 |
+
(proj): Conv2d(3, 1024, kernel_size=(14, 14), stride=(14, 14))
|
| 239 |
+
(norm): Identity()
|
| 240 |
+
)
|
| 241 |
+
(pos_drop): Dropout(p=0.0, inplace=False)
|
| 242 |
+
(norm_pre): Identity()
|
| 243 |
+
(blocks): ModuleList(
|
| 244 |
+
(0-23): 24 x EvaBlock(
|
| 245 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 246 |
+
(attn): EvaAttention(
|
| 247 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 248 |
+
(q_norm): Identity()
|
| 249 |
+
(k_norm): Identity()
|
| 250 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 251 |
+
(norm): Identity()
|
| 252 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=True)
|
| 253 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 254 |
+
)
|
| 255 |
+
(drop_path1): Identity()
|
| 256 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 257 |
+
(mlp): Mlp(
|
| 258 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=True)
|
| 259 |
+
(act): GELU(approximate='none')
|
| 260 |
+
(drop1): Dropout(p=0.0, inplace=False)
|
| 261 |
+
(norm): Identity()
|
| 262 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=True)
|
| 263 |
+
(drop2): Dropout(p=0.0, inplace=False)
|
| 264 |
+
)
|
| 265 |
+
(drop_path2): Identity()
|
| 266 |
+
)
|
| 267 |
+
)
|
| 268 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 269 |
+
(fc_norm): Identity()
|
| 270 |
+
(head_drop): Dropout(p=0.0, inplace=False)
|
| 271 |
+
(head): Identity()
|
| 272 |
+
(rope): _CapiPatchRoPE()
|
| 273 |
+
)
|
| 274 |
+
)
|
| 275 |
+
)
|
| 276 |
+
[INFO ][2026-05-14 09:43:23][root ][init_video_model ] Encoder number of parameters: 403136512
|
| 277 |
+
[INFO ][2026-05-14 09:43:23][root ][init_video_model ] Predictor number of parameters: 11416192
|
| 278 |
+
[INFO ][2026-05-14 09:43:23][root ][init_video_model ] Target encoder number of parameters: 0
|
| 279 |
+
[INFO ][2026-05-14 09:43:39][root ][make_videodataset ] VideoDataset dataset created
|
| 280 |
+
[INFO ][2026-05-14 09:43:39][WeightedSampler ][__init__ ] Using DistributedWeightedSampler with rank 1 / 16
|
| 281 |
+
[INFO ][2026-05-14 09:43:39][root ][make_videodataset ] VideoDataset unsupervised data loader created
|
| 282 |
+
[INFO ][2026-05-14 09:43:39][app.vjepa.train ][main ] iterations per epoch/dataset length: 300/399
|
| 283 |
+
[INFO ][2026-05-14 09:43:40][app.vjepa.train ][main ] Wrapping models in DDP (rank 1)...
|
| 284 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Initializing loader...
|
| 285 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 286 |
+
submitit WARNING (2026-05-14 13:38:23,907) - Bypassing signal SIGTERM
|
| 287 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGTERM
|
| 288 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGCONT
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_2_log.err
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/timm/models/layers/__init__.py:49: FutureWarning: Importing from timm.models.layers is deprecated, please import via timm.layers
|
| 2 |
+
warnings.warn(f"Importing from {__name__} is deprecated, please import via timm.layers", FutureWarning)
|
| 3 |
+
[rank2]:[W514 09:35:36.297380774 ProcessGroupNCCL.cpp:5138] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 4 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/utils.py:796: FutureWarning: `torch.cuda.amp.GradScaler(args...)` is deprecated. Please use `torch.amp.GradScaler('cuda', args...)` instead.
|
| 5 |
+
scaler = torch.cuda.amp.GradScaler() if mixed_precision else None
|
| 6 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 7 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 8 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 9 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 10 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 11 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 12 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 13 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 14 |
+
submitit WARNING (2026-05-14 13:38:23,807) - Bypassing signal SIGCONT
|
| 15 |
+
submitit WARNING (2026-05-14 13:38:23,937) - Bypassing signal SIGTERM
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_2_log.out
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
submitit INFO (2026-05-14 09:28:14,530) - Starting with JobEnvironment(job_id=22743106, hostname=gcn75.local.snellius.surf.nl, local_rank=2(4), node=0(4), global_rank=2(16))
|
| 2 |
+
submitit INFO (2026-05-14 09:28:14,530) - Loading pickle: /scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_submitted.pkl
|
| 3 |
+
INFO:root:loaded pretrain params...
|
| 4 |
+
{ 'app': 'vjepa',
|
| 5 |
+
'cpus_per_task': 16,
|
| 6 |
+
'data': { 'batch_size': 64,
|
| 7 |
+
'crop_size': 224,
|
| 8 |
+
'dataset_fpcs': [16, 16],
|
| 9 |
+
'dataset_type': 'VideoDataset',
|
| 10 |
+
'datasets': [ '/scratch-shared/dcanez/data/kinetics/k400/train.csv',
|
| 11 |
+
'/scratch-shared/dcanez/data/ssv2/train.csv'],
|
| 12 |
+
'datasets_weights': [0.65, 0.35],
|
| 13 |
+
'fps': 4,
|
| 14 |
+
'num_workers': 10,
|
| 15 |
+
'patch_size': 14,
|
| 16 |
+
'persistent_workers': True,
|
| 17 |
+
'pin_mem': True,
|
| 18 |
+
'stage': [ { 'dest': 'kinetics_240',
|
| 19 |
+
'format': 'targz_parts',
|
| 20 |
+
'src': '/scratch-shared/dcanez/data/kinetics/k400/tars_240/'},
|
| 21 |
+
{ 'dest': 'ssv2',
|
| 22 |
+
'format': 'multipart_tar',
|
| 23 |
+
'src': '/scratch-nvme/ml-datasets/something-something-v2/'}],
|
| 24 |
+
'tubelet_size': 1},
|
| 25 |
+
'data_aug': { 'auto_augment': False,
|
| 26 |
+
'motion_shift': False,
|
| 27 |
+
'random_resize_aspect_ratio': [0.75, 1.35],
|
| 28 |
+
'random_resize_scale': [0.3, 1.0],
|
| 29 |
+
'reprob': 0.0},
|
| 30 |
+
'folder': '/scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable',
|
| 31 |
+
'loss': {'loss_exp': 1.0},
|
| 32 |
+
'mask': [ { 'aspect_ratio': [0.75, 1.5],
|
| 33 |
+
'full_complement': False,
|
| 34 |
+
'max_keep': None,
|
| 35 |
+
'max_temporal_keep': 1.0,
|
| 36 |
+
'num_blocks': 8,
|
| 37 |
+
'spatial_scale': [0.15, 0.15],
|
| 38 |
+
'temporal_scale': [1.0, 1.0]},
|
| 39 |
+
{ 'aspect_ratio': [0.75, 1.5],
|
| 40 |
+
'full_complement': False,
|
| 41 |
+
'max_keep': None,
|
| 42 |
+
'max_temporal_keep': 1.0,
|
| 43 |
+
'num_blocks': 2,
|
| 44 |
+
'spatial_scale': [0.7, 0.7],
|
| 45 |
+
'temporal_scale': [1.0, 1.0]}],
|
| 46 |
+
'mem_per_gpu': '180G',
|
| 47 |
+
'meta': { 'dtype': 'bfloat16',
|
| 48 |
+
'knn_eval_epoch0': False,
|
| 49 |
+
'knn_eval_freq': 5,
|
| 50 |
+
'knn_eval_presets': [ { 'config': { 'batch_size': 64,
|
| 51 |
+
'dataset_train': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/train.csv',
|
| 52 |
+
'dataset_val': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/val.csv',
|
| 53 |
+
'eval_videos_per_class': 25,
|
| 54 |
+
'num_workers': 8,
|
| 55 |
+
'pool_type': 'slot_temporal_concat',
|
| 56 |
+
'train_videos_per_class': 100},
|
| 57 |
+
'preset': 'ucf101'},
|
| 58 |
+
{ 'config': { 'batch_size': 64,
|
| 59 |
+
'dataset_train': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/train_coarse10.csv',
|
| 60 |
+
'dataset_val': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/val_coarse10.csv',
|
| 61 |
+
'eval_videos_per_class': 100,
|
| 62 |
+
'linear_probe': True,
|
| 63 |
+
'num_workers': 8,
|
| 64 |
+
'pool_type': 'slot_temporal_concat',
|
| 65 |
+
'train_videos_per_class': 500},
|
| 66 |
+
'preset': 'ssv2_coarse10'}],
|
| 67 |
+
'load_checkpoint': True,
|
| 68 |
+
'read_checkpoint': None,
|
| 69 |
+
'save_every_freq': 5,
|
| 70 |
+
'seed': 239,
|
| 71 |
+
'use_sdpa': True,
|
| 72 |
+
'use_wandb': True,
|
| 73 |
+
'wandb_project': 'vjepa_ujepaside'},
|
| 74 |
+
'metrics': {'sigreg': {}, 'std': {}},
|
| 75 |
+
'model': { 'model_name': 'ujepaside_large_patch14_capi_lvd1689m',
|
| 76 |
+
'pred_depth': 6,
|
| 77 |
+
'pred_embed_dim': 384,
|
| 78 |
+
'pred_num_heads': 12,
|
| 79 |
+
'predictor': 'v2_cross',
|
| 80 |
+
'st_causal': False,
|
| 81 |
+
'st_drop_path': 0.2,
|
| 82 |
+
'st_flex_enable': False,
|
| 83 |
+
'st_layer_scale_init': 1e-05,
|
| 84 |
+
'st_num_slots': 16,
|
| 85 |
+
'st_side_block_type': 'factorized',
|
| 86 |
+
'st_slots_causal_within_frame': False,
|
| 87 |
+
'target_kind': 'frozen_2d',
|
| 88 |
+
'target_type': 'vit_large_patch14_capi.lvd1689m',
|
| 89 |
+
'temporal_spacing': 1.0,
|
| 90 |
+
'uniform_power': True,
|
| 91 |
+
'use_activation_checkpointing': True,
|
| 92 |
+
'use_mask_tokens': True,
|
| 93 |
+
'use_rope': True,
|
| 94 |
+
'use_sdpa': True,
|
| 95 |
+
'zero_init_mask_tokens': True},
|
| 96 |
+
'nodes': 4,
|
| 97 |
+
'optimization': { 'clip_grad': 3.0,
|
| 98 |
+
'ema': [0.99925, 0.99925],
|
| 99 |
+
'epochs': 100,
|
| 100 |
+
'final_lr': 0.0001,
|
| 101 |
+
'final_weight_decay': 0.04,
|
| 102 |
+
'ipe': 300,
|
| 103 |
+
'ipe_scale': 1.0,
|
| 104 |
+
'lr': 0.0005,
|
| 105 |
+
'start_lr': 0.0001,
|
| 106 |
+
'warmup': 10,
|
| 107 |
+
'weight_decay': 0.04},
|
| 108 |
+
'tasks_per_node': 4}
|
| 109 |
+
INFO:root:Running pre-training of app: vjepa
|
| 110 |
+
[INFO ][2026-05-14 09:32:27][app.vjepa.train ][main ] which_dtype='bfloat16'
|
| 111 |
+
[INFO ][2026-05-14 09:32:27][app.vjepa.train ][main ] Disabling persistent_workers (incompatible with KNN eval)
|
| 112 |
+
[INFO ][2026-05-14 09:32:27][app.vjepa.train ][main ] NCCL_SOCKET_IFNAME=eno
|
| 113 |
+
[INFO ][2026-05-14 09:32:28][app.vjepa.train ][main ] Initialized (rank/world-size) 2/16, tasks_per_node=4
|
| 114 |
+
[INFO ][2026-05-14 09:32:28][root ][stage_datasets ] [local_rank 2/4] Staging kinetics_240 (targz_parts)
|
| 115 |
+
[INFO ][2026-05-14 09:32:28][root ][_stage_targz_parts ] [rank 2] Extracting 26/103 tar.gz parts to /scratch-node/dcanez.22743106/kinetics_240
|
| 116 |
+
[INFO ][2026-05-14 09:32:43][root ][_stage_targz_parts ] [local_rank 2] Extracted 2/26 parts
|
| 117 |
+
[INFO ][2026-05-14 09:32:57][root ][_stage_targz_parts ] [local_rank 2] Extracted 4/26 parts
|
| 118 |
+
[INFO ][2026-05-14 09:33:12][root ][_stage_targz_parts ] [local_rank 2] Extracted 6/26 parts
|
| 119 |
+
[INFO ][2026-05-14 09:33:26][root ][_stage_targz_parts ] [local_rank 2] Extracted 8/26 parts
|
| 120 |
+
[INFO ][2026-05-14 09:33:40][root ][_stage_targz_parts ] [local_rank 2] Extracted 10/26 parts
|
| 121 |
+
[INFO ][2026-05-14 09:33:54][root ][_stage_targz_parts ] [local_rank 2] Extracted 12/26 parts
|
| 122 |
+
[INFO ][2026-05-14 09:34:08][root ][_stage_targz_parts ] [local_rank 2] Extracted 14/26 parts
|
| 123 |
+
[INFO ][2026-05-14 09:34:23][root ][_stage_targz_parts ] [local_rank 2] Extracted 16/26 parts
|
| 124 |
+
[INFO ][2026-05-14 09:34:38][root ][_stage_targz_parts ] [local_rank 2] Extracted 18/26 parts
|
| 125 |
+
[INFO ][2026-05-14 09:34:53][root ][_stage_targz_parts ] [local_rank 2] Extracted 20/26 parts
|
| 126 |
+
[INFO ][2026-05-14 09:35:07][root ][_stage_targz_parts ] [local_rank 2] Extracted 22/26 parts
|
| 127 |
+
[INFO ][2026-05-14 09:35:21][root ][_stage_targz_parts ] [local_rank 2] Extracted 24/26 parts
|
| 128 |
+
[INFO ][2026-05-14 09:35:36][root ][_stage_targz_parts ] [local_rank 2] Extracted 26/26 parts
|
| 129 |
+
[INFO ][2026-05-14 09:35:36][root ][stage_datasets ] [local_rank 2/4] Staging ssv2 (multipart_tar)
|
| 130 |
+
[INFO ][2026-05-14 09:36:49][app.vjepa.train ][main ] Staged datasets: ['/scratch-node/dcanez.22743106/kinetics_240/train.csv', '/scratch-node/dcanez.22743106/ssv2/train.csv']
|
| 131 |
+
[INFO ][2026-05-14 09:43:14][experiments.stmodels.vision_transformers_v3][vit_large_patch14_capi_lvd1689m] Loading pretrained weights for vit_large_patch14_capi_lvd1689m from https://dl.fbaipublicfiles.com/capi/capi_vitl14_lvd.pth
|
| 132 |
+
[INFO ][2026-05-14 09:43:23][root ][_build_frozen_2d_target ] Loaded frozen_2d target from timm: vit_large_patch14_capi.lvd1689m
|
| 133 |
+
[INFO ][2026-05-14 09:43:23][root ][init_video_model ] ViTMultiSeqWrapper(
|
| 134 |
+
(backbone): UJEPAside(
|
| 135 |
+
(patch_embed): PatchEmbed3D(
|
| 136 |
+
(proj): Conv3d(3, 1024, kernel_size=(1, 14, 14), stride=(1, 14, 14))
|
| 137 |
+
)
|
| 138 |
+
(rope): CAPI2DRoPE()
|
| 139 |
+
(blocks): ModuleList(
|
| 140 |
+
(0-23): 24 x Block(
|
| 141 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 142 |
+
(rope_impl): CAPI2DRoPE()
|
| 143 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 144 |
+
(drop_path1): Identity()
|
| 145 |
+
(drop_path2): Identity()
|
| 146 |
+
(attn): Attention(
|
| 147 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 148 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 149 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 150 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 151 |
+
(rope_impl): CAPI2DRoPE()
|
| 152 |
+
)
|
| 153 |
+
(mlp): MLP(
|
| 154 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 155 |
+
(act): GELU(approximate='none')
|
| 156 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 157 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 158 |
+
)
|
| 159 |
+
)
|
| 160 |
+
)
|
| 161 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=False)
|
| 162 |
+
(st_blocks): ModuleList(
|
| 163 |
+
(0-23): 24 x FactorizedSlotSideBlock(
|
| 164 |
+
(cross_attn): EfficientResidual(
|
| 165 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 166 |
+
(fn): Attention(
|
| 167 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 168 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 169 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 170 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 171 |
+
(rope_impl): CAPI3DRoPE()
|
| 172 |
+
)
|
| 173 |
+
)
|
| 174 |
+
(self_attn): EfficientResidual(
|
| 175 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 176 |
+
(fn): Attention(
|
| 177 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 178 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 179 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 180 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 181 |
+
(rope_impl): CAPI3DRoPE()
|
| 182 |
+
)
|
| 183 |
+
)
|
| 184 |
+
(mlp): EfficientResidual(
|
| 185 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 186 |
+
(fn): MLP(
|
| 187 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 188 |
+
(act): GELU(approximate='none')
|
| 189 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 190 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 191 |
+
)
|
| 192 |
+
)
|
| 193 |
+
)
|
| 194 |
+
)
|
| 195 |
+
(st_rope): CAPI3DRoPE()
|
| 196 |
+
)
|
| 197 |
+
)
|
| 198 |
+
[INFO ][2026-05-14 09:43:23][root ][init_video_model ] UJEPAsidePredictorMultiSeqWrapper(
|
| 199 |
+
(backbone): PredictorV2(
|
| 200 |
+
(predictor_embed): Linear(in_features=1024, out_features=384, bias=True)
|
| 201 |
+
(mask_tokens): ParameterList(
|
| 202 |
+
(0): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 203 |
+
(1): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 204 |
+
(2): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 205 |
+
(3): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 206 |
+
)
|
| 207 |
+
(predictor_blocks): ModuleList(
|
| 208 |
+
(0-5): 6 x Block(
|
| 209 |
+
(residual1): EfficientResidual(
|
| 210 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 211 |
+
(fn): Attention(
|
| 212 |
+
(q_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 213 |
+
(k_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 214 |
+
(v_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 215 |
+
(proj): Linear(in_features=384, out_features=384, bias=False)
|
| 216 |
+
(rope): Rope()
|
| 217 |
+
)
|
| 218 |
+
)
|
| 219 |
+
(residual2): EfficientResidual(
|
| 220 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 221 |
+
(fn): MLP(
|
| 222 |
+
(fc1): Linear(in_features=384, out_features=1536, bias=False)
|
| 223 |
+
(act): GELU(approximate='none')
|
| 224 |
+
(fc2): Linear(in_features=1536, out_features=384, bias=False)
|
| 225 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 226 |
+
)
|
| 227 |
+
)
|
| 228 |
+
)
|
| 229 |
+
)
|
| 230 |
+
(predictor_norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 231 |
+
(predictor_proj): Linear(in_features=384, out_features=1024, bias=True)
|
| 232 |
+
)
|
| 233 |
+
)
|
| 234 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] MultiSeqWrapper(
|
| 235 |
+
(backbone): Frozen2DTargetWrapper(
|
| 236 |
+
(backbone): Eva(
|
| 237 |
+
(patch_embed): PatchEmbed(
|
| 238 |
+
(proj): Conv2d(3, 1024, kernel_size=(14, 14), stride=(14, 14))
|
| 239 |
+
(norm): Identity()
|
| 240 |
+
)
|
| 241 |
+
(pos_drop): Dropout(p=0.0, inplace=False)
|
| 242 |
+
(norm_pre): Identity()
|
| 243 |
+
(blocks): ModuleList(
|
| 244 |
+
(0-23): 24 x EvaBlock(
|
| 245 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 246 |
+
(attn): EvaAttention(
|
| 247 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 248 |
+
(q_norm): Identity()
|
| 249 |
+
(k_norm): Identity()
|
| 250 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 251 |
+
(norm): Identity()
|
| 252 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=True)
|
| 253 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 254 |
+
)
|
| 255 |
+
(drop_path1): Identity()
|
| 256 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 257 |
+
(mlp): Mlp(
|
| 258 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=True)
|
| 259 |
+
(act): GELU(approximate='none')
|
| 260 |
+
(drop1): Dropout(p=0.0, inplace=False)
|
| 261 |
+
(norm): Identity()
|
| 262 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=True)
|
| 263 |
+
(drop2): Dropout(p=0.0, inplace=False)
|
| 264 |
+
)
|
| 265 |
+
(drop_path2): Identity()
|
| 266 |
+
)
|
| 267 |
+
)
|
| 268 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 269 |
+
(fc_norm): Identity()
|
| 270 |
+
(head_drop): Dropout(p=0.0, inplace=False)
|
| 271 |
+
(head): Identity()
|
| 272 |
+
(rope): _CapiPatchRoPE()
|
| 273 |
+
)
|
| 274 |
+
)
|
| 275 |
+
)
|
| 276 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] Encoder number of parameters: 403136512
|
| 277 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] Predictor number of parameters: 11416192
|
| 278 |
+
[INFO ][2026-05-14 09:43:24][root ][init_video_model ] Target encoder number of parameters: 0
|
| 279 |
+
[INFO ][2026-05-14 09:43:40][root ][make_videodataset ] VideoDataset dataset created
|
| 280 |
+
[INFO ][2026-05-14 09:43:40][WeightedSampler ][__init__ ] Using DistributedWeightedSampler with rank 2 / 16
|
| 281 |
+
[INFO ][2026-05-14 09:43:40][root ][make_videodataset ] VideoDataset unsupervised data loader created
|
| 282 |
+
[INFO ][2026-05-14 09:43:40][app.vjepa.train ][main ] iterations per epoch/dataset length: 300/399
|
| 283 |
+
[INFO ][2026-05-14 09:43:40][app.vjepa.train ][main ] Wrapping models in DDP (rank 2)...
|
| 284 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Initializing loader...
|
| 285 |
+
submitit WARNING (2026-05-14 13:38:23,807) - Bypassing signal SIGCONT
|
| 286 |
+
submitit WARNING (2026-05-14 13:38:23,937) - Bypassing signal SIGTERM
|
| 287 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGTERM
|
| 288 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGCONT
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_3_log.err
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/timm/models/layers/__init__.py:49: FutureWarning: Importing from timm.models.layers is deprecated, please import via timm.layers
|
| 2 |
+
warnings.warn(f"Importing from {__name__} is deprecated, please import via timm.layers", FutureWarning)
|
| 3 |
+
[rank3]:[W514 09:35:32.320490932 ProcessGroupNCCL.cpp:5138] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 4 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/utils.py:796: FutureWarning: `torch.cuda.amp.GradScaler(args...)` is deprecated. Please use `torch.amp.GradScaler('cuda', args...)` instead.
|
| 5 |
+
scaler = torch.cuda.amp.GradScaler() if mixed_precision else None
|
| 6 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 7 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 8 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 9 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 10 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 11 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 12 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 13 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 14 |
+
submitit WARNING (2026-05-14 13:38:23,807) - Bypassing signal SIGCONT
|
| 15 |
+
submitit WARNING (2026-05-14 13:38:23,819) - Bypassing signal SIGTERM
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_3_log.out
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
submitit INFO (2026-05-14 09:28:14,530) - Starting with JobEnvironment(job_id=22743106, hostname=gcn75.local.snellius.surf.nl, local_rank=3(4), node=0(4), global_rank=3(16))
|
| 2 |
+
submitit INFO (2026-05-14 09:28:14,530) - Loading pickle: /scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_submitted.pkl
|
| 3 |
+
INFO:root:loaded pretrain params...
|
| 4 |
+
{ 'app': 'vjepa',
|
| 5 |
+
'cpus_per_task': 16,
|
| 6 |
+
'data': { 'batch_size': 64,
|
| 7 |
+
'crop_size': 224,
|
| 8 |
+
'dataset_fpcs': [16, 16],
|
| 9 |
+
'dataset_type': 'VideoDataset',
|
| 10 |
+
'datasets': [ '/scratch-shared/dcanez/data/kinetics/k400/train.csv',
|
| 11 |
+
'/scratch-shared/dcanez/data/ssv2/train.csv'],
|
| 12 |
+
'datasets_weights': [0.65, 0.35],
|
| 13 |
+
'fps': 4,
|
| 14 |
+
'num_workers': 10,
|
| 15 |
+
'patch_size': 14,
|
| 16 |
+
'persistent_workers': True,
|
| 17 |
+
'pin_mem': True,
|
| 18 |
+
'stage': [ { 'dest': 'kinetics_240',
|
| 19 |
+
'format': 'targz_parts',
|
| 20 |
+
'src': '/scratch-shared/dcanez/data/kinetics/k400/tars_240/'},
|
| 21 |
+
{ 'dest': 'ssv2',
|
| 22 |
+
'format': 'multipart_tar',
|
| 23 |
+
'src': '/scratch-nvme/ml-datasets/something-something-v2/'}],
|
| 24 |
+
'tubelet_size': 1},
|
| 25 |
+
'data_aug': { 'auto_augment': False,
|
| 26 |
+
'motion_shift': False,
|
| 27 |
+
'random_resize_aspect_ratio': [0.75, 1.35],
|
| 28 |
+
'random_resize_scale': [0.3, 1.0],
|
| 29 |
+
'reprob': 0.0},
|
| 30 |
+
'folder': '/scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable',
|
| 31 |
+
'loss': {'loss_exp': 1.0},
|
| 32 |
+
'mask': [ { 'aspect_ratio': [0.75, 1.5],
|
| 33 |
+
'full_complement': False,
|
| 34 |
+
'max_keep': None,
|
| 35 |
+
'max_temporal_keep': 1.0,
|
| 36 |
+
'num_blocks': 8,
|
| 37 |
+
'spatial_scale': [0.15, 0.15],
|
| 38 |
+
'temporal_scale': [1.0, 1.0]},
|
| 39 |
+
{ 'aspect_ratio': [0.75, 1.5],
|
| 40 |
+
'full_complement': False,
|
| 41 |
+
'max_keep': None,
|
| 42 |
+
'max_temporal_keep': 1.0,
|
| 43 |
+
'num_blocks': 2,
|
| 44 |
+
'spatial_scale': [0.7, 0.7],
|
| 45 |
+
'temporal_scale': [1.0, 1.0]}],
|
| 46 |
+
'mem_per_gpu': '180G',
|
| 47 |
+
'meta': { 'dtype': 'bfloat16',
|
| 48 |
+
'knn_eval_epoch0': False,
|
| 49 |
+
'knn_eval_freq': 5,
|
| 50 |
+
'knn_eval_presets': [ { 'config': { 'batch_size': 64,
|
| 51 |
+
'dataset_train': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/train.csv',
|
| 52 |
+
'dataset_val': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/val.csv',
|
| 53 |
+
'eval_videos_per_class': 25,
|
| 54 |
+
'num_workers': 8,
|
| 55 |
+
'pool_type': 'slot_temporal_concat',
|
| 56 |
+
'train_videos_per_class': 100},
|
| 57 |
+
'preset': 'ucf101'},
|
| 58 |
+
{ 'config': { 'batch_size': 64,
|
| 59 |
+
'dataset_train': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/train_coarse10.csv',
|
| 60 |
+
'dataset_val': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/val_coarse10.csv',
|
| 61 |
+
'eval_videos_per_class': 100,
|
| 62 |
+
'linear_probe': True,
|
| 63 |
+
'num_workers': 8,
|
| 64 |
+
'pool_type': 'slot_temporal_concat',
|
| 65 |
+
'train_videos_per_class': 500},
|
| 66 |
+
'preset': 'ssv2_coarse10'}],
|
| 67 |
+
'load_checkpoint': True,
|
| 68 |
+
'read_checkpoint': None,
|
| 69 |
+
'save_every_freq': 5,
|
| 70 |
+
'seed': 239,
|
| 71 |
+
'use_sdpa': True,
|
| 72 |
+
'use_wandb': True,
|
| 73 |
+
'wandb_project': 'vjepa_ujepaside'},
|
| 74 |
+
'metrics': {'sigreg': {}, 'std': {}},
|
| 75 |
+
'model': { 'model_name': 'ujepaside_large_patch14_capi_lvd1689m',
|
| 76 |
+
'pred_depth': 6,
|
| 77 |
+
'pred_embed_dim': 384,
|
| 78 |
+
'pred_num_heads': 12,
|
| 79 |
+
'predictor': 'v2_cross',
|
| 80 |
+
'st_causal': False,
|
| 81 |
+
'st_drop_path': 0.2,
|
| 82 |
+
'st_flex_enable': False,
|
| 83 |
+
'st_layer_scale_init': 1e-05,
|
| 84 |
+
'st_num_slots': 16,
|
| 85 |
+
'st_side_block_type': 'factorized',
|
| 86 |
+
'st_slots_causal_within_frame': False,
|
| 87 |
+
'target_kind': 'frozen_2d',
|
| 88 |
+
'target_type': 'vit_large_patch14_capi.lvd1689m',
|
| 89 |
+
'temporal_spacing': 1.0,
|
| 90 |
+
'uniform_power': True,
|
| 91 |
+
'use_activation_checkpointing': True,
|
| 92 |
+
'use_mask_tokens': True,
|
| 93 |
+
'use_rope': True,
|
| 94 |
+
'use_sdpa': True,
|
| 95 |
+
'zero_init_mask_tokens': True},
|
| 96 |
+
'nodes': 4,
|
| 97 |
+
'optimization': { 'clip_grad': 3.0,
|
| 98 |
+
'ema': [0.99925, 0.99925],
|
| 99 |
+
'epochs': 100,
|
| 100 |
+
'final_lr': 0.0001,
|
| 101 |
+
'final_weight_decay': 0.04,
|
| 102 |
+
'ipe': 300,
|
| 103 |
+
'ipe_scale': 1.0,
|
| 104 |
+
'lr': 0.0005,
|
| 105 |
+
'start_lr': 0.0001,
|
| 106 |
+
'warmup': 10,
|
| 107 |
+
'weight_decay': 0.04},
|
| 108 |
+
'tasks_per_node': 4}
|
| 109 |
+
INFO:root:Running pre-training of app: vjepa
|
| 110 |
+
[INFO ][2026-05-14 09:32:27][app.vjepa.train ][main ] which_dtype='bfloat16'
|
| 111 |
+
[INFO ][2026-05-14 09:32:27][app.vjepa.train ][main ] Disabling persistent_workers (incompatible with KNN eval)
|
| 112 |
+
[INFO ][2026-05-14 09:32:27][app.vjepa.train ][main ] NCCL_SOCKET_IFNAME=eno
|
| 113 |
+
[INFO ][2026-05-14 09:32:29][app.vjepa.train ][main ] Initialized (rank/world-size) 3/16, tasks_per_node=4
|
| 114 |
+
[INFO ][2026-05-14 09:32:29][root ][stage_datasets ] [local_rank 3/4] Staging kinetics_240 (targz_parts)
|
| 115 |
+
[INFO ][2026-05-14 09:32:29][root ][_stage_targz_parts ] [rank 3] Extracting 25/103 tar.gz parts to /scratch-node/dcanez.22743106/kinetics_240
|
| 116 |
+
[INFO ][2026-05-14 09:32:44][root ][_stage_targz_parts ] [local_rank 3] Extracted 2/25 parts
|
| 117 |
+
[INFO ][2026-05-14 09:32:58][root ][_stage_targz_parts ] [local_rank 3] Extracted 4/25 parts
|
| 118 |
+
[INFO ][2026-05-14 09:33:13][root ][_stage_targz_parts ] [local_rank 3] Extracted 6/25 parts
|
| 119 |
+
[INFO ][2026-05-14 09:33:27][root ][_stage_targz_parts ] [local_rank 3] Extracted 8/25 parts
|
| 120 |
+
[INFO ][2026-05-14 09:33:42][root ][_stage_targz_parts ] [local_rank 3] Extracted 10/25 parts
|
| 121 |
+
[INFO ][2026-05-14 09:33:56][root ][_stage_targz_parts ] [local_rank 3] Extracted 12/25 parts
|
| 122 |
+
[INFO ][2026-05-14 09:34:11][root ][_stage_targz_parts ] [local_rank 3] Extracted 14/25 parts
|
| 123 |
+
[INFO ][2026-05-14 09:34:25][root ][_stage_targz_parts ] [local_rank 3] Extracted 16/25 parts
|
| 124 |
+
[INFO ][2026-05-14 09:34:39][root ][_stage_targz_parts ] [local_rank 3] Extracted 18/25 parts
|
| 125 |
+
[INFO ][2026-05-14 09:34:54][root ][_stage_targz_parts ] [local_rank 3] Extracted 20/25 parts
|
| 126 |
+
[INFO ][2026-05-14 09:35:09][root ][_stage_targz_parts ] [local_rank 3] Extracted 22/25 parts
|
| 127 |
+
[INFO ][2026-05-14 09:35:23][root ][_stage_targz_parts ] [local_rank 3] Extracted 24/25 parts
|
| 128 |
+
[INFO ][2026-05-14 09:35:30][root ][_stage_targz_parts ] [local_rank 3] Extracted 25/25 parts
|
| 129 |
+
[INFO ][2026-05-14 09:35:30][root ][stage_datasets ] [local_rank 3/4] Staging ssv2 (multipart_tar)
|
| 130 |
+
[INFO ][2026-05-14 09:36:49][app.vjepa.train ][main ] Staged datasets: ['/scratch-node/dcanez.22743106/kinetics_240/train.csv', '/scratch-node/dcanez.22743106/ssv2/train.csv']
|
| 131 |
+
[INFO ][2026-05-14 09:43:14][experiments.stmodels.vision_transformers_v3][vit_large_patch14_capi_lvd1689m] Loading pretrained weights for vit_large_patch14_capi_lvd1689m from https://dl.fbaipublicfiles.com/capi/capi_vitl14_lvd.pth
|
| 132 |
+
[INFO ][2026-05-14 09:43:23][root ][_build_frozen_2d_target ] Loaded frozen_2d target from timm: vit_large_patch14_capi.lvd1689m
|
| 133 |
+
[INFO ][2026-05-14 09:43:23][root ][init_video_model ] ViTMultiSeqWrapper(
|
| 134 |
+
(backbone): UJEPAside(
|
| 135 |
+
(patch_embed): PatchEmbed3D(
|
| 136 |
+
(proj): Conv3d(3, 1024, kernel_size=(1, 14, 14), stride=(1, 14, 14))
|
| 137 |
+
)
|
| 138 |
+
(rope): CAPI2DRoPE()
|
| 139 |
+
(blocks): ModuleList(
|
| 140 |
+
(0-23): 24 x Block(
|
| 141 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 142 |
+
(rope_impl): CAPI2DRoPE()
|
| 143 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 144 |
+
(drop_path1): Identity()
|
| 145 |
+
(drop_path2): Identity()
|
| 146 |
+
(attn): Attention(
|
| 147 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 148 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 149 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 150 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 151 |
+
(rope_impl): CAPI2DRoPE()
|
| 152 |
+
)
|
| 153 |
+
(mlp): MLP(
|
| 154 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 155 |
+
(act): GELU(approximate='none')
|
| 156 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 157 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 158 |
+
)
|
| 159 |
+
)
|
| 160 |
+
)
|
| 161 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=False)
|
| 162 |
+
(st_blocks): ModuleList(
|
| 163 |
+
(0-23): 24 x FactorizedSlotSideBlock(
|
| 164 |
+
(cross_attn): EfficientResidual(
|
| 165 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 166 |
+
(fn): Attention(
|
| 167 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 168 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 169 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 170 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 171 |
+
(rope_impl): CAPI3DRoPE()
|
| 172 |
+
)
|
| 173 |
+
)
|
| 174 |
+
(self_attn): EfficientResidual(
|
| 175 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 176 |
+
(fn): Attention(
|
| 177 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 178 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 179 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 180 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 181 |
+
(rope_impl): CAPI3DRoPE()
|
| 182 |
+
)
|
| 183 |
+
)
|
| 184 |
+
(mlp): EfficientResidual(
|
| 185 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 186 |
+
(fn): MLP(
|
| 187 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 188 |
+
(act): GELU(approximate='none')
|
| 189 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 190 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 191 |
+
)
|
| 192 |
+
)
|
| 193 |
+
)
|
| 194 |
+
)
|
| 195 |
+
(st_rope): CAPI3DRoPE()
|
| 196 |
+
)
|
| 197 |
+
)
|
| 198 |
+
[INFO ][2026-05-14 09:43:23][root ][init_video_model ] UJEPAsidePredictorMultiSeqWrapper(
|
| 199 |
+
(backbone): PredictorV2(
|
| 200 |
+
(predictor_embed): Linear(in_features=1024, out_features=384, bias=True)
|
| 201 |
+
(mask_tokens): ParameterList(
|
| 202 |
+
(0): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 203 |
+
(1): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 204 |
+
(2): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 205 |
+
(3): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 206 |
+
)
|
| 207 |
+
(predictor_blocks): ModuleList(
|
| 208 |
+
(0-5): 6 x Block(
|
| 209 |
+
(residual1): EfficientResidual(
|
| 210 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 211 |
+
(fn): Attention(
|
| 212 |
+
(q_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 213 |
+
(k_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 214 |
+
(v_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 215 |
+
(proj): Linear(in_features=384, out_features=384, bias=False)
|
| 216 |
+
(rope): Rope()
|
| 217 |
+
)
|
| 218 |
+
)
|
| 219 |
+
(residual2): EfficientResidual(
|
| 220 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 221 |
+
(fn): MLP(
|
| 222 |
+
(fc1): Linear(in_features=384, out_features=1536, bias=False)
|
| 223 |
+
(act): GELU(approximate='none')
|
| 224 |
+
(fc2): Linear(in_features=1536, out_features=384, bias=False)
|
| 225 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 226 |
+
)
|
| 227 |
+
)
|
| 228 |
+
)
|
| 229 |
+
)
|
| 230 |
+
(predictor_norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 231 |
+
(predictor_proj): Linear(in_features=384, out_features=1024, bias=True)
|
| 232 |
+
)
|
| 233 |
+
)
|
| 234 |
+
[INFO ][2026-05-14 09:43:23][root ][init_video_model ] MultiSeqWrapper(
|
| 235 |
+
(backbone): Frozen2DTargetWrapper(
|
| 236 |
+
(backbone): Eva(
|
| 237 |
+
(patch_embed): PatchEmbed(
|
| 238 |
+
(proj): Conv2d(3, 1024, kernel_size=(14, 14), stride=(14, 14))
|
| 239 |
+
(norm): Identity()
|
| 240 |
+
)
|
| 241 |
+
(pos_drop): Dropout(p=0.0, inplace=False)
|
| 242 |
+
(norm_pre): Identity()
|
| 243 |
+
(blocks): ModuleList(
|
| 244 |
+
(0-23): 24 x EvaBlock(
|
| 245 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 246 |
+
(attn): EvaAttention(
|
| 247 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 248 |
+
(q_norm): Identity()
|
| 249 |
+
(k_norm): Identity()
|
| 250 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 251 |
+
(norm): Identity()
|
| 252 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=True)
|
| 253 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 254 |
+
)
|
| 255 |
+
(drop_path1): Identity()
|
| 256 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 257 |
+
(mlp): Mlp(
|
| 258 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=True)
|
| 259 |
+
(act): GELU(approximate='none')
|
| 260 |
+
(drop1): Dropout(p=0.0, inplace=False)
|
| 261 |
+
(norm): Identity()
|
| 262 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=True)
|
| 263 |
+
(drop2): Dropout(p=0.0, inplace=False)
|
| 264 |
+
)
|
| 265 |
+
(drop_path2): Identity()
|
| 266 |
+
)
|
| 267 |
+
)
|
| 268 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 269 |
+
(fc_norm): Identity()
|
| 270 |
+
(head_drop): Dropout(p=0.0, inplace=False)
|
| 271 |
+
(head): Identity()
|
| 272 |
+
(rope): _CapiPatchRoPE()
|
| 273 |
+
)
|
| 274 |
+
)
|
| 275 |
+
)
|
| 276 |
+
[INFO ][2026-05-14 09:43:23][root ][init_video_model ] Encoder number of parameters: 403136512
|
| 277 |
+
[INFO ][2026-05-14 09:43:23][root ][init_video_model ] Predictor number of parameters: 11416192
|
| 278 |
+
[INFO ][2026-05-14 09:43:23][root ][init_video_model ] Target encoder number of parameters: 0
|
| 279 |
+
[INFO ][2026-05-14 09:43:39][root ][make_videodataset ] VideoDataset dataset created
|
| 280 |
+
[INFO ][2026-05-14 09:43:39][WeightedSampler ][__init__ ] Using DistributedWeightedSampler with rank 3 / 16
|
| 281 |
+
[INFO ][2026-05-14 09:43:39][root ][make_videodataset ] VideoDataset unsupervised data loader created
|
| 282 |
+
[INFO ][2026-05-14 09:43:39][app.vjepa.train ][main ] iterations per epoch/dataset length: 300/399
|
| 283 |
+
[INFO ][2026-05-14 09:43:40][app.vjepa.train ][main ] Wrapping models in DDP (rank 3)...
|
| 284 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Initializing loader...
|
| 285 |
+
submitit WARNING (2026-05-14 13:38:23,807) - Bypassing signal SIGCONT
|
| 286 |
+
submitit WARNING (2026-05-14 13:38:23,819) - Bypassing signal SIGTERM
|
| 287 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGTERM
|
| 288 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGCONT
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_4_log.err
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/timm/models/layers/__init__.py:49: FutureWarning: Importing from timm.models.layers is deprecated, please import via timm.layers
|
| 2 |
+
warnings.warn(f"Importing from {__name__} is deprecated, please import via timm.layers", FutureWarning)
|
| 3 |
+
[rank4]:[W514 09:36:37.824240654 ProcessGroupNCCL.cpp:5138] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 4 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/utils.py:796: FutureWarning: `torch.cuda.amp.GradScaler(args...)` is deprecated. Please use `torch.amp.GradScaler('cuda', args...)` instead.
|
| 5 |
+
scaler = torch.cuda.amp.GradScaler() if mixed_precision else None
|
| 6 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 7 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 8 |
+
submitit WARNING (2026-05-14 13:38:23,805) - Bypassing signal SIGCONT
|
| 9 |
+
submitit WARNING (2026-05-14 13:38:23,872) - Bypassing signal SIGTERM
|
| 10 |
+
[2026-05-14T13:38:55.995] error: namespace_p_join: open failed for /slurm/22743106/.ns: No such file or directory
|
| 11 |
+
[2026-05-14T13:38:55.995] error: namespace_g_join(JobId=22743106 SLUID=s8FXKRJP5N0000): No such file or directory
|
| 12 |
+
[2026-05-14T13:38:57.703] error: Failed to send MESSAGE_TASK_EXIT: Connection refused
|
| 13 |
+
[2026-05-14T13:38:59.416] error: namespace_p_join: open failed for /slurm/22743106/.ns: No such file or directory
|
| 14 |
+
[2026-05-14T13:38:59.416] error: namespace_g_join(JobId=22743106 SLUID=s8FXKRJP5N0000): No such file or directory
|
| 15 |
+
[2026-05-14T13:38:59.553] error: namespace_p_join: open failed for /slurm/22743106/.ns: No such file or directory
|
| 16 |
+
[2026-05-14T13:38:59.554] error: namespace_g_join(JobId=22743106 SLUID=s8FXKRJP5N0000): No such file or directory
|
| 17 |
+
[2026-05-14T13:38:59.591] error: namespace_p_join: open failed for /slurm/22743106/.ns: No such file or directory
|
| 18 |
+
[2026-05-14T13:38:59.591] error: namespace_g_join(JobId=22743106 SLUID=s8FXKRJP5N0000): No such file or directory
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_4_log.out
ADDED
|
@@ -0,0 +1,292 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
submitit INFO (2026-05-14 09:28:15,038) - Starting with JobEnvironment(job_id=22743106, hostname=gcn79.local.snellius.surf.nl, local_rank=0(4), node=1(4), global_rank=4(16))
|
| 2 |
+
submitit INFO (2026-05-14 09:28:15,038) - Loading pickle: /scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_submitted.pkl
|
| 3 |
+
INFO:root:loaded pretrain params...
|
| 4 |
+
{ 'app': 'vjepa',
|
| 5 |
+
'cpus_per_task': 16,
|
| 6 |
+
'data': { 'batch_size': 64,
|
| 7 |
+
'crop_size': 224,
|
| 8 |
+
'dataset_fpcs': [16, 16],
|
| 9 |
+
'dataset_type': 'VideoDataset',
|
| 10 |
+
'datasets': [ '/scratch-shared/dcanez/data/kinetics/k400/train.csv',
|
| 11 |
+
'/scratch-shared/dcanez/data/ssv2/train.csv'],
|
| 12 |
+
'datasets_weights': [0.65, 0.35],
|
| 13 |
+
'fps': 4,
|
| 14 |
+
'num_workers': 10,
|
| 15 |
+
'patch_size': 14,
|
| 16 |
+
'persistent_workers': True,
|
| 17 |
+
'pin_mem': True,
|
| 18 |
+
'stage': [ { 'dest': 'kinetics_240',
|
| 19 |
+
'format': 'targz_parts',
|
| 20 |
+
'src': '/scratch-shared/dcanez/data/kinetics/k400/tars_240/'},
|
| 21 |
+
{ 'dest': 'ssv2',
|
| 22 |
+
'format': 'multipart_tar',
|
| 23 |
+
'src': '/scratch-nvme/ml-datasets/something-something-v2/'}],
|
| 24 |
+
'tubelet_size': 1},
|
| 25 |
+
'data_aug': { 'auto_augment': False,
|
| 26 |
+
'motion_shift': False,
|
| 27 |
+
'random_resize_aspect_ratio': [0.75, 1.35],
|
| 28 |
+
'random_resize_scale': [0.3, 1.0],
|
| 29 |
+
'reprob': 0.0},
|
| 30 |
+
'folder': '/scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable',
|
| 31 |
+
'loss': {'loss_exp': 1.0},
|
| 32 |
+
'mask': [ { 'aspect_ratio': [0.75, 1.5],
|
| 33 |
+
'full_complement': False,
|
| 34 |
+
'max_keep': None,
|
| 35 |
+
'max_temporal_keep': 1.0,
|
| 36 |
+
'num_blocks': 8,
|
| 37 |
+
'spatial_scale': [0.15, 0.15],
|
| 38 |
+
'temporal_scale': [1.0, 1.0]},
|
| 39 |
+
{ 'aspect_ratio': [0.75, 1.5],
|
| 40 |
+
'full_complement': False,
|
| 41 |
+
'max_keep': None,
|
| 42 |
+
'max_temporal_keep': 1.0,
|
| 43 |
+
'num_blocks': 2,
|
| 44 |
+
'spatial_scale': [0.7, 0.7],
|
| 45 |
+
'temporal_scale': [1.0, 1.0]}],
|
| 46 |
+
'mem_per_gpu': '180G',
|
| 47 |
+
'meta': { 'dtype': 'bfloat16',
|
| 48 |
+
'knn_eval_epoch0': False,
|
| 49 |
+
'knn_eval_freq': 5,
|
| 50 |
+
'knn_eval_presets': [ { 'config': { 'batch_size': 64,
|
| 51 |
+
'dataset_train': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/train.csv',
|
| 52 |
+
'dataset_val': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/val.csv',
|
| 53 |
+
'eval_videos_per_class': 25,
|
| 54 |
+
'num_workers': 8,
|
| 55 |
+
'pool_type': 'slot_temporal_concat',
|
| 56 |
+
'train_videos_per_class': 100},
|
| 57 |
+
'preset': 'ucf101'},
|
| 58 |
+
{ 'config': { 'batch_size': 64,
|
| 59 |
+
'dataset_train': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/train_coarse10.csv',
|
| 60 |
+
'dataset_val': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/val_coarse10.csv',
|
| 61 |
+
'eval_videos_per_class': 100,
|
| 62 |
+
'linear_probe': True,
|
| 63 |
+
'num_workers': 8,
|
| 64 |
+
'pool_type': 'slot_temporal_concat',
|
| 65 |
+
'train_videos_per_class': 500},
|
| 66 |
+
'preset': 'ssv2_coarse10'}],
|
| 67 |
+
'load_checkpoint': True,
|
| 68 |
+
'read_checkpoint': None,
|
| 69 |
+
'save_every_freq': 5,
|
| 70 |
+
'seed': 239,
|
| 71 |
+
'use_sdpa': True,
|
| 72 |
+
'use_wandb': True,
|
| 73 |
+
'wandb_project': 'vjepa_ujepaside'},
|
| 74 |
+
'metrics': {'sigreg': {}, 'std': {}},
|
| 75 |
+
'model': { 'model_name': 'ujepaside_large_patch14_capi_lvd1689m',
|
| 76 |
+
'pred_depth': 6,
|
| 77 |
+
'pred_embed_dim': 384,
|
| 78 |
+
'pred_num_heads': 12,
|
| 79 |
+
'predictor': 'v2_cross',
|
| 80 |
+
'st_causal': False,
|
| 81 |
+
'st_drop_path': 0.2,
|
| 82 |
+
'st_flex_enable': False,
|
| 83 |
+
'st_layer_scale_init': 1e-05,
|
| 84 |
+
'st_num_slots': 16,
|
| 85 |
+
'st_side_block_type': 'factorized',
|
| 86 |
+
'st_slots_causal_within_frame': False,
|
| 87 |
+
'target_kind': 'frozen_2d',
|
| 88 |
+
'target_type': 'vit_large_patch14_capi.lvd1689m',
|
| 89 |
+
'temporal_spacing': 1.0,
|
| 90 |
+
'uniform_power': True,
|
| 91 |
+
'use_activation_checkpointing': True,
|
| 92 |
+
'use_mask_tokens': True,
|
| 93 |
+
'use_rope': True,
|
| 94 |
+
'use_sdpa': True,
|
| 95 |
+
'zero_init_mask_tokens': True},
|
| 96 |
+
'nodes': 4,
|
| 97 |
+
'optimization': { 'clip_grad': 3.0,
|
| 98 |
+
'ema': [0.99925, 0.99925],
|
| 99 |
+
'epochs': 100,
|
| 100 |
+
'final_lr': 0.0001,
|
| 101 |
+
'final_weight_decay': 0.04,
|
| 102 |
+
'ipe': 300,
|
| 103 |
+
'ipe_scale': 1.0,
|
| 104 |
+
'lr': 0.0005,
|
| 105 |
+
'start_lr': 0.0001,
|
| 106 |
+
'warmup': 10,
|
| 107 |
+
'weight_decay': 0.04},
|
| 108 |
+
'tasks_per_node': 4}
|
| 109 |
+
INFO:root:Running pre-training of app: vjepa
|
| 110 |
+
[INFO ][2026-05-14 09:32:14][app.vjepa.train ][main ] which_dtype='bfloat16'
|
| 111 |
+
[INFO ][2026-05-14 09:32:14][app.vjepa.train ][main ] Disabling persistent_workers (incompatible with KNN eval)
|
| 112 |
+
[INFO ][2026-05-14 09:32:14][app.vjepa.train ][main ] NCCL_SOCKET_IFNAME=eno
|
| 113 |
+
[INFO ][2026-05-14 09:32:32][app.vjepa.train ][main ] Initialized (rank/world-size) 4/16, tasks_per_node=4
|
| 114 |
+
[INFO ][2026-05-14 09:32:32][root ][stage_datasets ] [local_rank 0/4] Staging kinetics_240 (targz_parts)
|
| 115 |
+
[INFO ][2026-05-14 09:32:32][root ][_stage_targz_parts ] [rank 0] Extracting 26/103 tar.gz parts to /scratch-node/dcanez.22743106/kinetics_240
|
| 116 |
+
[INFO ][2026-05-14 09:32:46][root ][_stage_targz_parts ] [local_rank 0] Extracted 2/26 parts
|
| 117 |
+
[INFO ][2026-05-14 09:33:01][root ][_stage_targz_parts ] [local_rank 0] Extracted 4/26 parts
|
| 118 |
+
[INFO ][2026-05-14 09:33:15][root ][_stage_targz_parts ] [local_rank 0] Extracted 6/26 parts
|
| 119 |
+
[INFO ][2026-05-14 09:33:30][root ][_stage_targz_parts ] [local_rank 0] Extracted 8/26 parts
|
| 120 |
+
[INFO ][2026-05-14 09:33:44][root ][_stage_targz_parts ] [local_rank 0] Extracted 10/26 parts
|
| 121 |
+
[INFO ][2026-05-14 09:33:59][root ][_stage_targz_parts ] [local_rank 0] Extracted 12/26 parts
|
| 122 |
+
[INFO ][2026-05-14 09:34:13][root ][_stage_targz_parts ] [local_rank 0] Extracted 14/26 parts
|
| 123 |
+
[INFO ][2026-05-14 09:34:28][root ][_stage_targz_parts ] [local_rank 0] Extracted 16/26 parts
|
| 124 |
+
[INFO ][2026-05-14 09:34:43][root ][_stage_targz_parts ] [local_rank 0] Extracted 18/26 parts
|
| 125 |
+
[INFO ][2026-05-14 09:34:58][root ][_stage_targz_parts ] [local_rank 0] Extracted 20/26 parts
|
| 126 |
+
[INFO ][2026-05-14 09:35:12][root ][_stage_targz_parts ] [local_rank 0] Extracted 22/26 parts
|
| 127 |
+
[INFO ][2026-05-14 09:35:27][root ][_stage_targz_parts ] [local_rank 0] Extracted 24/26 parts
|
| 128 |
+
[INFO ][2026-05-14 09:35:42][root ][_stage_targz_parts ] [local_rank 0] Extracted 26/26 parts
|
| 129 |
+
[INFO ][2026-05-14 09:35:42][root ][stage_datasets ] [local_rank 0/4] Staging ssv2 (multipart_tar)
|
| 130 |
+
[INFO ][2026-05-14 09:35:42][root ][_stage_multipart_tar ] [rank 0] Extracting multipart tar (2 files) to /scratch-node/dcanez.22743106/ssv2
|
| 131 |
+
[INFO ][2026-05-14 09:36:48][root ][stage_datasets ] Data staging completed in 256.0s (4.3min)
|
| 132 |
+
[INFO ][2026-05-14 09:36:49][root ][_rewrite_csv ] Wrote local CSV: /scratch-node/dcanez.22743106/kinetics_240/train.csv (239789 entries)
|
| 133 |
+
[INFO ][2026-05-14 09:36:49][root ][_rewrite_csv ] Wrote local CSV: /scratch-node/dcanez.22743106/ssv2/train.csv (168913 entries)
|
| 134 |
+
[INFO ][2026-05-14 09:36:49][app.vjepa.train ][main ] Staged datasets: ['/scratch-node/dcanez.22743106/kinetics_240/train.csv', '/scratch-node/dcanez.22743106/ssv2/train.csv']
|
| 135 |
+
[INFO ][2026-05-14 09:43:14][experiments.stmodels.vision_transformers_v3][vit_large_patch14_capi_lvd1689m] Loading pretrained weights for vit_large_patch14_capi_lvd1689m from https://dl.fbaipublicfiles.com/capi/capi_vitl14_lvd.pth
|
| 136 |
+
[INFO ][2026-05-14 09:43:30][root ][_build_frozen_2d_target ] Loaded frozen_2d target from timm: vit_large_patch14_capi.lvd1689m
|
| 137 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] ViTMultiSeqWrapper(
|
| 138 |
+
(backbone): UJEPAside(
|
| 139 |
+
(patch_embed): PatchEmbed3D(
|
| 140 |
+
(proj): Conv3d(3, 1024, kernel_size=(1, 14, 14), stride=(1, 14, 14))
|
| 141 |
+
)
|
| 142 |
+
(rope): CAPI2DRoPE()
|
| 143 |
+
(blocks): ModuleList(
|
| 144 |
+
(0-23): 24 x Block(
|
| 145 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 146 |
+
(rope_impl): CAPI2DRoPE()
|
| 147 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 148 |
+
(drop_path1): Identity()
|
| 149 |
+
(drop_path2): Identity()
|
| 150 |
+
(attn): Attention(
|
| 151 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 152 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 153 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 154 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 155 |
+
(rope_impl): CAPI2DRoPE()
|
| 156 |
+
)
|
| 157 |
+
(mlp): MLP(
|
| 158 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 159 |
+
(act): GELU(approximate='none')
|
| 160 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 161 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 162 |
+
)
|
| 163 |
+
)
|
| 164 |
+
)
|
| 165 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=False)
|
| 166 |
+
(st_blocks): ModuleList(
|
| 167 |
+
(0-23): 24 x FactorizedSlotSideBlock(
|
| 168 |
+
(cross_attn): EfficientResidual(
|
| 169 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 170 |
+
(fn): Attention(
|
| 171 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 172 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 173 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 174 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 175 |
+
(rope_impl): CAPI3DRoPE()
|
| 176 |
+
)
|
| 177 |
+
)
|
| 178 |
+
(self_attn): EfficientResidual(
|
| 179 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 180 |
+
(fn): Attention(
|
| 181 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 182 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 183 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 184 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 185 |
+
(rope_impl): CAPI3DRoPE()
|
| 186 |
+
)
|
| 187 |
+
)
|
| 188 |
+
(mlp): EfficientResidual(
|
| 189 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 190 |
+
(fn): MLP(
|
| 191 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 192 |
+
(act): GELU(approximate='none')
|
| 193 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 194 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 195 |
+
)
|
| 196 |
+
)
|
| 197 |
+
)
|
| 198 |
+
)
|
| 199 |
+
(st_rope): CAPI3DRoPE()
|
| 200 |
+
)
|
| 201 |
+
)
|
| 202 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] UJEPAsidePredictorMultiSeqWrapper(
|
| 203 |
+
(backbone): PredictorV2(
|
| 204 |
+
(predictor_embed): Linear(in_features=1024, out_features=384, bias=True)
|
| 205 |
+
(mask_tokens): ParameterList(
|
| 206 |
+
(0): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 207 |
+
(1): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 208 |
+
(2): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 209 |
+
(3): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 210 |
+
)
|
| 211 |
+
(predictor_blocks): ModuleList(
|
| 212 |
+
(0-5): 6 x Block(
|
| 213 |
+
(residual1): EfficientResidual(
|
| 214 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 215 |
+
(fn): Attention(
|
| 216 |
+
(q_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 217 |
+
(k_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 218 |
+
(v_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 219 |
+
(proj): Linear(in_features=384, out_features=384, bias=False)
|
| 220 |
+
(rope): Rope()
|
| 221 |
+
)
|
| 222 |
+
)
|
| 223 |
+
(residual2): EfficientResidual(
|
| 224 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 225 |
+
(fn): MLP(
|
| 226 |
+
(fc1): Linear(in_features=384, out_features=1536, bias=False)
|
| 227 |
+
(act): GELU(approximate='none')
|
| 228 |
+
(fc2): Linear(in_features=1536, out_features=384, bias=False)
|
| 229 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 230 |
+
)
|
| 231 |
+
)
|
| 232 |
+
)
|
| 233 |
+
)
|
| 234 |
+
(predictor_norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 235 |
+
(predictor_proj): Linear(in_features=384, out_features=1024, bias=True)
|
| 236 |
+
)
|
| 237 |
+
)
|
| 238 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] MultiSeqWrapper(
|
| 239 |
+
(backbone): Frozen2DTargetWrapper(
|
| 240 |
+
(backbone): Eva(
|
| 241 |
+
(patch_embed): PatchEmbed(
|
| 242 |
+
(proj): Conv2d(3, 1024, kernel_size=(14, 14), stride=(14, 14))
|
| 243 |
+
(norm): Identity()
|
| 244 |
+
)
|
| 245 |
+
(pos_drop): Dropout(p=0.0, inplace=False)
|
| 246 |
+
(norm_pre): Identity()
|
| 247 |
+
(blocks): ModuleList(
|
| 248 |
+
(0-23): 24 x EvaBlock(
|
| 249 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 250 |
+
(attn): EvaAttention(
|
| 251 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 252 |
+
(q_norm): Identity()
|
| 253 |
+
(k_norm): Identity()
|
| 254 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 255 |
+
(norm): Identity()
|
| 256 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=True)
|
| 257 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 258 |
+
)
|
| 259 |
+
(drop_path1): Identity()
|
| 260 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 261 |
+
(mlp): Mlp(
|
| 262 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=True)
|
| 263 |
+
(act): GELU(approximate='none')
|
| 264 |
+
(drop1): Dropout(p=0.0, inplace=False)
|
| 265 |
+
(norm): Identity()
|
| 266 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=True)
|
| 267 |
+
(drop2): Dropout(p=0.0, inplace=False)
|
| 268 |
+
)
|
| 269 |
+
(drop_path2): Identity()
|
| 270 |
+
)
|
| 271 |
+
)
|
| 272 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 273 |
+
(fc_norm): Identity()
|
| 274 |
+
(head_drop): Dropout(p=0.0, inplace=False)
|
| 275 |
+
(head): Identity()
|
| 276 |
+
(rope): _CapiPatchRoPE()
|
| 277 |
+
)
|
| 278 |
+
)
|
| 279 |
+
)
|
| 280 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] Encoder number of parameters: 403136512
|
| 281 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] Predictor number of parameters: 11416192
|
| 282 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] Target encoder number of parameters: 0
|
| 283 |
+
[INFO ][2026-05-14 09:43:36][root ][make_videodataset ] VideoDataset dataset created
|
| 284 |
+
[INFO ][2026-05-14 09:43:36][WeightedSampler ][__init__ ] Using DistributedWeightedSampler with rank 4 / 16
|
| 285 |
+
[INFO ][2026-05-14 09:43:36][root ][make_videodataset ] VideoDataset unsupervised data loader created
|
| 286 |
+
[INFO ][2026-05-14 09:43:36][app.vjepa.train ][main ] iterations per epoch/dataset length: 300/399
|
| 287 |
+
[INFO ][2026-05-14 09:43:36][app.vjepa.train ][main ] Wrapping models in DDP (rank 4)...
|
| 288 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Initializing loader...
|
| 289 |
+
submitit WARNING (2026-05-14 13:38:23,805) - Bypassing signal SIGCONT
|
| 290 |
+
submitit WARNING (2026-05-14 13:38:23,872) - Bypassing signal SIGTERM
|
| 291 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGTERM
|
| 292 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGCONT
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_5_log.err
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/timm/models/layers/__init__.py:49: FutureWarning: Importing from timm.models.layers is deprecated, please import via timm.layers
|
| 2 |
+
warnings.warn(f"Importing from {__name__} is deprecated, please import via timm.layers", FutureWarning)
|
| 3 |
+
[rank5]:[W514 09:35:35.321909479 ProcessGroupNCCL.cpp:5138] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 4 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/utils.py:796: FutureWarning: `torch.cuda.amp.GradScaler(args...)` is deprecated. Please use `torch.amp.GradScaler('cuda', args...)` instead.
|
| 5 |
+
scaler = torch.cuda.amp.GradScaler() if mixed_precision else None
|
| 6 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 7 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 8 |
+
submitit WARNING (2026-05-14 13:38:23,805) - Bypassing signal SIGCONT
|
| 9 |
+
submitit WARNING (2026-05-14 13:38:23,814) - Bypassing signal SIGTERM
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_5_log.out
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
submitit INFO (2026-05-14 09:28:15,038) - Starting with JobEnvironment(job_id=22743106, hostname=gcn79.local.snellius.surf.nl, local_rank=1(4), node=1(4), global_rank=5(16))
|
| 2 |
+
submitit INFO (2026-05-14 09:28:15,038) - Loading pickle: /scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_submitted.pkl
|
| 3 |
+
INFO:root:loaded pretrain params...
|
| 4 |
+
{ 'app': 'vjepa',
|
| 5 |
+
'cpus_per_task': 16,
|
| 6 |
+
'data': { 'batch_size': 64,
|
| 7 |
+
'crop_size': 224,
|
| 8 |
+
'dataset_fpcs': [16, 16],
|
| 9 |
+
'dataset_type': 'VideoDataset',
|
| 10 |
+
'datasets': [ '/scratch-shared/dcanez/data/kinetics/k400/train.csv',
|
| 11 |
+
'/scratch-shared/dcanez/data/ssv2/train.csv'],
|
| 12 |
+
'datasets_weights': [0.65, 0.35],
|
| 13 |
+
'fps': 4,
|
| 14 |
+
'num_workers': 10,
|
| 15 |
+
'patch_size': 14,
|
| 16 |
+
'persistent_workers': True,
|
| 17 |
+
'pin_mem': True,
|
| 18 |
+
'stage': [ { 'dest': 'kinetics_240',
|
| 19 |
+
'format': 'targz_parts',
|
| 20 |
+
'src': '/scratch-shared/dcanez/data/kinetics/k400/tars_240/'},
|
| 21 |
+
{ 'dest': 'ssv2',
|
| 22 |
+
'format': 'multipart_tar',
|
| 23 |
+
'src': '/scratch-nvme/ml-datasets/something-something-v2/'}],
|
| 24 |
+
'tubelet_size': 1},
|
| 25 |
+
'data_aug': { 'auto_augment': False,
|
| 26 |
+
'motion_shift': False,
|
| 27 |
+
'random_resize_aspect_ratio': [0.75, 1.35],
|
| 28 |
+
'random_resize_scale': [0.3, 1.0],
|
| 29 |
+
'reprob': 0.0},
|
| 30 |
+
'folder': '/scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable',
|
| 31 |
+
'loss': {'loss_exp': 1.0},
|
| 32 |
+
'mask': [ { 'aspect_ratio': [0.75, 1.5],
|
| 33 |
+
'full_complement': False,
|
| 34 |
+
'max_keep': None,
|
| 35 |
+
'max_temporal_keep': 1.0,
|
| 36 |
+
'num_blocks': 8,
|
| 37 |
+
'spatial_scale': [0.15, 0.15],
|
| 38 |
+
'temporal_scale': [1.0, 1.0]},
|
| 39 |
+
{ 'aspect_ratio': [0.75, 1.5],
|
| 40 |
+
'full_complement': False,
|
| 41 |
+
'max_keep': None,
|
| 42 |
+
'max_temporal_keep': 1.0,
|
| 43 |
+
'num_blocks': 2,
|
| 44 |
+
'spatial_scale': [0.7, 0.7],
|
| 45 |
+
'temporal_scale': [1.0, 1.0]}],
|
| 46 |
+
'mem_per_gpu': '180G',
|
| 47 |
+
'meta': { 'dtype': 'bfloat16',
|
| 48 |
+
'knn_eval_epoch0': False,
|
| 49 |
+
'knn_eval_freq': 5,
|
| 50 |
+
'knn_eval_presets': [ { 'config': { 'batch_size': 64,
|
| 51 |
+
'dataset_train': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/train.csv',
|
| 52 |
+
'dataset_val': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/val.csv',
|
| 53 |
+
'eval_videos_per_class': 25,
|
| 54 |
+
'num_workers': 8,
|
| 55 |
+
'pool_type': 'slot_temporal_concat',
|
| 56 |
+
'train_videos_per_class': 100},
|
| 57 |
+
'preset': 'ucf101'},
|
| 58 |
+
{ 'config': { 'batch_size': 64,
|
| 59 |
+
'dataset_train': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/train_coarse10.csv',
|
| 60 |
+
'dataset_val': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/val_coarse10.csv',
|
| 61 |
+
'eval_videos_per_class': 100,
|
| 62 |
+
'linear_probe': True,
|
| 63 |
+
'num_workers': 8,
|
| 64 |
+
'pool_type': 'slot_temporal_concat',
|
| 65 |
+
'train_videos_per_class': 500},
|
| 66 |
+
'preset': 'ssv2_coarse10'}],
|
| 67 |
+
'load_checkpoint': True,
|
| 68 |
+
'read_checkpoint': None,
|
| 69 |
+
'save_every_freq': 5,
|
| 70 |
+
'seed': 239,
|
| 71 |
+
'use_sdpa': True,
|
| 72 |
+
'use_wandb': True,
|
| 73 |
+
'wandb_project': 'vjepa_ujepaside'},
|
| 74 |
+
'metrics': {'sigreg': {}, 'std': {}},
|
| 75 |
+
'model': { 'model_name': 'ujepaside_large_patch14_capi_lvd1689m',
|
| 76 |
+
'pred_depth': 6,
|
| 77 |
+
'pred_embed_dim': 384,
|
| 78 |
+
'pred_num_heads': 12,
|
| 79 |
+
'predictor': 'v2_cross',
|
| 80 |
+
'st_causal': False,
|
| 81 |
+
'st_drop_path': 0.2,
|
| 82 |
+
'st_flex_enable': False,
|
| 83 |
+
'st_layer_scale_init': 1e-05,
|
| 84 |
+
'st_num_slots': 16,
|
| 85 |
+
'st_side_block_type': 'factorized',
|
| 86 |
+
'st_slots_causal_within_frame': False,
|
| 87 |
+
'target_kind': 'frozen_2d',
|
| 88 |
+
'target_type': 'vit_large_patch14_capi.lvd1689m',
|
| 89 |
+
'temporal_spacing': 1.0,
|
| 90 |
+
'uniform_power': True,
|
| 91 |
+
'use_activation_checkpointing': True,
|
| 92 |
+
'use_mask_tokens': True,
|
| 93 |
+
'use_rope': True,
|
| 94 |
+
'use_sdpa': True,
|
| 95 |
+
'zero_init_mask_tokens': True},
|
| 96 |
+
'nodes': 4,
|
| 97 |
+
'optimization': { 'clip_grad': 3.0,
|
| 98 |
+
'ema': [0.99925, 0.99925],
|
| 99 |
+
'epochs': 100,
|
| 100 |
+
'final_lr': 0.0001,
|
| 101 |
+
'final_weight_decay': 0.04,
|
| 102 |
+
'ipe': 300,
|
| 103 |
+
'ipe_scale': 1.0,
|
| 104 |
+
'lr': 0.0005,
|
| 105 |
+
'start_lr': 0.0001,
|
| 106 |
+
'warmup': 10,
|
| 107 |
+
'weight_decay': 0.04},
|
| 108 |
+
'tasks_per_node': 4}
|
| 109 |
+
INFO:root:Running pre-training of app: vjepa
|
| 110 |
+
[INFO ][2026-05-14 09:32:17][app.vjepa.train ][main ] which_dtype='bfloat16'
|
| 111 |
+
[INFO ][2026-05-14 09:32:17][app.vjepa.train ][main ] Disabling persistent_workers (incompatible with KNN eval)
|
| 112 |
+
[INFO ][2026-05-14 09:32:17][app.vjepa.train ][main ] NCCL_SOCKET_IFNAME=eno
|
| 113 |
+
[INFO ][2026-05-14 09:32:31][app.vjepa.train ][main ] Initialized (rank/world-size) 5/16, tasks_per_node=4
|
| 114 |
+
[INFO ][2026-05-14 09:32:31][root ][stage_datasets ] [local_rank 1/4] Staging kinetics_240 (targz_parts)
|
| 115 |
+
[INFO ][2026-05-14 09:32:31][root ][_stage_targz_parts ] [rank 1] Extracting 26/103 tar.gz parts to /scratch-node/dcanez.22743106/kinetics_240
|
| 116 |
+
[INFO ][2026-05-14 09:32:43][root ][_stage_targz_parts ] [local_rank 1] Extracted 2/26 parts
|
| 117 |
+
[INFO ][2026-05-14 09:32:57][root ][_stage_targz_parts ] [local_rank 1] Extracted 4/26 parts
|
| 118 |
+
[INFO ][2026-05-14 09:33:11][root ][_stage_targz_parts ] [local_rank 1] Extracted 6/26 parts
|
| 119 |
+
[INFO ][2026-05-14 09:33:25][root ][_stage_targz_parts ] [local_rank 1] Extracted 8/26 parts
|
| 120 |
+
[INFO ][2026-05-14 09:33:39][root ][_stage_targz_parts ] [local_rank 1] Extracted 10/26 parts
|
| 121 |
+
[INFO ][2026-05-14 09:33:54][root ][_stage_targz_parts ] [local_rank 1] Extracted 12/26 parts
|
| 122 |
+
[INFO ][2026-05-14 09:34:07][root ][_stage_targz_parts ] [local_rank 1] Extracted 14/26 parts
|
| 123 |
+
[INFO ][2026-05-14 09:34:22][root ][_stage_targz_parts ] [local_rank 1] Extracted 16/26 parts
|
| 124 |
+
[INFO ][2026-05-14 09:34:36][root ][_stage_targz_parts ] [local_rank 1] Extracted 18/26 parts
|
| 125 |
+
[INFO ][2026-05-14 09:34:51][root ][_stage_targz_parts ] [local_rank 1] Extracted 20/26 parts
|
| 126 |
+
[INFO ][2026-05-14 09:35:06][root ][_stage_targz_parts ] [local_rank 1] Extracted 22/26 parts
|
| 127 |
+
[INFO ][2026-05-14 09:35:20][root ][_stage_targz_parts ] [local_rank 1] Extracted 24/26 parts
|
| 128 |
+
[INFO ][2026-05-14 09:35:35][root ][_stage_targz_parts ] [local_rank 1] Extracted 26/26 parts
|
| 129 |
+
[INFO ][2026-05-14 09:35:35][root ][stage_datasets ] [local_rank 1/4] Staging ssv2 (multipart_tar)
|
| 130 |
+
[INFO ][2026-05-14 09:36:49][app.vjepa.train ][main ] Staged datasets: ['/scratch-node/dcanez.22743106/kinetics_240/train.csv', '/scratch-node/dcanez.22743106/ssv2/train.csv']
|
| 131 |
+
[INFO ][2026-05-14 09:43:14][experiments.stmodels.vision_transformers_v3][vit_large_patch14_capi_lvd1689m] Loading pretrained weights for vit_large_patch14_capi_lvd1689m from https://dl.fbaipublicfiles.com/capi/capi_vitl14_lvd.pth
|
| 132 |
+
[INFO ][2026-05-14 09:43:30][root ][_build_frozen_2d_target ] Loaded frozen_2d target from timm: vit_large_patch14_capi.lvd1689m
|
| 133 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] ViTMultiSeqWrapper(
|
| 134 |
+
(backbone): UJEPAside(
|
| 135 |
+
(patch_embed): PatchEmbed3D(
|
| 136 |
+
(proj): Conv3d(3, 1024, kernel_size=(1, 14, 14), stride=(1, 14, 14))
|
| 137 |
+
)
|
| 138 |
+
(rope): CAPI2DRoPE()
|
| 139 |
+
(blocks): ModuleList(
|
| 140 |
+
(0-23): 24 x Block(
|
| 141 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 142 |
+
(rope_impl): CAPI2DRoPE()
|
| 143 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 144 |
+
(drop_path1): Identity()
|
| 145 |
+
(drop_path2): Identity()
|
| 146 |
+
(attn): Attention(
|
| 147 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 148 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 149 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 150 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 151 |
+
(rope_impl): CAPI2DRoPE()
|
| 152 |
+
)
|
| 153 |
+
(mlp): MLP(
|
| 154 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 155 |
+
(act): GELU(approximate='none')
|
| 156 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 157 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 158 |
+
)
|
| 159 |
+
)
|
| 160 |
+
)
|
| 161 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=False)
|
| 162 |
+
(st_blocks): ModuleList(
|
| 163 |
+
(0-23): 24 x FactorizedSlotSideBlock(
|
| 164 |
+
(cross_attn): EfficientResidual(
|
| 165 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 166 |
+
(fn): Attention(
|
| 167 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 168 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 169 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 170 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 171 |
+
(rope_impl): CAPI3DRoPE()
|
| 172 |
+
)
|
| 173 |
+
)
|
| 174 |
+
(self_attn): EfficientResidual(
|
| 175 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 176 |
+
(fn): Attention(
|
| 177 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 178 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 179 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 180 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 181 |
+
(rope_impl): CAPI3DRoPE()
|
| 182 |
+
)
|
| 183 |
+
)
|
| 184 |
+
(mlp): EfficientResidual(
|
| 185 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 186 |
+
(fn): MLP(
|
| 187 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 188 |
+
(act): GELU(approximate='none')
|
| 189 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 190 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 191 |
+
)
|
| 192 |
+
)
|
| 193 |
+
)
|
| 194 |
+
)
|
| 195 |
+
(st_rope): CAPI3DRoPE()
|
| 196 |
+
)
|
| 197 |
+
)
|
| 198 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] UJEPAsidePredictorMultiSeqWrapper(
|
| 199 |
+
(backbone): PredictorV2(
|
| 200 |
+
(predictor_embed): Linear(in_features=1024, out_features=384, bias=True)
|
| 201 |
+
(mask_tokens): ParameterList(
|
| 202 |
+
(0): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 203 |
+
(1): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 204 |
+
(2): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 205 |
+
(3): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 206 |
+
)
|
| 207 |
+
(predictor_blocks): ModuleList(
|
| 208 |
+
(0-5): 6 x Block(
|
| 209 |
+
(residual1): EfficientResidual(
|
| 210 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 211 |
+
(fn): Attention(
|
| 212 |
+
(q_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 213 |
+
(k_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 214 |
+
(v_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 215 |
+
(proj): Linear(in_features=384, out_features=384, bias=False)
|
| 216 |
+
(rope): Rope()
|
| 217 |
+
)
|
| 218 |
+
)
|
| 219 |
+
(residual2): EfficientResidual(
|
| 220 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 221 |
+
(fn): MLP(
|
| 222 |
+
(fc1): Linear(in_features=384, out_features=1536, bias=False)
|
| 223 |
+
(act): GELU(approximate='none')
|
| 224 |
+
(fc2): Linear(in_features=1536, out_features=384, bias=False)
|
| 225 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 226 |
+
)
|
| 227 |
+
)
|
| 228 |
+
)
|
| 229 |
+
)
|
| 230 |
+
(predictor_norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 231 |
+
(predictor_proj): Linear(in_features=384, out_features=1024, bias=True)
|
| 232 |
+
)
|
| 233 |
+
)
|
| 234 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] MultiSeqWrapper(
|
| 235 |
+
(backbone): Frozen2DTargetWrapper(
|
| 236 |
+
(backbone): Eva(
|
| 237 |
+
(patch_embed): PatchEmbed(
|
| 238 |
+
(proj): Conv2d(3, 1024, kernel_size=(14, 14), stride=(14, 14))
|
| 239 |
+
(norm): Identity()
|
| 240 |
+
)
|
| 241 |
+
(pos_drop): Dropout(p=0.0, inplace=False)
|
| 242 |
+
(norm_pre): Identity()
|
| 243 |
+
(blocks): ModuleList(
|
| 244 |
+
(0-23): 24 x EvaBlock(
|
| 245 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 246 |
+
(attn): EvaAttention(
|
| 247 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 248 |
+
(q_norm): Identity()
|
| 249 |
+
(k_norm): Identity()
|
| 250 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 251 |
+
(norm): Identity()
|
| 252 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=True)
|
| 253 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 254 |
+
)
|
| 255 |
+
(drop_path1): Identity()
|
| 256 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 257 |
+
(mlp): Mlp(
|
| 258 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=True)
|
| 259 |
+
(act): GELU(approximate='none')
|
| 260 |
+
(drop1): Dropout(p=0.0, inplace=False)
|
| 261 |
+
(norm): Identity()
|
| 262 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=True)
|
| 263 |
+
(drop2): Dropout(p=0.0, inplace=False)
|
| 264 |
+
)
|
| 265 |
+
(drop_path2): Identity()
|
| 266 |
+
)
|
| 267 |
+
)
|
| 268 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 269 |
+
(fc_norm): Identity()
|
| 270 |
+
(head_drop): Dropout(p=0.0, inplace=False)
|
| 271 |
+
(head): Identity()
|
| 272 |
+
(rope): _CapiPatchRoPE()
|
| 273 |
+
)
|
| 274 |
+
)
|
| 275 |
+
)
|
| 276 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] Encoder number of parameters: 403136512
|
| 277 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] Predictor number of parameters: 11416192
|
| 278 |
+
[INFO ][2026-05-14 09:43:31][root ][init_video_model ] Target encoder number of parameters: 0
|
| 279 |
+
[INFO ][2026-05-14 09:43:36][root ][make_videodataset ] VideoDataset dataset created
|
| 280 |
+
[INFO ][2026-05-14 09:43:36][WeightedSampler ][__init__ ] Using DistributedWeightedSampler with rank 5 / 16
|
| 281 |
+
[INFO ][2026-05-14 09:43:36][root ][make_videodataset ] VideoDataset unsupervised data loader created
|
| 282 |
+
[INFO ][2026-05-14 09:43:36][app.vjepa.train ][main ] iterations per epoch/dataset length: 300/399
|
| 283 |
+
[INFO ][2026-05-14 09:43:36][app.vjepa.train ][main ] Wrapping models in DDP (rank 5)...
|
| 284 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Initializing loader...
|
| 285 |
+
submitit WARNING (2026-05-14 13:38:23,805) - Bypassing signal SIGCONT
|
| 286 |
+
submitit WARNING (2026-05-14 13:38:23,814) - Bypassing signal SIGTERM
|
| 287 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGTERM
|
| 288 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGCONT
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_6_log.err
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/timm/models/layers/__init__.py:49: FutureWarning: Importing from timm.models.layers is deprecated, please import via timm.layers
|
| 2 |
+
warnings.warn(f"Importing from {__name__} is deprecated, please import via timm.layers", FutureWarning)
|
| 3 |
+
[rank6]:[W514 09:35:36.312177515 ProcessGroupNCCL.cpp:5138] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 4 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/utils.py:796: FutureWarning: `torch.cuda.amp.GradScaler(args...)` is deprecated. Please use `torch.amp.GradScaler('cuda', args...)` instead.
|
| 5 |
+
scaler = torch.cuda.amp.GradScaler() if mixed_precision else None
|
| 6 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 7 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 8 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 9 |
+
submitit WARNING (2026-05-14 13:38:23,836) - Bypassing signal SIGTERM
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_6_log.out
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
submitit INFO (2026-05-14 09:28:15,038) - Starting with JobEnvironment(job_id=22743106, hostname=gcn79.local.snellius.surf.nl, local_rank=2(4), node=1(4), global_rank=6(16))
|
| 2 |
+
submitit INFO (2026-05-14 09:28:15,038) - Loading pickle: /scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_submitted.pkl
|
| 3 |
+
INFO:root:loaded pretrain params...
|
| 4 |
+
{ 'app': 'vjepa',
|
| 5 |
+
'cpus_per_task': 16,
|
| 6 |
+
'data': { 'batch_size': 64,
|
| 7 |
+
'crop_size': 224,
|
| 8 |
+
'dataset_fpcs': [16, 16],
|
| 9 |
+
'dataset_type': 'VideoDataset',
|
| 10 |
+
'datasets': [ '/scratch-shared/dcanez/data/kinetics/k400/train.csv',
|
| 11 |
+
'/scratch-shared/dcanez/data/ssv2/train.csv'],
|
| 12 |
+
'datasets_weights': [0.65, 0.35],
|
| 13 |
+
'fps': 4,
|
| 14 |
+
'num_workers': 10,
|
| 15 |
+
'patch_size': 14,
|
| 16 |
+
'persistent_workers': True,
|
| 17 |
+
'pin_mem': True,
|
| 18 |
+
'stage': [ { 'dest': 'kinetics_240',
|
| 19 |
+
'format': 'targz_parts',
|
| 20 |
+
'src': '/scratch-shared/dcanez/data/kinetics/k400/tars_240/'},
|
| 21 |
+
{ 'dest': 'ssv2',
|
| 22 |
+
'format': 'multipart_tar',
|
| 23 |
+
'src': '/scratch-nvme/ml-datasets/something-something-v2/'}],
|
| 24 |
+
'tubelet_size': 1},
|
| 25 |
+
'data_aug': { 'auto_augment': False,
|
| 26 |
+
'motion_shift': False,
|
| 27 |
+
'random_resize_aspect_ratio': [0.75, 1.35],
|
| 28 |
+
'random_resize_scale': [0.3, 1.0],
|
| 29 |
+
'reprob': 0.0},
|
| 30 |
+
'folder': '/scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable',
|
| 31 |
+
'loss': {'loss_exp': 1.0},
|
| 32 |
+
'mask': [ { 'aspect_ratio': [0.75, 1.5],
|
| 33 |
+
'full_complement': False,
|
| 34 |
+
'max_keep': None,
|
| 35 |
+
'max_temporal_keep': 1.0,
|
| 36 |
+
'num_blocks': 8,
|
| 37 |
+
'spatial_scale': [0.15, 0.15],
|
| 38 |
+
'temporal_scale': [1.0, 1.0]},
|
| 39 |
+
{ 'aspect_ratio': [0.75, 1.5],
|
| 40 |
+
'full_complement': False,
|
| 41 |
+
'max_keep': None,
|
| 42 |
+
'max_temporal_keep': 1.0,
|
| 43 |
+
'num_blocks': 2,
|
| 44 |
+
'spatial_scale': [0.7, 0.7],
|
| 45 |
+
'temporal_scale': [1.0, 1.0]}],
|
| 46 |
+
'mem_per_gpu': '180G',
|
| 47 |
+
'meta': { 'dtype': 'bfloat16',
|
| 48 |
+
'knn_eval_epoch0': False,
|
| 49 |
+
'knn_eval_freq': 5,
|
| 50 |
+
'knn_eval_presets': [ { 'config': { 'batch_size': 64,
|
| 51 |
+
'dataset_train': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/train.csv',
|
| 52 |
+
'dataset_val': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/val.csv',
|
| 53 |
+
'eval_videos_per_class': 25,
|
| 54 |
+
'num_workers': 8,
|
| 55 |
+
'pool_type': 'slot_temporal_concat',
|
| 56 |
+
'train_videos_per_class': 100},
|
| 57 |
+
'preset': 'ucf101'},
|
| 58 |
+
{ 'config': { 'batch_size': 64,
|
| 59 |
+
'dataset_train': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/train_coarse10.csv',
|
| 60 |
+
'dataset_val': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/val_coarse10.csv',
|
| 61 |
+
'eval_videos_per_class': 100,
|
| 62 |
+
'linear_probe': True,
|
| 63 |
+
'num_workers': 8,
|
| 64 |
+
'pool_type': 'slot_temporal_concat',
|
| 65 |
+
'train_videos_per_class': 500},
|
| 66 |
+
'preset': 'ssv2_coarse10'}],
|
| 67 |
+
'load_checkpoint': True,
|
| 68 |
+
'read_checkpoint': None,
|
| 69 |
+
'save_every_freq': 5,
|
| 70 |
+
'seed': 239,
|
| 71 |
+
'use_sdpa': True,
|
| 72 |
+
'use_wandb': True,
|
| 73 |
+
'wandb_project': 'vjepa_ujepaside'},
|
| 74 |
+
'metrics': {'sigreg': {}, 'std': {}},
|
| 75 |
+
'model': { 'model_name': 'ujepaside_large_patch14_capi_lvd1689m',
|
| 76 |
+
'pred_depth': 6,
|
| 77 |
+
'pred_embed_dim': 384,
|
| 78 |
+
'pred_num_heads': 12,
|
| 79 |
+
'predictor': 'v2_cross',
|
| 80 |
+
'st_causal': False,
|
| 81 |
+
'st_drop_path': 0.2,
|
| 82 |
+
'st_flex_enable': False,
|
| 83 |
+
'st_layer_scale_init': 1e-05,
|
| 84 |
+
'st_num_slots': 16,
|
| 85 |
+
'st_side_block_type': 'factorized',
|
| 86 |
+
'st_slots_causal_within_frame': False,
|
| 87 |
+
'target_kind': 'frozen_2d',
|
| 88 |
+
'target_type': 'vit_large_patch14_capi.lvd1689m',
|
| 89 |
+
'temporal_spacing': 1.0,
|
| 90 |
+
'uniform_power': True,
|
| 91 |
+
'use_activation_checkpointing': True,
|
| 92 |
+
'use_mask_tokens': True,
|
| 93 |
+
'use_rope': True,
|
| 94 |
+
'use_sdpa': True,
|
| 95 |
+
'zero_init_mask_tokens': True},
|
| 96 |
+
'nodes': 4,
|
| 97 |
+
'optimization': { 'clip_grad': 3.0,
|
| 98 |
+
'ema': [0.99925, 0.99925],
|
| 99 |
+
'epochs': 100,
|
| 100 |
+
'final_lr': 0.0001,
|
| 101 |
+
'final_weight_decay': 0.04,
|
| 102 |
+
'ipe': 300,
|
| 103 |
+
'ipe_scale': 1.0,
|
| 104 |
+
'lr': 0.0005,
|
| 105 |
+
'start_lr': 0.0001,
|
| 106 |
+
'warmup': 10,
|
| 107 |
+
'weight_decay': 0.04},
|
| 108 |
+
'tasks_per_node': 4}
|
| 109 |
+
INFO:root:Running pre-training of app: vjepa
|
| 110 |
+
[INFO ][2026-05-14 09:32:18][app.vjepa.train ][main ] which_dtype='bfloat16'
|
| 111 |
+
[INFO ][2026-05-14 09:32:18][app.vjepa.train ][main ] Disabling persistent_workers (incompatible with KNN eval)
|
| 112 |
+
[INFO ][2026-05-14 09:32:18][app.vjepa.train ][main ] NCCL_SOCKET_IFNAME=eno
|
| 113 |
+
[INFO ][2026-05-14 09:32:29][app.vjepa.train ][main ] Initialized (rank/world-size) 6/16, tasks_per_node=4
|
| 114 |
+
[INFO ][2026-05-14 09:32:29][root ][stage_datasets ] [local_rank 2/4] Staging kinetics_240 (targz_parts)
|
| 115 |
+
[INFO ][2026-05-14 09:32:29][root ][_stage_targz_parts ] [rank 2] Extracting 26/103 tar.gz parts to /scratch-node/dcanez.22743106/kinetics_240
|
| 116 |
+
[INFO ][2026-05-14 09:32:43][root ][_stage_targz_parts ] [local_rank 2] Extracted 2/26 parts
|
| 117 |
+
[INFO ][2026-05-14 09:32:57][root ][_stage_targz_parts ] [local_rank 2] Extracted 4/26 parts
|
| 118 |
+
[INFO ][2026-05-14 09:33:12][root ][_stage_targz_parts ] [local_rank 2] Extracted 6/26 parts
|
| 119 |
+
[INFO ][2026-05-14 09:33:26][root ][_stage_targz_parts ] [local_rank 2] Extracted 8/26 parts
|
| 120 |
+
[INFO ][2026-05-14 09:33:40][root ][_stage_targz_parts ] [local_rank 2] Extracted 10/26 parts
|
| 121 |
+
[INFO ][2026-05-14 09:33:54][root ][_stage_targz_parts ] [local_rank 2] Extracted 12/26 parts
|
| 122 |
+
[INFO ][2026-05-14 09:34:08][root ][_stage_targz_parts ] [local_rank 2] Extracted 14/26 parts
|
| 123 |
+
[INFO ][2026-05-14 09:34:24][root ][_stage_targz_parts ] [local_rank 2] Extracted 16/26 parts
|
| 124 |
+
[INFO ][2026-05-14 09:34:38][root ][_stage_targz_parts ] [local_rank 2] Extracted 18/26 parts
|
| 125 |
+
[INFO ][2026-05-14 09:34:53][root ][_stage_targz_parts ] [local_rank 2] Extracted 20/26 parts
|
| 126 |
+
[INFO ][2026-05-14 09:35:07][root ][_stage_targz_parts ] [local_rank 2] Extracted 22/26 parts
|
| 127 |
+
[INFO ][2026-05-14 09:35:21][root ][_stage_targz_parts ] [local_rank 2] Extracted 24/26 parts
|
| 128 |
+
[INFO ][2026-05-14 09:35:36][root ][_stage_targz_parts ] [local_rank 2] Extracted 26/26 parts
|
| 129 |
+
[INFO ][2026-05-14 09:35:36][root ][stage_datasets ] [local_rank 2/4] Staging ssv2 (multipart_tar)
|
| 130 |
+
[INFO ][2026-05-14 09:36:49][app.vjepa.train ][main ] Staged datasets: ['/scratch-node/dcanez.22743106/kinetics_240/train.csv', '/scratch-node/dcanez.22743106/ssv2/train.csv']
|
| 131 |
+
[INFO ][2026-05-14 09:43:14][experiments.stmodels.vision_transformers_v3][vit_large_patch14_capi_lvd1689m] Loading pretrained weights for vit_large_patch14_capi_lvd1689m from https://dl.fbaipublicfiles.com/capi/capi_vitl14_lvd.pth
|
| 132 |
+
[INFO ][2026-05-14 09:43:30][root ][_build_frozen_2d_target ] Loaded frozen_2d target from timm: vit_large_patch14_capi.lvd1689m
|
| 133 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] ViTMultiSeqWrapper(
|
| 134 |
+
(backbone): UJEPAside(
|
| 135 |
+
(patch_embed): PatchEmbed3D(
|
| 136 |
+
(proj): Conv3d(3, 1024, kernel_size=(1, 14, 14), stride=(1, 14, 14))
|
| 137 |
+
)
|
| 138 |
+
(rope): CAPI2DRoPE()
|
| 139 |
+
(blocks): ModuleList(
|
| 140 |
+
(0-23): 24 x Block(
|
| 141 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 142 |
+
(rope_impl): CAPI2DRoPE()
|
| 143 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 144 |
+
(drop_path1): Identity()
|
| 145 |
+
(drop_path2): Identity()
|
| 146 |
+
(attn): Attention(
|
| 147 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 148 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 149 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 150 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 151 |
+
(rope_impl): CAPI2DRoPE()
|
| 152 |
+
)
|
| 153 |
+
(mlp): MLP(
|
| 154 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 155 |
+
(act): GELU(approximate='none')
|
| 156 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 157 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 158 |
+
)
|
| 159 |
+
)
|
| 160 |
+
)
|
| 161 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=False)
|
| 162 |
+
(st_blocks): ModuleList(
|
| 163 |
+
(0-23): 24 x FactorizedSlotSideBlock(
|
| 164 |
+
(cross_attn): EfficientResidual(
|
| 165 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 166 |
+
(fn): Attention(
|
| 167 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 168 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 169 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 170 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 171 |
+
(rope_impl): CAPI3DRoPE()
|
| 172 |
+
)
|
| 173 |
+
)
|
| 174 |
+
(self_attn): EfficientResidual(
|
| 175 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 176 |
+
(fn): Attention(
|
| 177 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 178 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 179 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 180 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 181 |
+
(rope_impl): CAPI3DRoPE()
|
| 182 |
+
)
|
| 183 |
+
)
|
| 184 |
+
(mlp): EfficientResidual(
|
| 185 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 186 |
+
(fn): MLP(
|
| 187 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 188 |
+
(act): GELU(approximate='none')
|
| 189 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 190 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 191 |
+
)
|
| 192 |
+
)
|
| 193 |
+
)
|
| 194 |
+
)
|
| 195 |
+
(st_rope): CAPI3DRoPE()
|
| 196 |
+
)
|
| 197 |
+
)
|
| 198 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] UJEPAsidePredictorMultiSeqWrapper(
|
| 199 |
+
(backbone): PredictorV2(
|
| 200 |
+
(predictor_embed): Linear(in_features=1024, out_features=384, bias=True)
|
| 201 |
+
(mask_tokens): ParameterList(
|
| 202 |
+
(0): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 203 |
+
(1): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 204 |
+
(2): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 205 |
+
(3): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 206 |
+
)
|
| 207 |
+
(predictor_blocks): ModuleList(
|
| 208 |
+
(0-5): 6 x Block(
|
| 209 |
+
(residual1): EfficientResidual(
|
| 210 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 211 |
+
(fn): Attention(
|
| 212 |
+
(q_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 213 |
+
(k_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 214 |
+
(v_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 215 |
+
(proj): Linear(in_features=384, out_features=384, bias=False)
|
| 216 |
+
(rope): Rope()
|
| 217 |
+
)
|
| 218 |
+
)
|
| 219 |
+
(residual2): EfficientResidual(
|
| 220 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 221 |
+
(fn): MLP(
|
| 222 |
+
(fc1): Linear(in_features=384, out_features=1536, bias=False)
|
| 223 |
+
(act): GELU(approximate='none')
|
| 224 |
+
(fc2): Linear(in_features=1536, out_features=384, bias=False)
|
| 225 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 226 |
+
)
|
| 227 |
+
)
|
| 228 |
+
)
|
| 229 |
+
)
|
| 230 |
+
(predictor_norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 231 |
+
(predictor_proj): Linear(in_features=384, out_features=1024, bias=True)
|
| 232 |
+
)
|
| 233 |
+
)
|
| 234 |
+
[INFO ][2026-05-14 09:43:31][root ][init_video_model ] MultiSeqWrapper(
|
| 235 |
+
(backbone): Frozen2DTargetWrapper(
|
| 236 |
+
(backbone): Eva(
|
| 237 |
+
(patch_embed): PatchEmbed(
|
| 238 |
+
(proj): Conv2d(3, 1024, kernel_size=(14, 14), stride=(14, 14))
|
| 239 |
+
(norm): Identity()
|
| 240 |
+
)
|
| 241 |
+
(pos_drop): Dropout(p=0.0, inplace=False)
|
| 242 |
+
(norm_pre): Identity()
|
| 243 |
+
(blocks): ModuleList(
|
| 244 |
+
(0-23): 24 x EvaBlock(
|
| 245 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 246 |
+
(attn): EvaAttention(
|
| 247 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 248 |
+
(q_norm): Identity()
|
| 249 |
+
(k_norm): Identity()
|
| 250 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 251 |
+
(norm): Identity()
|
| 252 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=True)
|
| 253 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 254 |
+
)
|
| 255 |
+
(drop_path1): Identity()
|
| 256 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 257 |
+
(mlp): Mlp(
|
| 258 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=True)
|
| 259 |
+
(act): GELU(approximate='none')
|
| 260 |
+
(drop1): Dropout(p=0.0, inplace=False)
|
| 261 |
+
(norm): Identity()
|
| 262 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=True)
|
| 263 |
+
(drop2): Dropout(p=0.0, inplace=False)
|
| 264 |
+
)
|
| 265 |
+
(drop_path2): Identity()
|
| 266 |
+
)
|
| 267 |
+
)
|
| 268 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 269 |
+
(fc_norm): Identity()
|
| 270 |
+
(head_drop): Dropout(p=0.0, inplace=False)
|
| 271 |
+
(head): Identity()
|
| 272 |
+
(rope): _CapiPatchRoPE()
|
| 273 |
+
)
|
| 274 |
+
)
|
| 275 |
+
)
|
| 276 |
+
[INFO ][2026-05-14 09:43:31][root ][init_video_model ] Encoder number of parameters: 403136512
|
| 277 |
+
[INFO ][2026-05-14 09:43:31][root ][init_video_model ] Predictor number of parameters: 11416192
|
| 278 |
+
[INFO ][2026-05-14 09:43:31][root ][init_video_model ] Target encoder number of parameters: 0
|
| 279 |
+
[INFO ][2026-05-14 09:43:36][root ][make_videodataset ] VideoDataset dataset created
|
| 280 |
+
[INFO ][2026-05-14 09:43:36][WeightedSampler ][__init__ ] Using DistributedWeightedSampler with rank 6 / 16
|
| 281 |
+
[INFO ][2026-05-14 09:43:36][root ][make_videodataset ] VideoDataset unsupervised data loader created
|
| 282 |
+
[INFO ][2026-05-14 09:43:36][app.vjepa.train ][main ] iterations per epoch/dataset length: 300/399
|
| 283 |
+
[INFO ][2026-05-14 09:43:36][app.vjepa.train ][main ] Wrapping models in DDP (rank 6)...
|
| 284 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Initializing loader...
|
| 285 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 286 |
+
submitit WARNING (2026-05-14 13:38:23,836) - Bypassing signal SIGTERM
|
| 287 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGTERM
|
| 288 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGCONT
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_7_log.err
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/timm/models/layers/__init__.py:49: FutureWarning: Importing from timm.models.layers is deprecated, please import via timm.layers
|
| 2 |
+
warnings.warn(f"Importing from {__name__} is deprecated, please import via timm.layers", FutureWarning)
|
| 3 |
+
[rank7]:[W514 09:35:32.299194984 ProcessGroupNCCL.cpp:5138] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 4 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/utils.py:796: FutureWarning: `torch.cuda.amp.GradScaler(args...)` is deprecated. Please use `torch.amp.GradScaler('cuda', args...)` instead.
|
| 5 |
+
scaler = torch.cuda.amp.GradScaler() if mixed_precision else None
|
| 6 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 7 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 8 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 9 |
+
submitit WARNING (2026-05-14 13:38:23,818) - Bypassing signal SIGTERM
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_7_log.out
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
submitit INFO (2026-05-14 09:28:15,038) - Starting with JobEnvironment(job_id=22743106, hostname=gcn79.local.snellius.surf.nl, local_rank=3(4), node=1(4), global_rank=7(16))
|
| 2 |
+
submitit INFO (2026-05-14 09:28:15,038) - Loading pickle: /scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_submitted.pkl
|
| 3 |
+
INFO:root:loaded pretrain params...
|
| 4 |
+
{ 'app': 'vjepa',
|
| 5 |
+
'cpus_per_task': 16,
|
| 6 |
+
'data': { 'batch_size': 64,
|
| 7 |
+
'crop_size': 224,
|
| 8 |
+
'dataset_fpcs': [16, 16],
|
| 9 |
+
'dataset_type': 'VideoDataset',
|
| 10 |
+
'datasets': [ '/scratch-shared/dcanez/data/kinetics/k400/train.csv',
|
| 11 |
+
'/scratch-shared/dcanez/data/ssv2/train.csv'],
|
| 12 |
+
'datasets_weights': [0.65, 0.35],
|
| 13 |
+
'fps': 4,
|
| 14 |
+
'num_workers': 10,
|
| 15 |
+
'patch_size': 14,
|
| 16 |
+
'persistent_workers': True,
|
| 17 |
+
'pin_mem': True,
|
| 18 |
+
'stage': [ { 'dest': 'kinetics_240',
|
| 19 |
+
'format': 'targz_parts',
|
| 20 |
+
'src': '/scratch-shared/dcanez/data/kinetics/k400/tars_240/'},
|
| 21 |
+
{ 'dest': 'ssv2',
|
| 22 |
+
'format': 'multipart_tar',
|
| 23 |
+
'src': '/scratch-nvme/ml-datasets/something-something-v2/'}],
|
| 24 |
+
'tubelet_size': 1},
|
| 25 |
+
'data_aug': { 'auto_augment': False,
|
| 26 |
+
'motion_shift': False,
|
| 27 |
+
'random_resize_aspect_ratio': [0.75, 1.35],
|
| 28 |
+
'random_resize_scale': [0.3, 1.0],
|
| 29 |
+
'reprob': 0.0},
|
| 30 |
+
'folder': '/scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable',
|
| 31 |
+
'loss': {'loss_exp': 1.0},
|
| 32 |
+
'mask': [ { 'aspect_ratio': [0.75, 1.5],
|
| 33 |
+
'full_complement': False,
|
| 34 |
+
'max_keep': None,
|
| 35 |
+
'max_temporal_keep': 1.0,
|
| 36 |
+
'num_blocks': 8,
|
| 37 |
+
'spatial_scale': [0.15, 0.15],
|
| 38 |
+
'temporal_scale': [1.0, 1.0]},
|
| 39 |
+
{ 'aspect_ratio': [0.75, 1.5],
|
| 40 |
+
'full_complement': False,
|
| 41 |
+
'max_keep': None,
|
| 42 |
+
'max_temporal_keep': 1.0,
|
| 43 |
+
'num_blocks': 2,
|
| 44 |
+
'spatial_scale': [0.7, 0.7],
|
| 45 |
+
'temporal_scale': [1.0, 1.0]}],
|
| 46 |
+
'mem_per_gpu': '180G',
|
| 47 |
+
'meta': { 'dtype': 'bfloat16',
|
| 48 |
+
'knn_eval_epoch0': False,
|
| 49 |
+
'knn_eval_freq': 5,
|
| 50 |
+
'knn_eval_presets': [ { 'config': { 'batch_size': 64,
|
| 51 |
+
'dataset_train': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/train.csv',
|
| 52 |
+
'dataset_val': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/val.csv',
|
| 53 |
+
'eval_videos_per_class': 25,
|
| 54 |
+
'num_workers': 8,
|
| 55 |
+
'pool_type': 'slot_temporal_concat',
|
| 56 |
+
'train_videos_per_class': 100},
|
| 57 |
+
'preset': 'ucf101'},
|
| 58 |
+
{ 'config': { 'batch_size': 64,
|
| 59 |
+
'dataset_train': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/train_coarse10.csv',
|
| 60 |
+
'dataset_val': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/val_coarse10.csv',
|
| 61 |
+
'eval_videos_per_class': 100,
|
| 62 |
+
'linear_probe': True,
|
| 63 |
+
'num_workers': 8,
|
| 64 |
+
'pool_type': 'slot_temporal_concat',
|
| 65 |
+
'train_videos_per_class': 500},
|
| 66 |
+
'preset': 'ssv2_coarse10'}],
|
| 67 |
+
'load_checkpoint': True,
|
| 68 |
+
'read_checkpoint': None,
|
| 69 |
+
'save_every_freq': 5,
|
| 70 |
+
'seed': 239,
|
| 71 |
+
'use_sdpa': True,
|
| 72 |
+
'use_wandb': True,
|
| 73 |
+
'wandb_project': 'vjepa_ujepaside'},
|
| 74 |
+
'metrics': {'sigreg': {}, 'std': {}},
|
| 75 |
+
'model': { 'model_name': 'ujepaside_large_patch14_capi_lvd1689m',
|
| 76 |
+
'pred_depth': 6,
|
| 77 |
+
'pred_embed_dim': 384,
|
| 78 |
+
'pred_num_heads': 12,
|
| 79 |
+
'predictor': 'v2_cross',
|
| 80 |
+
'st_causal': False,
|
| 81 |
+
'st_drop_path': 0.2,
|
| 82 |
+
'st_flex_enable': False,
|
| 83 |
+
'st_layer_scale_init': 1e-05,
|
| 84 |
+
'st_num_slots': 16,
|
| 85 |
+
'st_side_block_type': 'factorized',
|
| 86 |
+
'st_slots_causal_within_frame': False,
|
| 87 |
+
'target_kind': 'frozen_2d',
|
| 88 |
+
'target_type': 'vit_large_patch14_capi.lvd1689m',
|
| 89 |
+
'temporal_spacing': 1.0,
|
| 90 |
+
'uniform_power': True,
|
| 91 |
+
'use_activation_checkpointing': True,
|
| 92 |
+
'use_mask_tokens': True,
|
| 93 |
+
'use_rope': True,
|
| 94 |
+
'use_sdpa': True,
|
| 95 |
+
'zero_init_mask_tokens': True},
|
| 96 |
+
'nodes': 4,
|
| 97 |
+
'optimization': { 'clip_grad': 3.0,
|
| 98 |
+
'ema': [0.99925, 0.99925],
|
| 99 |
+
'epochs': 100,
|
| 100 |
+
'final_lr': 0.0001,
|
| 101 |
+
'final_weight_decay': 0.04,
|
| 102 |
+
'ipe': 300,
|
| 103 |
+
'ipe_scale': 1.0,
|
| 104 |
+
'lr': 0.0005,
|
| 105 |
+
'start_lr': 0.0001,
|
| 106 |
+
'warmup': 10,
|
| 107 |
+
'weight_decay': 0.04},
|
| 108 |
+
'tasks_per_node': 4}
|
| 109 |
+
INFO:root:Running pre-training of app: vjepa
|
| 110 |
+
[INFO ][2026-05-14 09:32:17][app.vjepa.train ][main ] which_dtype='bfloat16'
|
| 111 |
+
[INFO ][2026-05-14 09:32:17][app.vjepa.train ][main ] Disabling persistent_workers (incompatible with KNN eval)
|
| 112 |
+
[INFO ][2026-05-14 09:32:17][app.vjepa.train ][main ] NCCL_SOCKET_IFNAME=eno
|
| 113 |
+
[INFO ][2026-05-14 09:32:33][app.vjepa.train ][main ] Initialized (rank/world-size) 7/16, tasks_per_node=4
|
| 114 |
+
[INFO ][2026-05-14 09:32:33][root ][stage_datasets ] [local_rank 3/4] Staging kinetics_240 (targz_parts)
|
| 115 |
+
[INFO ][2026-05-14 09:32:33][root ][_stage_targz_parts ] [rank 3] Extracting 25/103 tar.gz parts to /scratch-node/dcanez.22743106/kinetics_240
|
| 116 |
+
[INFO ][2026-05-14 09:32:47][root ][_stage_targz_parts ] [local_rank 3] Extracted 2/25 parts
|
| 117 |
+
[INFO ][2026-05-14 09:33:01][root ][_stage_targz_parts ] [local_rank 3] Extracted 4/25 parts
|
| 118 |
+
[INFO ][2026-05-14 09:33:15][root ][_stage_targz_parts ] [local_rank 3] Extracted 6/25 parts
|
| 119 |
+
[INFO ][2026-05-14 09:33:29][root ][_stage_targz_parts ] [local_rank 3] Extracted 8/25 parts
|
| 120 |
+
[INFO ][2026-05-14 09:33:44][root ][_stage_targz_parts ] [local_rank 3] Extracted 10/25 parts
|
| 121 |
+
[INFO ][2026-05-14 09:33:58][root ][_stage_targz_parts ] [local_rank 3] Extracted 12/25 parts
|
| 122 |
+
[INFO ][2026-05-14 09:34:12][root ][_stage_targz_parts ] [local_rank 3] Extracted 14/25 parts
|
| 123 |
+
[INFO ][2026-05-14 09:34:27][root ][_stage_targz_parts ] [local_rank 3] Extracted 16/25 parts
|
| 124 |
+
[INFO ][2026-05-14 09:34:40][root ][_stage_targz_parts ] [local_rank 3] Extracted 18/25 parts
|
| 125 |
+
[INFO ][2026-05-14 09:34:55][root ][_stage_targz_parts ] [local_rank 3] Extracted 20/25 parts
|
| 126 |
+
[INFO ][2026-05-14 09:35:09][root ][_stage_targz_parts ] [local_rank 3] Extracted 22/25 parts
|
| 127 |
+
[INFO ][2026-05-14 09:35:23][root ][_stage_targz_parts ] [local_rank 3] Extracted 24/25 parts
|
| 128 |
+
[INFO ][2026-05-14 09:35:30][root ][_stage_targz_parts ] [local_rank 3] Extracted 25/25 parts
|
| 129 |
+
[INFO ][2026-05-14 09:35:30][root ][stage_datasets ] [local_rank 3/4] Staging ssv2 (multipart_tar)
|
| 130 |
+
[INFO ][2026-05-14 09:36:49][app.vjepa.train ][main ] Staged datasets: ['/scratch-node/dcanez.22743106/kinetics_240/train.csv', '/scratch-node/dcanez.22743106/ssv2/train.csv']
|
| 131 |
+
[INFO ][2026-05-14 09:43:14][experiments.stmodels.vision_transformers_v3][vit_large_patch14_capi_lvd1689m] Loading pretrained weights for vit_large_patch14_capi_lvd1689m from https://dl.fbaipublicfiles.com/capi/capi_vitl14_lvd.pth
|
| 132 |
+
[INFO ][2026-05-14 09:43:30][root ][_build_frozen_2d_target ] Loaded frozen_2d target from timm: vit_large_patch14_capi.lvd1689m
|
| 133 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] ViTMultiSeqWrapper(
|
| 134 |
+
(backbone): UJEPAside(
|
| 135 |
+
(patch_embed): PatchEmbed3D(
|
| 136 |
+
(proj): Conv3d(3, 1024, kernel_size=(1, 14, 14), stride=(1, 14, 14))
|
| 137 |
+
)
|
| 138 |
+
(rope): CAPI2DRoPE()
|
| 139 |
+
(blocks): ModuleList(
|
| 140 |
+
(0-23): 24 x Block(
|
| 141 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 142 |
+
(rope_impl): CAPI2DRoPE()
|
| 143 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 144 |
+
(drop_path1): Identity()
|
| 145 |
+
(drop_path2): Identity()
|
| 146 |
+
(attn): Attention(
|
| 147 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 148 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 149 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 150 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 151 |
+
(rope_impl): CAPI2DRoPE()
|
| 152 |
+
)
|
| 153 |
+
(mlp): MLP(
|
| 154 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 155 |
+
(act): GELU(approximate='none')
|
| 156 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 157 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 158 |
+
)
|
| 159 |
+
)
|
| 160 |
+
)
|
| 161 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=False)
|
| 162 |
+
(st_blocks): ModuleList(
|
| 163 |
+
(0-23): 24 x FactorizedSlotSideBlock(
|
| 164 |
+
(cross_attn): EfficientResidual(
|
| 165 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 166 |
+
(fn): Attention(
|
| 167 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 168 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 169 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 170 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 171 |
+
(rope_impl): CAPI3DRoPE()
|
| 172 |
+
)
|
| 173 |
+
)
|
| 174 |
+
(self_attn): EfficientResidual(
|
| 175 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 176 |
+
(fn): Attention(
|
| 177 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 178 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 179 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 180 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 181 |
+
(rope_impl): CAPI3DRoPE()
|
| 182 |
+
)
|
| 183 |
+
)
|
| 184 |
+
(mlp): EfficientResidual(
|
| 185 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 186 |
+
(fn): MLP(
|
| 187 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 188 |
+
(act): GELU(approximate='none')
|
| 189 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 190 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 191 |
+
)
|
| 192 |
+
)
|
| 193 |
+
)
|
| 194 |
+
)
|
| 195 |
+
(st_rope): CAPI3DRoPE()
|
| 196 |
+
)
|
| 197 |
+
)
|
| 198 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] UJEPAsidePredictorMultiSeqWrapper(
|
| 199 |
+
(backbone): PredictorV2(
|
| 200 |
+
(predictor_embed): Linear(in_features=1024, out_features=384, bias=True)
|
| 201 |
+
(mask_tokens): ParameterList(
|
| 202 |
+
(0): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 203 |
+
(1): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 204 |
+
(2): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 205 |
+
(3): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 206 |
+
)
|
| 207 |
+
(predictor_blocks): ModuleList(
|
| 208 |
+
(0-5): 6 x Block(
|
| 209 |
+
(residual1): EfficientResidual(
|
| 210 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 211 |
+
(fn): Attention(
|
| 212 |
+
(q_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 213 |
+
(k_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 214 |
+
(v_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 215 |
+
(proj): Linear(in_features=384, out_features=384, bias=False)
|
| 216 |
+
(rope): Rope()
|
| 217 |
+
)
|
| 218 |
+
)
|
| 219 |
+
(residual2): EfficientResidual(
|
| 220 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 221 |
+
(fn): MLP(
|
| 222 |
+
(fc1): Linear(in_features=384, out_features=1536, bias=False)
|
| 223 |
+
(act): GELU(approximate='none')
|
| 224 |
+
(fc2): Linear(in_features=1536, out_features=384, bias=False)
|
| 225 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 226 |
+
)
|
| 227 |
+
)
|
| 228 |
+
)
|
| 229 |
+
)
|
| 230 |
+
(predictor_norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 231 |
+
(predictor_proj): Linear(in_features=384, out_features=1024, bias=True)
|
| 232 |
+
)
|
| 233 |
+
)
|
| 234 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] MultiSeqWrapper(
|
| 235 |
+
(backbone): Frozen2DTargetWrapper(
|
| 236 |
+
(backbone): Eva(
|
| 237 |
+
(patch_embed): PatchEmbed(
|
| 238 |
+
(proj): Conv2d(3, 1024, kernel_size=(14, 14), stride=(14, 14))
|
| 239 |
+
(norm): Identity()
|
| 240 |
+
)
|
| 241 |
+
(pos_drop): Dropout(p=0.0, inplace=False)
|
| 242 |
+
(norm_pre): Identity()
|
| 243 |
+
(blocks): ModuleList(
|
| 244 |
+
(0-23): 24 x EvaBlock(
|
| 245 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 246 |
+
(attn): EvaAttention(
|
| 247 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 248 |
+
(q_norm): Identity()
|
| 249 |
+
(k_norm): Identity()
|
| 250 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 251 |
+
(norm): Identity()
|
| 252 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=True)
|
| 253 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 254 |
+
)
|
| 255 |
+
(drop_path1): Identity()
|
| 256 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 257 |
+
(mlp): Mlp(
|
| 258 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=True)
|
| 259 |
+
(act): GELU(approximate='none')
|
| 260 |
+
(drop1): Dropout(p=0.0, inplace=False)
|
| 261 |
+
(norm): Identity()
|
| 262 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=True)
|
| 263 |
+
(drop2): Dropout(p=0.0, inplace=False)
|
| 264 |
+
)
|
| 265 |
+
(drop_path2): Identity()
|
| 266 |
+
)
|
| 267 |
+
)
|
| 268 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 269 |
+
(fc_norm): Identity()
|
| 270 |
+
(head_drop): Dropout(p=0.0, inplace=False)
|
| 271 |
+
(head): Identity()
|
| 272 |
+
(rope): _CapiPatchRoPE()
|
| 273 |
+
)
|
| 274 |
+
)
|
| 275 |
+
)
|
| 276 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] Encoder number of parameters: 403136512
|
| 277 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] Predictor number of parameters: 11416192
|
| 278 |
+
[INFO ][2026-05-14 09:43:30][root ][init_video_model ] Target encoder number of parameters: 0
|
| 279 |
+
[INFO ][2026-05-14 09:43:36][root ][make_videodataset ] VideoDataset dataset created
|
| 280 |
+
[INFO ][2026-05-14 09:43:36][WeightedSampler ][__init__ ] Using DistributedWeightedSampler with rank 7 / 16
|
| 281 |
+
[INFO ][2026-05-14 09:43:36][root ][make_videodataset ] VideoDataset unsupervised data loader created
|
| 282 |
+
[INFO ][2026-05-14 09:43:36][app.vjepa.train ][main ] iterations per epoch/dataset length: 300/399
|
| 283 |
+
[INFO ][2026-05-14 09:43:36][app.vjepa.train ][main ] Wrapping models in DDP (rank 7)...
|
| 284 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Initializing loader...
|
| 285 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 286 |
+
submitit WARNING (2026-05-14 13:38:23,818) - Bypassing signal SIGTERM
|
| 287 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGTERM
|
| 288 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGCONT
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_8_log.err
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/timm/models/layers/__init__.py:49: FutureWarning: Importing from timm.models.layers is deprecated, please import via timm.layers
|
| 2 |
+
warnings.warn(f"Importing from {__name__} is deprecated, please import via timm.layers", FutureWarning)
|
| 3 |
+
[rank8]:[W514 09:36:45.075654955 ProcessGroupNCCL.cpp:5138] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 4 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/utils.py:796: FutureWarning: `torch.cuda.amp.GradScaler(args...)` is deprecated. Please use `torch.amp.GradScaler('cuda', args...)` instead.
|
| 5 |
+
scaler = torch.cuda.amp.GradScaler() if mixed_precision else None
|
| 6 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 7 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 8 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 9 |
+
submitit WARNING (2026-05-14 13:38:23,807) - Bypassing signal SIGTERM
|
| 10 |
+
[2026-05-14T13:38:58.610] error: namespace_p_join: open failed for /slurm/22743106/.ns: No such file or directory
|
| 11 |
+
[2026-05-14T13:38:58.611] error: namespace_g_join(JobId=22743106 SLUID=s8FXKRJP5N0000): No such file or directory
|
| 12 |
+
[2026-05-14T13:39:00.263] error: Failed to send MESSAGE_TASK_EXIT: Connection refused
|
| 13 |
+
[2026-05-14T13:39:00.265] error: namespace_p_join: open failed for /slurm/22743106/.ns: No such file or directory
|
| 14 |
+
[2026-05-14T13:39:00.265] error: namespace_g_join(JobId=22743106 SLUID=s8FXKRJP5N0000): No such file or directory
|
| 15 |
+
[2026-05-14T13:39:00.403] error: namespace_p_join: open failed for /slurm/22743106/.ns: No such file or directory
|
| 16 |
+
[2026-05-14T13:39:00.404] error: namespace_g_join(JobId=22743106 SLUID=s8FXKRJP5N0000): No such file or directory
|
| 17 |
+
[2026-05-14T13:39:00.441] error: namespace_p_join: open failed for /slurm/22743106/.ns: No such file or directory
|
| 18 |
+
[2026-05-14T13:39:00.442] error: namespace_g_join(JobId=22743106 SLUID=s8FXKRJP5N0000): No such file or directory
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_8_log.out
ADDED
|
@@ -0,0 +1,292 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
submitit INFO (2026-05-14 09:28:14,530) - Starting with JobEnvironment(job_id=22743106, hostname=gcn81.local.snellius.surf.nl, local_rank=0(4), node=2(4), global_rank=8(16))
|
| 2 |
+
submitit INFO (2026-05-14 09:28:14,530) - Loading pickle: /scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_submitted.pkl
|
| 3 |
+
INFO:root:loaded pretrain params...
|
| 4 |
+
{ 'app': 'vjepa',
|
| 5 |
+
'cpus_per_task': 16,
|
| 6 |
+
'data': { 'batch_size': 64,
|
| 7 |
+
'crop_size': 224,
|
| 8 |
+
'dataset_fpcs': [16, 16],
|
| 9 |
+
'dataset_type': 'VideoDataset',
|
| 10 |
+
'datasets': [ '/scratch-shared/dcanez/data/kinetics/k400/train.csv',
|
| 11 |
+
'/scratch-shared/dcanez/data/ssv2/train.csv'],
|
| 12 |
+
'datasets_weights': [0.65, 0.35],
|
| 13 |
+
'fps': 4,
|
| 14 |
+
'num_workers': 10,
|
| 15 |
+
'patch_size': 14,
|
| 16 |
+
'persistent_workers': True,
|
| 17 |
+
'pin_mem': True,
|
| 18 |
+
'stage': [ { 'dest': 'kinetics_240',
|
| 19 |
+
'format': 'targz_parts',
|
| 20 |
+
'src': '/scratch-shared/dcanez/data/kinetics/k400/tars_240/'},
|
| 21 |
+
{ 'dest': 'ssv2',
|
| 22 |
+
'format': 'multipart_tar',
|
| 23 |
+
'src': '/scratch-nvme/ml-datasets/something-something-v2/'}],
|
| 24 |
+
'tubelet_size': 1},
|
| 25 |
+
'data_aug': { 'auto_augment': False,
|
| 26 |
+
'motion_shift': False,
|
| 27 |
+
'random_resize_aspect_ratio': [0.75, 1.35],
|
| 28 |
+
'random_resize_scale': [0.3, 1.0],
|
| 29 |
+
'reprob': 0.0},
|
| 30 |
+
'folder': '/scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable',
|
| 31 |
+
'loss': {'loss_exp': 1.0},
|
| 32 |
+
'mask': [ { 'aspect_ratio': [0.75, 1.5],
|
| 33 |
+
'full_complement': False,
|
| 34 |
+
'max_keep': None,
|
| 35 |
+
'max_temporal_keep': 1.0,
|
| 36 |
+
'num_blocks': 8,
|
| 37 |
+
'spatial_scale': [0.15, 0.15],
|
| 38 |
+
'temporal_scale': [1.0, 1.0]},
|
| 39 |
+
{ 'aspect_ratio': [0.75, 1.5],
|
| 40 |
+
'full_complement': False,
|
| 41 |
+
'max_keep': None,
|
| 42 |
+
'max_temporal_keep': 1.0,
|
| 43 |
+
'num_blocks': 2,
|
| 44 |
+
'spatial_scale': [0.7, 0.7],
|
| 45 |
+
'temporal_scale': [1.0, 1.0]}],
|
| 46 |
+
'mem_per_gpu': '180G',
|
| 47 |
+
'meta': { 'dtype': 'bfloat16',
|
| 48 |
+
'knn_eval_epoch0': False,
|
| 49 |
+
'knn_eval_freq': 5,
|
| 50 |
+
'knn_eval_presets': [ { 'config': { 'batch_size': 64,
|
| 51 |
+
'dataset_train': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/train.csv',
|
| 52 |
+
'dataset_val': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/val.csv',
|
| 53 |
+
'eval_videos_per_class': 25,
|
| 54 |
+
'num_workers': 8,
|
| 55 |
+
'pool_type': 'slot_temporal_concat',
|
| 56 |
+
'train_videos_per_class': 100},
|
| 57 |
+
'preset': 'ucf101'},
|
| 58 |
+
{ 'config': { 'batch_size': 64,
|
| 59 |
+
'dataset_train': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/train_coarse10.csv',
|
| 60 |
+
'dataset_val': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/val_coarse10.csv',
|
| 61 |
+
'eval_videos_per_class': 100,
|
| 62 |
+
'linear_probe': True,
|
| 63 |
+
'num_workers': 8,
|
| 64 |
+
'pool_type': 'slot_temporal_concat',
|
| 65 |
+
'train_videos_per_class': 500},
|
| 66 |
+
'preset': 'ssv2_coarse10'}],
|
| 67 |
+
'load_checkpoint': True,
|
| 68 |
+
'read_checkpoint': None,
|
| 69 |
+
'save_every_freq': 5,
|
| 70 |
+
'seed': 239,
|
| 71 |
+
'use_sdpa': True,
|
| 72 |
+
'use_wandb': True,
|
| 73 |
+
'wandb_project': 'vjepa_ujepaside'},
|
| 74 |
+
'metrics': {'sigreg': {}, 'std': {}},
|
| 75 |
+
'model': { 'model_name': 'ujepaside_large_patch14_capi_lvd1689m',
|
| 76 |
+
'pred_depth': 6,
|
| 77 |
+
'pred_embed_dim': 384,
|
| 78 |
+
'pred_num_heads': 12,
|
| 79 |
+
'predictor': 'v2_cross',
|
| 80 |
+
'st_causal': False,
|
| 81 |
+
'st_drop_path': 0.2,
|
| 82 |
+
'st_flex_enable': False,
|
| 83 |
+
'st_layer_scale_init': 1e-05,
|
| 84 |
+
'st_num_slots': 16,
|
| 85 |
+
'st_side_block_type': 'factorized',
|
| 86 |
+
'st_slots_causal_within_frame': False,
|
| 87 |
+
'target_kind': 'frozen_2d',
|
| 88 |
+
'target_type': 'vit_large_patch14_capi.lvd1689m',
|
| 89 |
+
'temporal_spacing': 1.0,
|
| 90 |
+
'uniform_power': True,
|
| 91 |
+
'use_activation_checkpointing': True,
|
| 92 |
+
'use_mask_tokens': True,
|
| 93 |
+
'use_rope': True,
|
| 94 |
+
'use_sdpa': True,
|
| 95 |
+
'zero_init_mask_tokens': True},
|
| 96 |
+
'nodes': 4,
|
| 97 |
+
'optimization': { 'clip_grad': 3.0,
|
| 98 |
+
'ema': [0.99925, 0.99925],
|
| 99 |
+
'epochs': 100,
|
| 100 |
+
'final_lr': 0.0001,
|
| 101 |
+
'final_weight_decay': 0.04,
|
| 102 |
+
'ipe': 300,
|
| 103 |
+
'ipe_scale': 1.0,
|
| 104 |
+
'lr': 0.0005,
|
| 105 |
+
'start_lr': 0.0001,
|
| 106 |
+
'warmup': 10,
|
| 107 |
+
'weight_decay': 0.04},
|
| 108 |
+
'tasks_per_node': 4}
|
| 109 |
+
INFO:root:Running pre-training of app: vjepa
|
| 110 |
+
[INFO ][2026-05-14 09:32:44][app.vjepa.train ][main ] which_dtype='bfloat16'
|
| 111 |
+
[INFO ][2026-05-14 09:32:44][app.vjepa.train ][main ] Disabling persistent_workers (incompatible with KNN eval)
|
| 112 |
+
[INFO ][2026-05-14 09:32:44][app.vjepa.train ][main ] NCCL_SOCKET_IFNAME=eno
|
| 113 |
+
[INFO ][2026-05-14 09:32:46][app.vjepa.train ][main ] Initialized (rank/world-size) 8/16, tasks_per_node=4
|
| 114 |
+
[INFO ][2026-05-14 09:32:46][root ][stage_datasets ] [local_rank 0/4] Staging kinetics_240 (targz_parts)
|
| 115 |
+
[INFO ][2026-05-14 09:32:46][root ][_stage_targz_parts ] [rank 0] Extracting 26/103 tar.gz parts to /scratch-node/dcanez.22743106/kinetics_240
|
| 116 |
+
[INFO ][2026-05-14 09:33:00][root ][_stage_targz_parts ] [local_rank 0] Extracted 2/26 parts
|
| 117 |
+
[INFO ][2026-05-14 09:33:14][root ][_stage_targz_parts ] [local_rank 0] Extracted 4/26 parts
|
| 118 |
+
[INFO ][2026-05-14 09:33:27][root ][_stage_targz_parts ] [local_rank 0] Extracted 6/26 parts
|
| 119 |
+
[INFO ][2026-05-14 09:33:41][root ][_stage_targz_parts ] [local_rank 0] Extracted 8/26 parts
|
| 120 |
+
[INFO ][2026-05-14 09:33:55][root ][_stage_targz_parts ] [local_rank 0] Extracted 10/26 parts
|
| 121 |
+
[INFO ][2026-05-14 09:34:10][root ][_stage_targz_parts ] [local_rank 0] Extracted 12/26 parts
|
| 122 |
+
[INFO ][2026-05-14 09:34:24][root ][_stage_targz_parts ] [local_rank 0] Extracted 14/26 parts
|
| 123 |
+
[INFO ][2026-05-14 09:34:38][root ][_stage_targz_parts ] [local_rank 0] Extracted 16/26 parts
|
| 124 |
+
[INFO ][2026-05-14 09:34:52][root ][_stage_targz_parts ] [local_rank 0] Extracted 18/26 parts
|
| 125 |
+
[INFO ][2026-05-14 09:35:06][root ][_stage_targz_parts ] [local_rank 0] Extracted 20/26 parts
|
| 126 |
+
[INFO ][2026-05-14 09:35:21][root ][_stage_targz_parts ] [local_rank 0] Extracted 22/26 parts
|
| 127 |
+
[INFO ][2026-05-14 09:35:36][root ][_stage_targz_parts ] [local_rank 0] Extracted 24/26 parts
|
| 128 |
+
[INFO ][2026-05-14 09:35:50][root ][_stage_targz_parts ] [local_rank 0] Extracted 26/26 parts
|
| 129 |
+
[INFO ][2026-05-14 09:35:50][root ][stage_datasets ] [local_rank 0/4] Staging ssv2 (multipart_tar)
|
| 130 |
+
[INFO ][2026-05-14 09:35:50][root ][_stage_multipart_tar ] [rank 0] Extracting multipart tar (2 files) to /scratch-node/dcanez.22743106/ssv2
|
| 131 |
+
[INFO ][2026-05-14 09:36:48][root ][stage_datasets ] Data staging completed in 242.5s (4.0min)
|
| 132 |
+
[INFO ][2026-05-14 09:36:49][root ][_rewrite_csv ] Wrote local CSV: /scratch-node/dcanez.22743106/kinetics_240/train.csv (239789 entries)
|
| 133 |
+
[INFO ][2026-05-14 09:36:49][root ][_rewrite_csv ] Wrote local CSV: /scratch-node/dcanez.22743106/ssv2/train.csv (168913 entries)
|
| 134 |
+
[INFO ][2026-05-14 09:36:49][app.vjepa.train ][main ] Staged datasets: ['/scratch-node/dcanez.22743106/kinetics_240/train.csv', '/scratch-node/dcanez.22743106/ssv2/train.csv']
|
| 135 |
+
[INFO ][2026-05-14 09:43:14][experiments.stmodels.vision_transformers_v3][vit_large_patch14_capi_lvd1689m] Loading pretrained weights for vit_large_patch14_capi_lvd1689m from https://dl.fbaipublicfiles.com/capi/capi_vitl14_lvd.pth
|
| 136 |
+
[INFO ][2026-05-14 09:43:28][root ][_build_frozen_2d_target ] Loaded frozen_2d target from timm: vit_large_patch14_capi.lvd1689m
|
| 137 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] ViTMultiSeqWrapper(
|
| 138 |
+
(backbone): UJEPAside(
|
| 139 |
+
(patch_embed): PatchEmbed3D(
|
| 140 |
+
(proj): Conv3d(3, 1024, kernel_size=(1, 14, 14), stride=(1, 14, 14))
|
| 141 |
+
)
|
| 142 |
+
(rope): CAPI2DRoPE()
|
| 143 |
+
(blocks): ModuleList(
|
| 144 |
+
(0-23): 24 x Block(
|
| 145 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 146 |
+
(rope_impl): CAPI2DRoPE()
|
| 147 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 148 |
+
(drop_path1): Identity()
|
| 149 |
+
(drop_path2): Identity()
|
| 150 |
+
(attn): Attention(
|
| 151 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 152 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 153 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 154 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 155 |
+
(rope_impl): CAPI2DRoPE()
|
| 156 |
+
)
|
| 157 |
+
(mlp): MLP(
|
| 158 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 159 |
+
(act): GELU(approximate='none')
|
| 160 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 161 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 162 |
+
)
|
| 163 |
+
)
|
| 164 |
+
)
|
| 165 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=False)
|
| 166 |
+
(st_blocks): ModuleList(
|
| 167 |
+
(0-23): 24 x FactorizedSlotSideBlock(
|
| 168 |
+
(cross_attn): EfficientResidual(
|
| 169 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 170 |
+
(fn): Attention(
|
| 171 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 172 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 173 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 174 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 175 |
+
(rope_impl): CAPI3DRoPE()
|
| 176 |
+
)
|
| 177 |
+
)
|
| 178 |
+
(self_attn): EfficientResidual(
|
| 179 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 180 |
+
(fn): Attention(
|
| 181 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 182 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 183 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 184 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 185 |
+
(rope_impl): CAPI3DRoPE()
|
| 186 |
+
)
|
| 187 |
+
)
|
| 188 |
+
(mlp): EfficientResidual(
|
| 189 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 190 |
+
(fn): MLP(
|
| 191 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 192 |
+
(act): GELU(approximate='none')
|
| 193 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 194 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 195 |
+
)
|
| 196 |
+
)
|
| 197 |
+
)
|
| 198 |
+
)
|
| 199 |
+
(st_rope): CAPI3DRoPE()
|
| 200 |
+
)
|
| 201 |
+
)
|
| 202 |
+
[INFO ][2026-05-14 09:43:29][root ][init_video_model ] UJEPAsidePredictorMultiSeqWrapper(
|
| 203 |
+
(backbone): PredictorV2(
|
| 204 |
+
(predictor_embed): Linear(in_features=1024, out_features=384, bias=True)
|
| 205 |
+
(mask_tokens): ParameterList(
|
| 206 |
+
(0): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 207 |
+
(1): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 208 |
+
(2): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 209 |
+
(3): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 210 |
+
)
|
| 211 |
+
(predictor_blocks): ModuleList(
|
| 212 |
+
(0-5): 6 x Block(
|
| 213 |
+
(residual1): EfficientResidual(
|
| 214 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 215 |
+
(fn): Attention(
|
| 216 |
+
(q_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 217 |
+
(k_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 218 |
+
(v_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 219 |
+
(proj): Linear(in_features=384, out_features=384, bias=False)
|
| 220 |
+
(rope): Rope()
|
| 221 |
+
)
|
| 222 |
+
)
|
| 223 |
+
(residual2): EfficientResidual(
|
| 224 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 225 |
+
(fn): MLP(
|
| 226 |
+
(fc1): Linear(in_features=384, out_features=1536, bias=False)
|
| 227 |
+
(act): GELU(approximate='none')
|
| 228 |
+
(fc2): Linear(in_features=1536, out_features=384, bias=False)
|
| 229 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 230 |
+
)
|
| 231 |
+
)
|
| 232 |
+
)
|
| 233 |
+
)
|
| 234 |
+
(predictor_norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 235 |
+
(predictor_proj): Linear(in_features=384, out_features=1024, bias=True)
|
| 236 |
+
)
|
| 237 |
+
)
|
| 238 |
+
[INFO ][2026-05-14 09:43:29][root ][init_video_model ] MultiSeqWrapper(
|
| 239 |
+
(backbone): Frozen2DTargetWrapper(
|
| 240 |
+
(backbone): Eva(
|
| 241 |
+
(patch_embed): PatchEmbed(
|
| 242 |
+
(proj): Conv2d(3, 1024, kernel_size=(14, 14), stride=(14, 14))
|
| 243 |
+
(norm): Identity()
|
| 244 |
+
)
|
| 245 |
+
(pos_drop): Dropout(p=0.0, inplace=False)
|
| 246 |
+
(norm_pre): Identity()
|
| 247 |
+
(blocks): ModuleList(
|
| 248 |
+
(0-23): 24 x EvaBlock(
|
| 249 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 250 |
+
(attn): EvaAttention(
|
| 251 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 252 |
+
(q_norm): Identity()
|
| 253 |
+
(k_norm): Identity()
|
| 254 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 255 |
+
(norm): Identity()
|
| 256 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=True)
|
| 257 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 258 |
+
)
|
| 259 |
+
(drop_path1): Identity()
|
| 260 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 261 |
+
(mlp): Mlp(
|
| 262 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=True)
|
| 263 |
+
(act): GELU(approximate='none')
|
| 264 |
+
(drop1): Dropout(p=0.0, inplace=False)
|
| 265 |
+
(norm): Identity()
|
| 266 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=True)
|
| 267 |
+
(drop2): Dropout(p=0.0, inplace=False)
|
| 268 |
+
)
|
| 269 |
+
(drop_path2): Identity()
|
| 270 |
+
)
|
| 271 |
+
)
|
| 272 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 273 |
+
(fc_norm): Identity()
|
| 274 |
+
(head_drop): Dropout(p=0.0, inplace=False)
|
| 275 |
+
(head): Identity()
|
| 276 |
+
(rope): _CapiPatchRoPE()
|
| 277 |
+
)
|
| 278 |
+
)
|
| 279 |
+
)
|
| 280 |
+
[INFO ][2026-05-14 09:43:29][root ][init_video_model ] Encoder number of parameters: 403136512
|
| 281 |
+
[INFO ][2026-05-14 09:43:29][root ][init_video_model ] Predictor number of parameters: 11416192
|
| 282 |
+
[INFO ][2026-05-14 09:43:29][root ][init_video_model ] Target encoder number of parameters: 0
|
| 283 |
+
[INFO ][2026-05-14 09:43:41][root ][make_videodataset ] VideoDataset dataset created
|
| 284 |
+
[INFO ][2026-05-14 09:43:41][WeightedSampler ][__init__ ] Using DistributedWeightedSampler with rank 8 / 16
|
| 285 |
+
[INFO ][2026-05-14 09:43:41][root ][make_videodataset ] VideoDataset unsupervised data loader created
|
| 286 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] iterations per epoch/dataset length: 300/399
|
| 287 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Wrapping models in DDP (rank 8)...
|
| 288 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Initializing loader...
|
| 289 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 290 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGCONT
|
| 291 |
+
submitit WARNING (2026-05-14 13:38:23,807) - Bypassing signal SIGTERM
|
| 292 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGTERM
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_9_log.err
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
/scratch-shared/dcanez/.cache/uv5/virtualenvs/vd/lib/python3.13/site-packages/timm/models/layers/__init__.py:49: FutureWarning: Importing from timm.models.layers is deprecated, please import via timm.layers
|
| 2 |
+
warnings.warn(f"Importing from {__name__} is deprecated, please import via timm.layers", FutureWarning)
|
| 3 |
+
[rank9]:[W514 09:35:49.924907959 ProcessGroupNCCL.cpp:5138] Guessing device ID based on global rank. This can cause a hang if rank to GPU mapping is heterogeneous. You can specify device_id in init_process_group()
|
| 4 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/utils.py:796: FutureWarning: `torch.cuda.amp.GradScaler(args...)` is deprecated. Please use `torch.amp.GradScaler('cuda', args...)` instead.
|
| 5 |
+
scaler = torch.cuda.amp.GradScaler() if mixed_precision else None
|
| 6 |
+
/gpfs/scratch1/shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/code/app/vjepa/train.py:955: FutureWarning: `torch.cuda.amp.autocast(args...)` is deprecated. Please use `torch.amp.autocast('cuda', args...)` instead.
|
| 7 |
+
with torch.cuda.amp.autocast(dtype=dtype, enabled=mixed_precision):
|
| 8 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 9 |
+
submitit WARNING (2026-05-14 13:38:23,821) - Bypassing signal SIGTERM
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_9_log.out
ADDED
|
@@ -0,0 +1,288 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
submitit INFO (2026-05-14 09:28:14,530) - Starting with JobEnvironment(job_id=22743106, hostname=gcn81.local.snellius.surf.nl, local_rank=1(4), node=2(4), global_rank=9(16))
|
| 2 |
+
submitit INFO (2026-05-14 09:28:14,530) - Loading pickle: /scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/job_22743106/22743106_submitted.pkl
|
| 3 |
+
INFO:root:loaded pretrain params...
|
| 4 |
+
{ 'app': 'vjepa',
|
| 5 |
+
'cpus_per_task': 16,
|
| 6 |
+
'data': { 'batch_size': 64,
|
| 7 |
+
'crop_size': 224,
|
| 8 |
+
'dataset_fpcs': [16, 16],
|
| 9 |
+
'dataset_type': 'VideoDataset',
|
| 10 |
+
'datasets': [ '/scratch-shared/dcanez/data/kinetics/k400/train.csv',
|
| 11 |
+
'/scratch-shared/dcanez/data/ssv2/train.csv'],
|
| 12 |
+
'datasets_weights': [0.65, 0.35],
|
| 13 |
+
'fps': 4,
|
| 14 |
+
'num_workers': 10,
|
| 15 |
+
'patch_size': 14,
|
| 16 |
+
'persistent_workers': True,
|
| 17 |
+
'pin_mem': True,
|
| 18 |
+
'stage': [ { 'dest': 'kinetics_240',
|
| 19 |
+
'format': 'targz_parts',
|
| 20 |
+
'src': '/scratch-shared/dcanez/data/kinetics/k400/tars_240/'},
|
| 21 |
+
{ 'dest': 'ssv2',
|
| 22 |
+
'format': 'multipart_tar',
|
| 23 |
+
'src': '/scratch-nvme/ml-datasets/something-something-v2/'}],
|
| 24 |
+
'tubelet_size': 1},
|
| 25 |
+
'data_aug': { 'auto_augment': False,
|
| 26 |
+
'motion_shift': False,
|
| 27 |
+
'random_resize_aspect_ratio': [0.75, 1.35],
|
| 28 |
+
'random_resize_scale': [0.3, 1.0],
|
| 29 |
+
'reprob': 0.0},
|
| 30 |
+
'folder': '/scratch-shared/dcanez/runs/vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable',
|
| 31 |
+
'loss': {'loss_exp': 1.0},
|
| 32 |
+
'mask': [ { 'aspect_ratio': [0.75, 1.5],
|
| 33 |
+
'full_complement': False,
|
| 34 |
+
'max_keep': None,
|
| 35 |
+
'max_temporal_keep': 1.0,
|
| 36 |
+
'num_blocks': 8,
|
| 37 |
+
'spatial_scale': [0.15, 0.15],
|
| 38 |
+
'temporal_scale': [1.0, 1.0]},
|
| 39 |
+
{ 'aspect_ratio': [0.75, 1.5],
|
| 40 |
+
'full_complement': False,
|
| 41 |
+
'max_keep': None,
|
| 42 |
+
'max_temporal_keep': 1.0,
|
| 43 |
+
'num_blocks': 2,
|
| 44 |
+
'spatial_scale': [0.7, 0.7],
|
| 45 |
+
'temporal_scale': [1.0, 1.0]}],
|
| 46 |
+
'mem_per_gpu': '180G',
|
| 47 |
+
'meta': { 'dtype': 'bfloat16',
|
| 48 |
+
'knn_eval_epoch0': False,
|
| 49 |
+
'knn_eval_freq': 5,
|
| 50 |
+
'knn_eval_presets': [ { 'config': { 'batch_size': 64,
|
| 51 |
+
'dataset_train': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/train.csv',
|
| 52 |
+
'dataset_val': '/scratch-shared/mdorkenw/ucf101/ucfTrainTestlist/val.csv',
|
| 53 |
+
'eval_videos_per_class': 25,
|
| 54 |
+
'num_workers': 8,
|
| 55 |
+
'pool_type': 'slot_temporal_concat',
|
| 56 |
+
'train_videos_per_class': 100},
|
| 57 |
+
'preset': 'ucf101'},
|
| 58 |
+
{ 'config': { 'batch_size': 64,
|
| 59 |
+
'dataset_train': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/train_coarse10.csv',
|
| 60 |
+
'dataset_val': '/scratch-shared/mdorkenw/20bn-something-something-v2/something-something-v2-annotations/val_coarse10.csv',
|
| 61 |
+
'eval_videos_per_class': 100,
|
| 62 |
+
'linear_probe': True,
|
| 63 |
+
'num_workers': 8,
|
| 64 |
+
'pool_type': 'slot_temporal_concat',
|
| 65 |
+
'train_videos_per_class': 500},
|
| 66 |
+
'preset': 'ssv2_coarse10'}],
|
| 67 |
+
'load_checkpoint': True,
|
| 68 |
+
'read_checkpoint': None,
|
| 69 |
+
'save_every_freq': 5,
|
| 70 |
+
'seed': 239,
|
| 71 |
+
'use_sdpa': True,
|
| 72 |
+
'use_wandb': True,
|
| 73 |
+
'wandb_project': 'vjepa_ujepaside'},
|
| 74 |
+
'metrics': {'sigreg': {}, 'std': {}},
|
| 75 |
+
'model': { 'model_name': 'ujepaside_large_patch14_capi_lvd1689m',
|
| 76 |
+
'pred_depth': 6,
|
| 77 |
+
'pred_embed_dim': 384,
|
| 78 |
+
'pred_num_heads': 12,
|
| 79 |
+
'predictor': 'v2_cross',
|
| 80 |
+
'st_causal': False,
|
| 81 |
+
'st_drop_path': 0.2,
|
| 82 |
+
'st_flex_enable': False,
|
| 83 |
+
'st_layer_scale_init': 1e-05,
|
| 84 |
+
'st_num_slots': 16,
|
| 85 |
+
'st_side_block_type': 'factorized',
|
| 86 |
+
'st_slots_causal_within_frame': False,
|
| 87 |
+
'target_kind': 'frozen_2d',
|
| 88 |
+
'target_type': 'vit_large_patch14_capi.lvd1689m',
|
| 89 |
+
'temporal_spacing': 1.0,
|
| 90 |
+
'uniform_power': True,
|
| 91 |
+
'use_activation_checkpointing': True,
|
| 92 |
+
'use_mask_tokens': True,
|
| 93 |
+
'use_rope': True,
|
| 94 |
+
'use_sdpa': True,
|
| 95 |
+
'zero_init_mask_tokens': True},
|
| 96 |
+
'nodes': 4,
|
| 97 |
+
'optimization': { 'clip_grad': 3.0,
|
| 98 |
+
'ema': [0.99925, 0.99925],
|
| 99 |
+
'epochs': 100,
|
| 100 |
+
'final_lr': 0.0001,
|
| 101 |
+
'final_weight_decay': 0.04,
|
| 102 |
+
'ipe': 300,
|
| 103 |
+
'ipe_scale': 1.0,
|
| 104 |
+
'lr': 0.0005,
|
| 105 |
+
'start_lr': 0.0001,
|
| 106 |
+
'warmup': 10,
|
| 107 |
+
'weight_decay': 0.04},
|
| 108 |
+
'tasks_per_node': 4}
|
| 109 |
+
INFO:root:Running pre-training of app: vjepa
|
| 110 |
+
[INFO ][2026-05-14 09:32:44][app.vjepa.train ][main ] which_dtype='bfloat16'
|
| 111 |
+
[INFO ][2026-05-14 09:32:44][app.vjepa.train ][main ] Disabling persistent_workers (incompatible with KNN eval)
|
| 112 |
+
[INFO ][2026-05-14 09:32:44][app.vjepa.train ][main ] NCCL_SOCKET_IFNAME=eno
|
| 113 |
+
[INFO ][2026-05-14 09:32:46][app.vjepa.train ][main ] Initialized (rank/world-size) 9/16, tasks_per_node=4
|
| 114 |
+
[INFO ][2026-05-14 09:32:46][root ][stage_datasets ] [local_rank 1/4] Staging kinetics_240 (targz_parts)
|
| 115 |
+
[INFO ][2026-05-14 09:32:46][root ][_stage_targz_parts ] [rank 1] Extracting 26/103 tar.gz parts to /scratch-node/dcanez.22743106/kinetics_240
|
| 116 |
+
[INFO ][2026-05-14 09:32:59][root ][_stage_targz_parts ] [local_rank 1] Extracted 2/26 parts
|
| 117 |
+
[INFO ][2026-05-14 09:33:13][root ][_stage_targz_parts ] [local_rank 1] Extracted 4/26 parts
|
| 118 |
+
[INFO ][2026-05-14 09:33:27][root ][_stage_targz_parts ] [local_rank 1] Extracted 6/26 parts
|
| 119 |
+
[INFO ][2026-05-14 09:33:41][root ][_stage_targz_parts ] [local_rank 1] Extracted 8/26 parts
|
| 120 |
+
[INFO ][2026-05-14 09:33:55][root ][_stage_targz_parts ] [local_rank 1] Extracted 10/26 parts
|
| 121 |
+
[INFO ][2026-05-14 09:34:09][root ][_stage_targz_parts ] [local_rank 1] Extracted 12/26 parts
|
| 122 |
+
[INFO ][2026-05-14 09:34:23][root ][_stage_targz_parts ] [local_rank 1] Extracted 14/26 parts
|
| 123 |
+
[INFO ][2026-05-14 09:34:37][root ][_stage_targz_parts ] [local_rank 1] Extracted 16/26 parts
|
| 124 |
+
[INFO ][2026-05-14 09:34:51][root ][_stage_targz_parts ] [local_rank 1] Extracted 18/26 parts
|
| 125 |
+
[INFO ][2026-05-14 09:35:06][root ][_stage_targz_parts ] [local_rank 1] Extracted 20/26 parts
|
| 126 |
+
[INFO ][2026-05-14 09:35:20][root ][_stage_targz_parts ] [local_rank 1] Extracted 22/26 parts
|
| 127 |
+
[INFO ][2026-05-14 09:35:35][root ][_stage_targz_parts ] [local_rank 1] Extracted 24/26 parts
|
| 128 |
+
[INFO ][2026-05-14 09:35:49][root ][_stage_targz_parts ] [local_rank 1] Extracted 26/26 parts
|
| 129 |
+
[INFO ][2026-05-14 09:35:49][root ][stage_datasets ] [local_rank 1/4] Staging ssv2 (multipart_tar)
|
| 130 |
+
[INFO ][2026-05-14 09:36:49][app.vjepa.train ][main ] Staged datasets: ['/scratch-node/dcanez.22743106/kinetics_240/train.csv', '/scratch-node/dcanez.22743106/ssv2/train.csv']
|
| 131 |
+
[INFO ][2026-05-14 09:43:14][experiments.stmodels.vision_transformers_v3][vit_large_patch14_capi_lvd1689m] Loading pretrained weights for vit_large_patch14_capi_lvd1689m from https://dl.fbaipublicfiles.com/capi/capi_vitl14_lvd.pth
|
| 132 |
+
[INFO ][2026-05-14 09:43:28][root ][_build_frozen_2d_target ] Loaded frozen_2d target from timm: vit_large_patch14_capi.lvd1689m
|
| 133 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] ViTMultiSeqWrapper(
|
| 134 |
+
(backbone): UJEPAside(
|
| 135 |
+
(patch_embed): PatchEmbed3D(
|
| 136 |
+
(proj): Conv3d(3, 1024, kernel_size=(1, 14, 14), stride=(1, 14, 14))
|
| 137 |
+
)
|
| 138 |
+
(rope): CAPI2DRoPE()
|
| 139 |
+
(blocks): ModuleList(
|
| 140 |
+
(0-23): 24 x Block(
|
| 141 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 142 |
+
(rope_impl): CAPI2DRoPE()
|
| 143 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 144 |
+
(drop_path1): Identity()
|
| 145 |
+
(drop_path2): Identity()
|
| 146 |
+
(attn): Attention(
|
| 147 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 148 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 149 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 150 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 151 |
+
(rope_impl): CAPI2DRoPE()
|
| 152 |
+
)
|
| 153 |
+
(mlp): MLP(
|
| 154 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 155 |
+
(act): GELU(approximate='none')
|
| 156 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 157 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 158 |
+
)
|
| 159 |
+
)
|
| 160 |
+
)
|
| 161 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=False)
|
| 162 |
+
(st_blocks): ModuleList(
|
| 163 |
+
(0-23): 24 x FactorizedSlotSideBlock(
|
| 164 |
+
(cross_attn): EfficientResidual(
|
| 165 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 166 |
+
(fn): Attention(
|
| 167 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 168 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 169 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 170 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 171 |
+
(rope_impl): CAPI3DRoPE()
|
| 172 |
+
)
|
| 173 |
+
)
|
| 174 |
+
(self_attn): EfficientResidual(
|
| 175 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 176 |
+
(fn): Attention(
|
| 177 |
+
(q_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 178 |
+
(k_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 179 |
+
(v_proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 180 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=False)
|
| 181 |
+
(rope_impl): CAPI3DRoPE()
|
| 182 |
+
)
|
| 183 |
+
)
|
| 184 |
+
(mlp): EfficientResidual(
|
| 185 |
+
(norm): LayerNorm((1024,), eps=1e-06, elementwise_affine=True)
|
| 186 |
+
(fn): MLP(
|
| 187 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=False)
|
| 188 |
+
(act): GELU(approximate='none')
|
| 189 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=False)
|
| 190 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 191 |
+
)
|
| 192 |
+
)
|
| 193 |
+
)
|
| 194 |
+
)
|
| 195 |
+
(st_rope): CAPI3DRoPE()
|
| 196 |
+
)
|
| 197 |
+
)
|
| 198 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] UJEPAsidePredictorMultiSeqWrapper(
|
| 199 |
+
(backbone): PredictorV2(
|
| 200 |
+
(predictor_embed): Linear(in_features=1024, out_features=384, bias=True)
|
| 201 |
+
(mask_tokens): ParameterList(
|
| 202 |
+
(0): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 203 |
+
(1): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 204 |
+
(2): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 205 |
+
(3): Parameter containing: [torch.float32 of size 1x384 (cuda:0)]
|
| 206 |
+
)
|
| 207 |
+
(predictor_blocks): ModuleList(
|
| 208 |
+
(0-5): 6 x Block(
|
| 209 |
+
(residual1): EfficientResidual(
|
| 210 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 211 |
+
(fn): Attention(
|
| 212 |
+
(q_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 213 |
+
(k_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 214 |
+
(v_proj): Linear(in_features=384, out_features=384, bias=False)
|
| 215 |
+
(proj): Linear(in_features=384, out_features=384, bias=False)
|
| 216 |
+
(rope): Rope()
|
| 217 |
+
)
|
| 218 |
+
)
|
| 219 |
+
(residual2): EfficientResidual(
|
| 220 |
+
(norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 221 |
+
(fn): MLP(
|
| 222 |
+
(fc1): Linear(in_features=384, out_features=1536, bias=False)
|
| 223 |
+
(act): GELU(approximate='none')
|
| 224 |
+
(fc2): Linear(in_features=1536, out_features=384, bias=False)
|
| 225 |
+
(drop): Dropout(p=0.0, inplace=False)
|
| 226 |
+
)
|
| 227 |
+
)
|
| 228 |
+
)
|
| 229 |
+
)
|
| 230 |
+
(predictor_norm): LayerNorm((384,), eps=1e-05, elementwise_affine=True)
|
| 231 |
+
(predictor_proj): Linear(in_features=384, out_features=1024, bias=True)
|
| 232 |
+
)
|
| 233 |
+
)
|
| 234 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] MultiSeqWrapper(
|
| 235 |
+
(backbone): Frozen2DTargetWrapper(
|
| 236 |
+
(backbone): Eva(
|
| 237 |
+
(patch_embed): PatchEmbed(
|
| 238 |
+
(proj): Conv2d(3, 1024, kernel_size=(14, 14), stride=(14, 14))
|
| 239 |
+
(norm): Identity()
|
| 240 |
+
)
|
| 241 |
+
(pos_drop): Dropout(p=0.0, inplace=False)
|
| 242 |
+
(norm_pre): Identity()
|
| 243 |
+
(blocks): ModuleList(
|
| 244 |
+
(0-23): 24 x EvaBlock(
|
| 245 |
+
(norm1): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 246 |
+
(attn): EvaAttention(
|
| 247 |
+
(qkv): Linear(in_features=1024, out_features=3072, bias=False)
|
| 248 |
+
(q_norm): Identity()
|
| 249 |
+
(k_norm): Identity()
|
| 250 |
+
(attn_drop): Dropout(p=0.0, inplace=False)
|
| 251 |
+
(norm): Identity()
|
| 252 |
+
(proj): Linear(in_features=1024, out_features=1024, bias=True)
|
| 253 |
+
(proj_drop): Dropout(p=0.0, inplace=False)
|
| 254 |
+
)
|
| 255 |
+
(drop_path1): Identity()
|
| 256 |
+
(norm2): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 257 |
+
(mlp): Mlp(
|
| 258 |
+
(fc1): Linear(in_features=1024, out_features=4096, bias=True)
|
| 259 |
+
(act): GELU(approximate='none')
|
| 260 |
+
(drop1): Dropout(p=0.0, inplace=False)
|
| 261 |
+
(norm): Identity()
|
| 262 |
+
(fc2): Linear(in_features=4096, out_features=1024, bias=True)
|
| 263 |
+
(drop2): Dropout(p=0.0, inplace=False)
|
| 264 |
+
)
|
| 265 |
+
(drop_path2): Identity()
|
| 266 |
+
)
|
| 267 |
+
)
|
| 268 |
+
(norm): RMSNorm((1024,), eps=1e-05, elementwise_affine=True)
|
| 269 |
+
(fc_norm): Identity()
|
| 270 |
+
(head_drop): Dropout(p=0.0, inplace=False)
|
| 271 |
+
(head): Identity()
|
| 272 |
+
(rope): _CapiPatchRoPE()
|
| 273 |
+
)
|
| 274 |
+
)
|
| 275 |
+
)
|
| 276 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] Encoder number of parameters: 403136512
|
| 277 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] Predictor number of parameters: 11416192
|
| 278 |
+
[INFO ][2026-05-14 09:43:28][root ][init_video_model ] Target encoder number of parameters: 0
|
| 279 |
+
[INFO ][2026-05-14 09:43:41][root ][make_videodataset ] VideoDataset dataset created
|
| 280 |
+
[INFO ][2026-05-14 09:43:41][WeightedSampler ][__init__ ] Using DistributedWeightedSampler with rank 9 / 16
|
| 281 |
+
[INFO ][2026-05-14 09:43:41][root ][make_videodataset ] VideoDataset unsupervised data loader created
|
| 282 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] iterations per epoch/dataset length: 300/399
|
| 283 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Wrapping models in DDP (rank 9)...
|
| 284 |
+
[INFO ][2026-05-14 09:43:41][app.vjepa.train ][main ] Initializing loader...
|
| 285 |
+
submitit WARNING (2026-05-14 13:38:23,806) - Bypassing signal SIGCONT
|
| 286 |
+
submitit WARNING (2026-05-14 13:38:23,821) - Bypassing signal SIGTERM
|
| 287 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGTERM
|
| 288 |
+
[WARNING ][2026-05-14 13:38:23][submitit ][bypass ] Bypassing signal SIGCONT
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/latest.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:dd7a6989d7d0453d4de239cf6a9436e31d37e7a1df58195db5332c43e5d10c3e
|
| 3 |
+
size 7397341953
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r0.csv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r1.csv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r10.csv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r11.csv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r12.csv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r13.csv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r14.csv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r15.csv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r2.csv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r3.csv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r4.csv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r5.csv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r6.csv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r7.csv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
vjepa/ujepaside_capi_lvd/c003_vitl_k16_simple_cross_factorized_stable/log_r8.csv
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|