ckadirt commited on
Commit
4917273
·
verified ·
1 Parent(s): 095178e

Add files using upload-large-folder tool

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +5 -0
  2. fMRI-foundation-model/src/wandb/run-20241023_034538-HCPflat_large_gsrFalse__HCP_FT_83810/files/output.log +13 -0
  3. fMRI-foundation-model/src/wandb/run-20241023_034538-HCPflat_large_gsrFalse__HCP_FT_83810/files/requirements.txt +198 -0
  4. fMRI-foundation-model/src/wandb/run-20241023_034538-HCPflat_large_gsrFalse__HCP_FT_83810/files/wandb-metadata.json +144 -0
  5. fMRI-foundation-model/src/wandb/run-20241023_034538-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug-core.log +8 -0
  6. fMRI-foundation-model/src/wandb/run-20241023_034538-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug-internal.log +15 -0
  7. fMRI-foundation-model/src/wandb/run-20241023_034538-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug.log +34 -0
  8. fMRI-foundation-model/src/wandb/run-20241023_034538-HCPflat_large_gsrFalse__HCP_FT_83810/run-HCPflat_large_gsrFalse__HCP_FT_83810.wandb +0 -0
  9. fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/files/code/src/HCP_downstream_finetune.py +587 -0
  10. fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/files/output.log +1 -0
  11. fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/files/requirements.txt +198 -0
  12. fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/files/wandb-metadata.json +131 -0
  13. fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug-core.log +14 -0
  14. fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug-internal.log +11 -0
  15. fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug.log +26 -0
  16. fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/run-HCPflat_large_gsrFalse__HCP_FT_83810.wandb +0 -0
  17. fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/files/code/src/HCP_downstream_finetune.py +596 -0
  18. fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/files/output.log +1 -0
  19. fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/files/requirements.txt +198 -0
  20. fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/files/wandb-metadata.json +131 -0
  21. fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/logs/debug-core.log +14 -0
  22. fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/logs/debug-internal.log +11 -0
  23. fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/logs/debug.log +25 -0
  24. fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/run-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0.wandb +0 -0
  25. fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/files/code/src/HCP_downstream_finetune.py +596 -0
  26. fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/files/config.yaml +47 -0
  27. fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/files/output.log +147 -0
  28. fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/files/wandb-metadata.json +131 -0
  29. fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/files/wandb-summary.json +1 -0
  30. fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/logs/debug-core.log +14 -0
  31. fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/logs/debug-internal.log +19 -0
  32. fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/logs/debug.log +26 -0
  33. fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/run-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115.wandb +3 -0
  34. fMRI-foundation-model/src/wandb/run-20241023_150329-NSDflat_large_gsrFalse__HCP_FT_7920feb1-ec83-45eb-9cb6-844266415eba/logs/debug-core.log +7 -0
  35. fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/files/code/src/HCP_downstream_finetune.py +597 -0
  36. fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/files/config.yaml +47 -0
  37. fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/files/output.log +0 -0
  38. fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/files/wandb-metadata.json +131 -0
  39. fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/files/wandb-summary.json +1 -0
  40. fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/logs/debug-core.log +21 -0
  41. fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/logs/debug-internal.log +19 -0
  42. fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/logs/debug.log +26 -0
  43. fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/run-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7.wandb +3 -0
  44. fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/files/config.yaml +48 -0
  45. fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/files/output.log +209 -0
  46. fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/files/wandb-metadata.json +144 -0
  47. fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/files/wandb-summary.json +1 -0
  48. fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/logs/debug-core.log +16 -0
  49. fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/logs/debug-internal.log +21 -0
  50. fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/logs/debug.log +61 -0
.gitattributes CHANGED
@@ -5024,3 +5024,8 @@ fMRI-foundation-model/src/wandb/run-20241023_150216-HCPflat_large_gsrFalse__HCP_
5024
  fMRI-foundation-model/src/wandb/run-20241023_132722-NSDflat_large_gsrFalse__HCP_FT_30b3db5a-7076-4c47-a68a-ca00d1834e02/run-NSDflat_large_gsrFalse__HCP_FT_30b3db5a-7076-4c47-a68a-ca00d1834e02.wandb filter=lfs diff=lfs merge=lfs -text
5025
  fMRI-foundation-model/src/wandb/run-20241024_213905-NSDflat_large_gsrFalse__HCP_FT_5af28c50-b56e-467a-b00a-48ff6a643457/run-NSDflat_large_gsrFalse__HCP_FT_5af28c50-b56e-467a-b00a-48ff6a643457.wandb filter=lfs diff=lfs merge=lfs -text
5026
  fMRI-foundation-model/src/wandb/run-20241023_150329-NSDflat_large_gsrFalse__HCP_FT_7920feb1-ec83-45eb-9cb6-844266415eba/run-NSDflat_large_gsrFalse__HCP_FT_7920feb1-ec83-45eb-9cb6-844266415eba.wandb filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
5024
  fMRI-foundation-model/src/wandb/run-20241023_132722-NSDflat_large_gsrFalse__HCP_FT_30b3db5a-7076-4c47-a68a-ca00d1834e02/run-NSDflat_large_gsrFalse__HCP_FT_30b3db5a-7076-4c47-a68a-ca00d1834e02.wandb filter=lfs diff=lfs merge=lfs -text
5025
  fMRI-foundation-model/src/wandb/run-20241024_213905-NSDflat_large_gsrFalse__HCP_FT_5af28c50-b56e-467a-b00a-48ff6a643457/run-NSDflat_large_gsrFalse__HCP_FT_5af28c50-b56e-467a-b00a-48ff6a643457.wandb filter=lfs diff=lfs merge=lfs -text
5026
  fMRI-foundation-model/src/wandb/run-20241023_150329-NSDflat_large_gsrFalse__HCP_FT_7920feb1-ec83-45eb-9cb6-844266415eba/run-NSDflat_large_gsrFalse__HCP_FT_7920feb1-ec83-45eb-9cb6-844266415eba.wandb filter=lfs diff=lfs merge=lfs -text
5027
+ fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/run-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7.wandb filter=lfs diff=lfs merge=lfs -text
5028
+ fMRI-foundation-model/src/wandb/run-20241126_204427-HCPflat_raw_beta_sex_83810/run-HCPflat_raw_beta_sex_83810.wandb filter=lfs diff=lfs merge=lfs -text
5029
+ fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/run-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9.wandb filter=lfs diff=lfs merge=lfs -text
5030
+ fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/run-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115.wandb filter=lfs diff=lfs merge=lfs -text
5031
+ fMRI-foundation-model/src/wandb/run-20241126_221003-HCPflat_raw_beta_age_9a3e14f1-ec90-47c9-a06e-a395872f2271/run-HCPflat_raw_beta_age_9a3e14f1-ec90-47c9-a06e-a395872f2271.wandb filter=lfs diff=lfs merge=lfs -text
fMRI-foundation-model/src/wandb/run-20241023_034538-HCPflat_large_gsrFalse__HCP_FT_83810/files/output.log ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Step [100/13913] - Training Loss: 1.9649 - Training Accuracy: 59.38%
2
+ Step [200/13913] - Training Loss: 1.0904 - Training Accuracy: 71.19%
3
+ Step [300/13913] - Training Loss: 0.0775 - Training Accuracy: 75.50%
4
+ Step [400/13913] - Training Loss: 0.6052 - Training Accuracy: 78.84%
5
+ Step [500/13913] - Training Loss: 0.0226 - Training Accuracy: 80.95%
6
+ Step [600/13913] - Training Loss: 0.2728 - Training Accuracy: 82.54%
7
+ Step [700/13913] - Training Loss: 0.1662 - Training Accuracy: 83.70%
8
+ Exception ignored in: <bound method IPythonKernel._clean_thread_parent_frames of <ipykernel.ipkernel.IPythonKernel object at 0x7fd53caf2590>>
9
+ Traceback (most recent call last):
10
+ File "/admin/home-ckadirt/foundation_env/lib/python3.11/site-packages/ipykernel/ipkernel.py", line 775, in _clean_thread_parent_frames
11
+ def _clean_thread_parent_frames(
12
+
13
+ KeyboardInterrupt:
fMRI-foundation-model/src/wandb/run-20241023_034538-HCPflat_large_gsrFalse__HCP_FT_83810/files/requirements.txt ADDED
@@ -0,0 +1,198 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ protobuf==5.28.2
2
+ imageio==2.35.1
3
+ MarkupSafe==3.0.0
4
+ regex==2024.9.11
5
+ matplotlib==3.9.2
6
+ notebook==7.2.2
7
+ debugpy==1.8.6
8
+ aiosignal==1.3.1
9
+ jupyter_core==5.7.2
10
+ torchaudio==2.4.1+cu121
11
+ python-json-logger==2.0.7
12
+ six==1.16.0
13
+ scikit-image==0.24.0
14
+ types-python-dateutil==2.9.0.20241003
15
+ PyYAML==6.0.2
16
+ httpcore==1.0.6
17
+ clip==1.0
18
+ babel==2.16.0
19
+ webcolors==24.8.0
20
+ omegaconf==2.3.0
21
+ webencodings==0.5.1
22
+ kiwisolver==1.4.7
23
+ uri-template==1.3.0
24
+ diffusers==0.23.0
25
+ idna==3.10
26
+ fsspec==2024.9.0
27
+ parso==0.8.4
28
+ setuptools==65.5.0
29
+ tornado==6.4.1
30
+ webdataset==0.2.100
31
+ decord==0.6.0
32
+ nvidia-curand-cu12==10.3.2.106
33
+ ipykernel==6.29.5
34
+ jupyter==1.1.1
35
+ pexpect==4.9.0
36
+ kornia_rs==0.1.5
37
+ iopath==0.1.10
38
+ async-lru==2.0.4
39
+ future==1.0.0
40
+ torchvision==0.19.1+cu121
41
+ botocore==1.34.162
42
+ cycler==0.12.1
43
+ tzdata==2024.2
44
+ jupyter_server_terminals==0.5.3
45
+ click==8.1.7
46
+ einops==0.8.0
47
+ pyzmq==26.2.0
48
+ jupyter_client==8.6.3
49
+ nbconvert==7.16.4
50
+ scikit-learn==1.5.2
51
+ executing==2.1.0
52
+ asttokens==2.4.1
53
+ docker-pycreds==0.4.0
54
+ matplotlib-inline==0.1.7
55
+ overrides==7.7.0
56
+ websocket-client==1.8.0
57
+ nbformat==5.10.4
58
+ elbow==0.1.1
59
+ contourpy==1.3.0
60
+ nvidia-cudnn-cu12==9.1.0.70
61
+ transformers==4.44.2
62
+ gitdb==4.0.11
63
+ jupyterlab_nvdashboard==0.11.0
64
+ lazy_loader==0.4
65
+ jsonpointer==3.0.0
66
+ notebook_shim==0.2.4
67
+ nvidia-nccl-cu12==2.20.5
68
+ ffmpeg-python==0.2.0
69
+ triton==3.0.0
70
+ mistune==3.0.2
71
+ python-dateutil==2.9.0.post0
72
+ beautifulsoup4==4.12.3
73
+ nbclient==0.10.0
74
+ h5py==3.12.1
75
+ ftfy==6.2.3
76
+ zipp==3.20.2
77
+ ptyprocess==0.7.0
78
+ huggingface-hub==0.25.1
79
+ pytz==2024.2
80
+ jupyterlab_pygments==0.3.0
81
+ nvidia-cublas-cu12==12.1.3.1
82
+ pandocfilters==1.5.1
83
+ Jinja2==3.1.4
84
+ arrow==1.3.0
85
+ rpds-py==0.20.0
86
+ jupyter_server==2.14.2
87
+ simplejson==3.19.3
88
+ networkx==3.3
89
+ packaging==24.1
90
+ traitlets==5.14.3
91
+ pandas==2.2.3
92
+ xformers==0.0.22.post7
93
+ lightning-utilities==0.11.7
94
+ tifffile==2024.9.20
95
+ nvidia-cuda-cupti-cu12==12.1.105
96
+ mpmath==1.3.0
97
+ GitPython==3.1.43
98
+ scipy==1.14.1
99
+ jsonschema==4.23.0
100
+ prompt_toolkit==3.0.48
101
+ s3transfer==0.10.2
102
+ multidict==6.1.0
103
+ bleach==6.1.0
104
+ sentry-sdk==2.15.0
105
+ nibabel==5.2.1
106
+ accelerate==1.0.0
107
+ pyarrow==17.0.0
108
+ threadpoolctl==3.5.0
109
+ attrs==24.2.0
110
+ rfc3986-validator==0.1.1
111
+ nvidia-cuda-runtime-cu12==12.1.105
112
+ ipywidgets==8.1.5
113
+ frozenlist==1.4.1
114
+ pycparser==2.22
115
+ jupyterlab_server==2.27.3
116
+ nvidia-cuda-nvrtc-cu12==12.1.105
117
+ yarl==1.13.1
118
+ setproctitle==1.3.3
119
+ isoduration==20.11.0
120
+ Pygments==2.18.0
121
+ jedi==0.19.1
122
+ boto3==1.34.57
123
+ tokenizers==0.19.1
124
+ referencing==0.35.1
125
+ rfc3339-validator==0.1.4
126
+ pillow==10.4.0
127
+ jupyterlab==4.2.5
128
+ stack-data==0.6.3
129
+ h11==0.14.0
130
+ anyio==4.6.0
131
+ nilearn==0.10.4
132
+ nvidia-cusolver-cu12==11.4.5.107
133
+ tinycss2==1.3.0
134
+ defusedxml==0.7.1
135
+ argon2-cffi-bindings==21.2.0
136
+ soupsieve==2.6
137
+ nest-asyncio==1.6.0
138
+ torchmetrics==1.3.0.post0
139
+ tqdm==4.66.5
140
+ cffi==1.17.1
141
+ charset-normalizer==3.3.2
142
+ jsonschema-specifications==2023.12.1
143
+ decorator==5.1.1
144
+ open_clip_torch==2.26.1
145
+ jupyter-events==0.10.0
146
+ smart-open==7.0.5
147
+ antlr4-python3-runtime==4.9.3
148
+ prometheus_client==0.21.0
149
+ kornia==0.7.3
150
+ typing_extensions==4.12.2
151
+ sniffio==1.3.1
152
+ joblib==1.4.2
153
+ comm==0.2.2
154
+ aiohappyeyeballs==2.4.3
155
+ numpy==2.1.2
156
+ braceexpand==0.1.7
157
+ certifi==2024.8.30
158
+ psutil==6.0.0
159
+ pyparsing==3.1.4
160
+ pure_eval==0.2.3
161
+ nvidia-cusparse-cu12==12.1.0.106
162
+ wandb==0.18.3
163
+ urllib3==2.2.3
164
+ smmap==5.0.1
165
+ platformdirs==4.3.6
166
+ torch==2.4.1+cu121
167
+ requests==2.32.3
168
+ json5==0.9.25
169
+ nvidia-nvjitlink-cu12==12.6.77
170
+ jupyterlab_widgets==3.0.13
171
+ lxml==5.3.0
172
+ httpx==0.27.2
173
+ opencv-python==4.6.0.66
174
+ portalocker==2.10.1
175
+ pytorch-lightning==2.0.1
176
+ sympy==1.13.3
177
+ wcwidth==0.2.13
178
+ jmespath==1.0.1
179
+ fqdn==1.5.1
180
+ pynvml==11.5.3
181
+ pip==24.0
182
+ wrapt==1.16.0
183
+ aiohttp==3.10.9
184
+ filelock==3.16.1
185
+ fonttools==4.54.1
186
+ fastjsonschema==2.20.0
187
+ jupyter-console==6.6.3
188
+ widgetsnbextension==4.0.13
189
+ timm==1.0.9
190
+ nvidia-cufft-cu12==11.0.2.54
191
+ ipython==8.28.0
192
+ nvidia-nvtx-cu12==12.1.105
193
+ jupyter-lsp==2.2.5
194
+ safetensors==0.4.5
195
+ terminado==0.18.1
196
+ argon2-cffi==23.1.0
197
+ Send2Trash==1.8.3
198
+ importlib_metadata==8.5.0
fMRI-foundation-model/src/wandb/run-20241023_034538-HCPflat_large_gsrFalse__HCP_FT_83810/files/wandb-metadata.json ADDED
@@ -0,0 +1,144 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-5.15.0-1058-aws-x86_64-with-glibc2.31",
3
+ "python": "3.11.10",
4
+ "startedAt": "2024-10-23T03:45:38.205708Z",
5
+ "program": "ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.ipynb",
6
+ "git": {
7
+ "remote": "https://github.com/MedARC-AI/fMRI-foundation-model",
8
+ "commit": "b1ba684ae7a5cc4155cc046b0abe613de09bf700"
9
+ },
10
+ "email": "torrico.villanueva.cesar.kadir@gmail.com",
11
+ "root": "/weka/proj-fmri/ckadirt/fMRI-foundation-model/src",
12
+ "host": "ip-10-0-160-143",
13
+ "username": "ckadirt",
14
+ "executable": "/admin/home-ckadirt/foundation_env/bin/python",
15
+ "cpu_count": 96,
16
+ "cpu_count_logical": 192,
17
+ "gpu": "[NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3]",
18
+ "gpu_count": 8,
19
+ "disk": {
20
+ "/": {
21
+ "total": "249555763200",
22
+ "used": "185007661056"
23
+ }
24
+ },
25
+ "memory": {
26
+ "total": "2147443429376"
27
+ },
28
+ "cpu": {
29
+ "count": 96,
30
+ "countLogical": 192
31
+ },
32
+ "gpu_nvidia": [
33
+ {
34
+ "name": "NVIDIA H100 80GB HBM3",
35
+ "memoryTotal": "85520809984",
36
+ "cudaCores": 16896,
37
+ "architecture": "Hopper"
38
+ },
39
+ {
40
+ "name": "NVIDIA H100 80GB HBM3",
41
+ "memoryTotal": "85520809984",
42
+ "cudaCores": 16896,
43
+ "architecture": "Hopper"
44
+ },
45
+ {
46
+ "name": "NVIDIA H100 80GB HBM3",
47
+ "memoryTotal": "85520809984",
48
+ "cudaCores": 16896,
49
+ "architecture": "Hopper"
50
+ },
51
+ {
52
+ "name": "NVIDIA H100 80GB HBM3",
53
+ "memoryTotal": "85520809984",
54
+ "cudaCores": 16896,
55
+ "architecture": "Hopper"
56
+ },
57
+ {
58
+ "name": "NVIDIA H100 80GB HBM3",
59
+ "memoryTotal": "85520809984",
60
+ "cudaCores": 16896,
61
+ "architecture": "Hopper"
62
+ },
63
+ {
64
+ "name": "NVIDIA H100 80GB HBM3",
65
+ "memoryTotal": "85520809984",
66
+ "cudaCores": 16896,
67
+ "architecture": "Hopper"
68
+ },
69
+ {
70
+ "name": "NVIDIA H100 80GB HBM3",
71
+ "memoryTotal": "85520809984",
72
+ "cudaCores": 16896,
73
+ "architecture": "Hopper"
74
+ },
75
+ {
76
+ "name": "NVIDIA H100 80GB HBM3",
77
+ "memoryTotal": "85520809984",
78
+ "cudaCores": 16896,
79
+ "architecture": "Hopper"
80
+ }
81
+ ],
82
+ "slurm": {
83
+ "cluster_name": "sagemaker2",
84
+ "conf": "/opt/slurm/etc/slurm.conf",
85
+ "cpu_bind": "quiet,mask_cpu:0x00000000000000FFC000000000000000000000FFC0000000",
86
+ "cpu_bind_list": "0x00000000000000FFC000000000000000000000FFC0000000",
87
+ "cpu_bind_type": "mask_cpu:",
88
+ "cpu_bind_verbose": "quiet",
89
+ "cpus_on_node": "20",
90
+ "gpus": "1",
91
+ "gpus_on_node": "1",
92
+ "gtids": "0",
93
+ "job_account": "fmri",
94
+ "job_cpus_per_node": "20",
95
+ "job_end_time": "1729702756",
96
+ "job_gid": "1879800513",
97
+ "job_group": "Domain Users",
98
+ "job_id": "528040",
99
+ "job_name": "bash",
100
+ "job_nodelist": "ip-10-0-160-143",
101
+ "job_num_nodes": "1",
102
+ "job_partition": "p5",
103
+ "job_qos": "idle",
104
+ "job_start_time": "1729648756",
105
+ "job_uid": "1879804696",
106
+ "job_user": "ckadirt",
107
+ "jobid": "528040",
108
+ "launch_node_ipaddr": "172.17.12.61",
109
+ "localid": "0",
110
+ "mpi_type": "pmix_v3",
111
+ "nnodes": "1",
112
+ "nodeid": "0",
113
+ "nodelist": "ip-10-0-160-143",
114
+ "nprocs": "1",
115
+ "ntasks": "1",
116
+ "pmix_mapping_serv": "(vector,(0,1,1))",
117
+ "pmixp_abort_agent_port": "34923",
118
+ "prio_process": "0",
119
+ "procid": "0",
120
+ "pty_port": "45733",
121
+ "pty_win_col": "199",
122
+ "pty_win_row": "17",
123
+ "script_context": "prolog_task",
124
+ "srun_comm_host": "172.17.12.61",
125
+ "srun_comm_port": "39353",
126
+ "step_gpus": "3",
127
+ "step_id": "0",
128
+ "step_launcher_port": "39353",
129
+ "step_nodelist": "ip-10-0-160-143",
130
+ "step_num_nodes": "1",
131
+ "step_num_tasks": "1",
132
+ "step_tasks_per_node": "1",
133
+ "stepid": "0",
134
+ "submit_dir": "/weka/proj-fmri",
135
+ "submit_host": "ip-172-17-12-61",
136
+ "task_pid": "1032669",
137
+ "tasks_per_node": "1",
138
+ "topology_addr": "ip-10-0-160-143",
139
+ "topology_addr_pattern": "node",
140
+ "umask": "0022",
141
+ "working_cluster": "sagemaker2:ip-172-17-63-161:6817:9984:109"
142
+ },
143
+ "cudaVersion": "12.2"
144
+ }
fMRI-foundation-model/src/wandb/run-20241023_034538-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug-core.log ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {"time":"2024-10-23T03:45:37.696324079Z","level":"INFO","msg":"started logging, with flags","port-filename":"/tmp/tmpa2p5p65y/port-1071663.txt","pid":1071663,"debug":false,"disable-analytics":false}
2
+ {"time":"2024-10-23T03:45:37.69686141Z","level":"INFO","msg":"FeatureState","shutdownOnParentExitEnabled":false}
3
+ {"time":"2024-10-23T03:45:37.702133186Z","level":"INFO","msg":"Will exit if parent process dies.","ppid":1071663}
4
+ {"time":"2024-10-23T03:45:37.702124516Z","level":"INFO","msg":"server is running","addr":{"IP":"127.0.0.1","Port":37937,"Zone":""}}
5
+ {"time":"2024-10-23T03:45:37.71474024Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"127.0.0.1:38772"}
6
+ {"time":"2024-10-23T03:45:38.211534697Z","level":"INFO","msg":"handleInformInit: received","streamId":"HCPflat_large_gsrFalse__HCP_FT_83810","id":"127.0.0.1:38772"}
7
+ {"time":"2024-10-23T03:45:38.27080459Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"HCPflat_large_gsrFalse__HCP_FT_83810","id":"127.0.0.1:38772"}
8
+ {"time":"2024-10-23T03:56:22.055440969Z","level":"INFO","msg":"Parent process exited, terminating service process."}
fMRI-foundation-model/src/wandb/run-20241023_034538-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug-internal.log ADDED
@@ -0,0 +1,15 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2024-10-23T03:45:38.220044239Z","level":"INFO","msg":"using version","core version":"0.18.3"}
2
+ {"time":"2024-10-23T03:45:38.220064129Z","level":"INFO","msg":"created symlink","path":"/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241023_034538-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug-core.log"}
3
+ {"time":"2024-10-23T03:45:38.231848146Z","level":"ERROR","msg":"dialing: google: could not find default credentials. See https://cloud.google.com/docs/authentication/external/set-up-adc for more information"}
4
+ {"time":"2024-10-23T03:45:38.270767589Z","level":"INFO","msg":"created new stream","id":"HCPflat_large_gsrFalse__HCP_FT_83810"}
5
+ {"time":"2024-10-23T03:45:38.27079841Z","level":"INFO","msg":"stream: started","id":"HCPflat_large_gsrFalse__HCP_FT_83810"}
6
+ {"time":"2024-10-23T03:45:38.270827351Z","level":"INFO","msg":"sender: started","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_83810"}}
7
+ {"time":"2024-10-23T03:45:38.270836011Z","level":"INFO","msg":"handler: started","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_83810"}}
8
+ {"time":"2024-10-23T03:45:38.27081347Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_83810"}}
9
+ {"time":"2024-10-23T03:45:38.797445279Z","level":"INFO","msg":"wandb-core","!BADKEY":null}
10
+ {"time":"2024-10-23T03:45:38.803689894Z","level":"INFO","msg":"Starting system monitor"}
11
+ {"time":"2024-10-23T03:45:38.803706865Z","level":"WARN","msg":"handleCodeSave: program relative path is empty"}
12
+ {"time":"2024-10-23T03:45:38.805537671Z","level":"ERROR","msg":"git repo not found","error":"repository does not exist"}
13
+ {"time":"2024-10-23T03:45:39.178414895Z","level":"INFO","msg":"Pausing system monitor"}
14
+ {"time":"2024-10-23T03:45:39.238514165Z","level":"INFO","msg":"Resuming system monitor"}
15
+ {"time":"2024-10-23T03:50:56.294050139Z","level":"INFO","msg":"Pausing system monitor"}
fMRI-foundation-model/src/wandb/run-20241023_034538-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug.log ADDED
@@ -0,0 +1,34 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 2024-10-23 03:45:38,191 INFO MainThread:1071663 [wandb_setup.py:_flush():79] Current SDK version is 0.18.3
2
+ 2024-10-23 03:45:38,191 INFO MainThread:1071663 [wandb_setup.py:_flush():79] Configure stats pid to 1071663
3
+ 2024-10-23 03:45:38,191 INFO MainThread:1071663 [wandb_setup.py:_flush():79] Loading settings from /admin/home-ckadirt/.config/wandb/settings
4
+ 2024-10-23 03:45:38,191 INFO MainThread:1071663 [wandb_setup.py:_flush():79] Loading settings from /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/settings
5
+ 2024-10-23 03:45:38,191 INFO MainThread:1071663 [wandb_setup.py:_flush():79] Loading settings from environment variables: {}
6
+ 2024-10-23 03:45:38,191 INFO MainThread:1071663 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': None, '_disable_service': None}
7
+ 2024-10-23 03:45:38,191 INFO MainThread:1071663 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program': '<python with no main file>'}
8
+ 2024-10-23 03:45:38,191 INFO MainThread:1071663 [wandb_setup.py:_flush():79] Applying login settings: {}
9
+ 2024-10-23 03:45:38,192 INFO MainThread:1071663 [wandb_init.py:_log_setup():532] Logging user logs to /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241023_034538-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug.log
10
+ 2024-10-23 03:45:38,193 INFO MainThread:1071663 [wandb_init.py:_log_setup():533] Logging internal logs to /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241023_034538-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug-internal.log
11
+ 2024-10-23 03:45:38,193 INFO MainThread:1071663 [wandb_init.py:_jupyter_setup():478] configuring jupyter hooks <wandb.sdk.wandb_init._WandbInit object at 0x7fa8701eb110>
12
+ 2024-10-23 03:45:38,194 INFO MainThread:1071663 [wandb_init.py:init():617] calling init triggers
13
+ 2024-10-23 03:45:38,194 INFO MainThread:1071663 [wandb_init.py:init():624] wandb.init called with sweep_config: {}
14
+ config: {'model_name': 'HCPflat_large_gsrFalse__HCP_FT', 'batch_size': 8, 'learning_rate': 0.0001, 'weight_decay': 1e-05, 'num_epochs': 20, 'seed': 42}
15
+ 2024-10-23 03:45:38,194 INFO MainThread:1071663 [wandb_init.py:init():667] starting backend
16
+ 2024-10-23 03:45:38,194 INFO MainThread:1071663 [wandb_init.py:init():671] sending inform_init request
17
+ 2024-10-23 03:45:38,204 INFO MainThread:1071663 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
18
+ 2024-10-23 03:45:38,204 INFO MainThread:1071663 [wandb_init.py:init():684] backend started and connected
19
+ 2024-10-23 03:45:38,228 INFO MainThread:1071663 [wandb_run.py:_label_probe_notebook():1346] probe notebook
20
+ 2024-10-23 03:45:38,228 INFO MainThread:1071663 [wandb_run.py:_label_probe_notebook():1356] Unable to probe notebook: 'NoneType' object has no attribute 'get'
21
+ 2024-10-23 03:45:38,228 INFO MainThread:1071663 [wandb_init.py:init():779] updated telemetry
22
+ 2024-10-23 03:45:38,289 INFO MainThread:1071663 [wandb_init.py:init():812] communicating run to backend with 90.0 second timeout
23
+ 2024-10-23 03:45:38,736 INFO MainThread:1071663 [wandb_init.py:init():855] run resumed
24
+ 2024-10-23 03:45:38,781 INFO MainThread:1071663 [wandb_init.py:init():863] starting run threads in backend
25
+ 2024-10-23 03:45:39,138 INFO MainThread:1071663 [wandb_run.py:_console_start():2465] atexit reg
26
+ 2024-10-23 03:45:39,139 INFO MainThread:1071663 [wandb_run.py:_redirect():2313] redirect: wrap_raw
27
+ 2024-10-23 03:45:39,139 INFO MainThread:1071663 [wandb_run.py:_redirect():2378] Wrapping output streams.
28
+ 2024-10-23 03:45:39,139 INFO MainThread:1071663 [wandb_run.py:_redirect():2403] Redirects installed.
29
+ 2024-10-23 03:45:39,142 INFO MainThread:1071663 [wandb_init.py:init():907] run started, returning control to user process
30
+ 2024-10-23 03:45:39,148 INFO MainThread:1071663 [jupyter.py:_save_ipynb():398] looking for notebook: ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.ipynb
31
+ 2024-10-23 03:45:39,149 INFO MainThread:1071663 [wandb_init.py:_pause_backend():443] pausing backend
32
+ 2024-10-23 03:45:39,238 INFO MainThread:1071663 [wandb_init.py:_resume_backend():448] resuming backend
33
+ 2024-10-23 03:50:56,288 INFO MainThread:1071663 [jupyter.py:_save_ipynb():398] looking for notebook: ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.ipynb
34
+ 2024-10-23 03:50:56,291 INFO MainThread:1071663 [wandb_init.py:_pause_backend():443] pausing backend
fMRI-foundation-model/src/wandb/run-20241023_034538-HCPflat_large_gsrFalse__HCP_FT_83810/run-HCPflat_large_gsrFalse__HCP_FT_83810.wandb ADDED
Binary file (360 kB). View file
 
fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/files/code/src/HCP_downstream_finetune.py ADDED
@@ -0,0 +1,587 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python
2
+ # coding: utf-8
3
+
4
+ # In[1]:
5
+
6
+
7
+ # Import packages and setup gpu configuration.
8
+ # This code block shouldnt need to be adjusted!
9
+ import os
10
+ import sys
11
+ import json
12
+ import yaml
13
+ import numpy as np
14
+ import copy
15
+ import math
16
+ import time
17
+ import random
18
+ from tqdm.auto import tqdm
19
+ import webdataset as wds
20
+ import matplotlib.pyplot as plt
21
+
22
+ import torch
23
+ import torch.nn as nn
24
+ from torchvision import transforms
25
+ import utils
26
+ from mae_utils.flat_models import *
27
+ import h5py
28
+ from mae_utils import flat_models
29
+
30
+ # tf32 data type is faster than standard float32
31
+ torch.backends.cuda.matmul.allow_tf32 = True
32
+ # following fixes a Conv3D CUDNN_NOT_SUPPORTED error
33
+ torch.backends.cudnn.benchmark = True
34
+
35
+ # ## MODEL TO LOAD ##
36
+ if utils.is_interactive():
37
+ model_name = "HCPflat_large_gsrFalse_"
38
+ else:
39
+ model_name = sys.argv[1]
40
+
41
+
42
+ # outdir = os.path.abspath(f'checkpoints/{model_name}')
43
+ outdir = os.path.abspath(f'checkpoints/{model_name}')
44
+
45
+ print("outdir", outdir)
46
+ # Load previous config.yaml if available
47
+ if os.path.exists(f"{outdir}/config.yaml"):
48
+ config = yaml.load(open(f"{outdir}/config.yaml", 'r'), Loader=yaml.FullLoader)
49
+ print(f"Loaded config.yaml from ckpt folder {outdir}")
50
+ # create global variables from the config
51
+ print("\n__CONFIG__")
52
+ for attribute_name in config.keys():
53
+ print(f"{attribute_name} = {config[attribute_name]}")
54
+ globals()[attribute_name] = config[f'{attribute_name}']
55
+ print("\n")
56
+
57
+ world_size = os.getenv('WORLD_SIZE')
58
+ if world_size is None:
59
+ world_size = 1
60
+ else:
61
+ world_size = int(world_size)
62
+ print(f"WORLD_SIZE={world_size}")
63
+
64
+ if utils.is_interactive():
65
+ # Following allows you to change functions in models.py or utils.py and
66
+ # have this notebook automatically update with your revisions
67
+ get_ipython().run_line_magic('load_ext', 'autoreload')
68
+ get_ipython().run_line_magic('autoreload', '2')
69
+
70
+ batch_size = probe_batch_size
71
+ num_epochs = probe_num_epochs
72
+
73
+ data_type = torch.float32 # change depending on your mixed_precision
74
+ global_batch_size = batch_size * world_size
75
+
76
+ device = torch.device('cuda')
77
+
78
+ hcp_flat_path = "/weka/proj-medarc/shared/HCP-Flat"
79
+ # seed = 42
80
+ # num_frames = 16
81
+ # gsr = False
82
+ # num_workers = 10
83
+ # batch_size = 128
84
+
85
+ print("PID of this process =",os.getpid())
86
+ utils.seed_everything(seed)
87
+
88
+
89
+ # In[2]:
90
+
91
+
92
+ if os.getenv('global_pool') == "False":
93
+ global_pool = False
94
+ else:
95
+ global_pool = True
96
+ print(f"global_pool = {global_pool}")
97
+
98
+ try:
99
+ gsr
100
+ except:
101
+ gsr = True
102
+ print("set gsr to True")
103
+ print(f"gsr = {gsr}")
104
+
105
+
106
+ # In[3]:
107
+
108
+
109
+ #### UNCOMMENT THIS TO SAVE THE HCP-FLAT IN HDF5 FORMAT
110
+
111
+
112
+ # from torch.utils.data import default_collate
113
+ # from mae_utils.flat import load_hcp_flat_mask
114
+ # from mae_utils.flat import create_hcp_flat
115
+ # from mae_utils.flat import batch_unmask
116
+ # import mae_utils.visualize as vis
117
+
118
+
119
+ # batch_size = 26
120
+ # print(f"changed batch_size to {batch_size}")
121
+
122
+ # ## Test ##
123
+ # datasets_to_include = "HCP"
124
+ # assert "HCP" in datasets_to_include
125
+ # test_dataset = create_hcp_flat(root=hcp_flat_path,
126
+ # clip_mode="event", frames=num_frames, shuffle=False, gsr=gsr, sub_list = 'test')
127
+ # test_dl = wds.WebLoader(
128
+ # test_dataset.batched(batch_size, partial=False, collation_fn=default_collate),
129
+ # batch_size=None,
130
+ # shuffle=False,
131
+ # num_workers=num_workers,
132
+ # pin_memory=True,
133
+ # )
134
+
135
+ # ## Train ##
136
+ # assert "HCP" in datasets_to_include
137
+ # train_dataset = create_hcp_flat(root=hcp_flat_path,
138
+ # clip_mode="event", frames=num_frames, shuffle=False, gsr=gsr, sub_list = 'train')
139
+ # train_dl = wds.WebLoader(
140
+ # train_dataset.batched(batch_size, partial=False, collation_fn=default_collate),
141
+ # batch_size=None,
142
+ # shuffle=False,
143
+ # num_workers=num_workers,
144
+ # pin_memory=True,
145
+ # )
146
+
147
+ # def flatten_meta(meta_dict):
148
+ # """
149
+ # Flatten the meta dictionary by:
150
+ # - Replacing single-item lists with the item itself.
151
+ # - Converting tensors to scalar numbers.
152
+ # """
153
+ # flattened = {}
154
+ # for key, value in meta_dict.items():
155
+ # if isinstance(value, list):
156
+ # if len(value) == 1:
157
+ # flattened[key] = value[0] # Replace list with its single item
158
+ # else:
159
+ # flattened[key] = value # Keep as is if multiple items
160
+ # elif isinstance(value, torch.Tensor):
161
+ # # Convert tensor to scalar
162
+ # if value.numel() == 1:
163
+ # flattened[key] = value.item()
164
+ # else:
165
+ # flattened[key] = value.tolist() # Convert multi-element tensor to list
166
+ # else:
167
+ # flattened[key] = value # Keep the value as is
168
+ # return flattened
169
+
170
+ # import h5py
171
+ # meta_array = np.array([], dtype=object)
172
+ # # Open an HDF5 file in write mode
173
+ # with h5py.File('train_hcp.hdf5', 'w') as h5f:
174
+ # flatmaps_dset = None
175
+
176
+ # total_samples = 0
177
+
178
+ # for i, batch in tqdm(enumerate(train_dl), total = 120000):
179
+ # images = batch['image'][0]
180
+ # meta = batch['meta']
181
+ # batch_size = images.shape[0]
182
+ # meta_serializable = meta.copy()
183
+
184
+
185
+ # # Step 2: Serialize the dictionary to a JSON string
186
+ # meta_str = json.dumps(flatten_meta(meta_serializable), indent=4)
187
+ # meta_array = np.append(meta_array, meta_str)
188
+ # if flatmaps_dset is None:
189
+ # # Initialize datasets with unlimited (None) maxshape along the first axis
190
+ # flatmaps_shape = (0,) + images.shape[1:]
191
+ # flatmaps_maxshape = (None,) + images.shape[1:]
192
+
193
+ # flatmaps_dset = h5f.create_dataset(
194
+ # 'flatmaps',
195
+ # shape=flatmaps_shape,
196
+ # maxshape=flatmaps_maxshape,
197
+ # dtype=np.float16,
198
+ # chunks=True # Enable chunking for efficient resizing
199
+ # )
200
+
201
+ # # Resize datasets to accommodate new data
202
+ # flatmaps_dset.resize(total_samples + batch_size, axis=0)
203
+
204
+ # # Write data to the datasets
205
+ # flatmaps_dset[total_samples:total_samples + batch_size] = images.numpy().astype(np.float16)
206
+
207
+ # total_samples += batch_size
208
+
209
+ # print(f"Processed {total_samples} samples")
210
+ # np.save('metadata_test_HCP.npy', meta_array)
211
+
212
+
213
+ # import h5py
214
+ # meta_array = np.array([], dtype=object)
215
+ # # Open an HDF5 file in write mode
216
+ # with h5py.File('test_hcp.hdf5', 'w') as h5f:
217
+ # flatmaps_dset = None
218
+
219
+ # total_samples = 0
220
+
221
+ # for i, batch in tqdm(enumerate(test_dl), total = 12000):
222
+ # images = batch['image'][0]
223
+ # meta = batch['meta']
224
+ # batch_size = images.shape[0]
225
+ # meta_serializable = meta.copy()
226
+
227
+
228
+ # # Step 2: Serialize the dictionary to a JSON string
229
+ # meta_str = json.dumps(flatten_meta(meta_serializable), indent=4)
230
+ # meta_array = np.append(meta_array, meta_str)
231
+ # if flatmaps_dset is None:
232
+ # # Initialize datasets with unlimited (None) maxshape along the first axis
233
+ # flatmaps_shape = (0,) + images.shape[1:]
234
+ # flatmaps_maxshape = (None,) + images.shape[1:]
235
+
236
+ # flatmaps_dset = h5f.create_dataset(
237
+ # 'flatmaps',
238
+ # shape=flatmaps_shape,
239
+ # maxshape=flatmaps_maxshape,
240
+ # dtype=np.float16,
241
+ # chunks=True # Enable chunking for efficient resizing
242
+ # )
243
+
244
+ # # Resize datasets to accommodate new data
245
+ # flatmaps_dset.resize(total_samples + batch_size, axis=0)
246
+
247
+ # # Write data to the datasets
248
+ # flatmaps_dset[total_samples:total_samples + batch_size] = images.numpy().astype(np.float16)
249
+
250
+ # total_samples += batch_size
251
+
252
+ # print(f"Processed {total_samples} samples")
253
+ # np.save('metadata_train_HCP.npy', meta_array)
254
+
255
+
256
+ # ### Preparing data
257
+
258
+ # In[4]:
259
+
260
+
261
+ from sklearn.preprocessing import LabelEncoder
262
+
263
+ INCLUDE_CONDS = {
264
+ "fear",
265
+ "neut",
266
+ "math",
267
+ "story",
268
+ "lf",
269
+ "lh",
270
+ "rf",
271
+ "rh",
272
+ "t",
273
+ "match",
274
+ "relation",
275
+ "mental",
276
+ "rnd",
277
+ "0bk_body",
278
+ "2bk_body",
279
+ "0bk_faces",
280
+ "2bk_faces",
281
+ "0bk_places",
282
+ "2bk_places",
283
+ "0bk_tools",
284
+ "2bk_tools",
285
+ }
286
+
287
+ # test_data = []
288
+
289
+ # # Iterate over the DataLoader with a progress bar
290
+ # for sample in tqdm(train_dl, desc="Processing samples"):
291
+ # x = sample['image']
292
+ # y = sample['meta']['trial_type']
293
+ # key = sample['meta']['key']
294
+ # print(x.shape, y, key)
295
+ # break
296
+ # Initialize the label encoder
297
+ label_encoder = LabelEncoder()
298
+ label_encoder.fit(sorted(INCLUDE_CONDS)) # Ensure consistent ordering
299
+
300
+ num_classes = len(label_encoder.classes_)
301
+ print(f"Number of classes: {num_classes}")
302
+
303
+
304
+ # In[5]:
305
+
306
+
307
+ f_train = h5py.File('/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/train_hcp.hdf5', 'r')
308
+ flatmaps_train = f_train['flatmaps']
309
+
310
+ f_test = h5py.File('/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/test_hcp.hdf5', 'r')
311
+ flatmaps_test = f_test['flatmaps']
312
+
313
+ metadata_train = np.load('/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/metadata_train_HCP.npy', allow_pickle=True)
314
+ metadata_test = np.load('/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/metadata_test_HCP.npy', allow_pickle=True)
315
+
316
+
317
+ # In[6]:
318
+
319
+
320
+ from torch.utils.data import Dataset, DataLoader
321
+
322
+ class HCPFlatDataset(Dataset):
323
+ def __init__(self, flatmaps, metadata):
324
+ self.flatmaps = flatmaps
325
+ self.metadata = metadata
326
+
327
+ def __len__(self):
328
+ return len(self.metadata)
329
+
330
+ def __getitem__(self, idx):
331
+ return self.flatmaps[idx], json.loads(self.metadata[idx])
332
+ print("Moving datasets to ram")
333
+ # Loading to cpu for faster training, this can take several minutes. Remove this [:] if you want to move one at the time.
334
+ train_dataset = HCPFlatDataset(flatmaps_train, metadata_train)
335
+ train_dl = DataLoader(train_dataset, batch_size=batch_size, shuffle=True, num_workers=num_workers)
336
+
337
+ test_dataset = HCPFlatDataset(flatmaps_test, metadata_test)
338
+ test_dl = DataLoader(test_dataset, batch_size=batch_size, shuffle=False, num_workers=0)
339
+ print("Datasets ready")
340
+
341
+
342
+ # ### Creating and loading Model
343
+
344
+ # In[7]:
345
+
346
+
347
+ from mae_utils.flat import load_hcp_flat_mask
348
+ from mae_utils.flat import create_hcp_flat
349
+ from mae_utils.flat import batch_unmask
350
+ import mae_utils.visualize as vis
351
+
352
+ flat_mask = load_hcp_flat_mask(hcp_flat_path)
353
+
354
+ mae_model = flat_models.mae_vit_large_fmri(
355
+ patch_size=patch_size,
356
+ decoder_embed_dim=decoder_embed_dim,
357
+ t_patch_size=t_patch_size,
358
+ pred_t_dim=pred_t_dim,
359
+ decoder_depth=4,
360
+ cls_embed=cls_embed,
361
+ norm_pix_loss=norm_pix_loss,
362
+ no_qkv_bias=no_qkv_bias,
363
+ sep_pos_embed=sep_pos_embed,
364
+ trunc_init=trunc_init,
365
+ pct_masks_to_decode=pct_masks_to_decode,
366
+ img_mask=flat_mask,
367
+ )
368
+
369
+
370
+ # In[8]:
371
+
372
+
373
+ checkpoint_files = [f for f in os.listdir(outdir) if f.endswith('.pth')]
374
+
375
+ if utils.is_interactive():
376
+ latest_checkpoint = "epoch99.pth"
377
+ else:
378
+ latest_checkpoint = sys.argv[2]
379
+ print(f"latest_checkpoint: {latest_checkpoint}")
380
+
381
+ # Load the checkpoint
382
+ checkpoint_path = os.path.join(outdir, latest_checkpoint)
383
+
384
+ state = torch.load(checkpoint_path)
385
+ mae_model.load_state_dict(state["model_state_dict"], strict=False)
386
+ mae_model.to(device)
387
+
388
+ print(f"\nLoaded checkpoint {latest_checkpoint} from {outdir}\n")
389
+
390
+
391
+ # In[9]:
392
+
393
+
394
+ class LinearClassifier(nn.Module):
395
+ def __init__(self, input_dim, num_classes):
396
+ super(LinearClassifier, self).__init__()
397
+ self.linear = nn.Linear(input_dim, num_classes)
398
+
399
+ def forward(self, x):
400
+ # Flatten the input except for the batch dimension
401
+ x = x.view(x.size(0), -1)
402
+ out = self.linear(x)
403
+ return out # Raw logits
404
+
405
+ # Determine the input dimension from a single sample
406
+ # Assuming images are of shape [1, 16, 144, 320]
407
+ input_dim = np.prod(mae_model(torch.randn(1,1,16,144,320).to(device),global_pool=global_pool, forward_features = True).shape[1:])
408
+ print(f"Input dimension: {input_dim}")
409
+
410
+
411
+ # In[10]:
412
+
413
+
414
+ class FullModel(nn.Module):
415
+ def __init__(self, lc_model, mae_model):
416
+ super(FullModel, self).__init__()
417
+ self.lc_model = lc_model
418
+ self.mae_model = mae_model
419
+
420
+
421
+ def forward(self, x, gsr):
422
+ x = self.mae_model(x, global_pool=global_pool, forward_features = True)
423
+ x = self.lc_model(x)
424
+ return x
425
+
426
+
427
+ # In[11]:
428
+
429
+
430
+ # Initialize the model
431
+ lc_model = LinearClassifier(input_dim=input_dim, num_classes=num_classes)
432
+
433
+ model = FullModel(lc_model, mae_model)
434
+
435
+ # Move the model to the GPU
436
+ model.to(device)
437
+
438
+ # Define loss function
439
+ criterion = nn.CrossEntropyLoss()
440
+
441
+ # Define optimizer with L2 regularization (weight_decay)
442
+ learning_rate = 1e-4
443
+ weight_decay = 1e-5 # Adjust based on your needs
444
+ optimizer = torch.optim.Adam(model.parameters(), lr=learning_rate, weight_decay=weight_decay)
445
+ num_epochs = 20 # Adjust as needed
446
+
447
+
448
+ # ### Data
449
+
450
+ # In[12]:
451
+
452
+
453
+ import wandb
454
+
455
+ if utils.is_interactive():
456
+ print("Running in interactive notebook. Disabling W&B and ckpt saving.")
457
+ wandb_log = True
458
+ save_ckpt = True
459
+
460
+ if wandb_log:
461
+ wandb_project = 'fMRI-foundation-model'
462
+ wandb_config = {
463
+ "model_name": model_name+'_HCP_FT',
464
+ "batch_size": batch_size,
465
+ "learning_rate": learning_rate,
466
+ "weight_decay": weight_decay,
467
+ "num_epochs": num_epochs,
468
+ "seed": seed,
469
+ }
470
+ print("wandb_config:\n", wandb_config)
471
+ random_id = random.randint(0, 100000)
472
+ print("wandb_id:", "HCPflat_raw" + f"_{random_id}")
473
+ wandb.init(
474
+ id=model_name+'_HCP_FT' + f"_{random_id}",
475
+ project=wandb_project,
476
+ name=model_name+'_HCP_FT',
477
+ config=wandb_config,
478
+ resume="allow",
479
+ )
480
+
481
+
482
+ # In[13]:
483
+
484
+
485
+ for epoch in range(num_epochs):
486
+ running_train_loss = 0.0
487
+ correct_train = 0
488
+ total_train = 0
489
+ step = 0
490
+
491
+ # with torch.amp.autocast(device_type='cuda'):
492
+ # Training Phase
493
+ model.train()
494
+ for batch in tqdm(train_dl, desc=f"Epoch {epoch+1}/{num_epochs} - Training"):
495
+ optimizer.zero_grad()
496
+ images = batch[0].to(device).float().unsqueeze(1) #fix this # Shape: [batch_size, 1, 16, 144, 320]
497
+ labels = batch[1]['trial_type'] # List of labels
498
+
499
+ encoded_labels = label_encoder.transform(labels)
500
+ encoded_labels = torch.tensor(encoded_labels, dtype=torch.long).to(device) # Shape: [batch_size]
501
+
502
+ # Forward pass
503
+ outputs = model(images, gsr=gsr) # Shape: [num_train_samples, num_classes]
504
+
505
+ # Compute loss
506
+ loss = criterion(outputs, encoded_labels)
507
+
508
+ # Backward pass and optimization
509
+ loss.backward()
510
+ optimizer.step()
511
+
512
+ # Accumulate loss
513
+ running_train_loss += loss.item() * images.size(0)
514
+
515
+
516
+ # Calculate accuracy
517
+ _, predicted = torch.max(outputs, 1)
518
+
519
+ correct_train += (predicted == encoded_labels).sum().item()
520
+ total_train += encoded_labels.size(0)
521
+
522
+ step = step + 1
523
+ if step % 100 == 0:
524
+ print(f"Step [{step}/{len(train_dl)}] - Training Loss: {loss.item():.4f} - Training Accuracy: {100 * correct_train / total_train:.2f}%")
525
+ # thth
526
+
527
+ epoch_train_loss = running_train_loss / total_train if total_train > 0 else 0.0
528
+ train_accuracy = 100 * correct_train / total_train if total_train > 0 else 0.0
529
+
530
+ # Validation Phase
531
+ model.eval()
532
+ running_val_loss = 0.0
533
+ correct_val = 0
534
+ total_val = 0
535
+
536
+ with torch.no_grad():
537
+ for batch in tqdm(test_dl, desc=f"Epoch {epoch+1}/{num_epochs} - Validation"):
538
+
539
+ images = batch[0].to(device).float().unsqueeze(1) #fix this
540
+ labels = batch[1]['trial_type']
541
+
542
+ # Encode labels to integer indices
543
+ encoded_labels = label_encoder.transform(labels)
544
+ encoded_labels = torch.tensor(encoded_labels, dtype=torch.long).to(device)
545
+
546
+
547
+ # Forward pass
548
+ outputs = model(images, gsr=gsr)
549
+
550
+ # Compute loss
551
+ loss = criterion(outputs, encoded_labels)
552
+
553
+ # Accumulate loss
554
+ running_val_loss += loss.item() * images.size(0)
555
+
556
+ # Calculate accuracy
557
+ _, predicted = torch.max(outputs, 1)
558
+ correct_val += (predicted == encoded_labels).sum().item()
559
+ total_val += encoded_labels.size(0)
560
+
561
+
562
+
563
+ epoch_val_loss = running_val_loss / total_val if total_val > 0 else 0.0
564
+ val_accuracy = 100 * correct_val / total_val if total_val > 0 else 0.0
565
+
566
+ print(f"Epoch [{epoch+1}/{num_epochs}] "
567
+ f"- Training Loss: {epoch_train_loss:.4f}, Training Accuracy: {train_accuracy:.2f}% "
568
+ f"- Validation Loss: {epoch_val_loss:.4f}, Validation Accuracy: {val_accuracy:.2f}%")
569
+
570
+ if wandb_log:
571
+ wandb.log({
572
+ "epoch_train_loss": epoch_train_loss,
573
+ "epoch_val_loss": epoch_val_loss,
574
+ "train_accuracy": train_accuracy,
575
+ "val_accuracy": val_accuracy,
576
+ })
577
+ if save_ckpt:
578
+ outdir = os.path.abspath(f'checkpoints/{model_name+"HCP_FT"}')
579
+ os.makedirs(outdir, exist_ok=True)
580
+ print("outdir", outdir)
581
+ # Save model and config
582
+ torch.save(model.state_dict(), f"{outdir}/model.pth")
583
+ with open(f"{outdir}/config.yaml", 'w') as f:
584
+ yaml.dump(wandb_config, f)
585
+ print(f"Saved model and config to {outdir}")
586
+
587
+
fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/files/output.log ADDED
@@ -0,0 +1 @@
 
 
1
+ Epoch 1/20 - Training: 1%| | 97/13913 [00:53<1:21:27, 2.83it/s]
fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/files/requirements.txt ADDED
@@ -0,0 +1,198 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ protobuf==5.28.2
2
+ imageio==2.35.1
3
+ MarkupSafe==3.0.0
4
+ regex==2024.9.11
5
+ matplotlib==3.9.2
6
+ notebook==7.2.2
7
+ debugpy==1.8.6
8
+ aiosignal==1.3.1
9
+ jupyter_core==5.7.2
10
+ torchaudio==2.4.1+cu121
11
+ python-json-logger==2.0.7
12
+ six==1.16.0
13
+ scikit-image==0.24.0
14
+ types-python-dateutil==2.9.0.20241003
15
+ PyYAML==6.0.2
16
+ httpcore==1.0.6
17
+ clip==1.0
18
+ babel==2.16.0
19
+ webcolors==24.8.0
20
+ omegaconf==2.3.0
21
+ webencodings==0.5.1
22
+ kiwisolver==1.4.7
23
+ uri-template==1.3.0
24
+ diffusers==0.23.0
25
+ idna==3.10
26
+ fsspec==2024.9.0
27
+ parso==0.8.4
28
+ setuptools==65.5.0
29
+ tornado==6.4.1
30
+ webdataset==0.2.100
31
+ decord==0.6.0
32
+ nvidia-curand-cu12==10.3.2.106
33
+ ipykernel==6.29.5
34
+ jupyter==1.1.1
35
+ pexpect==4.9.0
36
+ kornia_rs==0.1.5
37
+ iopath==0.1.10
38
+ async-lru==2.0.4
39
+ future==1.0.0
40
+ torchvision==0.19.1+cu121
41
+ botocore==1.34.162
42
+ cycler==0.12.1
43
+ tzdata==2024.2
44
+ jupyter_server_terminals==0.5.3
45
+ click==8.1.7
46
+ einops==0.8.0
47
+ pyzmq==26.2.0
48
+ jupyter_client==8.6.3
49
+ nbconvert==7.16.4
50
+ scikit-learn==1.5.2
51
+ executing==2.1.0
52
+ asttokens==2.4.1
53
+ docker-pycreds==0.4.0
54
+ matplotlib-inline==0.1.7
55
+ overrides==7.7.0
56
+ websocket-client==1.8.0
57
+ nbformat==5.10.4
58
+ elbow==0.1.1
59
+ contourpy==1.3.0
60
+ nvidia-cudnn-cu12==9.1.0.70
61
+ transformers==4.44.2
62
+ gitdb==4.0.11
63
+ jupyterlab_nvdashboard==0.11.0
64
+ lazy_loader==0.4
65
+ jsonpointer==3.0.0
66
+ notebook_shim==0.2.4
67
+ nvidia-nccl-cu12==2.20.5
68
+ ffmpeg-python==0.2.0
69
+ triton==3.0.0
70
+ mistune==3.0.2
71
+ python-dateutil==2.9.0.post0
72
+ beautifulsoup4==4.12.3
73
+ nbclient==0.10.0
74
+ h5py==3.12.1
75
+ ftfy==6.2.3
76
+ zipp==3.20.2
77
+ ptyprocess==0.7.0
78
+ huggingface-hub==0.25.1
79
+ pytz==2024.2
80
+ jupyterlab_pygments==0.3.0
81
+ nvidia-cublas-cu12==12.1.3.1
82
+ pandocfilters==1.5.1
83
+ Jinja2==3.1.4
84
+ arrow==1.3.0
85
+ rpds-py==0.20.0
86
+ jupyter_server==2.14.2
87
+ simplejson==3.19.3
88
+ networkx==3.3
89
+ packaging==24.1
90
+ traitlets==5.14.3
91
+ pandas==2.2.3
92
+ xformers==0.0.22.post7
93
+ lightning-utilities==0.11.7
94
+ tifffile==2024.9.20
95
+ nvidia-cuda-cupti-cu12==12.1.105
96
+ mpmath==1.3.0
97
+ GitPython==3.1.43
98
+ scipy==1.14.1
99
+ jsonschema==4.23.0
100
+ prompt_toolkit==3.0.48
101
+ s3transfer==0.10.2
102
+ multidict==6.1.0
103
+ bleach==6.1.0
104
+ sentry-sdk==2.15.0
105
+ nibabel==5.2.1
106
+ accelerate==1.0.0
107
+ pyarrow==17.0.0
108
+ threadpoolctl==3.5.0
109
+ attrs==24.2.0
110
+ rfc3986-validator==0.1.1
111
+ nvidia-cuda-runtime-cu12==12.1.105
112
+ ipywidgets==8.1.5
113
+ frozenlist==1.4.1
114
+ pycparser==2.22
115
+ jupyterlab_server==2.27.3
116
+ nvidia-cuda-nvrtc-cu12==12.1.105
117
+ yarl==1.13.1
118
+ setproctitle==1.3.3
119
+ isoduration==20.11.0
120
+ Pygments==2.18.0
121
+ jedi==0.19.1
122
+ boto3==1.34.57
123
+ tokenizers==0.19.1
124
+ referencing==0.35.1
125
+ rfc3339-validator==0.1.4
126
+ pillow==10.4.0
127
+ jupyterlab==4.2.5
128
+ stack-data==0.6.3
129
+ h11==0.14.0
130
+ anyio==4.6.0
131
+ nilearn==0.10.4
132
+ nvidia-cusolver-cu12==11.4.5.107
133
+ tinycss2==1.3.0
134
+ defusedxml==0.7.1
135
+ argon2-cffi-bindings==21.2.0
136
+ soupsieve==2.6
137
+ nest-asyncio==1.6.0
138
+ torchmetrics==1.3.0.post0
139
+ tqdm==4.66.5
140
+ cffi==1.17.1
141
+ charset-normalizer==3.3.2
142
+ jsonschema-specifications==2023.12.1
143
+ decorator==5.1.1
144
+ open_clip_torch==2.26.1
145
+ jupyter-events==0.10.0
146
+ smart-open==7.0.5
147
+ antlr4-python3-runtime==4.9.3
148
+ prometheus_client==0.21.0
149
+ kornia==0.7.3
150
+ typing_extensions==4.12.2
151
+ sniffio==1.3.1
152
+ joblib==1.4.2
153
+ comm==0.2.2
154
+ aiohappyeyeballs==2.4.3
155
+ numpy==2.1.2
156
+ braceexpand==0.1.7
157
+ certifi==2024.8.30
158
+ psutil==6.0.0
159
+ pyparsing==3.1.4
160
+ pure_eval==0.2.3
161
+ nvidia-cusparse-cu12==12.1.0.106
162
+ wandb==0.18.3
163
+ urllib3==2.2.3
164
+ smmap==5.0.1
165
+ platformdirs==4.3.6
166
+ torch==2.4.1+cu121
167
+ requests==2.32.3
168
+ json5==0.9.25
169
+ nvidia-nvjitlink-cu12==12.6.77
170
+ jupyterlab_widgets==3.0.13
171
+ lxml==5.3.0
172
+ httpx==0.27.2
173
+ opencv-python==4.6.0.66
174
+ portalocker==2.10.1
175
+ pytorch-lightning==2.0.1
176
+ sympy==1.13.3
177
+ wcwidth==0.2.13
178
+ jmespath==1.0.1
179
+ fqdn==1.5.1
180
+ pynvml==11.5.3
181
+ pip==24.0
182
+ wrapt==1.16.0
183
+ aiohttp==3.10.9
184
+ filelock==3.16.1
185
+ fonttools==4.54.1
186
+ fastjsonschema==2.20.0
187
+ jupyter-console==6.6.3
188
+ widgetsnbextension==4.0.13
189
+ timm==1.0.9
190
+ nvidia-cufft-cu12==11.0.2.54
191
+ ipython==8.28.0
192
+ nvidia-nvtx-cu12==12.1.105
193
+ jupyter-lsp==2.2.5
194
+ safetensors==0.4.5
195
+ terminado==0.18.1
196
+ argon2-cffi==23.1.0
197
+ Send2Trash==1.8.3
198
+ importlib_metadata==8.5.0
fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/files/wandb-metadata.json ADDED
@@ -0,0 +1,131 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-5.15.0-1058-aws-x86_64-with-glibc2.31",
3
+ "python": "3.11.10",
4
+ "startedAt": "2024-10-23T04:08:30.347631Z",
5
+ "args": [
6
+ "HCPflat_large_gsrFalse_",
7
+ "epoch99.pth"
8
+ ],
9
+ "program": "/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.py",
10
+ "codePath": "src/HCP_downstream_finetune.py",
11
+ "git": {
12
+ "remote": "https://github.com/MedARC-AI/fMRI-foundation-model",
13
+ "commit": "b1ba684ae7a5cc4155cc046b0abe613de09bf700"
14
+ },
15
+ "email": "torrico.villanueva.cesar.kadir@gmail.com",
16
+ "root": "/weka/proj-fmri/ckadirt/fMRI-foundation-model/src",
17
+ "host": "ip-10-0-139-117",
18
+ "username": "ckadirt",
19
+ "executable": "/admin/home-ckadirt/foundation_env/bin/python",
20
+ "codePathLocal": "HCP_downstream_finetune.py",
21
+ "cpu_count": 96,
22
+ "cpu_count_logical": 192,
23
+ "gpu": "[NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3]",
24
+ "gpu_count": 8,
25
+ "disk": {
26
+ "/": {
27
+ "total": "249555763200",
28
+ "used": "181767704576"
29
+ }
30
+ },
31
+ "memory": {
32
+ "total": "2147443380224"
33
+ },
34
+ "cpu": {
35
+ "count": 96,
36
+ "countLogical": 192
37
+ },
38
+ "gpu_nvidia": [
39
+ {
40
+ "name": "NVIDIA H100 80GB HBM3",
41
+ "memoryTotal": "85520809984",
42
+ "cudaCores": 16896,
43
+ "architecture": "Hopper"
44
+ },
45
+ {
46
+ "name": "NVIDIA H100 80GB HBM3",
47
+ "memoryTotal": "85520809984",
48
+ "cudaCores": 16896,
49
+ "architecture": "Hopper"
50
+ },
51
+ {
52
+ "name": "NVIDIA H100 80GB HBM3",
53
+ "memoryTotal": "85520809984",
54
+ "cudaCores": 16896,
55
+ "architecture": "Hopper"
56
+ },
57
+ {
58
+ "name": "NVIDIA H100 80GB HBM3",
59
+ "memoryTotal": "85520809984",
60
+ "cudaCores": 16896,
61
+ "architecture": "Hopper"
62
+ },
63
+ {
64
+ "name": "NVIDIA H100 80GB HBM3",
65
+ "memoryTotal": "85520809984",
66
+ "cudaCores": 16896,
67
+ "architecture": "Hopper"
68
+ },
69
+ {
70
+ "name": "NVIDIA H100 80GB HBM3",
71
+ "memoryTotal": "85520809984",
72
+ "cudaCores": 16896,
73
+ "architecture": "Hopper"
74
+ },
75
+ {
76
+ "name": "NVIDIA H100 80GB HBM3",
77
+ "memoryTotal": "85520809984",
78
+ "cudaCores": 16896,
79
+ "architecture": "Hopper"
80
+ },
81
+ {
82
+ "name": "NVIDIA H100 80GB HBM3",
83
+ "memoryTotal": "85520809984",
84
+ "cudaCores": 16896,
85
+ "architecture": "Hopper"
86
+ }
87
+ ],
88
+ "slurm": {
89
+ "cluster_name": "sagemaker2",
90
+ "conf": "/opt/slurm/etc/slurm.conf",
91
+ "cpus_on_node": "20",
92
+ "gpus_on_node": "1",
93
+ "gpus_per_task": "1",
94
+ "gtids": "0",
95
+ "job_account": "fmri",
96
+ "job_cpus_per_node": "20",
97
+ "job_end_time": "1729699678",
98
+ "job_gid": "1879800513",
99
+ "job_gpus": "0",
100
+ "job_id": "528149",
101
+ "job_name": "finetuneHCP",
102
+ "job_nodelist": "ip-10-0-139-117",
103
+ "job_num_nodes": "1",
104
+ "job_partition": "p5",
105
+ "job_qos": "idle",
106
+ "job_start_time": "1729656478",
107
+ "job_uid": "1879804696",
108
+ "job_user": "ckadirt",
109
+ "jobid": "528149",
110
+ "localid": "0",
111
+ "mem_per_cpu": "11500",
112
+ "nnodes": "1",
113
+ "node_aliases": "(null)",
114
+ "nodeid": "0",
115
+ "nodelist": "ip-10-0-139-117",
116
+ "nprocs": "1",
117
+ "ntasks": "1",
118
+ "ntasks_per_node": "1",
119
+ "prio_process": "0",
120
+ "procid": "0",
121
+ "script_context": "prolog_task",
122
+ "submit_dir": "/weka/proj-fmri/ckadirt/fMRI-foundation-model/src",
123
+ "submit_host": "ip-172-17-12-61",
124
+ "task_pid": "3329544",
125
+ "tasks_per_node": "1",
126
+ "topology_addr": "ip-10-0-139-117",
127
+ "topology_addr_pattern": "node",
128
+ "working_cluster": "sagemaker2:ip-172-17-63-161:6817:9984:109"
129
+ },
130
+ "cudaVersion": "12.2"
131
+ }
fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug-core.log ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2024-10-23T04:08:29.759432555Z","level":"INFO","msg":"started logging, with flags","port-filename":"/tmp/tmpv0zbfmyf/port-3329596.txt","pid":3329596,"debug":false,"disable-analytics":false}
2
+ {"time":"2024-10-23T04:08:29.759434395Z","level":"INFO","msg":"started logging, with flags","port-filename":"/tmp/tmpdgtkqv8m/port-3329693.txt","pid":3329693,"debug":false,"disable-analytics":false}
3
+ {"time":"2024-10-23T04:08:29.75965549Z","level":"INFO","msg":"FeatureState","shutdownOnParentExitEnabled":false}
4
+ {"time":"2024-10-23T04:08:29.75967947Z","level":"INFO","msg":"FeatureState","shutdownOnParentExitEnabled":false}
5
+ {"time":"2024-10-23T04:08:29.762594934Z","level":"INFO","msg":"Will exit if parent process dies.","ppid":3329693}
6
+ {"time":"2024-10-23T04:08:29.762592144Z","level":"INFO","msg":"server is running","addr":{"IP":"127.0.0.1","Port":43169,"Zone":""}}
7
+ {"time":"2024-10-23T04:08:29.764024661Z","level":"INFO","msg":"Will exit if parent process dies.","ppid":3329596}
8
+ {"time":"2024-10-23T04:08:29.764057981Z","level":"INFO","msg":"server is running","addr":{"IP":"127.0.0.1","Port":45635,"Zone":""}}
9
+ {"time":"2024-10-23T04:08:29.893047528Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"127.0.0.1:44072"}
10
+ {"time":"2024-10-23T04:08:29.893048808Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"127.0.0.1:60086"}
11
+ {"time":"2024-10-23T04:08:30.348153132Z","level":"INFO","msg":"handleInformInit: received","streamId":"HCPflat_large_gsrFalse__HCP_FT_83810","id":"127.0.0.1:60086"}
12
+ {"time":"2024-10-23T04:08:30.390297655Z","level":"INFO","msg":"handleInformInit: received","streamId":"NSDflat_large_gsrFalse__HCP_FT_83810","id":"127.0.0.1:44072"}
13
+ {"time":"2024-10-23T04:08:30.392846482Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"HCPflat_large_gsrFalse__HCP_FT_83810","id":"127.0.0.1:60086"}
14
+ {"time":"2024-10-23T04:08:30.434661309Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"NSDflat_large_gsrFalse__HCP_FT_83810","id":"127.0.0.1:44072"}
fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug-internal.log ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2024-10-23T04:08:30.355853305Z","level":"INFO","msg":"using version","core version":"0.18.3"}
2
+ {"time":"2024-10-23T04:08:30.355871046Z","level":"INFO","msg":"created symlink","path":"/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug-core.log"}
3
+ {"time":"2024-10-23T04:08:30.358613097Z","level":"ERROR","msg":"dialing: google: could not find default credentials. See https://cloud.google.com/docs/authentication/external/set-up-adc for more information"}
4
+ {"time":"2024-10-23T04:08:30.392817732Z","level":"INFO","msg":"created new stream","id":"HCPflat_large_gsrFalse__HCP_FT_83810"}
5
+ {"time":"2024-10-23T04:08:30.392839382Z","level":"INFO","msg":"stream: started","id":"HCPflat_large_gsrFalse__HCP_FT_83810"}
6
+ {"time":"2024-10-23T04:08:30.392871663Z","level":"INFO","msg":"handler: started","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_83810"}}
7
+ {"time":"2024-10-23T04:08:30.392873713Z","level":"INFO","msg":"sender: started","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_83810"}}
8
+ {"time":"2024-10-23T04:08:30.392853283Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_83810"}}
9
+ {"time":"2024-10-23T04:08:30.892163378Z","level":"INFO","msg":"wandb-core","!BADKEY":null}
10
+ {"time":"2024-10-23T04:08:30.898577708Z","level":"INFO","msg":"Starting system monitor"}
11
+ {"time":"2024-10-23T04:08:30.932426706Z","level":"ERROR","msg":"git repo not found","error":"repository does not exist"}
fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug.log ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 2024-10-23 04:08:30,335 INFO MainThread:3329596 [wandb_setup.py:_flush():79] Current SDK version is 0.18.3
2
+ 2024-10-23 04:08:30,336 INFO MainThread:3329596 [wandb_setup.py:_flush():79] Configure stats pid to 3329596
3
+ 2024-10-23 04:08:30,336 INFO MainThread:3329596 [wandb_setup.py:_flush():79] Loading settings from /admin/home-ckadirt/.config/wandb/settings
4
+ 2024-10-23 04:08:30,336 INFO MainThread:3329596 [wandb_setup.py:_flush():79] Loading settings from /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/settings
5
+ 2024-10-23 04:08:30,336 INFO MainThread:3329596 [wandb_setup.py:_flush():79] Loading settings from environment variables: {}
6
+ 2024-10-23 04:08:30,336 INFO MainThread:3329596 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': None, '_disable_service': None}
7
+ 2024-10-23 04:08:30,336 INFO MainThread:3329596 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'src/HCP_downstream_finetune.py', 'program_abspath': '/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.py', 'program': '/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.py'}
8
+ 2024-10-23 04:08:30,336 INFO MainThread:3329596 [wandb_setup.py:_flush():79] Applying login settings: {}
9
+ 2024-10-23 04:08:30,336 INFO MainThread:3329596 [wandb_init.py:_log_setup():532] Logging user logs to /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug.log
10
+ 2024-10-23 04:08:30,337 INFO MainThread:3329596 [wandb_init.py:_log_setup():533] Logging internal logs to /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/logs/debug-internal.log
11
+ 2024-10-23 04:08:30,337 INFO MainThread:3329596 [wandb_init.py:init():617] calling init triggers
12
+ 2024-10-23 04:08:30,337 INFO MainThread:3329596 [wandb_init.py:init():624] wandb.init called with sweep_config: {}
13
+ config: {'model_name': 'HCPflat_large_gsrFalse__HCP_FT', 'batch_size': 8, 'learning_rate': 0.0001, 'weight_decay': 1e-05, 'num_epochs': 20, 'seed': 42}
14
+ 2024-10-23 04:08:30,337 INFO MainThread:3329596 [wandb_init.py:init():667] starting backend
15
+ 2024-10-23 04:08:30,337 INFO MainThread:3329596 [wandb_init.py:init():671] sending inform_init request
16
+ 2024-10-23 04:08:30,346 INFO MainThread:3329596 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
17
+ 2024-10-23 04:08:30,346 INFO MainThread:3329596 [wandb_init.py:init():684] backend started and connected
18
+ 2024-10-23 04:08:30,365 INFO MainThread:3329596 [wandb_init.py:init():779] updated telemetry
19
+ 2024-10-23 04:08:30,420 INFO MainThread:3329596 [wandb_init.py:init():812] communicating run to backend with 90.0 second timeout
20
+ 2024-10-23 04:08:30,843 INFO MainThread:3329596 [wandb_init.py:init():855] run resumed
21
+ 2024-10-23 04:08:30,876 INFO MainThread:3329596 [wandb_init.py:init():863] starting run threads in backend
22
+ 2024-10-23 04:08:31,304 INFO MainThread:3329596 [wandb_run.py:_console_start():2465] atexit reg
23
+ 2024-10-23 04:08:31,304 INFO MainThread:3329596 [wandb_run.py:_redirect():2313] redirect: wrap_raw
24
+ 2024-10-23 04:08:31,304 INFO MainThread:3329596 [wandb_run.py:_redirect():2378] Wrapping output streams.
25
+ 2024-10-23 04:08:31,304 INFO MainThread:3329596 [wandb_run.py:_redirect():2403] Redirects installed.
26
+ 2024-10-23 04:08:31,310 INFO MainThread:3329596 [wandb_init.py:init():907] run started, returning control to user process
fMRI-foundation-model/src/wandb/run-20241023_040830-HCPflat_large_gsrFalse__HCP_FT_83810/run-HCPflat_large_gsrFalse__HCP_FT_83810.wandb ADDED
Binary file (65.5 kB). View file
 
fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/files/code/src/HCP_downstream_finetune.py ADDED
@@ -0,0 +1,596 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python
2
+ # coding: utf-8
3
+
4
+ # In[1]:
5
+
6
+
7
+ # Import packages and setup gpu configuration.
8
+ # This code block shouldnt need to be adjusted!
9
+ import os
10
+ import sys
11
+ import json
12
+ import yaml
13
+ import numpy as np
14
+ import copy
15
+ import math
16
+ import time
17
+ import random
18
+ from tqdm.auto import tqdm
19
+ import webdataset as wds
20
+ import matplotlib.pyplot as plt
21
+
22
+ import torch
23
+ import torch.nn as nn
24
+ from torchvision import transforms
25
+ import utils
26
+ from mae_utils.flat_models import *
27
+ import h5py
28
+ from mae_utils import flat_models
29
+
30
+ # tf32 data type is faster than standard float32
31
+ torch.backends.cuda.matmul.allow_tf32 = True
32
+ # following fixes a Conv3D CUDNN_NOT_SUPPORTED error
33
+ torch.backends.cudnn.benchmark = True
34
+
35
+ # ## MODEL TO LOAD ##
36
+ if utils.is_interactive():
37
+ model_name = "HCPflat_large_gsrFalse_"
38
+ else:
39
+ model_name = sys.argv[1]
40
+
41
+
42
+ # outdir = os.path.abspath(f'checkpoints/{model_name}')
43
+ outdir = os.path.abspath(f'checkpoints/{model_name}')
44
+
45
+ print("outdir", outdir)
46
+ # Load previous config.yaml if available
47
+ if os.path.exists(f"{outdir}/config.yaml"):
48
+ config = yaml.load(open(f"{outdir}/config.yaml", 'r'), Loader=yaml.FullLoader)
49
+ print(f"Loaded config.yaml from ckpt folder {outdir}")
50
+ # create global variables from the config
51
+ print("\n__CONFIG__")
52
+ for attribute_name in config.keys():
53
+ print(f"{attribute_name} = {config[attribute_name]}")
54
+ globals()[attribute_name] = config[f'{attribute_name}']
55
+ print("\n")
56
+
57
+ world_size = os.getenv('WORLD_SIZE')
58
+ if world_size is None:
59
+ world_size = 1
60
+ else:
61
+ world_size = int(world_size)
62
+ print(f"WORLD_SIZE={world_size}")
63
+
64
+ if utils.is_interactive():
65
+ # Following allows you to change functions in models.py or utils.py and
66
+ # have this notebook automatically update with your revisions
67
+ get_ipython().run_line_magic('load_ext', 'autoreload')
68
+ get_ipython().run_line_magic('autoreload', '2')
69
+
70
+ batch_size = probe_batch_size
71
+ num_epochs = probe_num_epochs
72
+
73
+ data_type = torch.float32 # change depending on your mixed_precision
74
+ global_batch_size = batch_size * world_size
75
+
76
+ device = torch.device('cuda')
77
+
78
+ hcp_flat_path = "/weka/proj-medarc/shared/HCP-Flat"
79
+ # seed = 42
80
+ # num_frames = 16
81
+ # gsr = False
82
+ # num_workers = 10
83
+ # batch_size = 128
84
+
85
+ print("PID of this process =",os.getpid())
86
+ utils.seed_everything(seed)
87
+
88
+
89
+ # In[2]:
90
+
91
+
92
+ if os.getenv('global_pool') == "False":
93
+ global_pool = False
94
+ else:
95
+ global_pool = True
96
+ print(f"global_pool = {global_pool}")
97
+
98
+ try:
99
+ gsr
100
+ except:
101
+ gsr = True
102
+ print("set gsr to True")
103
+ print(f"gsr = {gsr}")
104
+
105
+
106
+ # In[3]:
107
+
108
+
109
+ #### UNCOMMENT THIS TO SAVE THE HCP-FLAT IN HDF5 FORMAT
110
+
111
+
112
+ # from torch.utils.data import default_collate
113
+ # from mae_utils.flat import load_hcp_flat_mask
114
+ # from mae_utils.flat import create_hcp_flat
115
+ # from mae_utils.flat import batch_unmask
116
+ # import mae_utils.visualize as vis
117
+
118
+
119
+ # batch_size = 26
120
+ # print(f"changed batch_size to {batch_size}")
121
+
122
+ # ## Test ##
123
+ # datasets_to_include = "HCP"
124
+ # assert "HCP" in datasets_to_include
125
+ # test_dataset = create_hcp_flat(root=hcp_flat_path,
126
+ # clip_mode="event", frames=num_frames, shuffle=False, gsr=gsr, sub_list = 'test')
127
+ # test_dl = wds.WebLoader(
128
+ # test_dataset.batched(batch_size, partial=False, collation_fn=default_collate),
129
+ # batch_size=None,
130
+ # shuffle=False,
131
+ # num_workers=num_workers,
132
+ # pin_memory=True,
133
+ # )
134
+
135
+ # ## Train ##
136
+ # assert "HCP" in datasets_to_include
137
+ # train_dataset = create_hcp_flat(root=hcp_flat_path,
138
+ # clip_mode="event", frames=num_frames, shuffle=False, gsr=gsr, sub_list = 'train')
139
+ # train_dl = wds.WebLoader(
140
+ # train_dataset.batched(batch_size, partial=False, collation_fn=default_collate),
141
+ # batch_size=None,
142
+ # shuffle=False,
143
+ # num_workers=num_workers,
144
+ # pin_memory=True,
145
+ # )
146
+
147
+ # def flatten_meta(meta_dict):
148
+ # """
149
+ # Flatten the meta dictionary by:
150
+ # - Replacing single-item lists with the item itself.
151
+ # - Converting tensors to scalar numbers.
152
+ # """
153
+ # flattened = {}
154
+ # for key, value in meta_dict.items():
155
+ # if isinstance(value, list):
156
+ # if len(value) == 1:
157
+ # flattened[key] = value[0] # Replace list with its single item
158
+ # else:
159
+ # flattened[key] = value # Keep as is if multiple items
160
+ # elif isinstance(value, torch.Tensor):
161
+ # # Convert tensor to scalar
162
+ # if value.numel() == 1:
163
+ # flattened[key] = value.item()
164
+ # else:
165
+ # flattened[key] = value.tolist() # Convert multi-element tensor to list
166
+ # else:
167
+ # flattened[key] = value # Keep the value as is
168
+ # return flattened
169
+
170
+ # import h5py
171
+ # meta_array = np.array([], dtype=object)
172
+ # # Open an HDF5 file in write mode
173
+ # with h5py.File('train_hcp.hdf5', 'w') as h5f:
174
+ # flatmaps_dset = None
175
+
176
+ # total_samples = 0
177
+
178
+ # for i, batch in tqdm(enumerate(train_dl), total = 120000):
179
+ # images = batch['image'][0]
180
+ # meta = batch['meta']
181
+ # batch_size = images.shape[0]
182
+ # meta_serializable = meta.copy()
183
+
184
+
185
+ # # Step 2: Serialize the dictionary to a JSON string
186
+ # meta_str = json.dumps(flatten_meta(meta_serializable), indent=4)
187
+ # meta_array = np.append(meta_array, meta_str)
188
+ # if flatmaps_dset is None:
189
+ # # Initialize datasets with unlimited (None) maxshape along the first axis
190
+ # flatmaps_shape = (0,) + images.shape[1:]
191
+ # flatmaps_maxshape = (None,) + images.shape[1:]
192
+
193
+ # flatmaps_dset = h5f.create_dataset(
194
+ # 'flatmaps',
195
+ # shape=flatmaps_shape,
196
+ # maxshape=flatmaps_maxshape,
197
+ # dtype=np.float16,
198
+ # chunks=True # Enable chunking for efficient resizing
199
+ # )
200
+
201
+ # # Resize datasets to accommodate new data
202
+ # flatmaps_dset.resize(total_samples + batch_size, axis=0)
203
+
204
+ # # Write data to the datasets
205
+ # flatmaps_dset[total_samples:total_samples + batch_size] = images.numpy().astype(np.float16)
206
+
207
+ # total_samples += batch_size
208
+
209
+ # print(f"Processed {total_samples} samples")
210
+ # np.save('metadata_test_HCP.npy', meta_array)
211
+
212
+
213
+ # import h5py
214
+ # meta_array = np.array([], dtype=object)
215
+ # # Open an HDF5 file in write mode
216
+ # with h5py.File('test_hcp.hdf5', 'w') as h5f:
217
+ # flatmaps_dset = None
218
+
219
+ # total_samples = 0
220
+
221
+ # for i, batch in tqdm(enumerate(test_dl), total = 12000):
222
+ # images = batch['image'][0]
223
+ # meta = batch['meta']
224
+ # batch_size = images.shape[0]
225
+ # meta_serializable = meta.copy()
226
+
227
+
228
+ # # Step 2: Serialize the dictionary to a JSON string
229
+ # meta_str = json.dumps(flatten_meta(meta_serializable), indent=4)
230
+ # meta_array = np.append(meta_array, meta_str)
231
+ # if flatmaps_dset is None:
232
+ # # Initialize datasets with unlimited (None) maxshape along the first axis
233
+ # flatmaps_shape = (0,) + images.shape[1:]
234
+ # flatmaps_maxshape = (None,) + images.shape[1:]
235
+
236
+ # flatmaps_dset = h5f.create_dataset(
237
+ # 'flatmaps',
238
+ # shape=flatmaps_shape,
239
+ # maxshape=flatmaps_maxshape,
240
+ # dtype=np.float16,
241
+ # chunks=True # Enable chunking for efficient resizing
242
+ # )
243
+
244
+ # # Resize datasets to accommodate new data
245
+ # flatmaps_dset.resize(total_samples + batch_size, axis=0)
246
+
247
+ # # Write data to the datasets
248
+ # flatmaps_dset[total_samples:total_samples + batch_size] = images.numpy().astype(np.float16)
249
+
250
+ # total_samples += batch_size
251
+
252
+ # print(f"Processed {total_samples} samples")
253
+ # np.save('metadata_train_HCP.npy', meta_array)
254
+
255
+
256
+ # ### Preparing data
257
+
258
+ # In[4]:
259
+
260
+
261
+ from sklearn.preprocessing import LabelEncoder
262
+
263
+ INCLUDE_CONDS = {
264
+ "fear",
265
+ "neut",
266
+ "math",
267
+ "story",
268
+ "lf",
269
+ "lh",
270
+ "rf",
271
+ "rh",
272
+ "t",
273
+ "match",
274
+ "relation",
275
+ "mental",
276
+ "rnd",
277
+ "0bk_body",
278
+ "2bk_body",
279
+ "0bk_faces",
280
+ "2bk_faces",
281
+ "0bk_places",
282
+ "2bk_places",
283
+ "0bk_tools",
284
+ "2bk_tools",
285
+ }
286
+
287
+ # test_data = []
288
+
289
+ # # Iterate over the DataLoader with a progress bar
290
+ # for sample in tqdm(train_dl, desc="Processing samples"):
291
+ # x = sample['image']
292
+ # y = sample['meta']['trial_type']
293
+ # key = sample['meta']['key']
294
+ # print(x.shape, y, key)
295
+ # break
296
+ # Initialize the label encoder
297
+ label_encoder = LabelEncoder()
298
+ label_encoder.fit(sorted(INCLUDE_CONDS)) # Ensure consistent ordering
299
+
300
+ num_classes = len(label_encoder.classes_)
301
+ print(f"Number of classes: {num_classes}")
302
+
303
+
304
+ # In[5]:
305
+
306
+
307
+ f_train = h5py.File('/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/train_hcp.hdf5', 'r')
308
+ flatmaps_train = f_train['flatmaps']
309
+
310
+ f_test = h5py.File('/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/test_hcp.hdf5', 'r')
311
+ flatmaps_test = f_test['flatmaps']
312
+
313
+ metadata_train = np.load('/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/metadata_train_HCP.npy', allow_pickle=True)
314
+ metadata_test = np.load('/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/metadata_test_HCP.npy', allow_pickle=True)
315
+
316
+
317
+ # In[6]:
318
+
319
+
320
+ from torch.utils.data import Dataset, DataLoader
321
+
322
+ class HCPFlatDataset(Dataset):
323
+ def __init__(self, flatmaps, metadata):
324
+ self.flatmaps = flatmaps
325
+ self.metadata = metadata
326
+
327
+ def __len__(self):
328
+ return len(self.metadata)
329
+
330
+ def __getitem__(self, idx):
331
+ return self.flatmaps[idx], json.loads(self.metadata[idx])
332
+ print("Moving datasets to ram")
333
+ # Loading to cpu for faster training, this can take several minutes. Remove this [:] if you want to move one at the time.
334
+ train_dataset = HCPFlatDataset(flatmaps_train, metadata_train)
335
+ train_dl = DataLoader(train_dataset, batch_size=batch_size, shuffle=True, num_workers=num_workers)
336
+
337
+ test_dataset = HCPFlatDataset(flatmaps_test, metadata_test)
338
+ test_dl = DataLoader(test_dataset, batch_size=batch_size, shuffle=False, num_workers=0)
339
+ print("Datasets ready")
340
+
341
+
342
+ # ### Creating and loading Model
343
+
344
+ # In[7]:
345
+
346
+
347
+ from mae_utils.flat import load_hcp_flat_mask
348
+ from mae_utils.flat import create_hcp_flat
349
+ from mae_utils.flat import batch_unmask
350
+ import mae_utils.visualize as vis
351
+
352
+ flat_mask = load_hcp_flat_mask(hcp_flat_path)
353
+
354
+ mae_model = flat_models.mae_vit_large_fmri(
355
+ patch_size=patch_size,
356
+ decoder_embed_dim=decoder_embed_dim,
357
+ t_patch_size=t_patch_size,
358
+ pred_t_dim=pred_t_dim,
359
+ decoder_depth=4,
360
+ cls_embed=cls_embed,
361
+ norm_pix_loss=norm_pix_loss,
362
+ no_qkv_bias=no_qkv_bias,
363
+ sep_pos_embed=sep_pos_embed,
364
+ trunc_init=trunc_init,
365
+ pct_masks_to_decode=pct_masks_to_decode,
366
+ img_mask=flat_mask,
367
+ )
368
+
369
+
370
+ # In[8]:
371
+
372
+
373
+ checkpoint_files = [f for f in os.listdir(outdir) if f.endswith('.pth')]
374
+
375
+ if utils.is_interactive():
376
+ latest_checkpoint = "epoch99.pth"
377
+ else:
378
+ latest_checkpoint = sys.argv[2]
379
+ print(f"latest_checkpoint: {latest_checkpoint}")
380
+
381
+ # Load the checkpoint
382
+ checkpoint_path = os.path.join(outdir, latest_checkpoint)
383
+
384
+ state = torch.load(checkpoint_path)
385
+ mae_model.load_state_dict(state["model_state_dict"], strict=False)
386
+ mae_model.to(device)
387
+
388
+ print(f"\nLoaded checkpoint {latest_checkpoint} from {outdir}\n")
389
+
390
+
391
+ # In[9]:
392
+
393
+
394
+ class LinearClassifier(nn.Module):
395
+ def __init__(self, input_dim, num_classes):
396
+ super(LinearClassifier, self).__init__()
397
+ self.linear = nn.Linear(input_dim, num_classes)
398
+
399
+ def forward(self, x):
400
+ # Flatten the input except for the batch dimension
401
+ x = x.view(x.size(0), -1)
402
+ out = self.linear(x)
403
+ return out # Raw logits
404
+
405
+ # Determine the input dimension from a single sample
406
+ # Assuming images are of shape [1, 16, 144, 320]
407
+ input_dim = np.prod(mae_model(torch.randn(1,1,16,144,320).to(device),global_pool=global_pool, forward_features = True).shape[1:])
408
+ print(f"Input dimension: {input_dim}")
409
+
410
+
411
+ # In[10]:
412
+
413
+
414
+ class FullModel(nn.Module):
415
+ def __init__(self, lc_model, mae_model):
416
+ super(FullModel, self).__init__()
417
+ self.lc_model = lc_model
418
+ self.mae_model = mae_model
419
+
420
+
421
+ def forward(self, x, gsr):
422
+ x = self.mae_model(x, global_pool=global_pool, forward_features = True)
423
+ x = self.lc_model(x)
424
+ return x
425
+
426
+
427
+ # In[11]:
428
+
429
+
430
+ # Initialize the model
431
+ lc_model = LinearClassifier(input_dim=input_dim, num_classes=num_classes)
432
+
433
+ model = FullModel(lc_model, mae_model)
434
+
435
+ # Move the model to the GPU
436
+ model.to(device)
437
+
438
+ # Define loss function
439
+ criterion = nn.CrossEntropyLoss()
440
+
441
+ # Define optimizer with L2 regularization (weight_decay)
442
+ learning_rate = 1e-4
443
+ weight_decay = 1e-5 # Adjust based on your needs
444
+ optimizer = torch.optim.Adam(model.parameters(), lr=learning_rate, weight_decay=weight_decay)
445
+ num_epochs = 20 # Adjust as needed
446
+
447
+
448
+ # ### Data
449
+
450
+ # In[16]:
451
+
452
+
453
+ import uuid
454
+
455
+ myuuid = uuid.uuid4()
456
+ str(myuuid)
457
+
458
+
459
+ # In[17]:
460
+
461
+
462
+ import wandb
463
+
464
+ if utils.is_interactive():
465
+ print("Running in interactive notebook. Disabling W&B and ckpt saving.")
466
+ wandb_log = True
467
+ save_ckpt = True
468
+
469
+ if wandb_log:
470
+ wandb_project = 'fMRI-foundation-model'
471
+ wandb_config = {
472
+ "model_name": model_name+'_HCP_FT',
473
+ "batch_size": batch_size,
474
+ "learning_rate": learning_rate,
475
+ "weight_decay": weight_decay,
476
+ "num_epochs": num_epochs,
477
+ "seed": seed,
478
+ }
479
+ print("wandb_config:\n", wandb_config)
480
+ random_id = str(uuid.uuid4())
481
+ print("wandb_id:", "HCPflat_raw" + f"_{random_id}")
482
+ wandb.init(
483
+ id=model_name+'_HCP_FT' + f"_{random_id}",
484
+ project=wandb_project,
485
+ name=model_name+'_HCP_FT',
486
+ config=wandb_config,
487
+ resume="allow",
488
+ )
489
+
490
+
491
+ # In[13]:
492
+
493
+
494
+ for epoch in range(num_epochs):
495
+ running_train_loss = 0.0
496
+ correct_train = 0
497
+ total_train = 0
498
+ step = 0
499
+
500
+ # with torch.amp.autocast(device_type='cuda'):
501
+ # Training Phase
502
+ model.train()
503
+ for batch in tqdm(train_dl, desc=f"Epoch {epoch+1}/{num_epochs} - Training"):
504
+ optimizer.zero_grad()
505
+ images = batch[0].to(device).float().unsqueeze(1) #fix this # Shape: [batch_size, 1, 16, 144, 320]
506
+ labels = batch[1]['trial_type'] # List of labels
507
+
508
+ encoded_labels = label_encoder.transform(labels)
509
+ encoded_labels = torch.tensor(encoded_labels, dtype=torch.long).to(device) # Shape: [batch_size]
510
+
511
+ # Forward pass
512
+ outputs = model(images, gsr=gsr) # Shape: [num_train_samples, num_classes]
513
+
514
+ # Compute loss
515
+ loss = criterion(outputs, encoded_labels)
516
+
517
+ # Backward pass and optimization
518
+ loss.backward()
519
+ optimizer.step()
520
+
521
+ # Accumulate loss
522
+ running_train_loss += loss.item() * images.size(0)
523
+
524
+
525
+ # Calculate accuracy
526
+ _, predicted = torch.max(outputs, 1)
527
+
528
+ correct_train += (predicted == encoded_labels).sum().item()
529
+ total_train += encoded_labels.size(0)
530
+
531
+ step = step + 1
532
+ if step % 100 == 0:
533
+ print(f"Step [{step}/{len(train_dl)}] - Training Loss: {loss.item():.4f} - Training Accuracy: {100 * correct_train / total_train:.2f}%")
534
+ # thth
535
+
536
+ epoch_train_loss = running_train_loss / total_train if total_train > 0 else 0.0
537
+ train_accuracy = 100 * correct_train / total_train if total_train > 0 else 0.0
538
+
539
+ # Validation Phase
540
+ model.eval()
541
+ running_val_loss = 0.0
542
+ correct_val = 0
543
+ total_val = 0
544
+
545
+ with torch.no_grad():
546
+ for batch in tqdm(test_dl, desc=f"Epoch {epoch+1}/{num_epochs} - Validation"):
547
+
548
+ images = batch[0].to(device).float().unsqueeze(1) #fix this
549
+ labels = batch[1]['trial_type']
550
+
551
+ # Encode labels to integer indices
552
+ encoded_labels = label_encoder.transform(labels)
553
+ encoded_labels = torch.tensor(encoded_labels, dtype=torch.long).to(device)
554
+
555
+
556
+ # Forward pass
557
+ outputs = model(images, gsr=gsr)
558
+
559
+ # Compute loss
560
+ loss = criterion(outputs, encoded_labels)
561
+
562
+ # Accumulate loss
563
+ running_val_loss += loss.item() * images.size(0)
564
+
565
+ # Calculate accuracy
566
+ _, predicted = torch.max(outputs, 1)
567
+ correct_val += (predicted == encoded_labels).sum().item()
568
+ total_val += encoded_labels.size(0)
569
+
570
+
571
+
572
+ epoch_val_loss = running_val_loss / total_val if total_val > 0 else 0.0
573
+ val_accuracy = 100 * correct_val / total_val if total_val > 0 else 0.0
574
+
575
+ print(f"Epoch [{epoch+1}/{num_epochs}] "
576
+ f"- Training Loss: {epoch_train_loss:.4f}, Training Accuracy: {train_accuracy:.2f}% "
577
+ f"- Validation Loss: {epoch_val_loss:.4f}, Validation Accuracy: {val_accuracy:.2f}%")
578
+
579
+ if wandb_log:
580
+ wandb.log({
581
+ "epoch_train_loss": epoch_train_loss,
582
+ "epoch_val_loss": epoch_val_loss,
583
+ "train_accuracy": train_accuracy,
584
+ "val_accuracy": val_accuracy,
585
+ })
586
+ if save_ckpt:
587
+ outdir = os.path.abspath(f'checkpoints/{model_name+"HCP_FT"}')
588
+ os.makedirs(outdir, exist_ok=True)
589
+ print("outdir", outdir)
590
+ # Save model and config
591
+ torch.save(model.state_dict(), f"{outdir}/model.pth")
592
+ with open(f"{outdir}/config.yaml", 'w') as f:
593
+ yaml.dump(wandb_config, f)
594
+ print(f"Saved model and config to {outdir}")
595
+
596
+
fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/files/output.log ADDED
@@ -0,0 +1 @@
 
 
1
+ Epoch 1/20 - Training: 1%| | 87/13913 [00:42<1:19:54, 2.88it/s]
fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/files/requirements.txt ADDED
@@ -0,0 +1,198 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ protobuf==5.28.2
2
+ imageio==2.35.1
3
+ MarkupSafe==3.0.0
4
+ regex==2024.9.11
5
+ matplotlib==3.9.2
6
+ notebook==7.2.2
7
+ debugpy==1.8.6
8
+ aiosignal==1.3.1
9
+ jupyter_core==5.7.2
10
+ torchaudio==2.4.1+cu121
11
+ python-json-logger==2.0.7
12
+ six==1.16.0
13
+ scikit-image==0.24.0
14
+ types-python-dateutil==2.9.0.20241003
15
+ PyYAML==6.0.2
16
+ httpcore==1.0.6
17
+ clip==1.0
18
+ babel==2.16.0
19
+ webcolors==24.8.0
20
+ omegaconf==2.3.0
21
+ webencodings==0.5.1
22
+ kiwisolver==1.4.7
23
+ uri-template==1.3.0
24
+ diffusers==0.23.0
25
+ idna==3.10
26
+ fsspec==2024.9.0
27
+ parso==0.8.4
28
+ setuptools==65.5.0
29
+ tornado==6.4.1
30
+ webdataset==0.2.100
31
+ decord==0.6.0
32
+ nvidia-curand-cu12==10.3.2.106
33
+ ipykernel==6.29.5
34
+ jupyter==1.1.1
35
+ pexpect==4.9.0
36
+ kornia_rs==0.1.5
37
+ iopath==0.1.10
38
+ async-lru==2.0.4
39
+ future==1.0.0
40
+ torchvision==0.19.1+cu121
41
+ botocore==1.34.162
42
+ cycler==0.12.1
43
+ tzdata==2024.2
44
+ jupyter_server_terminals==0.5.3
45
+ click==8.1.7
46
+ einops==0.8.0
47
+ pyzmq==26.2.0
48
+ jupyter_client==8.6.3
49
+ nbconvert==7.16.4
50
+ scikit-learn==1.5.2
51
+ executing==2.1.0
52
+ asttokens==2.4.1
53
+ docker-pycreds==0.4.0
54
+ matplotlib-inline==0.1.7
55
+ overrides==7.7.0
56
+ websocket-client==1.8.0
57
+ nbformat==5.10.4
58
+ elbow==0.1.1
59
+ contourpy==1.3.0
60
+ nvidia-cudnn-cu12==9.1.0.70
61
+ transformers==4.44.2
62
+ gitdb==4.0.11
63
+ jupyterlab_nvdashboard==0.11.0
64
+ lazy_loader==0.4
65
+ jsonpointer==3.0.0
66
+ notebook_shim==0.2.4
67
+ nvidia-nccl-cu12==2.20.5
68
+ ffmpeg-python==0.2.0
69
+ triton==3.0.0
70
+ mistune==3.0.2
71
+ python-dateutil==2.9.0.post0
72
+ beautifulsoup4==4.12.3
73
+ nbclient==0.10.0
74
+ h5py==3.12.1
75
+ ftfy==6.2.3
76
+ zipp==3.20.2
77
+ ptyprocess==0.7.0
78
+ huggingface-hub==0.25.1
79
+ pytz==2024.2
80
+ jupyterlab_pygments==0.3.0
81
+ nvidia-cublas-cu12==12.1.3.1
82
+ pandocfilters==1.5.1
83
+ Jinja2==3.1.4
84
+ arrow==1.3.0
85
+ rpds-py==0.20.0
86
+ jupyter_server==2.14.2
87
+ simplejson==3.19.3
88
+ networkx==3.3
89
+ packaging==24.1
90
+ traitlets==5.14.3
91
+ pandas==2.2.3
92
+ xformers==0.0.22.post7
93
+ lightning-utilities==0.11.7
94
+ tifffile==2024.9.20
95
+ nvidia-cuda-cupti-cu12==12.1.105
96
+ mpmath==1.3.0
97
+ GitPython==3.1.43
98
+ scipy==1.14.1
99
+ jsonschema==4.23.0
100
+ prompt_toolkit==3.0.48
101
+ s3transfer==0.10.2
102
+ multidict==6.1.0
103
+ bleach==6.1.0
104
+ sentry-sdk==2.15.0
105
+ nibabel==5.2.1
106
+ accelerate==1.0.0
107
+ pyarrow==17.0.0
108
+ threadpoolctl==3.5.0
109
+ attrs==24.2.0
110
+ rfc3986-validator==0.1.1
111
+ nvidia-cuda-runtime-cu12==12.1.105
112
+ ipywidgets==8.1.5
113
+ frozenlist==1.4.1
114
+ pycparser==2.22
115
+ jupyterlab_server==2.27.3
116
+ nvidia-cuda-nvrtc-cu12==12.1.105
117
+ yarl==1.13.1
118
+ setproctitle==1.3.3
119
+ isoduration==20.11.0
120
+ Pygments==2.18.0
121
+ jedi==0.19.1
122
+ boto3==1.34.57
123
+ tokenizers==0.19.1
124
+ referencing==0.35.1
125
+ rfc3339-validator==0.1.4
126
+ pillow==10.4.0
127
+ jupyterlab==4.2.5
128
+ stack-data==0.6.3
129
+ h11==0.14.0
130
+ anyio==4.6.0
131
+ nilearn==0.10.4
132
+ nvidia-cusolver-cu12==11.4.5.107
133
+ tinycss2==1.3.0
134
+ defusedxml==0.7.1
135
+ argon2-cffi-bindings==21.2.0
136
+ soupsieve==2.6
137
+ nest-asyncio==1.6.0
138
+ torchmetrics==1.3.0.post0
139
+ tqdm==4.66.5
140
+ cffi==1.17.1
141
+ charset-normalizer==3.3.2
142
+ jsonschema-specifications==2023.12.1
143
+ decorator==5.1.1
144
+ open_clip_torch==2.26.1
145
+ jupyter-events==0.10.0
146
+ smart-open==7.0.5
147
+ antlr4-python3-runtime==4.9.3
148
+ prometheus_client==0.21.0
149
+ kornia==0.7.3
150
+ typing_extensions==4.12.2
151
+ sniffio==1.3.1
152
+ joblib==1.4.2
153
+ comm==0.2.2
154
+ aiohappyeyeballs==2.4.3
155
+ numpy==2.1.2
156
+ braceexpand==0.1.7
157
+ certifi==2024.8.30
158
+ psutil==6.0.0
159
+ pyparsing==3.1.4
160
+ pure_eval==0.2.3
161
+ nvidia-cusparse-cu12==12.1.0.106
162
+ wandb==0.18.3
163
+ urllib3==2.2.3
164
+ smmap==5.0.1
165
+ platformdirs==4.3.6
166
+ torch==2.4.1+cu121
167
+ requests==2.32.3
168
+ json5==0.9.25
169
+ nvidia-nvjitlink-cu12==12.6.77
170
+ jupyterlab_widgets==3.0.13
171
+ lxml==5.3.0
172
+ httpx==0.27.2
173
+ opencv-python==4.6.0.66
174
+ portalocker==2.10.1
175
+ pytorch-lightning==2.0.1
176
+ sympy==1.13.3
177
+ wcwidth==0.2.13
178
+ jmespath==1.0.1
179
+ fqdn==1.5.1
180
+ pynvml==11.5.3
181
+ pip==24.0
182
+ wrapt==1.16.0
183
+ aiohttp==3.10.9
184
+ filelock==3.16.1
185
+ fonttools==4.54.1
186
+ fastjsonschema==2.20.0
187
+ jupyter-console==6.6.3
188
+ widgetsnbextension==4.0.13
189
+ timm==1.0.9
190
+ nvidia-cufft-cu12==11.0.2.54
191
+ ipython==8.28.0
192
+ nvidia-nvtx-cu12==12.1.105
193
+ jupyter-lsp==2.2.5
194
+ safetensors==0.4.5
195
+ terminado==0.18.1
196
+ argon2-cffi==23.1.0
197
+ Send2Trash==1.8.3
198
+ importlib_metadata==8.5.0
fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/files/wandb-metadata.json ADDED
@@ -0,0 +1,131 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-5.15.0-1058-aws-x86_64-with-glibc2.31",
3
+ "python": "3.11.10",
4
+ "startedAt": "2024-10-23T13:24:33.508901Z",
5
+ "args": [
6
+ "NSDflat_large_gsrFalse_",
7
+ "epoch99.pth"
8
+ ],
9
+ "program": "/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.py",
10
+ "codePath": "src/HCP_downstream_finetune.py",
11
+ "git": {
12
+ "remote": "https://github.com/MedARC-AI/fMRI-foundation-model",
13
+ "commit": "5908f2fe5945884e8a2044da614319dff358f6e5"
14
+ },
15
+ "email": "torrico.villanueva.cesar.kadir@gmail.com",
16
+ "root": "/weka/proj-fmri/ckadirt/fMRI-foundation-model/src",
17
+ "host": "ip-10-0-187-233",
18
+ "username": "ckadirt",
19
+ "executable": "/admin/home-ckadirt/foundation_env/bin/python",
20
+ "codePathLocal": "HCP_downstream_finetune.py",
21
+ "cpu_count": 96,
22
+ "cpu_count_logical": 192,
23
+ "gpu": "[NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3]",
24
+ "gpu_count": 8,
25
+ "disk": {
26
+ "/": {
27
+ "total": "249555763200",
28
+ "used": "182511673344"
29
+ }
30
+ },
31
+ "memory": {
32
+ "total": "2147443400704"
33
+ },
34
+ "cpu": {
35
+ "count": 96,
36
+ "countLogical": 192
37
+ },
38
+ "gpu_nvidia": [
39
+ {
40
+ "name": "NVIDIA H100 80GB HBM3",
41
+ "memoryTotal": "85520809984",
42
+ "cudaCores": 16896,
43
+ "architecture": "Hopper"
44
+ },
45
+ {
46
+ "name": "NVIDIA H100 80GB HBM3",
47
+ "memoryTotal": "85520809984",
48
+ "cudaCores": 16896,
49
+ "architecture": "Hopper"
50
+ },
51
+ {
52
+ "name": "NVIDIA H100 80GB HBM3",
53
+ "memoryTotal": "85520809984",
54
+ "cudaCores": 16896,
55
+ "architecture": "Hopper"
56
+ },
57
+ {
58
+ "name": "NVIDIA H100 80GB HBM3",
59
+ "memoryTotal": "85520809984",
60
+ "cudaCores": 16896,
61
+ "architecture": "Hopper"
62
+ },
63
+ {
64
+ "name": "NVIDIA H100 80GB HBM3",
65
+ "memoryTotal": "85520809984",
66
+ "cudaCores": 16896,
67
+ "architecture": "Hopper"
68
+ },
69
+ {
70
+ "name": "NVIDIA H100 80GB HBM3",
71
+ "memoryTotal": "85520809984",
72
+ "cudaCores": 16896,
73
+ "architecture": "Hopper"
74
+ },
75
+ {
76
+ "name": "NVIDIA H100 80GB HBM3",
77
+ "memoryTotal": "85520809984",
78
+ "cudaCores": 16896,
79
+ "architecture": "Hopper"
80
+ },
81
+ {
82
+ "name": "NVIDIA H100 80GB HBM3",
83
+ "memoryTotal": "85520809984",
84
+ "cudaCores": 16896,
85
+ "architecture": "Hopper"
86
+ }
87
+ ],
88
+ "slurm": {
89
+ "cluster_name": "sagemaker2",
90
+ "conf": "/opt/slurm/etc/slurm.conf",
91
+ "cpus_on_node": "20",
92
+ "gpus_on_node": "1",
93
+ "gpus_per_task": "1",
94
+ "gtids": "0",
95
+ "job_account": "fmri",
96
+ "job_cpus_per_node": "20",
97
+ "job_end_time": "1729733030",
98
+ "job_gid": "1879800513",
99
+ "job_gpus": "4",
100
+ "job_id": "528371",
101
+ "job_name": "finetuneHCP",
102
+ "job_nodelist": "ip-10-0-187-233",
103
+ "job_num_nodes": "1",
104
+ "job_partition": "p5",
105
+ "job_qos": "idle",
106
+ "job_start_time": "1729689830",
107
+ "job_uid": "1879804696",
108
+ "job_user": "ckadirt",
109
+ "jobid": "528371",
110
+ "localid": "0",
111
+ "mem_per_cpu": "11500",
112
+ "nnodes": "1",
113
+ "node_aliases": "(null)",
114
+ "nodeid": "0",
115
+ "nodelist": "ip-10-0-187-233",
116
+ "nprocs": "1",
117
+ "ntasks": "1",
118
+ "ntasks_per_node": "1",
119
+ "prio_process": "0",
120
+ "procid": "0",
121
+ "script_context": "prolog_task",
122
+ "submit_dir": "/weka/proj-fmri/ckadirt/fMRI-foundation-model/src",
123
+ "submit_host": "ip-172-17-12-61",
124
+ "task_pid": "978468",
125
+ "tasks_per_node": "1",
126
+ "topology_addr": "ip-10-0-187-233",
127
+ "topology_addr_pattern": "node",
128
+ "working_cluster": "sagemaker2:ip-172-17-63-161:6817:9984:109"
129
+ },
130
+ "cudaVersion": "12.2"
131
+ }
fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/logs/debug-core.log ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2024-10-23T13:24:32.854011185Z","level":"INFO","msg":"started logging, with flags","port-filename":"/tmp/tmpkg9us9hs/port-978737.txt","pid":978737,"debug":false,"disable-analytics":false}
2
+ {"time":"2024-10-23T13:24:32.854028775Z","level":"INFO","msg":"started logging, with flags","port-filename":"/tmp/tmp240fhy3c/port-978738.txt","pid":978738,"debug":false,"disable-analytics":false}
3
+ {"time":"2024-10-23T13:24:32.854411383Z","level":"INFO","msg":"FeatureState","shutdownOnParentExitEnabled":false}
4
+ {"time":"2024-10-23T13:24:32.854434863Z","level":"INFO","msg":"FeatureState","shutdownOnParentExitEnabled":false}
5
+ {"time":"2024-10-23T13:24:32.860638292Z","level":"INFO","msg":"Will exit if parent process dies.","ppid":978737}
6
+ {"time":"2024-10-23T13:24:32.860629982Z","level":"INFO","msg":"server is running","addr":{"IP":"127.0.0.1","Port":33033,"Zone":""}}
7
+ {"time":"2024-10-23T13:24:32.861562472Z","level":"INFO","msg":"Will exit if parent process dies.","ppid":978738}
8
+ {"time":"2024-10-23T13:24:32.861559652Z","level":"INFO","msg":"server is running","addr":{"IP":"127.0.0.1","Port":37409,"Zone":""}}
9
+ {"time":"2024-10-23T13:24:33.019117935Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"127.0.0.1:53870"}
10
+ {"time":"2024-10-23T13:24:33.019289548Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"127.0.0.1:58918"}
11
+ {"time":"2024-10-23T13:24:33.497469173Z","level":"INFO","msg":"handleInformInit: received","streamId":"HCPflat_large_gsrFalse__HCP_FT_ad26278b-8855-4c96-9acd-5892c21f250e","id":"127.0.0.1:53870"}
12
+ {"time":"2024-10-23T13:24:33.5093469Z","level":"INFO","msg":"handleInformInit: received","streamId":"NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0","id":"127.0.0.1:58918"}
13
+ {"time":"2024-10-23T13:24:33.555261534Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"HCPflat_large_gsrFalse__HCP_FT_ad26278b-8855-4c96-9acd-5892c21f250e","id":"127.0.0.1:53870"}
14
+ {"time":"2024-10-23T13:24:33.566389415Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0","id":"127.0.0.1:58918"}
fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/logs/debug-internal.log ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2024-10-23T13:24:33.526150099Z","level":"INFO","msg":"using version","core version":"0.18.3"}
2
+ {"time":"2024-10-23T13:24:33.526168419Z","level":"INFO","msg":"created symlink","path":"/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/logs/debug-core.log"}
3
+ {"time":"2024-10-23T13:24:33.532483661Z","level":"ERROR","msg":"dialing: google: could not find default credentials. See https://cloud.google.com/docs/authentication/external/set-up-adc for more information"}
4
+ {"time":"2024-10-23T13:24:33.566355274Z","level":"INFO","msg":"created new stream","id":"NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0"}
5
+ {"time":"2024-10-23T13:24:33.566382274Z","level":"INFO","msg":"stream: started","id":"NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0"}
6
+ {"time":"2024-10-23T13:24:33.566411135Z","level":"INFO","msg":"sender: started","stream_id":{"value":"NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0"}}
7
+ {"time":"2024-10-23T13:24:33.566402655Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0"}}
8
+ {"time":"2024-10-23T13:24:33.566416775Z","level":"INFO","msg":"handler: started","stream_id":{"value":"NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0"}}
9
+ {"time":"2024-10-23T13:24:34.141916511Z","level":"INFO","msg":"wandb-core","!BADKEY":null}
10
+ {"time":"2024-10-23T13:24:34.148171272Z","level":"INFO","msg":"Starting system monitor"}
11
+ {"time":"2024-10-23T13:24:34.218430101Z","level":"ERROR","msg":"git repo not found","error":"repository does not exist"}
fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/logs/debug.log ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 2024-10-23 13:24:33,497 INFO MainThread:978737 [wandb_setup.py:_flush():79] Current SDK version is 0.18.3
2
+ 2024-10-23 13:24:33,497 INFO MainThread:978737 [wandb_setup.py:_flush():79] Configure stats pid to 978737
3
+ 2024-10-23 13:24:33,497 INFO MainThread:978737 [wandb_setup.py:_flush():79] Loading settings from /admin/home-ckadirt/.config/wandb/settings
4
+ 2024-10-23 13:24:33,497 INFO MainThread:978737 [wandb_setup.py:_flush():79] Loading settings from /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/settings
5
+ 2024-10-23 13:24:33,497 INFO MainThread:978737 [wandb_setup.py:_flush():79] Loading settings from environment variables: {}
6
+ 2024-10-23 13:24:33,498 INFO MainThread:978737 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': None, '_disable_service': None}
7
+ 2024-10-23 13:24:33,498 INFO MainThread:978737 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'src/HCP_downstream_finetune.py', 'program_abspath': '/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.py', 'program': '/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.py'}
8
+ 2024-10-23 13:24:33,498 INFO MainThread:978737 [wandb_setup.py:_flush():79] Applying login settings: {}
9
+ 2024-10-23 13:24:33,498 INFO MainThread:978737 [wandb_init.py:_log_setup():532] Logging user logs to /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/logs/debug.log
10
+ 2024-10-23 13:24:33,499 INFO MainThread:978737 [wandb_init.py:_log_setup():533] Logging internal logs to /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/logs/debug-internal.log
11
+ 2024-10-23 13:24:33,499 INFO MainThread:978737 [wandb_init.py:init():617] calling init triggers
12
+ 2024-10-23 13:24:33,499 INFO MainThread:978737 [wandb_init.py:init():624] wandb.init called with sweep_config: {}
13
+ config: {'model_name': 'NSDflat_large_gsrFalse__HCP_FT', 'batch_size': 8, 'learning_rate': 0.0001, 'weight_decay': 1e-05, 'num_epochs': 20, 'seed': 42}
14
+ 2024-10-23 13:24:33,499 INFO MainThread:978737 [wandb_init.py:init():667] starting backend
15
+ 2024-10-23 13:24:33,499 INFO MainThread:978737 [wandb_init.py:init():671] sending inform_init request
16
+ 2024-10-23 13:24:33,508 INFO MainThread:978737 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
17
+ 2024-10-23 13:24:33,508 INFO MainThread:978737 [wandb_init.py:init():684] backend started and connected
18
+ 2024-10-23 13:24:33,532 INFO MainThread:978737 [wandb_init.py:init():779] updated telemetry
19
+ 2024-10-23 13:24:33,576 INFO MainThread:978737 [wandb_init.py:init():812] communicating run to backend with 90.0 second timeout
20
+ 2024-10-23 13:24:34,127 INFO MainThread:978737 [wandb_init.py:init():863] starting run threads in backend
21
+ 2024-10-23 13:24:34,702 INFO MainThread:978737 [wandb_run.py:_console_start():2465] atexit reg
22
+ 2024-10-23 13:24:34,702 INFO MainThread:978737 [wandb_run.py:_redirect():2313] redirect: wrap_raw
23
+ 2024-10-23 13:24:34,702 INFO MainThread:978737 [wandb_run.py:_redirect():2378] Wrapping output streams.
24
+ 2024-10-23 13:24:34,703 INFO MainThread:978737 [wandb_run.py:_redirect():2403] Redirects installed.
25
+ 2024-10-23 13:24:34,711 INFO MainThread:978737 [wandb_init.py:init():907] run started, returning control to user process
fMRI-foundation-model/src/wandb/run-20241023_132433-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0/run-NSDflat_large_gsrFalse__HCP_FT_1adf6af0-9660-46b3-9162-41befd06f8c0.wandb ADDED
Binary file (32.8 kB). View file
 
fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/files/code/src/HCP_downstream_finetune.py ADDED
@@ -0,0 +1,596 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python
2
+ # coding: utf-8
3
+
4
+ # In[1]:
5
+
6
+
7
+ # Import packages and setup gpu configuration.
8
+ # This code block shouldnt need to be adjusted!
9
+ import os
10
+ import sys
11
+ import json
12
+ import yaml
13
+ import numpy as np
14
+ import copy
15
+ import math
16
+ import time
17
+ import random
18
+ from tqdm.auto import tqdm
19
+ import webdataset as wds
20
+ import matplotlib.pyplot as plt
21
+
22
+ import torch
23
+ import torch.nn as nn
24
+ from torchvision import transforms
25
+ import utils
26
+ from mae_utils.flat_models import *
27
+ import h5py
28
+ from mae_utils import flat_models
29
+
30
+ # tf32 data type is faster than standard float32
31
+ torch.backends.cuda.matmul.allow_tf32 = True
32
+ # following fixes a Conv3D CUDNN_NOT_SUPPORTED error
33
+ torch.backends.cudnn.benchmark = True
34
+
35
+ # ## MODEL TO LOAD ##
36
+ if utils.is_interactive():
37
+ model_name = "HCPflat_large_gsrFalse_"
38
+ else:
39
+ model_name = sys.argv[1]
40
+
41
+
42
+ # outdir = os.path.abspath(f'checkpoints/{model_name}')
43
+ outdir = os.path.abspath(f'checkpoints/{model_name}')
44
+
45
+ print("outdir", outdir)
46
+ # Load previous config.yaml if available
47
+ if os.path.exists(f"{outdir}/config.yaml"):
48
+ config = yaml.load(open(f"{outdir}/config.yaml", 'r'), Loader=yaml.FullLoader)
49
+ print(f"Loaded config.yaml from ckpt folder {outdir}")
50
+ # create global variables from the config
51
+ print("\n__CONFIG__")
52
+ for attribute_name in config.keys():
53
+ print(f"{attribute_name} = {config[attribute_name]}")
54
+ globals()[attribute_name] = config[f'{attribute_name}']
55
+ print("\n")
56
+
57
+ world_size = os.getenv('WORLD_SIZE')
58
+ if world_size is None:
59
+ world_size = 1
60
+ else:
61
+ world_size = int(world_size)
62
+ print(f"WORLD_SIZE={world_size}")
63
+
64
+ if utils.is_interactive():
65
+ # Following allows you to change functions in models.py or utils.py and
66
+ # have this notebook automatically update with your revisions
67
+ get_ipython().run_line_magic('load_ext', 'autoreload')
68
+ get_ipython().run_line_magic('autoreload', '2')
69
+
70
+ batch_size = probe_batch_size
71
+ num_epochs = probe_num_epochs
72
+
73
+ data_type = torch.float32 # change depending on your mixed_precision
74
+ global_batch_size = batch_size * world_size
75
+
76
+ device = torch.device('cuda')
77
+
78
+ hcp_flat_path = "/weka/proj-medarc/shared/HCP-Flat"
79
+ # seed = 42
80
+ # num_frames = 16
81
+ # gsr = False
82
+ # num_workers = 10
83
+ # batch_size = 128
84
+
85
+ print("PID of this process =",os.getpid())
86
+ utils.seed_everything(seed)
87
+
88
+
89
+ # In[2]:
90
+
91
+
92
+ if os.getenv('global_pool') == "False":
93
+ global_pool = False
94
+ else:
95
+ global_pool = True
96
+ print(f"global_pool = {global_pool}")
97
+
98
+ try:
99
+ gsr
100
+ except:
101
+ gsr = True
102
+ print("set gsr to True")
103
+ print(f"gsr = {gsr}")
104
+
105
+
106
+ # In[3]:
107
+
108
+
109
+ #### UNCOMMENT THIS TO SAVE THE HCP-FLAT IN HDF5 FORMAT
110
+
111
+
112
+ # from torch.utils.data import default_collate
113
+ # from mae_utils.flat import load_hcp_flat_mask
114
+ # from mae_utils.flat import create_hcp_flat
115
+ # from mae_utils.flat import batch_unmask
116
+ # import mae_utils.visualize as vis
117
+
118
+
119
+ # batch_size = 26
120
+ # print(f"changed batch_size to {batch_size}")
121
+
122
+ # ## Test ##
123
+ # datasets_to_include = "HCP"
124
+ # assert "HCP" in datasets_to_include
125
+ # test_dataset = create_hcp_flat(root=hcp_flat_path,
126
+ # clip_mode="event", frames=num_frames, shuffle=False, gsr=gsr, sub_list = 'test')
127
+ # test_dl = wds.WebLoader(
128
+ # test_dataset.batched(batch_size, partial=False, collation_fn=default_collate),
129
+ # batch_size=None,
130
+ # shuffle=False,
131
+ # num_workers=num_workers,
132
+ # pin_memory=True,
133
+ # )
134
+
135
+ # ## Train ##
136
+ # assert "HCP" in datasets_to_include
137
+ # train_dataset = create_hcp_flat(root=hcp_flat_path,
138
+ # clip_mode="event", frames=num_frames, shuffle=False, gsr=gsr, sub_list = 'train')
139
+ # train_dl = wds.WebLoader(
140
+ # train_dataset.batched(batch_size, partial=False, collation_fn=default_collate),
141
+ # batch_size=None,
142
+ # shuffle=False,
143
+ # num_workers=num_workers,
144
+ # pin_memory=True,
145
+ # )
146
+
147
+ # def flatten_meta(meta_dict):
148
+ # """
149
+ # Flatten the meta dictionary by:
150
+ # - Replacing single-item lists with the item itself.
151
+ # - Converting tensors to scalar numbers.
152
+ # """
153
+ # flattened = {}
154
+ # for key, value in meta_dict.items():
155
+ # if isinstance(value, list):
156
+ # if len(value) == 1:
157
+ # flattened[key] = value[0] # Replace list with its single item
158
+ # else:
159
+ # flattened[key] = value # Keep as is if multiple items
160
+ # elif isinstance(value, torch.Tensor):
161
+ # # Convert tensor to scalar
162
+ # if value.numel() == 1:
163
+ # flattened[key] = value.item()
164
+ # else:
165
+ # flattened[key] = value.tolist() # Convert multi-element tensor to list
166
+ # else:
167
+ # flattened[key] = value # Keep the value as is
168
+ # return flattened
169
+
170
+ # import h5py
171
+ # meta_array = np.array([], dtype=object)
172
+ # # Open an HDF5 file in write mode
173
+ # with h5py.File('train_hcp.hdf5', 'w') as h5f:
174
+ # flatmaps_dset = None
175
+
176
+ # total_samples = 0
177
+
178
+ # for i, batch in tqdm(enumerate(train_dl), total = 120000):
179
+ # images = batch['image'][0]
180
+ # meta = batch['meta']
181
+ # batch_size = images.shape[0]
182
+ # meta_serializable = meta.copy()
183
+
184
+
185
+ # # Step 2: Serialize the dictionary to a JSON string
186
+ # meta_str = json.dumps(flatten_meta(meta_serializable), indent=4)
187
+ # meta_array = np.append(meta_array, meta_str)
188
+ # if flatmaps_dset is None:
189
+ # # Initialize datasets with unlimited (None) maxshape along the first axis
190
+ # flatmaps_shape = (0,) + images.shape[1:]
191
+ # flatmaps_maxshape = (None,) + images.shape[1:]
192
+
193
+ # flatmaps_dset = h5f.create_dataset(
194
+ # 'flatmaps',
195
+ # shape=flatmaps_shape,
196
+ # maxshape=flatmaps_maxshape,
197
+ # dtype=np.float16,
198
+ # chunks=True # Enable chunking for efficient resizing
199
+ # )
200
+
201
+ # # Resize datasets to accommodate new data
202
+ # flatmaps_dset.resize(total_samples + batch_size, axis=0)
203
+
204
+ # # Write data to the datasets
205
+ # flatmaps_dset[total_samples:total_samples + batch_size] = images.numpy().astype(np.float16)
206
+
207
+ # total_samples += batch_size
208
+
209
+ # print(f"Processed {total_samples} samples")
210
+ # np.save('metadata_test_HCP.npy', meta_array)
211
+
212
+
213
+ # import h5py
214
+ # meta_array = np.array([], dtype=object)
215
+ # # Open an HDF5 file in write mode
216
+ # with h5py.File('test_hcp.hdf5', 'w') as h5f:
217
+ # flatmaps_dset = None
218
+
219
+ # total_samples = 0
220
+
221
+ # for i, batch in tqdm(enumerate(test_dl), total = 12000):
222
+ # images = batch['image'][0]
223
+ # meta = batch['meta']
224
+ # batch_size = images.shape[0]
225
+ # meta_serializable = meta.copy()
226
+
227
+
228
+ # # Step 2: Serialize the dictionary to a JSON string
229
+ # meta_str = json.dumps(flatten_meta(meta_serializable), indent=4)
230
+ # meta_array = np.append(meta_array, meta_str)
231
+ # if flatmaps_dset is None:
232
+ # # Initialize datasets with unlimited (None) maxshape along the first axis
233
+ # flatmaps_shape = (0,) + images.shape[1:]
234
+ # flatmaps_maxshape = (None,) + images.shape[1:]
235
+
236
+ # flatmaps_dset = h5f.create_dataset(
237
+ # 'flatmaps',
238
+ # shape=flatmaps_shape,
239
+ # maxshape=flatmaps_maxshape,
240
+ # dtype=np.float16,
241
+ # chunks=True # Enable chunking for efficient resizing
242
+ # )
243
+
244
+ # # Resize datasets to accommodate new data
245
+ # flatmaps_dset.resize(total_samples + batch_size, axis=0)
246
+
247
+ # # Write data to the datasets
248
+ # flatmaps_dset[total_samples:total_samples + batch_size] = images.numpy().astype(np.float16)
249
+
250
+ # total_samples += batch_size
251
+
252
+ # print(f"Processed {total_samples} samples")
253
+ # np.save('metadata_train_HCP.npy', meta_array)
254
+
255
+
256
+ # ### Preparing data
257
+
258
+ # In[4]:
259
+
260
+
261
+ from sklearn.preprocessing import LabelEncoder
262
+
263
+ INCLUDE_CONDS = {
264
+ "fear",
265
+ "neut",
266
+ "math",
267
+ "story",
268
+ "lf",
269
+ "lh",
270
+ "rf",
271
+ "rh",
272
+ "t",
273
+ "match",
274
+ "relation",
275
+ "mental",
276
+ "rnd",
277
+ "0bk_body",
278
+ "2bk_body",
279
+ "0bk_faces",
280
+ "2bk_faces",
281
+ "0bk_places",
282
+ "2bk_places",
283
+ "0bk_tools",
284
+ "2bk_tools",
285
+ }
286
+
287
+ # test_data = []
288
+
289
+ # # Iterate over the DataLoader with a progress bar
290
+ # for sample in tqdm(train_dl, desc="Processing samples"):
291
+ # x = sample['image']
292
+ # y = sample['meta']['trial_type']
293
+ # key = sample['meta']['key']
294
+ # print(x.shape, y, key)
295
+ # break
296
+ # Initialize the label encoder
297
+ label_encoder = LabelEncoder()
298
+ label_encoder.fit(sorted(INCLUDE_CONDS)) # Ensure consistent ordering
299
+
300
+ num_classes = len(label_encoder.classes_)
301
+ print(f"Number of classes: {num_classes}")
302
+
303
+
304
+ # In[5]:
305
+
306
+
307
+ f_train = h5py.File('/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/train_hcp.hdf5', 'r')
308
+ flatmaps_train = f_train['flatmaps']
309
+
310
+ f_test = h5py.File('/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/test_hcp.hdf5', 'r')
311
+ flatmaps_test = f_test['flatmaps']
312
+
313
+ metadata_train = np.load('/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/metadata_train_HCP.npy', allow_pickle=True)
314
+ metadata_test = np.load('/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/metadata_test_HCP.npy', allow_pickle=True)
315
+
316
+
317
+ # In[6]:
318
+
319
+
320
+ from torch.utils.data import Dataset, DataLoader
321
+
322
+ class HCPFlatDataset(Dataset):
323
+ def __init__(self, flatmaps, metadata):
324
+ self.flatmaps = flatmaps
325
+ self.metadata = metadata
326
+
327
+ def __len__(self):
328
+ return len(self.metadata)
329
+
330
+ def __getitem__(self, idx):
331
+ return self.flatmaps[idx], json.loads(self.metadata[idx])
332
+ print("Moving datasets to ram")
333
+ # Loading to cpu for faster training, this can take several minutes. Remove this [:] if you want to move one at the time.
334
+ train_dataset = HCPFlatDataset(flatmaps_train, metadata_train)
335
+ train_dl = DataLoader(train_dataset, batch_size=batch_size, shuffle=True, num_workers=num_workers)
336
+
337
+ test_dataset = HCPFlatDataset(flatmaps_test, metadata_test)
338
+ test_dl = DataLoader(test_dataset, batch_size=batch_size, shuffle=False, num_workers=0)
339
+ print("Datasets ready")
340
+
341
+
342
+ # ### Creating and loading Model
343
+
344
+ # In[7]:
345
+
346
+
347
+ from mae_utils.flat import load_hcp_flat_mask
348
+ from mae_utils.flat import create_hcp_flat
349
+ from mae_utils.flat import batch_unmask
350
+ import mae_utils.visualize as vis
351
+
352
+ flat_mask = load_hcp_flat_mask(hcp_flat_path)
353
+
354
+ mae_model = flat_models.mae_vit_large_fmri(
355
+ patch_size=patch_size,
356
+ decoder_embed_dim=decoder_embed_dim,
357
+ t_patch_size=t_patch_size,
358
+ pred_t_dim=pred_t_dim,
359
+ decoder_depth=4,
360
+ cls_embed=cls_embed,
361
+ norm_pix_loss=norm_pix_loss,
362
+ no_qkv_bias=no_qkv_bias,
363
+ sep_pos_embed=sep_pos_embed,
364
+ trunc_init=trunc_init,
365
+ pct_masks_to_decode=pct_masks_to_decode,
366
+ img_mask=flat_mask,
367
+ )
368
+
369
+
370
+ # In[8]:
371
+
372
+
373
+ checkpoint_files = [f for f in os.listdir(outdir) if f.endswith('.pth')]
374
+
375
+ if utils.is_interactive():
376
+ latest_checkpoint = "epoch99.pth"
377
+ else:
378
+ latest_checkpoint = sys.argv[2]
379
+ print(f"latest_checkpoint: {latest_checkpoint}")
380
+
381
+ # Load the checkpoint
382
+ checkpoint_path = os.path.join(outdir, latest_checkpoint)
383
+
384
+ state = torch.load(checkpoint_path)
385
+ mae_model.load_state_dict(state["model_state_dict"], strict=False)
386
+ mae_model.to(device)
387
+
388
+ print(f"\nLoaded checkpoint {latest_checkpoint} from {outdir}\n")
389
+
390
+
391
+ # In[9]:
392
+
393
+
394
+ class LinearClassifier(nn.Module):
395
+ def __init__(self, input_dim, num_classes):
396
+ super(LinearClassifier, self).__init__()
397
+ self.linear = nn.Linear(input_dim, num_classes)
398
+
399
+ def forward(self, x):
400
+ # Flatten the input except for the batch dimension
401
+ x = x.view(x.size(0), -1)
402
+ out = self.linear(x)
403
+ return out # Raw logits
404
+
405
+ # Determine the input dimension from a single sample
406
+ # Assuming images are of shape [1, 16, 144, 320]
407
+ input_dim = np.prod(mae_model(torch.randn(1,1,16,144,320).to(device),global_pool=global_pool, forward_features = True).shape[1:])
408
+ print(f"Input dimension: {input_dim}")
409
+
410
+
411
+ # In[10]:
412
+
413
+
414
+ class FullModel(nn.Module):
415
+ def __init__(self, lc_model, mae_model):
416
+ super(FullModel, self).__init__()
417
+ self.lc_model = lc_model
418
+ self.mae_model = mae_model
419
+
420
+
421
+ def forward(self, x, gsr):
422
+ x = self.mae_model(x, global_pool=global_pool, forward_features = True)
423
+ x = self.lc_model(x)
424
+ return x
425
+
426
+
427
+ # In[11]:
428
+
429
+
430
+ # Initialize the model
431
+ lc_model = LinearClassifier(input_dim=input_dim, num_classes=num_classes)
432
+
433
+ model = FullModel(lc_model, mae_model)
434
+
435
+ # Move the model to the GPU
436
+ model.to(device)
437
+
438
+ # Define loss function
439
+ criterion = nn.CrossEntropyLoss()
440
+
441
+ # Define optimizer with L2 regularization (weight_decay)
442
+ learning_rate = 1e-4
443
+ weight_decay = 1e-5 # Adjust based on your needs
444
+ optimizer = torch.optim.Adam(model.parameters(), lr=learning_rate, weight_decay=weight_decay)
445
+ num_epochs = 20 # Adjust as needed
446
+
447
+
448
+ # ### Data
449
+
450
+ # In[16]:
451
+
452
+
453
+ import uuid
454
+
455
+ myuuid = uuid.uuid4()
456
+ str(myuuid)
457
+
458
+
459
+ # In[17]:
460
+
461
+
462
+ import wandb
463
+
464
+ if utils.is_interactive():
465
+ print("Running in interactive notebook. Disabling W&B and ckpt saving.")
466
+ wandb_log = True
467
+ save_ckpt = True
468
+
469
+ if wandb_log:
470
+ wandb_project = 'fMRI-foundation-model'
471
+ wandb_config = {
472
+ "model_name": model_name+'_HCP_FT',
473
+ "batch_size": batch_size,
474
+ "learning_rate": learning_rate,
475
+ "weight_decay": weight_decay,
476
+ "num_epochs": num_epochs,
477
+ "seed": seed,
478
+ }
479
+ print("wandb_config:\n", wandb_config)
480
+ random_id = str(uuid.uuid4())
481
+ print("wandb_id:", "HCPflat_raw" + f"_{random_id}")
482
+ wandb.init(
483
+ id=model_name+'_HCP_FT' + f"_{random_id}",
484
+ project=wandb_project,
485
+ name=model_name+'_HCP_FT',
486
+ config=wandb_config,
487
+ resume="allow",
488
+ )
489
+
490
+
491
+ # In[13]:
492
+
493
+
494
+ for epoch in range(num_epochs):
495
+ running_train_loss = 0.0
496
+ correct_train = 0
497
+ total_train = 0
498
+ step = 0
499
+
500
+ # with torch.amp.autocast(device_type='cuda'):
501
+ # Training Phase
502
+ model.train()
503
+ for batch in tqdm(train_dl, desc=f"Epoch {epoch+1}/{num_epochs} - Training"):
504
+ optimizer.zero_grad()
505
+ images = batch[0].to(device).float().unsqueeze(1) #fix this # Shape: [batch_size, 1, 16, 144, 320]
506
+ labels = batch[1]['trial_type'] # List of labels
507
+
508
+ encoded_labels = label_encoder.transform(labels)
509
+ encoded_labels = torch.tensor(encoded_labels, dtype=torch.long).to(device) # Shape: [batch_size]
510
+
511
+ # Forward pass
512
+ outputs = model(images, gsr=gsr) # Shape: [num_train_samples, num_classes]
513
+
514
+ # Compute loss
515
+ loss = criterion(outputs, encoded_labels)
516
+
517
+ # Backward pass and optimization
518
+ loss.backward()
519
+ optimizer.step()
520
+
521
+ # Accumulate loss
522
+ running_train_loss += loss.item() * images.size(0)
523
+
524
+
525
+ # Calculate accuracy
526
+ _, predicted = torch.max(outputs, 1)
527
+
528
+ correct_train += (predicted == encoded_labels).sum().item()
529
+ total_train += encoded_labels.size(0)
530
+
531
+ step = step + 1
532
+ if step % 100 == 0:
533
+ print(f"Step [{step}/{len(train_dl)}] - Training Loss: {loss.item():.4f} - Training Accuracy: {100 * correct_train / total_train:.2f}%")
534
+ # thth
535
+
536
+ epoch_train_loss = running_train_loss / total_train if total_train > 0 else 0.0
537
+ train_accuracy = 100 * correct_train / total_train if total_train > 0 else 0.0
538
+
539
+ # Validation Phase
540
+ model.eval()
541
+ running_val_loss = 0.0
542
+ correct_val = 0
543
+ total_val = 0
544
+
545
+ with torch.no_grad():
546
+ for batch in tqdm(test_dl, desc=f"Epoch {epoch+1}/{num_epochs} - Validation"):
547
+
548
+ images = batch[0].to(device).float().unsqueeze(1) #fix this
549
+ labels = batch[1]['trial_type']
550
+
551
+ # Encode labels to integer indices
552
+ encoded_labels = label_encoder.transform(labels)
553
+ encoded_labels = torch.tensor(encoded_labels, dtype=torch.long).to(device)
554
+
555
+
556
+ # Forward pass
557
+ outputs = model(images, gsr=gsr)
558
+
559
+ # Compute loss
560
+ loss = criterion(outputs, encoded_labels)
561
+
562
+ # Accumulate loss
563
+ running_val_loss += loss.item() * images.size(0)
564
+
565
+ # Calculate accuracy
566
+ _, predicted = torch.max(outputs, 1)
567
+ correct_val += (predicted == encoded_labels).sum().item()
568
+ total_val += encoded_labels.size(0)
569
+
570
+
571
+
572
+ epoch_val_loss = running_val_loss / total_val if total_val > 0 else 0.0
573
+ val_accuracy = 100 * correct_val / total_val if total_val > 0 else 0.0
574
+
575
+ print(f"Epoch [{epoch+1}/{num_epochs}] "
576
+ f"- Training Loss: {epoch_train_loss:.4f}, Training Accuracy: {train_accuracy:.2f}% "
577
+ f"- Validation Loss: {epoch_val_loss:.4f}, Validation Accuracy: {val_accuracy:.2f}%")
578
+
579
+ if wandb_log:
580
+ wandb.log({
581
+ "epoch_train_loss": epoch_train_loss,
582
+ "epoch_val_loss": epoch_val_loss,
583
+ "train_accuracy": train_accuracy,
584
+ "val_accuracy": val_accuracy,
585
+ })
586
+ if save_ckpt:
587
+ outdir = os.path.abspath(f'checkpoints/{model_name+"HCP_FT"}')
588
+ os.makedirs(outdir, exist_ok=True)
589
+ print("outdir", outdir)
590
+ # Save model and config
591
+ torch.save(model.state_dict(), f"{outdir}/model.pth")
592
+ with open(f"{outdir}/config.yaml", 'w') as f:
593
+ yaml.dump(wandb_config, f)
594
+ print(f"Saved model and config to {outdir}")
595
+
596
+
fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/files/config.yaml ADDED
@@ -0,0 +1,47 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ _wandb:
2
+ value:
3
+ cli_version: 0.18.3
4
+ code_path: code/src/HCP_downstream_finetune.py
5
+ m: []
6
+ python_version: 3.11.10
7
+ t:
8
+ "1":
9
+ - 1
10
+ - 5
11
+ - 41
12
+ - 49
13
+ - 53
14
+ - 55
15
+ - 63
16
+ "2":
17
+ - 1
18
+ - 5
19
+ - 41
20
+ - 49
21
+ - 53
22
+ - 55
23
+ - 63
24
+ "3":
25
+ - 13
26
+ - 14
27
+ - 16
28
+ - 23
29
+ - 55
30
+ "4": 3.11.10
31
+ "5": 0.18.3
32
+ "8":
33
+ - 5
34
+ "12": 0.18.3
35
+ "13": linux-x86_64
36
+ batch_size:
37
+ value: 8
38
+ learning_rate:
39
+ value: 0.0001
40
+ model_name:
41
+ value: HCPflat_large_gsrFalse__HCP_FT
42
+ num_epochs:
43
+ value: 20
44
+ seed:
45
+ value: 42
46
+ weight_decay:
47
+ value: 1e-05
fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/files/output.log ADDED
@@ -0,0 +1,147 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Epoch 1/20 - Training: 23%|██▎ | 3199/13913 [18:29<1:02:00, 2.88it/s]
2
+ Step [100/13913] - Training Loss: 1.9649 - Training Accuracy: 59.38%
3
+ Step [200/13913] - Training Loss: 1.0904 - Training Accuracy: 71.19%
4
+ Step [300/13913] - Training Loss: 0.0775 - Training Accuracy: 75.50%
5
+ Step [400/13913] - Training Loss: 0.6052 - Training Accuracy: 78.84%
6
+ Step [500/13913] - Training Loss: 0.0226 - Training Accuracy: 80.95%
7
+ Step [600/13913] - Training Loss: 0.2728 - Training Accuracy: 82.54%
8
+ Step [700/13913] - Training Loss: 0.1662 - Training Accuracy: 83.70%
9
+ Step [800/13913] - Training Loss: 0.0385 - Training Accuracy: 84.89%
10
+ Step [900/13913] - Training Loss: 0.2377 - Training Accuracy: 85.61%
11
+ Step [1000/13913] - Training Loss: 0.8172 - Training Accuracy: 85.83%
12
+ Step [1100/13913] - Training Loss: 0.2276 - Training Accuracy: 86.68%
13
+ Step [1200/13913] - Training Loss: 0.0118 - Training Accuracy: 87.36%
14
+ Step [1300/13913] - Training Loss: 1.0419 - Training Accuracy: 87.68%
15
+ Step [1400/13913] - Training Loss: 0.8943 - Training Accuracy: 87.96%
16
+ Step [1500/13913] - Training Loss: 0.2801 - Training Accuracy: 88.23%
17
+ Step [1600/13913] - Training Loss: 0.6734 - Training Accuracy: 88.58%
18
+ Step [1700/13913] - Training Loss: 0.6202 - Training Accuracy: 88.88%
19
+ Step [1800/13913] - Training Loss: 0.0159 - Training Accuracy: 89.12%
20
+ Step [1900/13913] - Training Loss: 0.0682 - Training Accuracy: 89.37%
21
+ Step [2000/13913] - Training Loss: 0.3378 - Training Accuracy: 89.52%
22
+ Step [2100/13913] - Training Loss: 0.0509 - Training Accuracy: 89.77%
23
+ Step [2200/13913] - Training Loss: 0.1161 - Training Accuracy: 89.99%
24
+ Step [2300/13913] - Training Loss: 0.0025 - Training Accuracy: 90.12%
25
+ Step [2400/13913] - Training Loss: 0.5385 - Training Accuracy: 90.25%
26
+ Step [2500/13913] - Training Loss: 0.0003 - Training Accuracy: 90.47%
27
+ Step [2600/13913] - Training Loss: 0.0962 - Training Accuracy: 90.52%
28
+ Step [2700/13913] - Training Loss: 0.0480 - Training Accuracy: 90.62%
29
+ Step [2800/13913] - Training Loss: 0.0004 - Training Accuracy: 90.71%
30
+ Step [2900/13913] - Training Loss: 0.3049 - Training Accuracy: 90.84%
31
+ Step [3000/13913] - Training Loss: 0.0339 - Training Accuracy: 90.95%
32
+ Step [3100/13913] - Training Loss: 0.5572 - Training Accuracy: 91.02%
33
+ Step [3200/13913] - Training Loss: 0.0673 - Training Accuracy: 91.11%
34
+ Step [3300/13913] - Training Loss: 0.0018 - Training Accuracy: 91.21%
35
+ Step [3400/13913] - Training Loss: 0.0048 - Training Accuracy: 91.28%
36
+ Step [3500/13913] - Training Loss: 1.2120 - Training Accuracy: 91.33%
37
+ Step [3600/13913] - Training Loss: 0.0542 - Training Accuracy: 91.37%
38
+ Step [3700/13913] - Training Loss: 0.0016 - Training Accuracy: 91.43%
39
+ Step [3800/13913] - Training Loss: 0.0582 - Training Accuracy: 91.52%
40
+ Step [3900/13913] - Training Loss: 0.0960 - Training Accuracy: 91.60%
41
+ Step [4000/13913] - Training Loss: 0.0020 - Training Accuracy: 91.68%
42
+ Step [4100/13913] - Training Loss: 0.0043 - Training Accuracy: 91.76%
43
+ Step [4200/13913] - Training Loss: 0.0029 - Training Accuracy: 91.77%
44
+ Step [4300/13913] - Training Loss: 0.0107 - Training Accuracy: 91.83%
45
+ Step [4400/13913] - Training Loss: 0.1122 - Training Accuracy: 91.86%
46
+ Step [4500/13913] - Training Loss: 0.1595 - Training Accuracy: 91.92%
47
+ Step [4600/13913] - Training Loss: 0.0453 - Training Accuracy: 91.97%
48
+ Step [4700/13913] - Training Loss: 0.2770 - Training Accuracy: 92.05%
49
+ Step [4800/13913] - Training Loss: 0.0057 - Training Accuracy: 92.09%
50
+ Step [4900/13913] - Training Loss: 0.0120 - Training Accuracy: 92.16%
51
+ Step [5000/13913] - Training Loss: 0.0235 - Training Accuracy: 92.24%
52
+ Step [5100/13913] - Training Loss: 0.3907 - Training Accuracy: 92.32%
53
+ Step [5200/13913] - Training Loss: 0.4558 - Training Accuracy: 92.34%
54
+ Step [5300/13913] - Training Loss: 0.0051 - Training Accuracy: 92.40%
55
+ Step [5400/13913] - Training Loss: 0.0017 - Training Accuracy: 92.46%
56
+ Step [5500/13913] - Training Loss: 1.3554 - Training Accuracy: 92.50%
57
+ Step [5600/13913] - Training Loss: 0.0617 - Training Accuracy: 92.55%
58
+ Step [5700/13913] - Training Loss: 0.2618 - Training Accuracy: 92.60%
59
+ Step [5800/13913] - Training Loss: 0.0192 - Training Accuracy: 92.60%
60
+ Step [5900/13913] - Training Loss: 0.3865 - Training Accuracy: 92.67%
61
+ Step [6000/13913] - Training Loss: 0.0139 - Training Accuracy: 92.70%
62
+ Step [6100/13913] - Training Loss: 0.1493 - Training Accuracy: 92.72%
63
+ Step [6200/13913] - Training Loss: 0.3629 - Training Accuracy: 92.77%
64
+ Step [6300/13913] - Training Loss: 0.4069 - Training Accuracy: 92.81%
65
+ Step [6400/13913] - Training Loss: 0.4954 - Training Accuracy: 92.85%
66
+ Step [6500/13913] - Training Loss: 0.0061 - Training Accuracy: 92.88%
67
+ Step [6600/13913] - Training Loss: 0.0373 - Training Accuracy: 92.91%
68
+ Step [6700/13913] - Training Loss: 0.0690 - Training Accuracy: 92.93%
69
+ Step [6800/13913] - Training Loss: 0.0158 - Training Accuracy: 92.97%
70
+ Step [6900/13913] - Training Loss: 0.3957 - Training Accuracy: 92.99%
71
+ Step [7000/13913] - Training Loss: 0.0615 - Training Accuracy: 93.03%
72
+ Step [7100/13913] - Training Loss: 0.0017 - Training Accuracy: 93.06%
73
+ Step [7200/13913] - Training Loss: 0.1726 - Training Accuracy: 93.07%
74
+ Step [7300/13913] - Training Loss: 0.0145 - Training Accuracy: 93.11%
75
+ Step [7400/13913] - Training Loss: 0.1883 - Training Accuracy: 93.15%
76
+ Step [7500/13913] - Training Loss: 0.0286 - Training Accuracy: 93.17%
77
+ Step [7600/13913] - Training Loss: 0.0591 - Training Accuracy: 93.21%
78
+ Step [7700/13913] - Training Loss: 0.4116 - Training Accuracy: 93.22%
79
+ Step [7800/13913] - Training Loss: 0.0143 - Training Accuracy: 93.25%
80
+ Step [7900/13913] - Training Loss: 0.0108 - Training Accuracy: 93.28%
81
+ Step [8000/13913] - Training Loss: 0.0013 - Training Accuracy: 93.30%
82
+ Step [8100/13913] - Training Loss: 0.1299 - Training Accuracy: 93.31%
83
+ Step [8200/13913] - Training Loss: 0.0535 - Training Accuracy: 93.36%
84
+ Step [8300/13913] - Training Loss: 0.1179 - Training Accuracy: 93.37%
85
+ Step [8400/13913] - Training Loss: 0.0817 - Training Accuracy: 93.38%
86
+ Step [8500/13913] - Training Loss: 0.0000 - Training Accuracy: 93.41%
87
+ Step [8600/13913] - Training Loss: 0.1190 - Training Accuracy: 93.45%
88
+ Step [8700/13913] - Training Loss: 0.4036 - Training Accuracy: 93.46%
89
+ Step [8800/13913] - Training Loss: 0.1972 - Training Accuracy: 93.49%
90
+ Step [8900/13913] - Training Loss: 0.0570 - Training Accuracy: 93.51%
91
+ Step [9000/13913] - Training Loss: 0.0005 - Training Accuracy: 93.55%
92
+ Step [9100/13913] - Training Loss: 0.0023 - Training Accuracy: 93.58%
93
+ Step [9200/13913] - Training Loss: 0.0509 - Training Accuracy: 93.60%
94
+ Step [9300/13913] - Training Loss: 0.5032 - Training Accuracy: 93.63%
95
+ Step [9400/13913] - Training Loss: 0.0022 - Training Accuracy: 93.66%
96
+ Step [9500/13913] - Training Loss: 0.1065 - Training Accuracy: 93.68%
97
+ Step [9600/13913] - Training Loss: 0.0017 - Training Accuracy: 93.69%
98
+ Step [9700/13913] - Training Loss: 0.0000 - Training Accuracy: 93.72%
99
+ Step [9800/13913] - Training Loss: 0.1971 - Training Accuracy: 93.72%
100
+ Step [9900/13913] - Training Loss: 0.0001 - Training Accuracy: 93.74%
101
+ Step [10000/13913] - Training Loss: 0.4162 - Training Accuracy: 93.74%
102
+ Step [10100/13913] - Training Loss: 0.0123 - Training Accuracy: 93.77%
103
+ Step [10200/13913] - Training Loss: 0.0439 - Training Accuracy: 93.80%
104
+ Step [10300/13913] - Training Loss: 0.2364 - Training Accuracy: 93.82%
105
+ Step [10400/13913] - Training Loss: 0.0197 - Training Accuracy: 93.84%
106
+ Step [10500/13913] - Training Loss: 0.3435 - Training Accuracy: 93.86%
107
+ Step [10600/13913] - Training Loss: 0.0243 - Training Accuracy: 93.87%
108
+ Step [10700/13913] - Training Loss: 0.0080 - Training Accuracy: 93.88%
109
+ Step [10800/13913] - Training Loss: 0.0018 - Training Accuracy: 93.92%
110
+ Step [10900/13913] - Training Loss: 0.1508 - Training Accuracy: 93.92%
111
+ Step [11000/13913] - Training Loss: 0.0002 - Training Accuracy: 93.94%
112
+ Step [11100/13913] - Training Loss: 0.5014 - Training Accuracy: 93.96%
113
+ Step [11200/13913] - Training Loss: 0.3636 - Training Accuracy: 93.98%
114
+ Step [11300/13913] - Training Loss: 0.1294 - Training Accuracy: 94.00%
115
+ Step [11400/13913] - Training Loss: 0.8976 - Training Accuracy: 94.01%
116
+ Step [11500/13913] - Training Loss: 0.0029 - Training Accuracy: 94.03%
117
+ Step [11600/13913] - Training Loss: 0.0006 - Training Accuracy: 94.05%
118
+ Step [11700/13913] - Training Loss: 0.1646 - Training Accuracy: 94.08%
119
+ Step [11800/13913] - Training Loss: 0.5143 - Training Accuracy: 94.08%
120
+ Step [11900/13913] - Training Loss: 0.1646 - Training Accuracy: 94.10%
121
+ Step [12000/13913] - Training Loss: 0.0189 - Training Accuracy: 94.13%
122
+ Step [12100/13913] - Training Loss: 0.0067 - Training Accuracy: 94.14%
123
+ Step [12200/13913] - Training Loss: 0.2044 - Training Accuracy: 94.15%
124
+ Step [12300/13913] - Training Loss: 0.0247 - Training Accuracy: 94.15%
125
+ Step [12400/13913] - Training Loss: 0.5848 - Training Accuracy: 94.17%
126
+ Step [12500/13913] - Training Loss: 0.0024 - Training Accuracy: 94.20%
127
+ Step [12600/13913] - Training Loss: 0.0015 - Training Accuracy: 94.21%
128
+ Step [12700/13913] - Training Loss: 0.6383 - Training Accuracy: 94.23%
129
+ Step [12800/13913] - Training Loss: 0.3236 - Training Accuracy: 94.24%
130
+ Step [12900/13913] - Training Loss: 0.4483 - Training Accuracy: 94.25%
131
+ Step [13000/13913] - Training Loss: 0.0402 - Training Accuracy: 94.25%
132
+ Step [13100/13913] - Training Loss: 0.0103 - Training Accuracy: 94.26%
133
+ Step [13200/13913] - Training Loss: 0.0004 - Training Accuracy: 94.27%
134
+ Step [13300/13913] - Training Loss: 0.0488 - Training Accuracy: 94.28%
135
+ Step [13400/13913] - Training Loss: 0.0009 - Training Accuracy: 94.30%
136
+ Step [13500/13913] - Training Loss: 0.0705 - Training Accuracy: 94.32%
137
+ Step [13600/13913] - Training Loss: 0.0214 - Training Accuracy: 94.33%
138
+ Step [13700/13913] - Training Loss: 0.0036 - Training Accuracy: 94.33%
139
+ Step [13800/13913] - Training Loss: 0.0102 - Training Accuracy: 94.34%
140
+ Step [13900/13913] - Training Loss: 0.0112 - Training Accuracy: 94.36%
141
+ Epoch 1/20 - Validation: 100%|██████████| 1511/1511 [06:36<00:00, 3.81it/s]
142
+ Epoch [1/20] - Training Loss: 0.1951, Training Accuracy: 94.36% - Validation Loss: 0.1229, Validation Accuracy: 96.41%
143
+ Traceback (most recent call last):
144
+ File "/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.py", line 586, in <module>
145
+ if save_ckpt:
146
+ ^^^^^^^^^
147
+ NameError: name 'save_ckpt' is not defined
fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/files/wandb-metadata.json ADDED
@@ -0,0 +1,131 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-5.15.0-1058-aws-x86_64-with-glibc2.31",
3
+ "python": "3.11.10",
4
+ "startedAt": "2024-10-23T13:28:08.819065Z",
5
+ "args": [
6
+ "HCPflat_large_gsrFalse_",
7
+ "epoch99.pth"
8
+ ],
9
+ "program": "/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.py",
10
+ "codePath": "src/HCP_downstream_finetune.py",
11
+ "git": {
12
+ "remote": "https://github.com/MedARC-AI/fMRI-foundation-model",
13
+ "commit": "5908f2fe5945884e8a2044da614319dff358f6e5"
14
+ },
15
+ "email": "torrico.villanueva.cesar.kadir@gmail.com",
16
+ "root": "/weka/proj-fmri/ckadirt/fMRI-foundation-model/src",
17
+ "host": "ip-10-0-156-184",
18
+ "username": "ckadirt",
19
+ "executable": "/admin/home-ckadirt/foundation_env/bin/python",
20
+ "codePathLocal": "HCP_downstream_finetune.py",
21
+ "cpu_count": 96,
22
+ "cpu_count_logical": 192,
23
+ "gpu": "[NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3]",
24
+ "gpu_count": 8,
25
+ "disk": {
26
+ "/": {
27
+ "total": "249555763200",
28
+ "used": "184972439552"
29
+ }
30
+ },
31
+ "memory": {
32
+ "total": "2147443380224"
33
+ },
34
+ "cpu": {
35
+ "count": 96,
36
+ "countLogical": 192
37
+ },
38
+ "gpu_nvidia": [
39
+ {
40
+ "name": "NVIDIA H100 80GB HBM3",
41
+ "memoryTotal": "85520809984",
42
+ "cudaCores": 16896,
43
+ "architecture": "Hopper"
44
+ },
45
+ {
46
+ "name": "NVIDIA H100 80GB HBM3",
47
+ "memoryTotal": "85520809984",
48
+ "cudaCores": 16896,
49
+ "architecture": "Hopper"
50
+ },
51
+ {
52
+ "name": "NVIDIA H100 80GB HBM3",
53
+ "memoryTotal": "85520809984",
54
+ "cudaCores": 16896,
55
+ "architecture": "Hopper"
56
+ },
57
+ {
58
+ "name": "NVIDIA H100 80GB HBM3",
59
+ "memoryTotal": "85520809984",
60
+ "cudaCores": 16896,
61
+ "architecture": "Hopper"
62
+ },
63
+ {
64
+ "name": "NVIDIA H100 80GB HBM3",
65
+ "memoryTotal": "85520809984",
66
+ "cudaCores": 16896,
67
+ "architecture": "Hopper"
68
+ },
69
+ {
70
+ "name": "NVIDIA H100 80GB HBM3",
71
+ "memoryTotal": "85520809984",
72
+ "cudaCores": 16896,
73
+ "architecture": "Hopper"
74
+ },
75
+ {
76
+ "name": "NVIDIA H100 80GB HBM3",
77
+ "memoryTotal": "85520809984",
78
+ "cudaCores": 16896,
79
+ "architecture": "Hopper"
80
+ },
81
+ {
82
+ "name": "NVIDIA H100 80GB HBM3",
83
+ "memoryTotal": "85520809984",
84
+ "cudaCores": 16896,
85
+ "architecture": "Hopper"
86
+ }
87
+ ],
88
+ "slurm": {
89
+ "cluster_name": "sagemaker2",
90
+ "conf": "/opt/slurm/etc/slurm.conf",
91
+ "cpus_on_node": "20",
92
+ "gpus_on_node": "1",
93
+ "gpus_per_task": "1",
94
+ "gtids": "0",
95
+ "job_account": "fmri",
96
+ "job_cpus_per_node": "20",
97
+ "job_end_time": "1729733268",
98
+ "job_gid": "1879800513",
99
+ "job_gpus": "3",
100
+ "job_id": "528374",
101
+ "job_name": "finetuneHCP",
102
+ "job_nodelist": "ip-10-0-156-184",
103
+ "job_num_nodes": "1",
104
+ "job_partition": "p5",
105
+ "job_qos": "normal",
106
+ "job_start_time": "1729690068",
107
+ "job_uid": "1879804696",
108
+ "job_user": "ckadirt",
109
+ "jobid": "528374",
110
+ "localid": "0",
111
+ "mem_per_cpu": "11500",
112
+ "nnodes": "1",
113
+ "node_aliases": "(null)",
114
+ "nodeid": "0",
115
+ "nodelist": "ip-10-0-156-184",
116
+ "nprocs": "1",
117
+ "ntasks": "1",
118
+ "ntasks_per_node": "1",
119
+ "prio_process": "0",
120
+ "procid": "0",
121
+ "script_context": "prolog_task",
122
+ "submit_dir": "/weka/proj-fmri/ckadirt/fMRI-foundation-model/src",
123
+ "submit_host": "ip-172-17-12-61",
124
+ "task_pid": "1410338",
125
+ "tasks_per_node": "1",
126
+ "topology_addr": "ip-10-0-156-184",
127
+ "topology_addr_pattern": "node",
128
+ "working_cluster": "sagemaker2:ip-172-17-63-161:6817:9984:109"
129
+ },
130
+ "cudaVersion": "12.2"
131
+ }
fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/files/wandb-summary.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"epoch_train_loss":0.19505918743580747,"epoch_val_loss":0.12290618471104833,"train_accuracy":94.35949039550053,"val_accuracy":96.40787949015063,"_timestamp":1.7296953058000546e+09,"_wandb":{"runtime":5216},"_runtime":5216.981460538,"_step":0}
fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/logs/debug-core.log ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2024-10-23T13:28:08.144965691Z","level":"INFO","msg":"started logging, with flags","port-filename":"/tmp/tmp4x4jqd26/port-1410365.txt","pid":1410365,"debug":false,"disable-analytics":false}
2
+ {"time":"2024-10-23T13:28:08.145292217Z","level":"INFO","msg":"FeatureState","shutdownOnParentExitEnabled":false}
3
+ {"time":"2024-10-23T13:28:08.148163723Z","level":"INFO","msg":"Will exit if parent process dies.","ppid":1410365}
4
+ {"time":"2024-10-23T13:28:08.148096622Z","level":"INFO","msg":"server is running","addr":{"IP":"127.0.0.1","Port":34171,"Zone":""}}
5
+ {"time":"2024-10-23T13:28:08.337555914Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"127.0.0.1:39994"}
6
+ {"time":"2024-10-23T13:28:08.818968785Z","level":"INFO","msg":"handleInformInit: received","streamId":"HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115","id":"127.0.0.1:39994"}
7
+ {"time":"2024-10-23T13:28:08.897522616Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115","id":"127.0.0.1:39994"}
8
+ {"time":"2024-10-23T14:55:05.8123362Z","level":"INFO","msg":"handleInformTeardown: server teardown initiated","id":"127.0.0.1:39994"}
9
+ {"time":"2024-10-23T14:55:05.813756916Z","level":"INFO","msg":"server is shutting down"}
10
+ {"time":"2024-10-23T14:55:05.813745416Z","level":"INFO","msg":"connection: Close: initiating connection closure","id":"127.0.0.1:39994"}
11
+ {"time":"2024-10-23T14:55:05.814054132Z","level":"INFO","msg":"connection: Close: connection successfully closed","id":"127.0.0.1:39994"}
12
+ {"time":"2024-10-23T14:55:06.748804634Z","level":"INFO","msg":"handleInformTeardown: server shutdown complete","id":"127.0.0.1:39994"}
13
+ {"time":"2024-10-23T14:55:06.748896936Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"127.0.0.1:39994"}
14
+ {"time":"2024-10-23T14:55:06.748913656Z","level":"INFO","msg":"server is closed"}
fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/logs/debug-internal.log ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2024-10-23T13:28:08.826877069Z","level":"INFO","msg":"using version","core version":"0.18.3"}
2
+ {"time":"2024-10-23T13:28:08.826894269Z","level":"INFO","msg":"created symlink","path":"/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/logs/debug-core.log"}
3
+ {"time":"2024-10-23T13:28:08.829725954Z","level":"ERROR","msg":"dialing: google: could not find default credentials. See https://cloud.google.com/docs/authentication/external/set-up-adc for more information"}
4
+ {"time":"2024-10-23T13:28:08.897483695Z","level":"INFO","msg":"created new stream","id":"HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115"}
5
+ {"time":"2024-10-23T13:28:08.897515195Z","level":"INFO","msg":"stream: started","id":"HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115"}
6
+ {"time":"2024-10-23T13:28:08.897522565Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115"}}
7
+ {"time":"2024-10-23T13:28:08.897537076Z","level":"INFO","msg":"handler: started","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115"}}
8
+ {"time":"2024-10-23T13:28:08.897556526Z","level":"INFO","msg":"sender: started","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115"}}
9
+ {"time":"2024-10-23T13:28:09.42017246Z","level":"INFO","msg":"wandb-core","!BADKEY":null}
10
+ {"time":"2024-10-23T13:28:09.424963983Z","level":"INFO","msg":"Starting system monitor"}
11
+ {"time":"2024-10-23T13:28:09.456835134Z","level":"ERROR","msg":"git repo not found","error":"repository does not exist"}
12
+ {"time":"2024-10-23T14:55:05.813759006Z","level":"INFO","msg":"stream: closing","id":"HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115"}
13
+ {"time":"2024-10-23T14:55:05.813798617Z","level":"INFO","msg":"Stopping system monitor"}
14
+ {"time":"2024-10-23T14:55:05.825470716Z","level":"INFO","msg":"Stopped system monitor"}
15
+ {"time":"2024-10-23T14:55:06.401455789Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
16
+ {"time":"2024-10-23T14:55:06.743665588Z","level":"INFO","msg":"handler: closed","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115"}}
17
+ {"time":"2024-10-23T14:55:06.743715539Z","level":"INFO","msg":"writer: Close: closed","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115"}}
18
+ {"time":"2024-10-23T14:55:06.74373361Z","level":"INFO","msg":"sender: closed","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115"}}
19
+ {"time":"2024-10-23T14:55:06.743832761Z","level":"INFO","msg":"stream: closed","id":"HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115"}
fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/logs/debug.log ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 2024-10-23 13:28:08,806 INFO MainThread:1410365 [wandb_setup.py:_flush():79] Current SDK version is 0.18.3
2
+ 2024-10-23 13:28:08,806 INFO MainThread:1410365 [wandb_setup.py:_flush():79] Configure stats pid to 1410365
3
+ 2024-10-23 13:28:08,806 INFO MainThread:1410365 [wandb_setup.py:_flush():79] Loading settings from /admin/home-ckadirt/.config/wandb/settings
4
+ 2024-10-23 13:28:08,806 INFO MainThread:1410365 [wandb_setup.py:_flush():79] Loading settings from /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/settings
5
+ 2024-10-23 13:28:08,806 INFO MainThread:1410365 [wandb_setup.py:_flush():79] Loading settings from environment variables: {}
6
+ 2024-10-23 13:28:08,806 INFO MainThread:1410365 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': None, '_disable_service': None}
7
+ 2024-10-23 13:28:08,806 INFO MainThread:1410365 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'src/HCP_downstream_finetune.py', 'program_abspath': '/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.py', 'program': '/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.py'}
8
+ 2024-10-23 13:28:08,806 INFO MainThread:1410365 [wandb_setup.py:_flush():79] Applying login settings: {}
9
+ 2024-10-23 13:28:08,807 INFO MainThread:1410365 [wandb_init.py:_log_setup():532] Logging user logs to /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/logs/debug.log
10
+ 2024-10-23 13:28:08,807 INFO MainThread:1410365 [wandb_init.py:_log_setup():533] Logging internal logs to /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/logs/debug-internal.log
11
+ 2024-10-23 13:28:08,807 INFO MainThread:1410365 [wandb_init.py:init():617] calling init triggers
12
+ 2024-10-23 13:28:08,807 INFO MainThread:1410365 [wandb_init.py:init():624] wandb.init called with sweep_config: {}
13
+ config: {'model_name': 'HCPflat_large_gsrFalse__HCP_FT', 'batch_size': 8, 'learning_rate': 0.0001, 'weight_decay': 1e-05, 'num_epochs': 20, 'seed': 42}
14
+ 2024-10-23 13:28:08,807 INFO MainThread:1410365 [wandb_init.py:init():667] starting backend
15
+ 2024-10-23 13:28:08,807 INFO MainThread:1410365 [wandb_init.py:init():671] sending inform_init request
16
+ 2024-10-23 13:28:08,817 INFO MainThread:1410365 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
17
+ 2024-10-23 13:28:08,817 INFO MainThread:1410365 [wandb_init.py:init():684] backend started and connected
18
+ 2024-10-23 13:28:08,839 INFO MainThread:1410365 [wandb_init.py:init():779] updated telemetry
19
+ 2024-10-23 13:28:08,925 INFO MainThread:1410365 [wandb_init.py:init():812] communicating run to backend with 90.0 second timeout
20
+ 2024-10-23 13:28:09,405 INFO MainThread:1410365 [wandb_init.py:init():863] starting run threads in backend
21
+ 2024-10-23 13:28:09,719 INFO MainThread:1410365 [wandb_run.py:_console_start():2465] atexit reg
22
+ 2024-10-23 13:28:09,719 INFO MainThread:1410365 [wandb_run.py:_redirect():2313] redirect: wrap_raw
23
+ 2024-10-23 13:28:09,719 INFO MainThread:1410365 [wandb_run.py:_redirect():2378] Wrapping output streams.
24
+ 2024-10-23 13:28:09,719 INFO MainThread:1410365 [wandb_run.py:_redirect():2403] Redirects installed.
25
+ 2024-10-23 13:28:09,721 INFO MainThread:1410365 [wandb_init.py:init():907] run started, returning control to user process
26
+ 2024-10-23 14:55:05,814 WARNING MsgRouterThr:1410365 [router.py:message_loop():77] message_loop has been closed
fMRI-foundation-model/src/wandb/run-20241023_132808-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115/run-HCPflat_large_gsrFalse__HCP_FT_a1fd5808-55ff-41af-bf18-5f6191368115.wandb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f5e376a29abf5cdc0800ff5f3a6f92c2d8161c946e73a0d8691cb5e29d262ac0
3
+ size 9238580
fMRI-foundation-model/src/wandb/run-20241023_150329-NSDflat_large_gsrFalse__HCP_FT_7920feb1-ec83-45eb-9cb6-844266415eba/logs/debug-core.log ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {"time":"2024-10-23T15:03:28.525679796Z","level":"INFO","msg":"started logging, with flags","port-filename":"/tmp/tmp5rkyf405/port-1581516.txt","pid":1581516,"debug":false,"disable-analytics":false}
2
+ {"time":"2024-10-23T15:03:28.525964251Z","level":"INFO","msg":"FeatureState","shutdownOnParentExitEnabled":false}
3
+ {"time":"2024-10-23T15:03:28.529402966Z","level":"INFO","msg":"Will exit if parent process dies.","ppid":1581516}
4
+ {"time":"2024-10-23T15:03:28.529392426Z","level":"INFO","msg":"server is running","addr":{"IP":"127.0.0.1","Port":35683,"Zone":""}}
5
+ {"time":"2024-10-23T15:03:28.717833825Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"127.0.0.1:34738"}
6
+ {"time":"2024-10-23T15:03:29.295039806Z","level":"INFO","msg":"handleInformInit: received","streamId":"NSDflat_large_gsrFalse__HCP_FT_7920feb1-ec83-45eb-9cb6-844266415eba","id":"127.0.0.1:34738"}
7
+ {"time":"2024-10-23T15:03:29.43168708Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"NSDflat_large_gsrFalse__HCP_FT_7920feb1-ec83-45eb-9cb6-844266415eba","id":"127.0.0.1:34738"}
fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/files/code/src/HCP_downstream_finetune.py ADDED
@@ -0,0 +1,597 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python
2
+ # coding: utf-8
3
+
4
+ # In[1]:
5
+
6
+
7
+ # Import packages and setup gpu configuration.
8
+ # This code block shouldnt need to be adjusted!
9
+ import os
10
+ import sys
11
+ import json
12
+ import yaml
13
+ import numpy as np
14
+ import copy
15
+ import math
16
+ import time
17
+ import random
18
+ from tqdm.auto import tqdm
19
+ import webdataset as wds
20
+ import matplotlib.pyplot as plt
21
+
22
+ import torch
23
+ import torch.nn as nn
24
+ from torchvision import transforms
25
+ import utils
26
+ from mae_utils.flat_models import *
27
+ import h5py
28
+ from mae_utils import flat_models
29
+
30
+ # tf32 data type is faster than standard float32
31
+ torch.backends.cuda.matmul.allow_tf32 = True
32
+ # following fixes a Conv3D CUDNN_NOT_SUPPORTED error
33
+ torch.backends.cudnn.benchmark = True
34
+
35
+ # ## MODEL TO LOAD ##
36
+ if utils.is_interactive():
37
+ model_name = "HCPflat_large_gsrFalse_"
38
+ else:
39
+ model_name = sys.argv[1]
40
+
41
+
42
+ # outdir = os.path.abspath(f'checkpoints/{model_name}')
43
+ outdir = os.path.abspath(f'checkpoints/{model_name}')
44
+
45
+ print("outdir", outdir)
46
+ # Load previous config.yaml if available
47
+ if os.path.exists(f"{outdir}/config.yaml"):
48
+ config = yaml.load(open(f"{outdir}/config.yaml", 'r'), Loader=yaml.FullLoader)
49
+ print(f"Loaded config.yaml from ckpt folder {outdir}")
50
+ # create global variables from the config
51
+ print("\n__CONFIG__")
52
+ for attribute_name in config.keys():
53
+ print(f"{attribute_name} = {config[attribute_name]}")
54
+ globals()[attribute_name] = config[f'{attribute_name}']
55
+ print("\n")
56
+
57
+ world_size = os.getenv('WORLD_SIZE')
58
+ if world_size is None:
59
+ world_size = 1
60
+ else:
61
+ world_size = int(world_size)
62
+ print(f"WORLD_SIZE={world_size}")
63
+
64
+ if utils.is_interactive():
65
+ # Following allows you to change functions in models.py or utils.py and
66
+ # have this notebook automatically update with your revisions
67
+ get_ipython().run_line_magic('load_ext', 'autoreload')
68
+ get_ipython().run_line_magic('autoreload', '2')
69
+
70
+ batch_size = probe_batch_size
71
+ num_epochs = probe_num_epochs
72
+
73
+ data_type = torch.float32 # change depending on your mixed_precision
74
+ global_batch_size = batch_size * world_size
75
+
76
+ device = torch.device('cuda')
77
+
78
+ hcp_flat_path = "/weka/proj-medarc/shared/HCP-Flat"
79
+ # seed = 42
80
+ # num_frames = 16
81
+ # gsr = False
82
+ # num_workers = 10
83
+ # batch_size = 128
84
+ save_ckpt = True
85
+ wandb_log = True
86
+ print("PID of this process =",os.getpid())
87
+ utils.seed_everything(seed)
88
+
89
+
90
+ # In[2]:
91
+
92
+
93
+ if os.getenv('global_pool') == "False":
94
+ global_pool = False
95
+ else:
96
+ global_pool = True
97
+ print(f"global_pool = {global_pool}")
98
+
99
+ try:
100
+ gsr
101
+ except:
102
+ gsr = True
103
+ print("set gsr to True")
104
+ print(f"gsr = {gsr}")
105
+
106
+
107
+ # In[3]:
108
+
109
+
110
+ #### UNCOMMENT THIS TO SAVE THE HCP-FLAT IN HDF5 FORMAT
111
+
112
+
113
+ # from torch.utils.data import default_collate
114
+ # from mae_utils.flat import load_hcp_flat_mask
115
+ # from mae_utils.flat import create_hcp_flat
116
+ # from mae_utils.flat import batch_unmask
117
+ # import mae_utils.visualize as vis
118
+
119
+
120
+ # batch_size = 26
121
+ # print(f"changed batch_size to {batch_size}")
122
+
123
+ # ## Test ##
124
+ # datasets_to_include = "HCP"
125
+ # assert "HCP" in datasets_to_include
126
+ # test_dataset = create_hcp_flat(root=hcp_flat_path,
127
+ # clip_mode="event", frames=num_frames, shuffle=False, gsr=gsr, sub_list = 'test')
128
+ # test_dl = wds.WebLoader(
129
+ # test_dataset.batched(batch_size, partial=False, collation_fn=default_collate),
130
+ # batch_size=None,
131
+ # shuffle=False,
132
+ # num_workers=num_workers,
133
+ # pin_memory=True,
134
+ # )
135
+
136
+ # ## Train ##
137
+ # assert "HCP" in datasets_to_include
138
+ # train_dataset = create_hcp_flat(root=hcp_flat_path,
139
+ # clip_mode="event", frames=num_frames, shuffle=False, gsr=gsr, sub_list = 'train')
140
+ # train_dl = wds.WebLoader(
141
+ # train_dataset.batched(batch_size, partial=False, collation_fn=default_collate),
142
+ # batch_size=None,
143
+ # shuffle=False,
144
+ # num_workers=num_workers,
145
+ # pin_memory=True,
146
+ # )
147
+
148
+ # def flatten_meta(meta_dict):
149
+ # """
150
+ # Flatten the meta dictionary by:
151
+ # - Replacing single-item lists with the item itself.
152
+ # - Converting tensors to scalar numbers.
153
+ # """
154
+ # flattened = {}
155
+ # for key, value in meta_dict.items():
156
+ # if isinstance(value, list):
157
+ # if len(value) == 1:
158
+ # flattened[key] = value[0] # Replace list with its single item
159
+ # else:
160
+ # flattened[key] = value # Keep as is if multiple items
161
+ # elif isinstance(value, torch.Tensor):
162
+ # # Convert tensor to scalar
163
+ # if value.numel() == 1:
164
+ # flattened[key] = value.item()
165
+ # else:
166
+ # flattened[key] = value.tolist() # Convert multi-element tensor to list
167
+ # else:
168
+ # flattened[key] = value # Keep the value as is
169
+ # return flattened
170
+
171
+ # import h5py
172
+ # meta_array = np.array([], dtype=object)
173
+ # # Open an HDF5 file in write mode
174
+ # with h5py.File('train_hcp.hdf5', 'w') as h5f:
175
+ # flatmaps_dset = None
176
+
177
+ # total_samples = 0
178
+
179
+ # for i, batch in tqdm(enumerate(train_dl), total = 120000):
180
+ # images = batch['image'][0]
181
+ # meta = batch['meta']
182
+ # batch_size = images.shape[0]
183
+ # meta_serializable = meta.copy()
184
+
185
+
186
+ # # Step 2: Serialize the dictionary to a JSON string
187
+ # meta_str = json.dumps(flatten_meta(meta_serializable), indent=4)
188
+ # meta_array = np.append(meta_array, meta_str)
189
+ # if flatmaps_dset is None:
190
+ # # Initialize datasets with unlimited (None) maxshape along the first axis
191
+ # flatmaps_shape = (0,) + images.shape[1:]
192
+ # flatmaps_maxshape = (None,) + images.shape[1:]
193
+
194
+ # flatmaps_dset = h5f.create_dataset(
195
+ # 'flatmaps',
196
+ # shape=flatmaps_shape,
197
+ # maxshape=flatmaps_maxshape,
198
+ # dtype=np.float16,
199
+ # chunks=True # Enable chunking for efficient resizing
200
+ # )
201
+
202
+ # # Resize datasets to accommodate new data
203
+ # flatmaps_dset.resize(total_samples + batch_size, axis=0)
204
+
205
+ # # Write data to the datasets
206
+ # flatmaps_dset[total_samples:total_samples + batch_size] = images.numpy().astype(np.float16)
207
+
208
+ # total_samples += batch_size
209
+
210
+ # print(f"Processed {total_samples} samples")
211
+ # np.save('metadata_test_HCP.npy', meta_array)
212
+
213
+
214
+ # import h5py
215
+ # meta_array = np.array([], dtype=object)
216
+ # # Open an HDF5 file in write mode
217
+ # with h5py.File('test_hcp.hdf5', 'w') as h5f:
218
+ # flatmaps_dset = None
219
+
220
+ # total_samples = 0
221
+
222
+ # for i, batch in tqdm(enumerate(test_dl), total = 12000):
223
+ # images = batch['image'][0]
224
+ # meta = batch['meta']
225
+ # batch_size = images.shape[0]
226
+ # meta_serializable = meta.copy()
227
+
228
+
229
+ # # Step 2: Serialize the dictionary to a JSON string
230
+ # meta_str = json.dumps(flatten_meta(meta_serializable), indent=4)
231
+ # meta_array = np.append(meta_array, meta_str)
232
+ # if flatmaps_dset is None:
233
+ # # Initialize datasets with unlimited (None) maxshape along the first axis
234
+ # flatmaps_shape = (0,) + images.shape[1:]
235
+ # flatmaps_maxshape = (None,) + images.shape[1:]
236
+
237
+ # flatmaps_dset = h5f.create_dataset(
238
+ # 'flatmaps',
239
+ # shape=flatmaps_shape,
240
+ # maxshape=flatmaps_maxshape,
241
+ # dtype=np.float16,
242
+ # chunks=True # Enable chunking for efficient resizing
243
+ # )
244
+
245
+ # # Resize datasets to accommodate new data
246
+ # flatmaps_dset.resize(total_samples + batch_size, axis=0)
247
+
248
+ # # Write data to the datasets
249
+ # flatmaps_dset[total_samples:total_samples + batch_size] = images.numpy().astype(np.float16)
250
+
251
+ # total_samples += batch_size
252
+
253
+ # print(f"Processed {total_samples} samples")
254
+ # np.save('metadata_train_HCP.npy', meta_array)
255
+
256
+
257
+ # ### Preparing data
258
+
259
+ # In[4]:
260
+
261
+
262
+ from sklearn.preprocessing import LabelEncoder
263
+
264
+ INCLUDE_CONDS = {
265
+ "fear",
266
+ "neut",
267
+ "math",
268
+ "story",
269
+ "lf",
270
+ "lh",
271
+ "rf",
272
+ "rh",
273
+ "t",
274
+ "match",
275
+ "relation",
276
+ "mental",
277
+ "rnd",
278
+ "0bk_body",
279
+ "2bk_body",
280
+ "0bk_faces",
281
+ "2bk_faces",
282
+ "0bk_places",
283
+ "2bk_places",
284
+ "0bk_tools",
285
+ "2bk_tools",
286
+ }
287
+
288
+ # test_data = []
289
+
290
+ # # Iterate over the DataLoader with a progress bar
291
+ # for sample in tqdm(train_dl, desc="Processing samples"):
292
+ # x = sample['image']
293
+ # y = sample['meta']['trial_type']
294
+ # key = sample['meta']['key']
295
+ # print(x.shape, y, key)
296
+ # break
297
+ # Initialize the label encoder
298
+ label_encoder = LabelEncoder()
299
+ label_encoder.fit(sorted(INCLUDE_CONDS)) # Ensure consistent ordering
300
+
301
+ num_classes = len(label_encoder.classes_)
302
+ print(f"Number of classes: {num_classes}")
303
+
304
+
305
+ # In[5]:
306
+
307
+
308
+ f_train = h5py.File('/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/train_hcp.hdf5', 'r')
309
+ flatmaps_train = f_train['flatmaps']
310
+
311
+ f_test = h5py.File('/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/test_hcp.hdf5', 'r')
312
+ flatmaps_test = f_test['flatmaps']
313
+
314
+ metadata_train = np.load('/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/metadata_train_HCP.npy', allow_pickle=True)
315
+ metadata_test = np.load('/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/metadata_test_HCP.npy', allow_pickle=True)
316
+
317
+
318
+ # In[6]:
319
+
320
+
321
+ from torch.utils.data import Dataset, DataLoader
322
+
323
+ class HCPFlatDataset(Dataset):
324
+ def __init__(self, flatmaps, metadata):
325
+ self.flatmaps = flatmaps
326
+ self.metadata = metadata
327
+
328
+ def __len__(self):
329
+ return len(self.metadata)
330
+
331
+ def __getitem__(self, idx):
332
+ return self.flatmaps[idx], json.loads(self.metadata[idx])
333
+ print("Moving datasets to ram")
334
+ # Loading to cpu for faster training, this can take several minutes. Remove this [:] if you want to move one at the time.
335
+ train_dataset = HCPFlatDataset(flatmaps_train, metadata_train)
336
+ train_dl = DataLoader(train_dataset, batch_size=batch_size, shuffle=True, num_workers=num_workers)
337
+
338
+ test_dataset = HCPFlatDataset(flatmaps_test, metadata_test)
339
+ test_dl = DataLoader(test_dataset, batch_size=batch_size, shuffle=False, num_workers=0)
340
+ print("Datasets ready")
341
+
342
+
343
+ # ### Creating and loading Model
344
+
345
+ # In[7]:
346
+
347
+
348
+ from mae_utils.flat import load_hcp_flat_mask
349
+ from mae_utils.flat import create_hcp_flat
350
+ from mae_utils.flat import batch_unmask
351
+ import mae_utils.visualize as vis
352
+
353
+ flat_mask = load_hcp_flat_mask(hcp_flat_path)
354
+
355
+ mae_model = flat_models.mae_vit_large_fmri(
356
+ patch_size=patch_size,
357
+ decoder_embed_dim=decoder_embed_dim,
358
+ t_patch_size=t_patch_size,
359
+ pred_t_dim=pred_t_dim,
360
+ decoder_depth=4,
361
+ cls_embed=cls_embed,
362
+ norm_pix_loss=norm_pix_loss,
363
+ no_qkv_bias=no_qkv_bias,
364
+ sep_pos_embed=sep_pos_embed,
365
+ trunc_init=trunc_init,
366
+ pct_masks_to_decode=pct_masks_to_decode,
367
+ img_mask=flat_mask,
368
+ )
369
+
370
+
371
+ # In[8]:
372
+
373
+
374
+ checkpoint_files = [f for f in os.listdir(outdir) if f.endswith('.pth')]
375
+
376
+ if utils.is_interactive():
377
+ latest_checkpoint = "epoch99.pth"
378
+ else:
379
+ latest_checkpoint = sys.argv[2]
380
+ print(f"latest_checkpoint: {latest_checkpoint}")
381
+
382
+ # Load the checkpoint
383
+ checkpoint_path = os.path.join(outdir, latest_checkpoint)
384
+
385
+ state = torch.load(checkpoint_path)
386
+ mae_model.load_state_dict(state["model_state_dict"], strict=False)
387
+ mae_model.to(device)
388
+
389
+ print(f"\nLoaded checkpoint {latest_checkpoint} from {outdir}\n")
390
+
391
+
392
+ # In[9]:
393
+
394
+
395
+ class LinearClassifier(nn.Module):
396
+ def __init__(self, input_dim, num_classes):
397
+ super(LinearClassifier, self).__init__()
398
+ self.linear = nn.Linear(input_dim, num_classes)
399
+
400
+ def forward(self, x):
401
+ # Flatten the input except for the batch dimension
402
+ x = x.view(x.size(0), -1)
403
+ out = self.linear(x)
404
+ return out # Raw logits
405
+
406
+ # Determine the input dimension from a single sample
407
+ # Assuming images are of shape [1, 16, 144, 320]
408
+ input_dim = np.prod(mae_model(torch.randn(1,1,16,144,320).to(device),global_pool=global_pool, forward_features = True).shape[1:])
409
+ print(f"Input dimension: {input_dim}")
410
+
411
+
412
+ # In[10]:
413
+
414
+
415
+ class FullModel(nn.Module):
416
+ def __init__(self, lc_model, mae_model):
417
+ super(FullModel, self).__init__()
418
+ self.lc_model = lc_model
419
+ self.mae_model = mae_model
420
+
421
+
422
+ def forward(self, x, gsr):
423
+ x = self.mae_model(x, global_pool=global_pool, forward_features = True)
424
+ x = self.lc_model(x)
425
+ return x
426
+
427
+
428
+ # In[11]:
429
+
430
+
431
+ # Initialize the model
432
+ lc_model = LinearClassifier(input_dim=input_dim, num_classes=num_classes)
433
+
434
+ model = FullModel(lc_model, mae_model)
435
+
436
+ # Move the model to the GPU
437
+ model.to(device)
438
+
439
+ # Define loss function
440
+ criterion = nn.CrossEntropyLoss()
441
+
442
+ # Define optimizer with L2 regularization (weight_decay)
443
+ learning_rate = 1e-4
444
+ weight_decay = 1e-5 # Adjust based on your needs
445
+ optimizer = torch.optim.Adam(model.parameters(), lr=learning_rate, weight_decay=weight_decay)
446
+ num_epochs = 20 # Adjust as needed
447
+
448
+
449
+ # ### Data
450
+
451
+ # In[16]:
452
+
453
+
454
+ import uuid
455
+
456
+ myuuid = uuid.uuid4()
457
+ str(myuuid)
458
+
459
+
460
+ # In[17]:
461
+
462
+
463
+ import wandb
464
+
465
+ if utils.is_interactive():
466
+ print("Running in interactive notebook. Disabling W&B and ckpt saving.")
467
+ wandb_log = True
468
+ save_ckpt = True
469
+
470
+ if wandb_log:
471
+ wandb_project = 'fMRI-foundation-model'
472
+ wandb_config = {
473
+ "model_name": model_name+'_HCP_FT',
474
+ "batch_size": batch_size,
475
+ "learning_rate": learning_rate,
476
+ "weight_decay": weight_decay,
477
+ "num_epochs": num_epochs,
478
+ "seed": seed,
479
+ }
480
+ print("wandb_config:\n", wandb_config)
481
+ random_id = str(uuid.uuid4())
482
+ print("wandb_id:", "HCPflat_raw" + f"_{random_id}")
483
+ wandb.init(
484
+ id=model_name+'_HCP_FT' + f"_{random_id}",
485
+ project=wandb_project,
486
+ name=model_name+'_HCP_FT',
487
+ config=wandb_config,
488
+ resume="allow",
489
+ )
490
+
491
+
492
+ # In[13]:
493
+
494
+
495
+ for epoch in range(num_epochs):
496
+ running_train_loss = 0.0
497
+ correct_train = 0
498
+ total_train = 0
499
+ step = 0
500
+
501
+ # with torch.amp.autocast(device_type='cuda'):
502
+ # Training Phase
503
+ model.train()
504
+ for batch in tqdm(train_dl, desc=f"Epoch {epoch+1}/{num_epochs} - Training"):
505
+ optimizer.zero_grad()
506
+ images = batch[0].to(device).float().unsqueeze(1) #fix this # Shape: [batch_size, 1, 16, 144, 320]
507
+ labels = batch[1]['trial_type'] # List of labels
508
+
509
+ encoded_labels = label_encoder.transform(labels)
510
+ encoded_labels = torch.tensor(encoded_labels, dtype=torch.long).to(device) # Shape: [batch_size]
511
+
512
+ # Forward pass
513
+ outputs = model(images, gsr=gsr) # Shape: [num_train_samples, num_classes]
514
+
515
+ # Compute loss
516
+ loss = criterion(outputs, encoded_labels)
517
+
518
+ # Backward pass and optimization
519
+ loss.backward()
520
+ optimizer.step()
521
+
522
+ # Accumulate loss
523
+ running_train_loss += loss.item() * images.size(0)
524
+
525
+
526
+ # Calculate accuracy
527
+ _, predicted = torch.max(outputs, 1)
528
+
529
+ correct_train += (predicted == encoded_labels).sum().item()
530
+ total_train += encoded_labels.size(0)
531
+
532
+ step = step + 1
533
+ if step % 100 == 0:
534
+ print(f"Step [{step}/{len(train_dl)}] - Training Loss: {loss.item():.4f} - Training Accuracy: {100 * correct_train / total_train:.2f}%")
535
+ # thth
536
+
537
+ epoch_train_loss = running_train_loss / total_train if total_train > 0 else 0.0
538
+ train_accuracy = 100 * correct_train / total_train if total_train > 0 else 0.0
539
+
540
+ # Validation Phase
541
+ model.eval()
542
+ running_val_loss = 0.0
543
+ correct_val = 0
544
+ total_val = 0
545
+
546
+ with torch.no_grad():
547
+ for batch in tqdm(test_dl, desc=f"Epoch {epoch+1}/{num_epochs} - Validation"):
548
+
549
+ images = batch[0].to(device).float().unsqueeze(1) #fix this
550
+ labels = batch[1]['trial_type']
551
+
552
+ # Encode labels to integer indices
553
+ encoded_labels = label_encoder.transform(labels)
554
+ encoded_labels = torch.tensor(encoded_labels, dtype=torch.long).to(device)
555
+
556
+
557
+ # Forward pass
558
+ outputs = model(images, gsr=gsr)
559
+
560
+ # Compute loss
561
+ loss = criterion(outputs, encoded_labels)
562
+
563
+ # Accumulate loss
564
+ running_val_loss += loss.item() * images.size(0)
565
+
566
+ # Calculate accuracy
567
+ _, predicted = torch.max(outputs, 1)
568
+ correct_val += (predicted == encoded_labels).sum().item()
569
+ total_val += encoded_labels.size(0)
570
+
571
+
572
+
573
+ epoch_val_loss = running_val_loss / total_val if total_val > 0 else 0.0
574
+ val_accuracy = 100 * correct_val / total_val if total_val > 0 else 0.0
575
+
576
+ print(f"Epoch [{epoch+1}/{num_epochs}] "
577
+ f"- Training Loss: {epoch_train_loss:.4f}, Training Accuracy: {train_accuracy:.2f}% "
578
+ f"- Validation Loss: {epoch_val_loss:.4f}, Validation Accuracy: {val_accuracy:.2f}%")
579
+
580
+ if wandb_log:
581
+ wandb.log({
582
+ "epoch_train_loss": epoch_train_loss,
583
+ "epoch_val_loss": epoch_val_loss,
584
+ "train_accuracy": train_accuracy,
585
+ "val_accuracy": val_accuracy,
586
+ })
587
+ if save_ckpt:
588
+ outdir = os.path.abspath(f'checkpoints/{model_name+"HCP_FT"}')
589
+ os.makedirs(outdir, exist_ok=True)
590
+ print("outdir", outdir)
591
+ # Save model and config
592
+ torch.save(model.state_dict(), f"{outdir}/model.pth")
593
+ with open(f"{outdir}/config.yaml", 'w') as f:
594
+ yaml.dump(wandb_config, f)
595
+ print(f"Saved model and config to {outdir}")
596
+
597
+
fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/files/config.yaml ADDED
@@ -0,0 +1,47 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ _wandb:
2
+ value:
3
+ cli_version: 0.18.3
4
+ code_path: code/src/HCP_downstream_finetune.py
5
+ m: []
6
+ python_version: 3.11.10
7
+ t:
8
+ "1":
9
+ - 1
10
+ - 5
11
+ - 41
12
+ - 49
13
+ - 53
14
+ - 55
15
+ - 63
16
+ "2":
17
+ - 1
18
+ - 5
19
+ - 41
20
+ - 49
21
+ - 53
22
+ - 55
23
+ - 63
24
+ "3":
25
+ - 13
26
+ - 14
27
+ - 16
28
+ - 23
29
+ - 55
30
+ "4": 3.11.10
31
+ "5": 0.18.3
32
+ "8":
33
+ - 5
34
+ "12": 0.18.3
35
+ "13": linux-x86_64
36
+ batch_size:
37
+ value: 8
38
+ learning_rate:
39
+ value: 0.0001
40
+ model_name:
41
+ value: HCPflat_large_gsrFalse__HCP_FT
42
+ num_epochs:
43
+ value: 20
44
+ seed:
45
+ value: 42
46
+ weight_decay:
47
+ value: 1e-05
fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/files/output.log ADDED
The diff for this file is too large to render. See raw diff
 
fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/files/wandb-metadata.json ADDED
@@ -0,0 +1,131 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-5.15.0-1058-aws-x86_64-with-glibc2.31",
3
+ "python": "3.11.10",
4
+ "startedAt": "2024-10-24T21:38:35.063615Z",
5
+ "args": [
6
+ "HCPflat_large_gsrFalse_",
7
+ "epoch99.pth"
8
+ ],
9
+ "program": "/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.py",
10
+ "codePath": "src/HCP_downstream_finetune.py",
11
+ "git": {
12
+ "remote": "https://github.com/MedARC-AI/fMRI-foundation-model",
13
+ "commit": "cf8214d4ebe437188b68b4ee5a34c5211a810db0"
14
+ },
15
+ "email": "torrico.villanueva.cesar.kadir@gmail.com",
16
+ "root": "/weka/proj-fmri/ckadirt/fMRI-foundation-model/src",
17
+ "host": "ip-10-0-152-216",
18
+ "username": "ckadirt",
19
+ "executable": "/admin/home-ckadirt/foundation_env/bin/python",
20
+ "codePathLocal": "HCP_downstream_finetune.py",
21
+ "cpu_count": 96,
22
+ "cpu_count_logical": 192,
23
+ "gpu": "[NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3]",
24
+ "gpu_count": 8,
25
+ "disk": {
26
+ "/": {
27
+ "total": "249555763200",
28
+ "used": "182052564992"
29
+ }
30
+ },
31
+ "memory": {
32
+ "total": "2147443372032"
33
+ },
34
+ "cpu": {
35
+ "count": 96,
36
+ "countLogical": 192
37
+ },
38
+ "gpu_nvidia": [
39
+ {
40
+ "name": "NVIDIA H100 80GB HBM3",
41
+ "memoryTotal": "85520809984",
42
+ "cudaCores": 16896,
43
+ "architecture": "Hopper"
44
+ },
45
+ {
46
+ "name": "NVIDIA H100 80GB HBM3",
47
+ "memoryTotal": "85520809984",
48
+ "cudaCores": 16896,
49
+ "architecture": "Hopper"
50
+ },
51
+ {
52
+ "name": "NVIDIA H100 80GB HBM3",
53
+ "memoryTotal": "85520809984",
54
+ "cudaCores": 16896,
55
+ "architecture": "Hopper"
56
+ },
57
+ {
58
+ "name": "NVIDIA H100 80GB HBM3",
59
+ "memoryTotal": "85520809984",
60
+ "cudaCores": 16896,
61
+ "architecture": "Hopper"
62
+ },
63
+ {
64
+ "name": "NVIDIA H100 80GB HBM3",
65
+ "memoryTotal": "85520809984",
66
+ "cudaCores": 16896,
67
+ "architecture": "Hopper"
68
+ },
69
+ {
70
+ "name": "NVIDIA H100 80GB HBM3",
71
+ "memoryTotal": "85520809984",
72
+ "cudaCores": 16896,
73
+ "architecture": "Hopper"
74
+ },
75
+ {
76
+ "name": "NVIDIA H100 80GB HBM3",
77
+ "memoryTotal": "85520809984",
78
+ "cudaCores": 16896,
79
+ "architecture": "Hopper"
80
+ },
81
+ {
82
+ "name": "NVIDIA H100 80GB HBM3",
83
+ "memoryTotal": "85520809984",
84
+ "cudaCores": 16896,
85
+ "architecture": "Hopper"
86
+ }
87
+ ],
88
+ "slurm": {
89
+ "cluster_name": "sagemaker2",
90
+ "conf": "/opt/slurm/etc/slurm.conf",
91
+ "cpus_on_node": "20",
92
+ "gpus_on_node": "1",
93
+ "gpus_per_task": "1",
94
+ "gtids": "0",
95
+ "job_account": "fmri",
96
+ "job_cpus_per_node": "20",
97
+ "job_end_time": "1729921091",
98
+ "job_gid": "1879800513",
99
+ "job_gpus": "5",
100
+ "job_id": "529189",
101
+ "job_name": "finetuneHCP",
102
+ "job_nodelist": "ip-10-0-152-216",
103
+ "job_num_nodes": "1",
104
+ "job_partition": "p5",
105
+ "job_qos": "normal",
106
+ "job_start_time": "1729805891",
107
+ "job_uid": "1879804696",
108
+ "job_user": "ckadirt",
109
+ "jobid": "529189",
110
+ "localid": "0",
111
+ "mem_per_cpu": "11500",
112
+ "nnodes": "1",
113
+ "node_aliases": "(null)",
114
+ "nodeid": "0",
115
+ "nodelist": "ip-10-0-152-216",
116
+ "nprocs": "1",
117
+ "ntasks": "1",
118
+ "ntasks_per_node": "1",
119
+ "prio_process": "0",
120
+ "procid": "0",
121
+ "script_context": "prolog_task",
122
+ "submit_dir": "/weka/proj-fmri/ckadirt/fMRI-foundation-model/src",
123
+ "submit_host": "ip-172-17-12-61",
124
+ "task_pid": "337561",
125
+ "tasks_per_node": "1",
126
+ "topology_addr": "ip-10-0-152-216",
127
+ "topology_addr_pattern": "node",
128
+ "working_cluster": "sagemaker2:ip-172-17-63-161:6817:9984:109"
129
+ },
130
+ "cudaVersion": "12.2"
131
+ }
fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/files/wandb-summary.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"_runtime":103260.163117955,"_step":19,"epoch_train_loss":0.02495735058172655,"epoch_val_loss":0.10187672240381022,"_wandb":{"runtime":103261},"train_accuracy":99.20935832240211,"val_accuracy":97.4921370634001,"_timestamp":1.7299091752265117e+09}
fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/logs/debug-core.log ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2024-10-24T21:38:34.366295935Z","level":"INFO","msg":"started logging, with flags","port-filename":"/tmp/tmpwn5d9jvk/port-337263.txt","pid":337263,"debug":false,"disable-analytics":false}
2
+ {"time":"2024-10-24T21:38:34.366306785Z","level":"INFO","msg":"started logging, with flags","port-filename":"/tmp/tmp_63dqr1q/port-337587.txt","pid":337587,"debug":false,"disable-analytics":false}
3
+ {"time":"2024-10-24T21:38:34.366577281Z","level":"INFO","msg":"FeatureState","shutdownOnParentExitEnabled":false}
4
+ {"time":"2024-10-24T21:38:34.366603112Z","level":"INFO","msg":"FeatureState","shutdownOnParentExitEnabled":false}
5
+ {"time":"2024-10-24T21:38:34.372814525Z","level":"INFO","msg":"Will exit if parent process dies.","ppid":337587}
6
+ {"time":"2024-10-24T21:38:34.372817685Z","level":"INFO","msg":"server is running","addr":{"IP":"127.0.0.1","Port":43157,"Zone":""}}
7
+ {"time":"2024-10-24T21:38:34.373811796Z","level":"INFO","msg":"Will exit if parent process dies.","ppid":337263}
8
+ {"time":"2024-10-24T21:38:34.373817646Z","level":"INFO","msg":"server is running","addr":{"IP":"127.0.0.1","Port":44789,"Zone":""}}
9
+ {"time":"2024-10-24T21:38:34.528156376Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"127.0.0.1:38070"}
10
+ {"time":"2024-10-24T21:38:34.528235848Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"127.0.0.1:51210"}
11
+ {"time":"2024-10-24T21:38:35.06181568Z","level":"INFO","msg":"handleInformInit: received","streamId":"HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7","id":"127.0.0.1:38070"}
12
+ {"time":"2024-10-24T21:38:35.093628892Z","level":"INFO","msg":"handleInformInit: received","streamId":"HCPflat_large_gsrFalse__HCP_FT_d07870e8-3c53-420c-b734-f32a1c5c3e5a","id":"127.0.0.1:51210"}
13
+ {"time":"2024-10-24T21:38:35.140878185Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7","id":"127.0.0.1:38070"}
14
+ {"time":"2024-10-24T21:38:35.145483394Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"HCPflat_large_gsrFalse__HCP_FT_d07870e8-3c53-420c-b734-f32a1c5c3e5a","id":"127.0.0.1:51210"}
15
+ {"time":"2024-10-26T02:19:36.419933572Z","level":"INFO","msg":"handleInformTeardown: server teardown initiated","id":"127.0.0.1:38070"}
16
+ {"time":"2024-10-26T02:19:36.421381993Z","level":"INFO","msg":"server is shutting down"}
17
+ {"time":"2024-10-26T02:19:36.421375983Z","level":"INFO","msg":"connection: Close: initiating connection closure","id":"127.0.0.1:38070"}
18
+ {"time":"2024-10-26T02:19:36.421592698Z","level":"INFO","msg":"connection: Close: connection successfully closed","id":"127.0.0.1:38070"}
19
+ {"time":"2024-10-26T02:19:38.264827227Z","level":"INFO","msg":"handleInformTeardown: server shutdown complete","id":"127.0.0.1:38070"}
20
+ {"time":"2024-10-26T02:19:38.264913688Z","level":"INFO","msg":"connection: ManageConnectionData: connection closed","id":"127.0.0.1:38070"}
21
+ {"time":"2024-10-26T02:19:38.264935459Z","level":"INFO","msg":"server is closed"}
fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/logs/debug-internal.log ADDED
@@ -0,0 +1,19 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2024-10-24T21:38:35.081361619Z","level":"INFO","msg":"using version","core version":"0.18.3"}
2
+ {"time":"2024-10-24T21:38:35.081379619Z","level":"INFO","msg":"created symlink","path":"/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/logs/debug-core.log"}
3
+ {"time":"2024-10-24T21:38:35.09309359Z","level":"ERROR","msg":"dialing: google: could not find default credentials. See https://cloud.google.com/docs/authentication/external/set-up-adc for more information"}
4
+ {"time":"2024-10-24T21:38:35.140844725Z","level":"INFO","msg":"created new stream","id":"HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7"}
5
+ {"time":"2024-10-24T21:38:35.140872365Z","level":"INFO","msg":"stream: started","id":"HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7"}
6
+ {"time":"2024-10-24T21:38:35.140886525Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7"}}
7
+ {"time":"2024-10-24T21:38:35.140891135Z","level":"INFO","msg":"sender: started","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7"}}
8
+ {"time":"2024-10-24T21:38:35.140894655Z","level":"INFO","msg":"handler: started","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7"}}
9
+ {"time":"2024-10-24T21:38:35.713911263Z","level":"INFO","msg":"wandb-core","!BADKEY":null}
10
+ {"time":"2024-10-24T21:38:35.719170496Z","level":"INFO","msg":"Starting system monitor"}
11
+ {"time":"2024-10-24T21:38:35.766834168Z","level":"ERROR","msg":"git repo not found","error":"repository does not exist"}
12
+ {"time":"2024-10-26T02:19:36.421403224Z","level":"INFO","msg":"stream: closing","id":"HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7"}
13
+ {"time":"2024-10-26T02:19:36.421508436Z","level":"INFO","msg":"Stopping system monitor"}
14
+ {"time":"2024-10-26T02:19:36.446202186Z","level":"INFO","msg":"Stopped system monitor"}
15
+ {"time":"2024-10-26T02:19:37.733741018Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
16
+ {"time":"2024-10-26T02:19:37.995379307Z","level":"INFO","msg":"handler: closed","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7"}}
17
+ {"time":"2024-10-26T02:19:37.995451438Z","level":"INFO","msg":"sender: closed","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7"}}
18
+ {"time":"2024-10-26T02:19:37.995438468Z","level":"INFO","msg":"writer: Close: closed","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7"}}
19
+ {"time":"2024-10-26T02:19:38.012118493Z","level":"INFO","msg":"stream: closed","id":"HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7"}
fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/logs/debug.log ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 2024-10-24 21:38:35,040 INFO MainThread:337587 [wandb_setup.py:_flush():79] Current SDK version is 0.18.3
2
+ 2024-10-24 21:38:35,040 INFO MainThread:337587 [wandb_setup.py:_flush():79] Configure stats pid to 337587
3
+ 2024-10-24 21:38:35,040 INFO MainThread:337587 [wandb_setup.py:_flush():79] Loading settings from /admin/home-ckadirt/.config/wandb/settings
4
+ 2024-10-24 21:38:35,040 INFO MainThread:337587 [wandb_setup.py:_flush():79] Loading settings from /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/settings
5
+ 2024-10-24 21:38:35,041 INFO MainThread:337587 [wandb_setup.py:_flush():79] Loading settings from environment variables: {}
6
+ 2024-10-24 21:38:35,041 INFO MainThread:337587 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': None, '_disable_service': None}
7
+ 2024-10-24 21:38:35,041 INFO MainThread:337587 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'src/HCP_downstream_finetune.py', 'program_abspath': '/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.py', 'program': '/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.py'}
8
+ 2024-10-24 21:38:35,041 INFO MainThread:337587 [wandb_setup.py:_flush():79] Applying login settings: {}
9
+ 2024-10-24 21:38:35,042 INFO MainThread:337587 [wandb_init.py:_log_setup():532] Logging user logs to /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/logs/debug.log
10
+ 2024-10-24 21:38:35,042 INFO MainThread:337587 [wandb_init.py:_log_setup():533] Logging internal logs to /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/logs/debug-internal.log
11
+ 2024-10-24 21:38:35,042 INFO MainThread:337587 [wandb_init.py:init():617] calling init triggers
12
+ 2024-10-24 21:38:35,042 INFO MainThread:337587 [wandb_init.py:init():624] wandb.init called with sweep_config: {}
13
+ config: {'model_name': 'HCPflat_large_gsrFalse__HCP_FT', 'batch_size': 8, 'learning_rate': 0.0001, 'weight_decay': 1e-05, 'num_epochs': 20, 'seed': 42}
14
+ 2024-10-24 21:38:35,042 INFO MainThread:337587 [wandb_init.py:init():667] starting backend
15
+ 2024-10-24 21:38:35,042 INFO MainThread:337587 [wandb_init.py:init():671] sending inform_init request
16
+ 2024-10-24 21:38:35,058 INFO MainThread:337587 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
17
+ 2024-10-24 21:38:35,058 INFO MainThread:337587 [wandb_init.py:init():684] backend started and connected
18
+ 2024-10-24 21:38:35,090 INFO MainThread:337587 [wandb_init.py:init():779] updated telemetry
19
+ 2024-10-24 21:38:35,163 INFO MainThread:337587 [wandb_init.py:init():812] communicating run to backend with 90.0 second timeout
20
+ 2024-10-24 21:38:35,693 INFO MainThread:337587 [wandb_init.py:init():863] starting run threads in backend
21
+ 2024-10-24 21:38:36,532 INFO MainThread:337587 [wandb_run.py:_console_start():2465] atexit reg
22
+ 2024-10-24 21:38:36,532 INFO MainThread:337587 [wandb_run.py:_redirect():2313] redirect: wrap_raw
23
+ 2024-10-24 21:38:36,532 INFO MainThread:337587 [wandb_run.py:_redirect():2378] Wrapping output streams.
24
+ 2024-10-24 21:38:36,532 INFO MainThread:337587 [wandb_run.py:_redirect():2403] Redirects installed.
25
+ 2024-10-24 21:38:36,548 INFO MainThread:337587 [wandb_init.py:init():907] run started, returning control to user process
26
+ 2024-10-26 02:19:36,421 WARNING MsgRouterThr:337587 [router.py:message_loop():77] message_loop has been closed
fMRI-foundation-model/src/wandb/run-20241024_213834-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7/run-HCPflat_large_gsrFalse__HCP_FT_cdc7b70b-b155-4483-907d-91148a52e1d7.wandb ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e555af263e6470b6c6d672ad54c976ab3187de98b3853038e90487c2d49646bd
3
+ size 182858202
fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/files/config.yaml ADDED
@@ -0,0 +1,48 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ _wandb:
2
+ value:
3
+ cli_version: 0.18.3
4
+ m: []
5
+ python_version: 3.11.9
6
+ t:
7
+ "1":
8
+ - 1
9
+ - 5
10
+ - 41
11
+ - 49
12
+ - 53
13
+ - 55
14
+ - 63
15
+ "2":
16
+ - 1
17
+ - 5
18
+ - 41
19
+ - 49
20
+ - 53
21
+ - 55
22
+ - 63
23
+ "3":
24
+ - 2
25
+ - 13
26
+ - 14
27
+ - 16
28
+ - 23
29
+ - 55
30
+ "4": 3.11.9
31
+ "5": 0.18.3
32
+ "8":
33
+ - 1
34
+ - 5
35
+ "12": 0.18.3
36
+ "13": linux-x86_64
37
+ batch_size:
38
+ value: 8
39
+ learning_rate:
40
+ value: 0.0001
41
+ model_name:
42
+ value: HCPflat_large_gsrFalse__HCP_FT
43
+ num_epochs:
44
+ value: 20
45
+ seed:
46
+ value: 42
47
+ weight_decay:
48
+ value: 1e-05
fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/files/output.log ADDED
@@ -0,0 +1,209 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ outdir /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/checkpoints/HCPflat_large_gsrFalse_
2
+ Loaded config.yaml from ckpt folder /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/checkpoints/HCPflat_large_gsrFalse_
3
+
4
+ __CONFIG__
5
+ base_lr = 0.001
6
+ batch_size = 32
7
+ ckpt_interval = 5
8
+ ckpt_saving = True
9
+ cls_embed = True
10
+ contrastive_loss_weight = 1.0
11
+ datasets_to_include = HCP
12
+ decoder_embed_dim = 512
13
+ grad_accumulation_steps = 1
14
+ grad_clip = 1.0
15
+ gsr = False
16
+ hcp_flat_path = /weka/proj-medarc/shared/HCP-Flat
17
+ mask_ratio = 0.75
18
+ model_name = HCPflat_large_gsrFalse_
19
+ no_qkv_bias = False
20
+ norm_pix_loss = False
21
+ nsd_flat_path = /weka/proj-medarc/shared/NSD-Flat
22
+ num_epochs = 100
23
+ num_frames = 16
24
+ num_samples_per_epoch = 200000
25
+ num_workers = 10
26
+ patch_size = 16
27
+ pct_masks_to_decode = 1
28
+ plotting = True
29
+ pred_t_dim = 8
30
+ print_interval = 20
31
+ probe_base_lr = 0.0003
32
+ probe_batch_size = 8
33
+ probe_num_epochs = 30
34
+ probe_num_samples_per_epoch = 100000
35
+ resume_from_ckpt = True
36
+ seed = 42
37
+ sep_pos_embed = True
38
+ t_patch_size = 2
39
+ test_num_samples_per_epoch = 50000
40
+ test_set = False
41
+ trunc_init = False
42
+ use_contrastive_loss = False
43
+ wandb_log = True
44
+
45
+
46
+ WORLD_SIZE=1
47
+ The autoreload extension is already loaded. To reload it, use:
48
+ %reload_ext autoreload
49
+ PID of this process = 1863430
50
+ outdir /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/checkpoints/HCPflat_large_gsrFalse_
51
+ Loaded config.yaml from ckpt folder /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/checkpoints/HCPflat_large_gsrFalse_
52
+
53
+ __CONFIG__
54
+ base_lr = 0.001
55
+ batch_size = 32
56
+ ckpt_interval = 5
57
+ ckpt_saving = True
58
+ cls_embed = True
59
+ contrastive_loss_weight = 1.0
60
+ datasets_to_include = HCP
61
+ decoder_embed_dim = 512
62
+ grad_accumulation_steps = 1
63
+ grad_clip = 1.0
64
+ gsr = False
65
+ hcp_flat_path = /weka/proj-medarc/shared/HCP-Flat
66
+ mask_ratio = 0.75
67
+ model_name = HCPflat_large_gsrFalse_
68
+ no_qkv_bias = False
69
+ norm_pix_loss = False
70
+ nsd_flat_path = /weka/proj-medarc/shared/NSD-Flat
71
+ num_epochs = 100
72
+ num_frames = 16
73
+ num_samples_per_epoch = 200000
74
+ num_workers = 10
75
+ patch_size = 16
76
+ pct_masks_to_decode = 1
77
+ plotting = True
78
+ pred_t_dim = 8
79
+ print_interval = 20
80
+ probe_base_lr = 0.0003
81
+ probe_batch_size = 8
82
+ probe_num_epochs = 30
83
+ probe_num_samples_per_epoch = 100000
84
+ resume_from_ckpt = True
85
+ seed = 42
86
+ sep_pos_embed = True
87
+ t_patch_size = 2
88
+ test_num_samples_per_epoch = 50000
89
+ test_set = False
90
+ trunc_init = False
91
+ use_contrastive_loss = False
92
+ wandb_log = True
93
+
94
+
95
+ WORLD_SIZE=1
96
+ The autoreload extension is already loaded. To reload it, use:
97
+ %reload_ext autoreload
98
+ PID of this process = 1863430
99
+ outdir /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/checkpoints/HCPflat_large_gsrFalse_
100
+ Loaded config.yaml from ckpt folder /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/checkpoints/HCPflat_large_gsrFalse_
101
+
102
+ __CONFIG__
103
+ base_lr = 0.001
104
+ batch_size = 32
105
+ ckpt_interval = 5
106
+ ckpt_saving = True
107
+ cls_embed = True
108
+ contrastive_loss_weight = 1.0
109
+ datasets_to_include = HCP
110
+ decoder_embed_dim = 512
111
+ grad_accumulation_steps = 1
112
+ grad_clip = 1.0
113
+ gsr = False
114
+ hcp_flat_path = /weka/proj-medarc/shared/HCP-Flat
115
+ mask_ratio = 0.75
116
+ model_name = HCPflat_large_gsrFalse_
117
+ no_qkv_bias = False
118
+ norm_pix_loss = False
119
+ nsd_flat_path = /weka/proj-medarc/shared/NSD-Flat
120
+ num_epochs = 100
121
+ num_frames = 16
122
+ num_samples_per_epoch = 200000
123
+ num_workers = 10
124
+ patch_size = 16
125
+ pct_masks_to_decode = 1
126
+ plotting = True
127
+ pred_t_dim = 8
128
+ print_interval = 20
129
+ probe_base_lr = 0.0003
130
+ probe_batch_size = 8
131
+ probe_num_epochs = 30
132
+ probe_num_samples_per_epoch = 100000
133
+ resume_from_ckpt = True
134
+ seed = 42
135
+ sep_pos_embed = True
136
+ t_patch_size = 2
137
+ test_num_samples_per_epoch = 50000
138
+ test_set = False
139
+ trunc_init = False
140
+ use_contrastive_loss = False
141
+ wandb_log = True
142
+
143
+
144
+ WORLD_SIZE=1
145
+ The autoreload extension is already loaded. To reload it, use:
146
+ %reload_ext autoreload
147
+ PID of this process = 1863430
148
+ outdir /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/checkpoints/HCPflat_large_gsrFalse_
149
+ Loaded config.yaml from ckpt folder /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/checkpoints/HCPflat_large_gsrFalse_
150
+
151
+ __CONFIG__
152
+ base_lr = 0.001
153
+ batch_size = 32
154
+ ckpt_interval = 5
155
+ ckpt_saving = True
156
+ cls_embed = True
157
+ contrastive_loss_weight = 1.0
158
+ datasets_to_include = HCP
159
+ decoder_embed_dim = 512
160
+ grad_accumulation_steps = 1
161
+ grad_clip = 1.0
162
+ gsr = False
163
+ hcp_flat_path = /weka/proj-medarc/shared/HCP-Flat
164
+ mask_ratio = 0.75
165
+ model_name = HCPflat_large_gsrFalse_
166
+ no_qkv_bias = False
167
+ norm_pix_loss = False
168
+ nsd_flat_path = /weka/proj-medarc/shared/NSD-Flat
169
+ num_epochs = 100
170
+ num_frames = 16
171
+ num_samples_per_epoch = 200000
172
+ num_workers = 10
173
+ patch_size = 16
174
+ pct_masks_to_decode = 1
175
+ plotting = True
176
+ pred_t_dim = 8
177
+ print_interval = 20
178
+ probe_base_lr = 0.0003
179
+ probe_batch_size = 8
180
+ probe_num_epochs = 30
181
+ probe_num_samples_per_epoch = 100000
182
+ resume_from_ckpt = True
183
+ seed = 42
184
+ sep_pos_embed = True
185
+ t_patch_size = 2
186
+ test_num_samples_per_epoch = 50000
187
+ test_set = False
188
+ trunc_init = False
189
+ use_contrastive_loss = False
190
+ wandb_log = True
191
+
192
+
193
+ WORLD_SIZE=1
194
+ The autoreload extension is already loaded. To reload it, use:
195
+ %reload_ext autoreload
196
+ PID of this process = 1863430
197
+ img_size (144, 320) patch_size (16, 16) frames 16 t_patch_size 2
198
+ model initialized
199
+ latest_checkpoint: epoch99.pth
200
+ /tmp/ipykernel_1863430/781257473.py:12: FutureWarning: You are using `torch.load` with `weights_only=False` (the current default value), which uses the default pickle module implicitly. It is possible to construct malicious pickle data which will execute arbitrary code during unpickling (See https://github.com/pytorch/pytorch/blob/main/SECURITY.md#untrusted-models for more details). In a future release, the default value for `weights_only` will be flipped to `True`. This limits the functions that could be executed during unpickling. Arbitrary objects will no longer be allowed to be loaded via this mode unless they are explicitly allowlisted by the user via `torch.serialization.add_safe_globals`. We recommend you start setting `weights_only=True` for any use case where you don't have full control of the loaded file. Please open an issue on GitHub for any issues related to this experimental feature.
201
+ state = torch.load(checkpoint_path)
202
+
203
+ Loaded checkpoint epoch99.pth from /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/checkpoints/HCPflat_large_gsrFalse_
204
+
205
+ Input dimension: 1024
206
+ Running in interactive notebook. Disabling W&B and ckpt saving.
207
+ wandb_config:
208
+ {'model_name': 'HCPflat_large_gsrFalse__HCP_FT', 'batch_size': 8, 'learning_rate': 0.0001, 'weight_decay': 1e-05, 'num_epochs': 20, 'seed': 42}
209
+ wandb_id: HCPflat_raw_79bf330c-a53f-43d5-86dd-b4bb676b9b78
fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/files/wandb-metadata.json ADDED
@@ -0,0 +1,144 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "os": "Linux-5.15.0-1058-aws-x86_64-with-glibc2.31",
3
+ "python": "3.11.9",
4
+ "startedAt": "2024-11-26T14:15:28.574035Z",
5
+ "program": "ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.ipynb",
6
+ "git": {
7
+ "remote": "https://github.com/MedARC-AI/fMRI-foundation-model",
8
+ "commit": "7c9bb03314a9f929bb8f0fc0ce92c85ea1a2e495"
9
+ },
10
+ "email": "torrico.villanueva.cesar.kadir@gmail.com",
11
+ "root": "/weka/proj-fmri/ckadirt/fMRI-foundation-model/src",
12
+ "host": "ip-10-0-135-126",
13
+ "username": "ckadirt",
14
+ "executable": "/admin/home-ckadirt/foundation_env/bin/python",
15
+ "cpu_count": 96,
16
+ "cpu_count_logical": 192,
17
+ "gpu": "[NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3, NVIDIA H100 80GB HBM3]",
18
+ "gpu_count": 8,
19
+ "disk": {
20
+ "/": {
21
+ "total": "249555763200",
22
+ "used": "184980103168"
23
+ }
24
+ },
25
+ "memory": {
26
+ "total": "2147443412992"
27
+ },
28
+ "cpu": {
29
+ "count": 96,
30
+ "countLogical": 192
31
+ },
32
+ "gpu_nvidia": [
33
+ {
34
+ "name": "NVIDIA H100 80GB HBM3",
35
+ "memoryTotal": "85520809984",
36
+ "cudaCores": 16896,
37
+ "architecture": "Hopper"
38
+ },
39
+ {
40
+ "name": "NVIDIA H100 80GB HBM3",
41
+ "memoryTotal": "85520809984",
42
+ "cudaCores": 16896,
43
+ "architecture": "Hopper"
44
+ },
45
+ {
46
+ "name": "NVIDIA H100 80GB HBM3",
47
+ "memoryTotal": "85520809984",
48
+ "cudaCores": 16896,
49
+ "architecture": "Hopper"
50
+ },
51
+ {
52
+ "name": "NVIDIA H100 80GB HBM3",
53
+ "memoryTotal": "85520809984",
54
+ "cudaCores": 16896,
55
+ "architecture": "Hopper"
56
+ },
57
+ {
58
+ "name": "NVIDIA H100 80GB HBM3",
59
+ "memoryTotal": "85520809984",
60
+ "cudaCores": 16896,
61
+ "architecture": "Hopper"
62
+ },
63
+ {
64
+ "name": "NVIDIA H100 80GB HBM3",
65
+ "memoryTotal": "85520809984",
66
+ "cudaCores": 16896,
67
+ "architecture": "Hopper"
68
+ },
69
+ {
70
+ "name": "NVIDIA H100 80GB HBM3",
71
+ "memoryTotal": "85520809984",
72
+ "cudaCores": 16896,
73
+ "architecture": "Hopper"
74
+ },
75
+ {
76
+ "name": "NVIDIA H100 80GB HBM3",
77
+ "memoryTotal": "85520809984",
78
+ "cudaCores": 16896,
79
+ "architecture": "Hopper"
80
+ }
81
+ ],
82
+ "slurm": {
83
+ "cluster_name": "sagemaker2",
84
+ "conf": "/opt/slurm/etc/slurm.conf",
85
+ "cpu_bind": "quiet,mask_cpu:0x00000000000000003C00FC0000000000000000003C00FC00",
86
+ "cpu_bind_list": "0x00000000000000003C00FC0000000000000000003C00FC00",
87
+ "cpu_bind_type": "mask_cpu:",
88
+ "cpu_bind_verbose": "quiet",
89
+ "cpus_on_node": "20",
90
+ "gpus": "1",
91
+ "gpus_on_node": "1",
92
+ "gtids": "0",
93
+ "job_account": "fmri",
94
+ "job_cpus_per_node": "20",
95
+ "job_end_time": "1732683404",
96
+ "job_gid": "1879800513",
97
+ "job_group": "Domain Users",
98
+ "job_id": "541182",
99
+ "job_name": "bash",
100
+ "job_nodelist": "ip-10-0-135-126",
101
+ "job_num_nodes": "1",
102
+ "job_partition": "p5",
103
+ "job_qos": "idle",
104
+ "job_start_time": "1732629404",
105
+ "job_uid": "1879804696",
106
+ "job_user": "ckadirt",
107
+ "jobid": "541182",
108
+ "launch_node_ipaddr": "172.17.12.61",
109
+ "localid": "0",
110
+ "mpi_type": "pmix_v3",
111
+ "nnodes": "1",
112
+ "nodeid": "0",
113
+ "nodelist": "ip-10-0-135-126",
114
+ "nprocs": "1",
115
+ "ntasks": "1",
116
+ "pmix_mapping_serv": "(vector,(0,1,1))",
117
+ "pmixp_abort_agent_port": "37569",
118
+ "prio_process": "0",
119
+ "procid": "0",
120
+ "pty_port": "34527",
121
+ "pty_win_col": "205",
122
+ "pty_win_row": "21",
123
+ "script_context": "prolog_task",
124
+ "srun_comm_host": "172.17.12.61",
125
+ "srun_comm_port": "42777",
126
+ "step_gpus": "1",
127
+ "step_id": "0",
128
+ "step_launcher_port": "42777",
129
+ "step_nodelist": "ip-10-0-135-126",
130
+ "step_num_nodes": "1",
131
+ "step_num_tasks": "1",
132
+ "step_tasks_per_node": "1",
133
+ "stepid": "0",
134
+ "submit_dir": "/weka/proj-fmri",
135
+ "submit_host": "ip-172-17-12-61",
136
+ "task_pid": "1856467",
137
+ "tasks_per_node": "1",
138
+ "topology_addr": "ip-10-0-135-126",
139
+ "topology_addr_pattern": "node",
140
+ "umask": "0022",
141
+ "working_cluster": "sagemaker2:ip-172-17-63-161:6817:9984:109"
142
+ },
143
+ "cudaVersion": "12.2"
144
+ }
fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/files/wandb-summary.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"_wandb":{"runtime":1301}}
fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/logs/debug-core.log ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2024-11-26T14:15:15.826910446Z","level":"INFO","msg":"started logging, with flags","port-filename":"/tmp/tmpawnk3nqm/port-1863430.txt","pid":1863430,"debug":false,"disable-analytics":false}
2
+ {"time":"2024-11-26T14:15:15.82724423Z","level":"INFO","msg":"FeatureState","shutdownOnParentExitEnabled":false}
3
+ {"time":"2024-11-26T14:15:15.832997851Z","level":"INFO","msg":"Will exit if parent process dies.","ppid":1863430}
4
+ {"time":"2024-11-26T14:15:15.83295411Z","level":"INFO","msg":"server is running","addr":{"IP":"127.0.0.1","Port":42111,"Zone":""}}
5
+ {"time":"2024-11-26T14:15:15.943657315Z","level":"INFO","msg":"connection: ManageConnectionData: new connection created","id":"127.0.0.1:42876"}
6
+ {"time":"2024-11-26T14:15:16.41249604Z","level":"INFO","msg":"handleInformInit: received","streamId":"HCPflat_large_gsrFalse__HCP_FT_72f5af42-9465-4e84-8949-8372cb087275","id":"127.0.0.1:42876"}
7
+ {"time":"2024-11-26T14:15:16.452884654Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"HCPflat_large_gsrFalse__HCP_FT_72f5af42-9465-4e84-8949-8372cb087275","id":"127.0.0.1:42876"}
8
+ {"time":"2024-11-26T14:15:28.525648106Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"HCPflat_large_gsrFalse__HCP_FT_72f5af42-9465-4e84-8949-8372cb087275","id":"127.0.0.1:42876"}
9
+ {"time":"2024-11-26T14:15:28.52594631Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"HCPflat_large_gsrFalse__HCP_FT_72f5af42-9465-4e84-8949-8372cb087275","id":"127.0.0.1:42876"}
10
+ {"time":"2024-11-26T14:15:28.573694737Z","level":"INFO","msg":"handleInformInit: received","streamId":"HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9","id":"127.0.0.1:42876"}
11
+ {"time":"2024-11-26T14:15:28.594206293Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9","id":"127.0.0.1:42876"}
12
+ {"time":"2024-11-26T14:37:12.211752238Z","level":"INFO","msg":"handleInformFinish: finish message received","streamId":"HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9","id":"127.0.0.1:42876"}
13
+ {"time":"2024-11-26T14:37:12.212190204Z","level":"INFO","msg":"handleInformFinish: stream closed","streamId":"HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9","id":"127.0.0.1:42876"}
14
+ {"time":"2024-11-26T14:37:12.303091089Z","level":"INFO","msg":"handleInformInit: received","streamId":"HCPflat_large_gsrFalse__HCP_FT_79bf330c-a53f-43d5-86dd-b4bb676b9b78","id":"127.0.0.1:42876"}
15
+ {"time":"2024-11-26T14:37:12.332059235Z","level":"INFO","msg":"handleInformInit: stream started","streamId":"HCPflat_large_gsrFalse__HCP_FT_79bf330c-a53f-43d5-86dd-b4bb676b9b78","id":"127.0.0.1:42876"}
16
+ {"time":"2024-11-26T22:09:54.413039355Z","level":"INFO","msg":"Parent process exited, terminating service process."}
fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/logs/debug-internal.log ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"time":"2024-11-26T14:15:28.576474946Z","level":"INFO","msg":"using version","core version":"0.18.3"}
2
+ {"time":"2024-11-26T14:15:28.576498496Z","level":"INFO","msg":"created symlink","path":"/weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/logs/debug-core.log"}
3
+ {"time":"2024-11-26T14:15:28.577184315Z","level":"ERROR","msg":"dialing: google: could not find default credentials. See https://cloud.google.com/docs/authentication/external/set-up-adc for more information"}
4
+ {"time":"2024-11-26T14:15:28.594167422Z","level":"INFO","msg":"created new stream","id":"HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9"}
5
+ {"time":"2024-11-26T14:15:28.594200033Z","level":"INFO","msg":"stream: started","id":"HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9"}
6
+ {"time":"2024-11-26T14:15:28.594247034Z","level":"INFO","msg":"sender: started","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9"}}
7
+ {"time":"2024-11-26T14:15:28.594215853Z","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9"}}
8
+ {"time":"2024-11-26T14:15:28.594240183Z","level":"INFO","msg":"handler: started","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9"}}
9
+ {"time":"2024-11-26T14:15:29.086640847Z","level":"INFO","msg":"wandb-core","!BADKEY":null}
10
+ {"time":"2024-11-26T14:15:29.090676043Z","level":"INFO","msg":"Starting system monitor"}
11
+ {"time":"2024-11-26T14:15:29.090708263Z","level":"WARN","msg":"handleCodeSave: program relative path is empty"}
12
+ {"time":"2024-11-26T14:15:29.090963677Z","level":"ERROR","msg":"git repo not found","error":"repository does not exist"}
13
+ {"time":"2024-11-26T14:37:10.290633768Z","level":"INFO","msg":"Stopping system monitor"}
14
+ {"time":"2024-11-26T14:37:10.305423635Z","level":"INFO","msg":"Stopped system monitor"}
15
+ {"time":"2024-11-26T14:37:10.859005011Z","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
16
+ {"time":"2024-11-26T14:37:12.212055063Z","level":"INFO","msg":"stream: closing","id":"HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9"}
17
+ {"time":"2024-11-26T14:37:12.212084293Z","level":"INFO","msg":"handler: closed","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9"}}
18
+ {"time":"2024-11-26T14:37:12.212114673Z","level":"INFO","msg":"writer: Close: closed","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9"}}
19
+ {"time":"2024-11-26T14:37:12.212130973Z","level":"INFO","msg":"sender: closed","stream_id":{"value":"HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9"}}
20
+ {"time":"2024-11-26T14:37:12.212182504Z","level":"INFO","msg":"stream: closed","id":"HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9"}
21
+ {"time":"2024-11-26T14:37:15.293524941Z","level":"ERROR","msg":"monitor: gpu: timeout waiting for process to exit"}
fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/logs/debug.log ADDED
@@ -0,0 +1,61 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 2024-11-26 14:15:25,764 INFO MainThread:1863430 [wandb_setup.py:_flush():79] Current SDK version is 0.18.3
2
+ 2024-11-26 14:15:25,765 INFO MainThread:1863430 [wandb_setup.py:_flush():79] Configure stats pid to 1863430
3
+ 2024-11-26 14:15:25,765 INFO MainThread:1863430 [wandb_setup.py:_flush():79] Loading settings from /admin/home-ckadirt/.config/wandb/settings
4
+ 2024-11-26 14:15:25,766 INFO MainThread:1863430 [wandb_setup.py:_flush():79] Loading settings from /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/settings
5
+ 2024-11-26 14:15:25,766 INFO MainThread:1863430 [wandb_setup.py:_flush():79] Loading settings from environment variables: {}
6
+ 2024-11-26 14:15:25,766 INFO MainThread:1863430 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': None, '_disable_service': None}
7
+ 2024-11-26 14:15:25,766 INFO MainThread:1863430 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program': '<python with no main file>'}
8
+ 2024-11-26 14:15:25,766 INFO MainThread:1863430 [wandb_setup.py:_flush():79] Applying login settings: {}
9
+ 2024-11-26 14:15:25,766 INFO MainThread:1863430 [wandb_init.py:_log_setup():532] Logging user logs to /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/logs/debug.log
10
+ 2024-11-26 14:15:25,767 INFO MainThread:1863430 [wandb_init.py:_log_setup():533] Logging internal logs to /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241126_141525-HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9/logs/debug-internal.log
11
+ 2024-11-26 14:15:25,768 INFO MainThread:1863430 [wandb_init.py:init():617] calling init triggers
12
+ 2024-11-26 14:15:25,768 INFO MainThread:1863430 [wandb_init.py:init():624] wandb.init called with sweep_config: {}
13
+ config: {'model_name': 'HCPflat_large_gsrFalse__HCP_FT', 'batch_size': 8, 'learning_rate': 0.0001, 'weight_decay': 1e-05, 'num_epochs': 20, 'seed': 42}
14
+ 2024-11-26 14:15:25,768 INFO MainThread:1863430 [wandb_init.py:init():642] re-initializing run, found existing run on stack: HCPflat_large_gsrFalse__HCP_FT_72f5af42-9465-4e84-8949-8372cb087275
15
+ 2024-11-26 14:15:25,774 INFO MainThread:1863430 [wandb_run.py:_finish():2164] finishing run ckadirt/fMRI-foundation-model/HCPflat_large_gsrFalse__HCP_FT_72f5af42-9465-4e84-8949-8372cb087275
16
+ 2024-11-26 14:15:25,850 INFO MainThread:1863430 [jupyter.py:save_history():488] saving 13 cells to _session_history.ipynb
17
+ 2024-11-26 14:15:25,850 INFO MainThread:1863430 [wandb_run.py:_config_callback():1394] config_cb ('_wandb', 'session_history') code/_session_history.ipynb None
18
+ 2024-11-26 14:15:25,863 INFO MainThread:1863430 [jupyter.py:_save_ipynb():398] looking for notebook: ckadirt/fMRI-foundation-model/src/HCP_downstream_finetune.ipynb
19
+ 2024-11-26 14:15:25,863 INFO MainThread:1863430 [wandb_init.py:_jupyter_teardown():460] cleaning up jupyter logic
20
+ 2024-11-26 14:15:25,863 INFO MainThread:1863430 [wandb_run.py:_atexit_cleanup():2428] got exitcode: 0
21
+ 2024-11-26 14:15:25,863 INFO MainThread:1863430 [wandb_run.py:_restore():2410] restore
22
+ 2024-11-26 14:15:25,864 INFO MainThread:1863430 [wandb_run.py:_restore():2416] restore done
23
+ 2024-11-26 14:15:28,512 INFO MainThread:1863430 [wandb_run.py:_footer_history_summary_info():4049] rendering history
24
+ 2024-11-26 14:15:28,512 INFO MainThread:1863430 [wandb_run.py:_footer_history_summary_info():4081] rendering summary
25
+ 2024-11-26 14:15:28,522 INFO MainThread:1863430 [wandb_run.py:_footer_sync_info():4008] logging synced files
26
+ 2024-11-26 14:15:28,569 INFO MainThread:1863430 [wandb_init.py:init():667] starting backend
27
+ 2024-11-26 14:15:28,569 INFO MainThread:1863430 [wandb_init.py:init():671] sending inform_init request
28
+ 2024-11-26 14:15:28,573 INFO MainThread:1863430 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
29
+ 2024-11-26 14:15:28,573 INFO MainThread:1863430 [wandb_init.py:init():684] backend started and connected
30
+ 2024-11-26 14:15:28,588 INFO MainThread:1863430 [wandb_run.py:_label_probe_notebook():1346] probe notebook
31
+ 2024-11-26 14:15:28,589 INFO MainThread:1863430 [wandb_run.py:_label_probe_notebook():1356] Unable to probe notebook: 'NoneType' object has no attribute 'get'
32
+ 2024-11-26 14:15:28,589 INFO MainThread:1863430 [wandb_init.py:init():779] updated telemetry
33
+ 2024-11-26 14:15:28,605 INFO MainThread:1863430 [wandb_init.py:init():812] communicating run to backend with 90.0 second timeout
34
+ 2024-11-26 14:15:29,080 INFO MainThread:1863430 [wandb_init.py:init():863] starting run threads in backend
35
+ 2024-11-26 14:15:29,397 INFO MainThread:1863430 [wandb_run.py:_console_start():2465] atexit reg
36
+ 2024-11-26 14:15:29,397 INFO MainThread:1863430 [wandb_run.py:_redirect():2313] redirect: wrap_raw
37
+ 2024-11-26 14:15:29,397 INFO MainThread:1863430 [wandb_run.py:_redirect():2378] Wrapping output streams.
38
+ 2024-11-26 14:15:29,397 INFO MainThread:1863430 [wandb_run.py:_redirect():2403] Redirects installed.
39
+ 2024-11-26 14:15:29,398 INFO MainThread:1863430 [wandb_init.py:init():907] run started, returning control to user process
40
+ 2024-11-26 14:37:10,287 INFO MainThread:1863430 [wandb_setup.py:_flush():79] Current SDK version is 0.18.3
41
+ 2024-11-26 14:37:10,287 INFO MainThread:1863430 [wandb_setup.py:_flush():79] Configure stats pid to 1863430
42
+ 2024-11-26 14:37:10,287 INFO MainThread:1863430 [wandb_setup.py:_flush():79] Loading settings from /admin/home-ckadirt/.config/wandb/settings
43
+ 2024-11-26 14:37:10,287 INFO MainThread:1863430 [wandb_setup.py:_flush():79] Loading settings from /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/settings
44
+ 2024-11-26 14:37:10,287 INFO MainThread:1863430 [wandb_setup.py:_flush():79] Loading settings from environment variables: {}
45
+ 2024-11-26 14:37:10,287 INFO MainThread:1863430 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': None, '_disable_service': None}
46
+ 2024-11-26 14:37:10,287 INFO MainThread:1863430 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program': '<python with no main file>'}
47
+ 2024-11-26 14:37:10,287 INFO MainThread:1863430 [wandb_setup.py:_flush():79] Applying login settings: {}
48
+ 2024-11-26 14:37:10,288 INFO MainThread:1863430 [wandb_init.py:_log_setup():532] Logging user logs to /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241126_143710-HCPflat_large_gsrFalse__HCP_FT_79bf330c-a53f-43d5-86dd-b4bb676b9b78/logs/debug.log
49
+ 2024-11-26 14:37:10,288 INFO MainThread:1863430 [wandb_init.py:_log_setup():533] Logging internal logs to /weka/proj-fmri/ckadirt/fMRI-foundation-model/src/wandb/run-20241126_143710-HCPflat_large_gsrFalse__HCP_FT_79bf330c-a53f-43d5-86dd-b4bb676b9b78/logs/debug-internal.log
50
+ 2024-11-26 14:37:10,288 INFO MainThread:1863430 [wandb_init.py:_jupyter_setup():478] configuring jupyter hooks <wandb.sdk.wandb_init._WandbInit object at 0x7fea7f22a610>
51
+ 2024-11-26 14:37:10,288 INFO MainThread:1863430 [wandb_init.py:init():617] calling init triggers
52
+ 2024-11-26 14:37:10,288 INFO MainThread:1863430 [wandb_init.py:init():624] wandb.init called with sweep_config: {}
53
+ config: {'model_name': 'HCPflat_large_gsrFalse__HCP_FT', 'batch_size': 8, 'learning_rate': 0.0001, 'weight_decay': 1e-05, 'num_epochs': 20, 'seed': 42}
54
+ 2024-11-26 14:37:10,288 INFO MainThread:1863430 [wandb_init.py:init():642] re-initializing run, found existing run on stack: HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9
55
+ 2024-11-26 14:37:10,289 INFO MainThread:1863430 [wandb_run.py:_finish():2164] finishing run ckadirt/fMRI-foundation-model/HCPflat_large_gsrFalse__HCP_FT_06a43d89-5346-4bb5-ac55-1b000bfb55d9
56
+ 2024-11-26 14:37:10,289 INFO MainThread:1863430 [wandb_run.py:_atexit_cleanup():2428] got exitcode: 0
57
+ 2024-11-26 14:37:10,290 INFO MainThread:1863430 [wandb_run.py:_restore():2410] restore
58
+ 2024-11-26 14:37:10,290 INFO MainThread:1863430 [wandb_run.py:_restore():2416] restore done
59
+ 2024-11-26 14:37:12,191 INFO MainThread:1863430 [wandb_run.py:_footer_history_summary_info():4049] rendering history
60
+ 2024-11-26 14:37:12,192 INFO MainThread:1863430 [wandb_run.py:_footer_history_summary_info():4081] rendering summary
61
+ 2024-11-26 14:37:12,209 INFO MainThread:1863430 [wandb_run.py:_footer_sync_info():4008] logging synced files