Add files using upload-large-folder tool
Browse filesThis view is limited to 50 files because it contains too many changes.
See raw diff
- .gitattributes +5 -0
- wandb/run-20241030_010305-4ki7693c/files/config.yaml +47 -0
- wandb/run-20241030_010305-4ki7693c/files/output.log +4 -0
- wandb/run-20241030_010305-4ki7693c/files/wandb-metadata.json +29 -0
- wandb/run-20241030_010305-4ki7693c/files/wandb-summary.json +1 -0
- wandb/run-20241030_010305-4ki7693c/logs/debug-internal.log +16 -0
- wandb/run-20241030_010305-4ki7693c/logs/debug.log +27 -0
- wandb/run-20241030_010305-4ki7693c/run-4ki7693c.wandb +0 -0
- wandb/run-20241030_011509-zmlu7388/run-zmlu7388.wandb +3 -0
- wandb/run-20241030_013339-dgadwxty/run-dgadwxty.wandb +3 -0
- wandb/run-20241030_013339-s77qk5li/run-s77qk5li.wandb +3 -0
- wandb/run-20241030_112852-qf0srieq/files/config.yaml +48 -0
- wandb/run-20241030_112852-qf0srieq/files/output.log +17 -0
- wandb/run-20241030_112852-qf0srieq/files/wandb-metadata.json +97 -0
- wandb/run-20241030_112852-qf0srieq/files/wandb-summary.json +1 -0
- wandb/run-20241030_112852-qf0srieq/logs/debug-internal.log +107 -0
- wandb/run-20241030_112852-qf0srieq/logs/debug.log +33 -0
- wandb/run-20241030_112853-ognjedxv/files/config.yaml +48 -0
- wandb/run-20241030_112853-ognjedxv/files/output.log +19 -0
- wandb/run-20241030_112853-ognjedxv/files/wandb-metadata.json +97 -0
- wandb/run-20241030_112853-ognjedxv/files/wandb-summary.json +1 -0
- wandb/run-20241030_112853-ognjedxv/logs/debug-internal.log +108 -0
- wandb/run-20241030_112853-ognjedxv/logs/debug.log +33 -0
- wandb/run-20241030_233740-anh3ext7/files/output.log +16 -0
- wandb/run-20241030_233740-anh3ext7/files/requirements.txt +147 -0
- wandb/run-20241030_233740-anh3ext7/files/wandb-metadata.json +97 -0
- wandb/run-20241030_233740-anh3ext7/logs/debug-internal.log +8 -0
- wandb/run-20241030_233740-anh3ext7/logs/debug.log +26 -0
- wandb/run-20241031_114700-78zg7gu4/files/output.log +48 -0
- wandb/run-20241031_114700-78zg7gu4/files/requirements.txt +147 -0
- wandb/run-20241031_114700-78zg7gu4/files/wandb-metadata.json +97 -0
- wandb/run-20241031_114700-78zg7gu4/logs/debug-internal.log +8 -0
- wandb/run-20241031_114700-78zg7gu4/logs/debug.log +26 -0
- wandb/run-20241101_094656-v2rxhny6/files/output.log +13 -0
- wandb/run-20241101_094656-v2rxhny6/files/requirements.txt +147 -0
- wandb/run-20241101_094656-v2rxhny6/logs/debug-internal.log +8 -0
- wandb/run-20241101_094656-v2rxhny6/logs/debug.log +26 -0
- wandb/run-20241101_200535-6xsf0vem/run-6xsf0vem.wandb +3 -0
- wandb/run-20241101_200535-hnfjoqai/run-hnfjoqai.wandb +3 -0
- wandb/run-20241105_155954-wehwcr47/files/config.yaml +49 -0
- wandb/run-20241105_155954-wehwcr47/files/output.log +19 -0
- wandb/run-20241105_155954-wehwcr47/files/requirements.txt +147 -0
- wandb/run-20241105_155954-wehwcr47/files/wandb-metadata.json +44 -0
- wandb/run-20241105_155954-wehwcr47/files/wandb-summary.json +1 -0
- wandb/run-20241105_155954-wehwcr47/logs/debug-internal.log +17 -0
- wandb/run-20241105_155954-wehwcr47/logs/debug.log +27 -0
- wandb/run-20241105_155954-wehwcr47/run-wehwcr47.wandb +0 -0
- wandb/run-20241105_161113-xd1fe9ua/files/config.yaml +49 -0
- wandb/run-20241105_161113-xd1fe9ua/files/wandb-metadata.json +97 -0
- wandb/run-20241105_161113-xd1fe9ua/run-xd1fe9ua.wandb +0 -0
.gitattributes
CHANGED
|
@@ -110,3 +110,8 @@ wandb/run-20241105_163248-thalxhcd/run-thalxhcd.wandb filter=lfs diff=lfs merge=
|
|
| 110 |
wandb/run-20241031_002020-q6ot1vz6/run-q6ot1vz6.wandb filter=lfs diff=lfs merge=lfs -text
|
| 111 |
wandb/run-20241106_234348-l3eig11b/run-l3eig11b.wandb filter=lfs diff=lfs merge=lfs -text
|
| 112 |
wandb/run-20241031_122114-2k9672ya/run-2k9672ya.wandb filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 110 |
wandb/run-20241031_002020-q6ot1vz6/run-q6ot1vz6.wandb filter=lfs diff=lfs merge=lfs -text
|
| 111 |
wandb/run-20241106_234348-l3eig11b/run-l3eig11b.wandb filter=lfs diff=lfs merge=lfs -text
|
| 112 |
wandb/run-20241031_122114-2k9672ya/run-2k9672ya.wandb filter=lfs diff=lfs merge=lfs -text
|
| 113 |
+
wandb/run-20241030_013339-dgadwxty/run-dgadwxty.wandb filter=lfs diff=lfs merge=lfs -text
|
| 114 |
+
wandb/run-20241101_200535-hnfjoqai/run-hnfjoqai.wandb filter=lfs diff=lfs merge=lfs -text
|
| 115 |
+
wandb/run-20241030_013339-s77qk5li/run-s77qk5li.wandb filter=lfs diff=lfs merge=lfs -text
|
| 116 |
+
wandb/run-20241101_200535-6xsf0vem/run-6xsf0vem.wandb filter=lfs diff=lfs merge=lfs -text
|
| 117 |
+
wandb/run-20241030_011509-zmlu7388/run-zmlu7388.wandb filter=lfs diff=lfs merge=lfs -text
|
wandb/run-20241030_010305-4ki7693c/files/config.yaml
ADDED
|
@@ -0,0 +1,47 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
_wandb:
|
| 2 |
+
value:
|
| 3 |
+
cli_version: 0.18.5
|
| 4 |
+
m: []
|
| 5 |
+
python_version: 3.9.19
|
| 6 |
+
t:
|
| 7 |
+
"1":
|
| 8 |
+
- 1
|
| 9 |
+
- 5
|
| 10 |
+
- 11
|
| 11 |
+
- 49
|
| 12 |
+
- 51
|
| 13 |
+
- 53
|
| 14 |
+
- 55
|
| 15 |
+
- 71
|
| 16 |
+
- 98
|
| 17 |
+
"2":
|
| 18 |
+
- 1
|
| 19 |
+
- 5
|
| 20 |
+
- 11
|
| 21 |
+
- 49
|
| 22 |
+
- 51
|
| 23 |
+
- 53
|
| 24 |
+
- 55
|
| 25 |
+
- 71
|
| 26 |
+
- 98
|
| 27 |
+
"3":
|
| 28 |
+
- 13
|
| 29 |
+
- 23
|
| 30 |
+
- 55
|
| 31 |
+
"4": 3.9.19
|
| 32 |
+
"5": 0.18.5
|
| 33 |
+
"6": 4.45.1
|
| 34 |
+
"8":
|
| 35 |
+
- 5
|
| 36 |
+
"12": 0.18.5
|
| 37 |
+
"13": linux-x86_64
|
| 38 |
+
batch_size:
|
| 39 |
+
value: 3
|
| 40 |
+
epoch:
|
| 41 |
+
value: 7
|
| 42 |
+
perturbation:
|
| 43 |
+
value: reverse_control
|
| 44 |
+
seed:
|
| 45 |
+
value: 0
|
| 46 |
+
train_set:
|
| 47 |
+
value: 10M
|
wandb/run-20241030_010305-4ki7693c/files/output.log
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Traceback (most recent call last):
|
| 2 |
+
File "/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py", line 162, in <module>
|
| 3 |
+
dataset_name = f"babylm_{args.perturbation}_{args.train_zset}_seed{args.seed}"
|
| 4 |
+
AttributeError: 'Namespace' object has no attribute 'train_zset'
|
wandb/run-20241030_010305-4ki7693c/files/wandb-metadata.json
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"os": "Linux-5.4.0-162-generic-x86_64-with-glibc2.31",
|
| 3 |
+
"python": "3.9.19",
|
| 4 |
+
"startedAt": "2024-10-30T05:03:05.639656Z",
|
| 5 |
+
"args": [
|
| 6 |
+
"--perturbation",
|
| 7 |
+
"reverse_control",
|
| 8 |
+
"--train_set",
|
| 9 |
+
"10M",
|
| 10 |
+
"--batch_size",
|
| 11 |
+
"3",
|
| 12 |
+
"--epoch",
|
| 13 |
+
"7",
|
| 14 |
+
"--seed",
|
| 15 |
+
"0"
|
| 16 |
+
],
|
| 17 |
+
"program": "/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py",
|
| 18 |
+
"codePath": "train/train_deep_wandb.py",
|
| 19 |
+
"git": {
|
| 20 |
+
"remote": "git@hf.co:Yaning1001/Impossible_llm.git",
|
| 21 |
+
"commit": "ed716cdcfcdea02b67f7ed0f3504c2b1c8b737c4"
|
| 22 |
+
},
|
| 23 |
+
"email": "yaning1001@gmail.com",
|
| 24 |
+
"root": "/mnt/ssd3/chunhui/yaning/project/impossible_llm/train",
|
| 25 |
+
"host": "mms-large-2",
|
| 26 |
+
"username": "chunhui",
|
| 27 |
+
"executable": "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/bin/python",
|
| 28 |
+
"codePathLocal": "train_deep_wandb.py"
|
| 29 |
+
}
|
wandb/run-20241030_010305-4ki7693c/files/wandb-summary.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"_wandb":{"runtime":0}}
|
wandb/run-20241030_010305-4ki7693c/logs/debug-internal.log
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"time":"2024-10-30T01:03:05.642979784-04:00","level":"INFO","msg":"using version","core version":"0.18.5"}
|
| 2 |
+
{"time":"2024-10-30T01:03:05.643001845-04:00","level":"INFO","msg":"created symlink","path":"/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241030_010305-4ki7693c/logs/debug-core.log"}
|
| 3 |
+
{"time":"2024-10-30T01:03:05.751540951-04:00","level":"INFO","msg":"created new stream","id":"4ki7693c"}
|
| 4 |
+
{"time":"2024-10-30T01:03:05.751579781-04:00","level":"INFO","msg":"stream: started","id":"4ki7693c"}
|
| 5 |
+
{"time":"2024-10-30T01:03:05.751623931-04:00","level":"INFO","msg":"sender: started","stream_id":"4ki7693c"}
|
| 6 |
+
{"time":"2024-10-30T01:03:05.751594441-04:00","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"4ki7693c"}}
|
| 7 |
+
{"time":"2024-10-30T01:03:05.751612271-04:00","level":"INFO","msg":"handler: started","stream_id":{"value":"4ki7693c"}}
|
| 8 |
+
{"time":"2024-10-30T01:03:05.943856433-04:00","level":"INFO","msg":"Starting system monitor"}
|
| 9 |
+
{"time":"2024-10-30T01:03:06.036999384-04:00","level":"INFO","msg":"stream: closing","id":"4ki7693c"}
|
| 10 |
+
{"time":"2024-10-30T01:03:06.037043194-04:00","level":"INFO","msg":"Stopping system monitor"}
|
| 11 |
+
{"time":"2024-10-30T01:03:06.051196751-04:00","level":"INFO","msg":"Stopped system monitor"}
|
| 12 |
+
{"time":"2024-10-30T01:03:07.432977083-04:00","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
|
| 13 |
+
{"time":"2024-10-30T01:03:07.547907803-04:00","level":"INFO","msg":"handler: closed","stream_id":{"value":"4ki7693c"}}
|
| 14 |
+
{"time":"2024-10-30T01:03:07.547982493-04:00","level":"INFO","msg":"sender: closed","stream_id":"4ki7693c"}
|
| 15 |
+
{"time":"2024-10-30T01:03:07.547981023-04:00","level":"INFO","msg":"writer: Close: closed","stream_id":{"value":"4ki7693c"}}
|
| 16 |
+
{"time":"2024-10-30T01:03:07.548091724-04:00","level":"INFO","msg":"stream: closed","id":"4ki7693c"}
|
wandb/run-20241030_010305-4ki7693c/logs/debug.log
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2024-10-30 01:03:05,637 INFO MainThread:320659 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5
|
| 2 |
+
2024-10-30 01:03:05,638 INFO MainThread:320659 [wandb_setup.py:_flush():79] Configure stats pid to 320659
|
| 3 |
+
2024-10-30 01:03:05,638 INFO MainThread:320659 [wandb_setup.py:_flush():79] Loading settings from /home/chunhui/.config/wandb/settings
|
| 4 |
+
2024-10-30 01:03:05,638 INFO MainThread:320659 [wandb_setup.py:_flush():79] Loading settings from /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/settings
|
| 5 |
+
2024-10-30 01:03:05,638 INFO MainThread:320659 [wandb_setup.py:_flush():79] Loading settings from environment variables: {}
|
| 6 |
+
2024-10-30 01:03:05,638 INFO MainThread:320659 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': None, '_disable_service': None}
|
| 7 |
+
2024-10-30 01:03:05,638 INFO MainThread:320659 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'train/train_deep_wandb.py', 'program_abspath': '/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py', 'program': '/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py'}
|
| 8 |
+
2024-10-30 01:03:05,638 INFO MainThread:320659 [wandb_setup.py:_flush():79] Applying login settings: {}
|
| 9 |
+
2024-10-30 01:03:05,638 INFO MainThread:320659 [wandb_init.py:_log_setup():534] Logging user logs to /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241030_010305-4ki7693c/logs/debug.log
|
| 10 |
+
2024-10-30 01:03:05,638 INFO MainThread:320659 [wandb_init.py:_log_setup():535] Logging internal logs to /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241030_010305-4ki7693c/logs/debug-internal.log
|
| 11 |
+
2024-10-30 01:03:05,638 INFO MainThread:320659 [wandb_init.py:init():621] calling init triggers
|
| 12 |
+
2024-10-30 01:03:05,638 INFO MainThread:320659 [wandb_init.py:init():628] wandb.init called with sweep_config: {}
|
| 13 |
+
config: {}
|
| 14 |
+
2024-10-30 01:03:05,638 INFO MainThread:320659 [wandb_init.py:init():671] starting backend
|
| 15 |
+
2024-10-30 01:03:05,638 INFO MainThread:320659 [wandb_init.py:init():675] sending inform_init request
|
| 16 |
+
2024-10-30 01:03:05,639 INFO MainThread:320659 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
|
| 17 |
+
2024-10-30 01:03:05,639 INFO MainThread:320659 [wandb_init.py:init():688] backend started and connected
|
| 18 |
+
2024-10-30 01:03:05,641 INFO MainThread:320659 [wandb_init.py:init():783] updated telemetry
|
| 19 |
+
2024-10-30 01:03:05,664 INFO MainThread:320659 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout
|
| 20 |
+
2024-10-30 01:03:05,940 INFO MainThread:320659 [wandb_init.py:init():867] starting run threads in backend
|
| 21 |
+
2024-10-30 01:03:06,034 INFO MainThread:320659 [wandb_run.py:_console_start():2463] atexit reg
|
| 22 |
+
2024-10-30 01:03:06,034 INFO MainThread:320659 [wandb_run.py:_redirect():2311] redirect: wrap_raw
|
| 23 |
+
2024-10-30 01:03:06,034 INFO MainThread:320659 [wandb_run.py:_redirect():2376] Wrapping output streams.
|
| 24 |
+
2024-10-30 01:03:06,034 INFO MainThread:320659 [wandb_run.py:_redirect():2401] Redirects installed.
|
| 25 |
+
2024-10-30 01:03:06,035 INFO MainThread:320659 [wandb_init.py:init():911] run started, returning control to user process
|
| 26 |
+
2024-10-30 01:03:06,036 INFO MainThread:320659 [wandb_run.py:_config_callback():1390] config_cb None None {'perturbation': 'reverse_control', 'train_set': '10M', 'batch_size': 3, 'epoch': 7, 'seed': 0}
|
| 27 |
+
2024-10-30 01:03:06,037 WARNING MsgRouterThr:320659 [router.py:message_loop():77] message_loop has been closed
|
wandb/run-20241030_010305-4ki7693c/run-4ki7693c.wandb
ADDED
|
Binary file (1.56 kB). View file
|
|
|
wandb/run-20241030_011509-zmlu7388/run-zmlu7388.wandb
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:040fe78d8ec4b781ed5ef6142b4febcf6226f4d06df169e2ee75ef4a7b3149d1
|
| 3 |
+
size 163840
|
wandb/run-20241030_013339-dgadwxty/run-dgadwxty.wandb
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:589f69288e34d7adbf24b1cbd3ee3d6189d8cd39e1c0aa92d97593459f8c1f11
|
| 3 |
+
size 20119552
|
wandb/run-20241030_013339-s77qk5li/run-s77qk5li.wandb
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6c67abff0d802c9d8d917c03111d5b98030ac30597afe446dadeb16459c0339b
|
| 3 |
+
size 20119552
|
wandb/run-20241030_112852-qf0srieq/files/config.yaml
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
_wandb:
|
| 2 |
+
value:
|
| 3 |
+
cli_version: 0.18.5
|
| 4 |
+
m: []
|
| 5 |
+
python_version: 3.9.19
|
| 6 |
+
t:
|
| 7 |
+
"1":
|
| 8 |
+
- 1
|
| 9 |
+
- 5
|
| 10 |
+
- 11
|
| 11 |
+
- 49
|
| 12 |
+
- 51
|
| 13 |
+
- 53
|
| 14 |
+
- 55
|
| 15 |
+
- 71
|
| 16 |
+
- 98
|
| 17 |
+
"2":
|
| 18 |
+
- 1
|
| 19 |
+
- 5
|
| 20 |
+
- 11
|
| 21 |
+
- 49
|
| 22 |
+
- 51
|
| 23 |
+
- 53
|
| 24 |
+
- 55
|
| 25 |
+
- 71
|
| 26 |
+
- 98
|
| 27 |
+
"3":
|
| 28 |
+
- 2
|
| 29 |
+
- 13
|
| 30 |
+
- 23
|
| 31 |
+
- 55
|
| 32 |
+
"4": 3.9.19
|
| 33 |
+
"5": 0.18.5
|
| 34 |
+
"6": 4.45.1
|
| 35 |
+
"8":
|
| 36 |
+
- 5
|
| 37 |
+
"12": 0.18.5
|
| 38 |
+
"13": linux-x86_64
|
| 39 |
+
batch_size:
|
| 40 |
+
value: 3
|
| 41 |
+
epoch:
|
| 42 |
+
value: 3
|
| 43 |
+
perturbation:
|
| 44 |
+
value: reverse_control
|
| 45 |
+
seed:
|
| 46 |
+
value: 0
|
| 47 |
+
train_set:
|
| 48 |
+
value: 10M
|
wandb/run-20241030_112852-qf0srieq/files/output.log
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
model-00001-of-00002.safetensors: 100%|███████████████████████████████████████████████████████████████████████████████████████████████████████████| 4.97G/4.97G [01:33<00:00, 41.3MB/s]
|
| 2 |
+
Downloading shards: 100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 2/2 [02:08<00:00, 64.25s/it]
|
| 3 |
+
Loading checkpoint shards: 100%|█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 2/2 [00:04<00:00, 2.30s/it]
|
| 4 |
+
Map: 100%|███████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 18140/18140 [00:48<00:00, 372.82 examples/s]
|
| 5 |
+
tokenized_valid: Dataset({
|
| 6 |
+
features: ['input_ids', 'attention_mask'],
|
| 7 |
+
num_rows: 600
|
| 8 |
+
})
|
| 9 |
+
/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/transformers/training_args.py:1545: FutureWarning: `evaluation_strategy` is deprecated and will be removed in version 4.46 of 🤗 Transformers. Use `eval_strategy` instead
|
| 10 |
+
warnings.warn(
|
| 11 |
+
[2024-10-30 11:31:56,623] [INFO] [real_accelerator.py:219:get_accelerator] Setting ds_accelerator to cuda (auto detect)
|
| 12 |
+
[2024-10-30 11:32:04,855] [INFO] [comm.py:652:init_distributed] cdb=None
|
| 13 |
+
Installed CUDA version 11.8 does not match the version torch was compiled with 11.7 but since the APIs are compatible, accepting this combination
|
| 14 |
+
Using /home/chunhui/.cache/torch_extensions/py39_cu117 as PyTorch extensions root...
|
| 15 |
+
Loading extension module cpu_adam...
|
| 16 |
+
Time to load cpu_adam op: 4.642748594284058 seconds
|
| 17 |
+
[34m[1mwandb[0m: [33mWARNING[0m Fatal error while uploading data. Some run data will not be synced, but it will still be written to disk. Use `wandb sync` at the end of the run to try uploading.
|
wandb/run-20241030_112852-qf0srieq/files/wandb-metadata.json
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"os": "Linux-5.4.0-162-generic-x86_64-with-glibc2.31",
|
| 3 |
+
"python": "3.9.19",
|
| 4 |
+
"startedAt": "2024-10-30T15:28:52.808778Z",
|
| 5 |
+
"args": [
|
| 6 |
+
"--perturbation",
|
| 7 |
+
"reverse_control",
|
| 8 |
+
"--train_set",
|
| 9 |
+
"10M",
|
| 10 |
+
"--batch_size",
|
| 11 |
+
"3",
|
| 12 |
+
"--epoch",
|
| 13 |
+
"3",
|
| 14 |
+
"--seed",
|
| 15 |
+
"0"
|
| 16 |
+
],
|
| 17 |
+
"program": "/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py",
|
| 18 |
+
"codePath": "train/train_deep_wandb.py",
|
| 19 |
+
"git": {
|
| 20 |
+
"remote": "git@hf.co:Yaning1001/Impossible_llm.git",
|
| 21 |
+
"commit": "ed716cdcfcdea02b67f7ed0f3504c2b1c8b737c4"
|
| 22 |
+
},
|
| 23 |
+
"email": "yaning1001@gmail.com",
|
| 24 |
+
"root": "/mnt/ssd3/chunhui/yaning/project/impossible_llm/train",
|
| 25 |
+
"host": "mms-large-2",
|
| 26 |
+
"username": "chunhui",
|
| 27 |
+
"executable": "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/bin/python",
|
| 28 |
+
"codePathLocal": "train_deep_wandb.py",
|
| 29 |
+
"cpu_count": 32,
|
| 30 |
+
"cpu_count_logical": 64,
|
| 31 |
+
"gpu": "NVIDIA RTX A6000",
|
| 32 |
+
"gpu_count": 8,
|
| 33 |
+
"disk": {
|
| 34 |
+
"/": {
|
| 35 |
+
"total": "1888559353856",
|
| 36 |
+
"used": "1710831611904"
|
| 37 |
+
}
|
| 38 |
+
},
|
| 39 |
+
"memory": {
|
| 40 |
+
"total": "202617098240"
|
| 41 |
+
},
|
| 42 |
+
"cpu": {
|
| 43 |
+
"count": 32,
|
| 44 |
+
"countLogical": 64
|
| 45 |
+
},
|
| 46 |
+
"gpu_nvidia": [
|
| 47 |
+
{
|
| 48 |
+
"name": "NVIDIA RTX A6000",
|
| 49 |
+
"memoryTotal": "51527024640",
|
| 50 |
+
"cudaCores": 10752,
|
| 51 |
+
"architecture": "Ampere"
|
| 52 |
+
},
|
| 53 |
+
{
|
| 54 |
+
"name": "NVIDIA RTX A6000",
|
| 55 |
+
"memoryTotal": "51527024640",
|
| 56 |
+
"cudaCores": 10752,
|
| 57 |
+
"architecture": "Ampere"
|
| 58 |
+
},
|
| 59 |
+
{
|
| 60 |
+
"name": "NVIDIA RTX A6000",
|
| 61 |
+
"memoryTotal": "51527024640",
|
| 62 |
+
"cudaCores": 10752,
|
| 63 |
+
"architecture": "Ampere"
|
| 64 |
+
},
|
| 65 |
+
{
|
| 66 |
+
"name": "NVIDIA RTX A6000",
|
| 67 |
+
"memoryTotal": "51527024640",
|
| 68 |
+
"cudaCores": 10752,
|
| 69 |
+
"architecture": "Ampere"
|
| 70 |
+
},
|
| 71 |
+
{
|
| 72 |
+
"name": "NVIDIA RTX A6000",
|
| 73 |
+
"memoryTotal": "51527024640",
|
| 74 |
+
"cudaCores": 10752,
|
| 75 |
+
"architecture": "Ampere"
|
| 76 |
+
},
|
| 77 |
+
{
|
| 78 |
+
"name": "NVIDIA RTX A6000",
|
| 79 |
+
"memoryTotal": "51527024640",
|
| 80 |
+
"cudaCores": 10752,
|
| 81 |
+
"architecture": "Ampere"
|
| 82 |
+
},
|
| 83 |
+
{
|
| 84 |
+
"name": "NVIDIA RTX A6000",
|
| 85 |
+
"memoryTotal": "51527024640",
|
| 86 |
+
"cudaCores": 10752,
|
| 87 |
+
"architecture": "Ampere"
|
| 88 |
+
},
|
| 89 |
+
{
|
| 90 |
+
"name": "NVIDIA RTX A6000",
|
| 91 |
+
"memoryTotal": "51527024640",
|
| 92 |
+
"cudaCores": 10752,
|
| 93 |
+
"architecture": "Ampere"
|
| 94 |
+
}
|
| 95 |
+
],
|
| 96 |
+
"cudaVersion": "11.8"
|
| 97 |
+
}
|
wandb/run-20241030_112852-qf0srieq/files/wandb-summary.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"_wandb":{"runtime":23503}}
|
wandb/run-20241030_112852-qf0srieq/logs/debug-internal.log
ADDED
|
@@ -0,0 +1,107 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"time":"2024-10-30T11:28:52.811130791-04:00","level":"INFO","msg":"using version","core version":"0.18.5"}
|
| 2 |
+
{"time":"2024-10-30T11:28:52.811141081-04:00","level":"INFO","msg":"created symlink","path":"/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241030_112852-qf0srieq/logs/debug-core.log"}
|
| 3 |
+
{"time":"2024-10-30T11:28:52.917206018-04:00","level":"INFO","msg":"created new stream","id":"qf0srieq"}
|
| 4 |
+
{"time":"2024-10-30T11:28:52.917234588-04:00","level":"INFO","msg":"stream: started","id":"qf0srieq"}
|
| 5 |
+
{"time":"2024-10-30T11:28:52.917337019-04:00","level":"INFO","msg":"sender: started","stream_id":"qf0srieq"}
|
| 6 |
+
{"time":"2024-10-30T11:28:52.917290668-04:00","level":"INFO","msg":"handler: started","stream_id":{"value":"qf0srieq"}}
|
| 7 |
+
{"time":"2024-10-30T11:28:52.917283768-04:00","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"qf0srieq"}}
|
| 8 |
+
{"time":"2024-10-30T11:28:53.127590893-04:00","level":"INFO","msg":"Starting system monitor"}
|
| 9 |
+
{"time":"2024-10-30T14:03:23.466556608-04:00","level":"ERROR","msg":"HTTP error","status":404,"method":"POST","url":"https://api.wandb.ai/files/yaning1001-dartmouth-college/impossible_llm_reverse/qf0srieq/file_stream"}
|
| 10 |
+
{"time":"2024-10-30T14:03:23.470904408-04:00","level":"ERROR+4","msg":"filestream: fatal error: filestream: failed to upload: 404 Not Found path=files/yaning1001-dartmouth-college/impossible_llm_reverse/qf0srieq/file_stream: {\"error\":\"run impossible_llm_reverse/qf0srieq not found while streaming file\"}"}
|
| 11 |
+
{"time":"2024-10-30T18:00:35.968510571-04:00","level":"INFO","msg":"Stopping system monitor"}
|
| 12 |
+
{"time":"2024-10-30T18:00:35.984835172-04:00","level":"INFO","msg":"Stopped system monitor"}
|
| 13 |
+
{"time":"2024-10-30T18:00:36.01388325-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 14 |
+
{"time":"2024-10-30T18:00:36.968916927-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":1.010176656,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 15 |
+
{"time":"2024-10-30T18:00:38.18070286-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 16 |
+
{"time":"2024-10-30T18:00:43.234030085-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 17 |
+
{"time":"2024-10-30T18:00:51.514450735-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 18 |
+
{"time":"2024-10-30T18:01:08.632732116-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 19 |
+
{"time":"2024-10-30T18:01:36.992239115-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":61.033489373,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 20 |
+
{"time":"2024-10-30T18:01:43.980932764-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 21 |
+
{"time":"2024-10-30T18:02:37.012879837-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":121.054135236,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 22 |
+
{"time":"2024-10-30T18:02:44.036787297-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 23 |
+
{"time":"2024-10-30T18:03:37.032386632-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":181.07363706,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 24 |
+
{"time":"2024-10-30T18:03:44.087114894-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 25 |
+
{"time":"2024-10-30T18:04:37.058210791-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":241.09946743,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 26 |
+
{"time":"2024-10-30T18:04:44.146732173-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 27 |
+
{"time":"2024-10-30T18:05:37.08031714-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":301.121564228,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 28 |
+
{"time":"2024-10-30T18:05:44.202578641-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 29 |
+
{"time":"2024-10-30T18:06:37.103155122-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":361.14441734,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 30 |
+
{"time":"2024-10-30T18:06:44.254811138-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 31 |
+
{"time":"2024-10-30T18:07:37.129370298-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":421.170622677,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 32 |
+
{"time":"2024-10-30T18:07:44.304687106-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 33 |
+
{"time":"2024-10-30T18:08:37.187762317-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":481.229015945,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 34 |
+
{"time":"2024-10-30T18:08:44.353692443-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 35 |
+
{"time":"2024-10-30T18:09:37.205365399-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":541.246616438,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 36 |
+
{"time":"2024-10-30T18:09:44.410062874-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 37 |
+
{"time":"2024-10-30T18:10:35.95923303-04:00","level":"WARN","msg":"sender: taking a long time","seconds":600.000334216,"work":"WorkRecord(*service_go_proto.Record_Telemetry); Control(connection_id:\"127.0.0.1:54966\")"}
|
| 38 |
+
{"time":"2024-10-30T18:10:37.228433012-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":601.269685091,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 39 |
+
{"time":"2024-10-30T18:10:44.460819633-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 40 |
+
{"time":"2024-10-30T18:11:37.252924346-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":661.294174825,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 41 |
+
{"time":"2024-10-30T18:11:44.510745999-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 42 |
+
{"time":"2024-10-30T18:12:37.279579159-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":721.320828287,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 43 |
+
{"time":"2024-10-30T18:12:44.561934485-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 44 |
+
{"time":"2024-10-30T18:13:37.30132052-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":781.342568848,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 45 |
+
{"time":"2024-10-30T18:13:44.612933316-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 46 |
+
{"time":"2024-10-30T18:14:37.326717713-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":841.367977471,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 47 |
+
{"time":"2024-10-30T18:14:44.662885212-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 48 |
+
{"time":"2024-10-30T18:15:37.352013001-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":901.39327482,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 49 |
+
{"time":"2024-10-30T18:15:44.719534816-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 50 |
+
{"time":"2024-10-30T18:16:37.377043784-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":961.418290923,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 51 |
+
{"time":"2024-10-30T18:16:44.771072869-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 52 |
+
{"time":"2024-10-30T18:16:44.771190161-04:00","level":"ERROR","msg":"sender: sendConfig:","error":"api: failed sending: POST https://api.wandb.ai/graphql giving up after 21 attempt(s)"}
|
| 53 |
+
{"time":"2024-10-30T18:16:44.771395633-04:00","level":"INFO","msg":"sender: succeeded after taking longer than expected","seconds":968.812598021,"work":"WorkRecord(*service_go_proto.Record_Telemetry); Control(connection_id:\"127.0.0.1:54966\")"}
|
| 54 |
+
{"time":"2024-10-30T18:16:44.827636949-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 55 |
+
{"time":"2024-10-30T18:16:44.870800532-04:00","level":"ERROR","msg":"HTTP error","status":404,"method":"POST","url":"https://api.wandb.ai/graphql"}
|
| 56 |
+
{"time":"2024-10-30T18:16:44.870865293-04:00","level":"ERROR","msg":"runfiles: CreateRunFiles returned error: returned error 404 Not Found: {\"errors\":[{\"message\":\"run impossible_llm_reverse/qf0srieq not found during createRunFiles\",\"path\":[\"createRunFiles\"]}],\"data\":{\"createRunFiles\":null}}"}
|
| 57 |
+
{"time":"2024-10-30T18:16:47.272026705-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 58 |
+
{"time":"2024-10-30T18:16:52.216994148-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 59 |
+
{"time":"2024-10-30T18:17:01.180898323-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 60 |
+
{"time":"2024-10-30T18:17:17.315501289-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 61 |
+
{"time":"2024-10-30T18:17:37.396983698-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":52.62529182,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 62 |
+
{"time":"2024-10-30T18:17:53.545827597-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 63 |
+
{"time":"2024-10-30T18:18:37.418983081-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":112.647293623,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 64 |
+
{"time":"2024-10-30T18:18:53.600763399-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 65 |
+
{"time":"2024-10-30T18:19:37.437849459-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":172.666171092,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 66 |
+
{"time":"2024-10-30T18:19:53.656455438-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 67 |
+
{"time":"2024-10-30T18:20:37.467570843-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":232.695889516,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 68 |
+
{"time":"2024-10-30T18:20:53.722959862-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 69 |
+
{"time":"2024-10-30T18:21:37.493223385-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":292.721535378,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 70 |
+
{"time":"2024-10-30T18:21:53.77677964-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 71 |
+
{"time":"2024-10-30T18:22:37.516133857-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":352.744444299,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 72 |
+
{"time":"2024-10-30T18:22:53.831863788-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 73 |
+
{"time":"2024-10-30T18:23:37.542829635-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":412.771142798,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 74 |
+
{"time":"2024-10-30T18:23:53.88651383-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 75 |
+
{"time":"2024-10-30T18:24:37.566591857-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":472.7949116,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 76 |
+
{"time":"2024-10-30T18:24:53.943522118-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 77 |
+
{"time":"2024-10-30T18:25:37.588022475-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":532.816332248,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 78 |
+
{"time":"2024-10-30T18:25:54.011175643-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 79 |
+
{"time":"2024-10-30T18:26:37.616856963-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":592.845168906,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 80 |
+
{"time":"2024-10-30T18:26:44.77275561-04:00","level":"WARN","msg":"sender: taking a long time","seconds":600.000793211,"work":"WorkRecord(*service_go_proto.Request_Defer); Control(local:true always_send:true)"}
|
| 81 |
+
{"time":"2024-10-30T18:26:54.082149884-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 82 |
+
{"time":"2024-10-30T18:27:37.638631965-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":652.866944558,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 83 |
+
{"time":"2024-10-30T18:27:54.135251151-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 84 |
+
{"time":"2024-10-30T18:28:37.658947741-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":712.887259204,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 85 |
+
{"time":"2024-10-30T18:28:54.198855109-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 86 |
+
{"time":"2024-10-30T18:29:37.681597392-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":772.909908875,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 87 |
+
{"time":"2024-10-30T18:29:54.26606069-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 88 |
+
{"time":"2024-10-30T18:30:37.704016353-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":832.932328576,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 89 |
+
{"time":"2024-10-30T18:30:54.329584432-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 90 |
+
{"time":"2024-10-30T18:31:37.730566652-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":892.958878965,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 91 |
+
{"time":"2024-10-30T18:31:54.417597707-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 92 |
+
{"time":"2024-10-30T18:32:37.760964501-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":952.989275424,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 93 |
+
{"time":"2024-10-30T18:32:54.47204599-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 94 |
+
{"time":"2024-10-30T18:32:54.472190042-04:00","level":"ERROR","msg":"sender: sendConfig:","error":"api: failed sending: POST https://api.wandb.ai/graphql giving up after 21 attempt(s)"}
|
| 95 |
+
{"time":"2024-10-30T18:32:54.472742607-04:00","level":"INFO","msg":"sender: succeeded after taking longer than expected","seconds":969.700851498,"work":"WorkRecord(*service_go_proto.Request_Defer); Control(local:true always_send:true)"}
|
| 96 |
+
{"time":"2024-10-30T18:32:54.572982549-04:00","level":"ERROR","msg":"HTTP error","status":404,"method":"POST","url":"https://api.wandb.ai/graphql"}
|
| 97 |
+
{"time":"2024-10-30T18:32:54.5730265-04:00","level":"ERROR","msg":"runfiles: CreateRunFiles returned error: returned error 404 Not Found: {\"errors\":[{\"message\":\"run impossible_llm_reverse/qf0srieq not found during createRunFiles\",\"path\":[\"createRunFiles\"]}],\"data\":{\"createRunFiles\":null}}"}
|
| 98 |
+
{"time":"2024-10-30T18:32:54.659812312-04:00","level":"ERROR","msg":"HTTP error","status":404,"method":"POST","url":"https://api.wandb.ai/graphql"}
|
| 99 |
+
{"time":"2024-10-30T18:32:54.660095935-04:00","level":"ERROR","msg":"sender: failed to save job artifact: ArtifactSaver.createManifest: returned error 404 Not Found: {\"errors\":[{\"message\":\"failed to find run impossible_llm_reverse/qf0srieq\",\"path\":[\"createArtifactManifest\"]}],\"data\":{\"createArtifactManifest\":null}}"}
|
| 100 |
+
{"time":"2024-10-30T18:32:54.711516723-04:00","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
|
| 101 |
+
{"time":"2024-10-30T18:32:54.759969425-04:00","level":"ERROR","msg":"HTTP error","status":404,"method":"POST","url":"https://api.wandb.ai/graphql"}
|
| 102 |
+
{"time":"2024-10-30T18:32:54.760011335-04:00","level":"ERROR","msg":"runfiles: CreateRunFiles returned error: returned error 404 Not Found: {\"errors\":[{\"message\":\"run impossible_llm_reverse/qf0srieq not found during createRunFiles\",\"path\":[\"createRunFiles\"]}],\"data\":{\"createRunFiles\":null}}"}
|
| 103 |
+
{"time":"2024-10-30T18:32:55.719242378-04:00","level":"INFO","msg":"stream: closing","id":"qf0srieq"}
|
| 104 |
+
{"time":"2024-10-30T18:32:55.719270409-04:00","level":"INFO","msg":"handler: closed","stream_id":{"value":"qf0srieq"}}
|
| 105 |
+
{"time":"2024-10-30T18:32:55.719310959-04:00","level":"INFO","msg":"sender: closed","stream_id":"qf0srieq"}
|
| 106 |
+
{"time":"2024-10-30T18:32:55.719298099-04:00","level":"INFO","msg":"writer: Close: closed","stream_id":{"value":"qf0srieq"}}
|
| 107 |
+
{"time":"2024-10-30T18:32:55.71947517-04:00","level":"INFO","msg":"stream: closed","id":"qf0srieq"}
|
wandb/run-20241030_112852-qf0srieq/logs/debug.log
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2024-10-30 11:28:52,806 INFO MainThread:367767 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5
|
| 2 |
+
2024-10-30 11:28:52,807 INFO MainThread:367767 [wandb_setup.py:_flush():79] Configure stats pid to 367767
|
| 3 |
+
2024-10-30 11:28:52,807 INFO MainThread:367767 [wandb_setup.py:_flush():79] Loading settings from /home/chunhui/.config/wandb/settings
|
| 4 |
+
2024-10-30 11:28:52,807 INFO MainThread:367767 [wandb_setup.py:_flush():79] Loading settings from /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/settings
|
| 5 |
+
2024-10-30 11:28:52,807 INFO MainThread:367767 [wandb_setup.py:_flush():79] Loading settings from environment variables: {}
|
| 6 |
+
2024-10-30 11:28:52,807 INFO MainThread:367767 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': None, '_disable_service': None}
|
| 7 |
+
2024-10-30 11:28:52,807 INFO MainThread:367767 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'train/train_deep_wandb.py', 'program_abspath': '/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py', 'program': '/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py'}
|
| 8 |
+
2024-10-30 11:28:52,807 INFO MainThread:367767 [wandb_setup.py:_flush():79] Applying login settings: {}
|
| 9 |
+
2024-10-30 11:28:52,807 INFO MainThread:367767 [wandb_init.py:_log_setup():534] Logging user logs to /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241030_112852-qf0srieq/logs/debug.log
|
| 10 |
+
2024-10-30 11:28:52,807 INFO MainThread:367767 [wandb_init.py:_log_setup():535] Logging internal logs to /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241030_112852-qf0srieq/logs/debug-internal.log
|
| 11 |
+
2024-10-30 11:28:52,807 INFO MainThread:367767 [wandb_init.py:init():621] calling init triggers
|
| 12 |
+
2024-10-30 11:28:52,807 INFO MainThread:367767 [wandb_init.py:init():628] wandb.init called with sweep_config: {}
|
| 13 |
+
config: {}
|
| 14 |
+
2024-10-30 11:28:52,807 INFO MainThread:367767 [wandb_init.py:init():671] starting backend
|
| 15 |
+
2024-10-30 11:28:52,807 INFO MainThread:367767 [wandb_init.py:init():675] sending inform_init request
|
| 16 |
+
2024-10-30 11:28:52,808 INFO MainThread:367767 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
|
| 17 |
+
2024-10-30 11:28:52,808 INFO MainThread:367767 [wandb_init.py:init():688] backend started and connected
|
| 18 |
+
2024-10-30 11:28:52,811 INFO MainThread:367767 [wandb_init.py:init():783] updated telemetry
|
| 19 |
+
2024-10-30 11:28:52,828 INFO MainThread:367767 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout
|
| 20 |
+
2024-10-30 11:28:53,124 INFO MainThread:367767 [wandb_init.py:init():867] starting run threads in backend
|
| 21 |
+
2024-10-30 11:28:53,211 INFO MainThread:367767 [wandb_run.py:_console_start():2463] atexit reg
|
| 22 |
+
2024-10-30 11:28:53,211 INFO MainThread:367767 [wandb_run.py:_redirect():2311] redirect: wrap_raw
|
| 23 |
+
2024-10-30 11:28:53,211 INFO MainThread:367767 [wandb_run.py:_redirect():2376] Wrapping output streams.
|
| 24 |
+
2024-10-30 11:28:53,211 INFO MainThread:367767 [wandb_run.py:_redirect():2401] Redirects installed.
|
| 25 |
+
2024-10-30 11:28:53,213 INFO MainThread:367767 [wandb_init.py:init():911] run started, returning control to user process
|
| 26 |
+
2024-10-30 11:28:53,213 INFO MainThread:367767 [wandb_run.py:_config_callback():1390] config_cb None None {'perturbation': 'reverse_control', 'train_set': '10M', 'batch_size': 3, 'epoch': 3, 'seed': 0}
|
| 27 |
+
2024-10-30 18:00:35,944 INFO MainThread:367767 [wandb_run.py:_finish():2158] finishing run yaning1001-dartmouth-college/impossible_llm_reverse/qf0srieq
|
| 28 |
+
2024-10-30 18:00:35,951 INFO MainThread:367767 [wandb_run.py:_atexit_cleanup():2426] got exitcode: 0
|
| 29 |
+
2024-10-30 18:00:35,951 INFO MainThread:367767 [wandb_run.py:_restore():2408] restore
|
| 30 |
+
2024-10-30 18:00:35,968 INFO MainThread:367767 [wandb_run.py:_restore():2414] restore done
|
| 31 |
+
2024-10-30 18:32:55,714 INFO MainThread:367767 [wandb_run.py:_footer_history_summary_info():3975] rendering history
|
| 32 |
+
2024-10-30 18:32:55,714 INFO MainThread:367767 [wandb_run.py:_footer_history_summary_info():4007] rendering summary
|
| 33 |
+
2024-10-30 18:32:55,718 INFO MainThread:367767 [wandb_run.py:_footer_sync_info():3934] logging synced files
|
wandb/run-20241030_112853-ognjedxv/files/config.yaml
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
_wandb:
|
| 2 |
+
value:
|
| 3 |
+
cli_version: 0.18.5
|
| 4 |
+
m: []
|
| 5 |
+
python_version: 3.9.19
|
| 6 |
+
t:
|
| 7 |
+
"1":
|
| 8 |
+
- 1
|
| 9 |
+
- 5
|
| 10 |
+
- 11
|
| 11 |
+
- 49
|
| 12 |
+
- 51
|
| 13 |
+
- 53
|
| 14 |
+
- 55
|
| 15 |
+
- 71
|
| 16 |
+
- 98
|
| 17 |
+
"2":
|
| 18 |
+
- 1
|
| 19 |
+
- 5
|
| 20 |
+
- 11
|
| 21 |
+
- 49
|
| 22 |
+
- 51
|
| 23 |
+
- 53
|
| 24 |
+
- 55
|
| 25 |
+
- 71
|
| 26 |
+
- 98
|
| 27 |
+
"3":
|
| 28 |
+
- 2
|
| 29 |
+
- 13
|
| 30 |
+
- 23
|
| 31 |
+
- 55
|
| 32 |
+
"4": 3.9.19
|
| 33 |
+
"5": 0.18.5
|
| 34 |
+
"6": 4.45.1
|
| 35 |
+
"8":
|
| 36 |
+
- 5
|
| 37 |
+
"12": 0.18.5
|
| 38 |
+
"13": linux-x86_64
|
| 39 |
+
batch_size:
|
| 40 |
+
value: 3
|
| 41 |
+
epoch:
|
| 42 |
+
value: 3
|
| 43 |
+
perturbation:
|
| 44 |
+
value: reverse_control
|
| 45 |
+
seed:
|
| 46 |
+
value: 0
|
| 47 |
+
train_set:
|
| 48 |
+
value: 10M
|
wandb/run-20241030_112853-ognjedxv/files/output.log
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Downloading shards: 100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 2/2 [02:08<00:00, 64.05s/it]
|
| 2 |
+
Loading checkpoint shards: 100%|█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 2/2 [00:04<00:00, 2.47s/it]
|
| 3 |
+
Map: 100%|███████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 18140/18140 [00:48<00:00, 372.44 examples/s]
|
| 4 |
+
tokenized_valid: Dataset({
|
| 5 |
+
features: ['input_ids', 'attention_mask'],
|
| 6 |
+
num_rows: 600
|
| 7 |
+
})
|
| 8 |
+
/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/transformers/training_args.py:1545: FutureWarning: `evaluation_strategy` is deprecated and will be removed in version 4.46 of 🤗 Transformers. Use `eval_strategy` instead
|
| 9 |
+
warnings.warn(
|
| 10 |
+
[2024-10-30 11:31:56,739] [INFO] [real_accelerator.py:219:get_accelerator] Setting ds_accelerator to cuda (auto detect)
|
| 11 |
+
[2024-10-30 11:32:04,178] [INFO] [comm.py:652:init_distributed] cdb=None
|
| 12 |
+
Installed CUDA version 11.8 does not match the version torch was compiled with 11.7 but since the APIs are compatible, accepting this combination
|
| 13 |
+
Using /home/chunhui/.cache/torch_extensions/py39_cu117 as PyTorch extensions root...
|
| 14 |
+
Emitting ninja build file /home/chunhui/.cache/torch_extensions/py39_cu117/cpu_adam/build.ninja...
|
| 15 |
+
Building extension module cpu_adam...
|
| 16 |
+
Allowing ninja to set a default number of workers... (overridable by setting the environment variable MAX_JOBS=N)
|
| 17 |
+
Loading extension module cpu_adam...
|
| 18 |
+
Time to load cpu_adam op: 4.685936212539673 seconds
|
| 19 |
+
[34m[1mwandb[0m: [33mWARNING[0m Fatal error while uploading data. Some run data will not be synced, but it will still be written to disk. Use `wandb sync` at the end of the run to try uploading.
|
wandb/run-20241030_112853-ognjedxv/files/wandb-metadata.json
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"os": "Linux-5.4.0-162-generic-x86_64-with-glibc2.31",
|
| 3 |
+
"python": "3.9.19",
|
| 4 |
+
"startedAt": "2024-10-30T15:28:53.133377Z",
|
| 5 |
+
"args": [
|
| 6 |
+
"--perturbation",
|
| 7 |
+
"reverse_control",
|
| 8 |
+
"--train_set",
|
| 9 |
+
"10M",
|
| 10 |
+
"--batch_size",
|
| 11 |
+
"3",
|
| 12 |
+
"--epoch",
|
| 13 |
+
"3",
|
| 14 |
+
"--seed",
|
| 15 |
+
"0"
|
| 16 |
+
],
|
| 17 |
+
"program": "/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py",
|
| 18 |
+
"codePath": "train/train_deep_wandb.py",
|
| 19 |
+
"git": {
|
| 20 |
+
"remote": "git@hf.co:Yaning1001/Impossible_llm.git",
|
| 21 |
+
"commit": "ed716cdcfcdea02b67f7ed0f3504c2b1c8b737c4"
|
| 22 |
+
},
|
| 23 |
+
"email": "yaning1001@gmail.com",
|
| 24 |
+
"root": "/mnt/ssd3/chunhui/yaning/project/impossible_llm/train",
|
| 25 |
+
"host": "mms-large-2",
|
| 26 |
+
"username": "chunhui",
|
| 27 |
+
"executable": "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/bin/python",
|
| 28 |
+
"codePathLocal": "train_deep_wandb.py",
|
| 29 |
+
"cpu_count": 32,
|
| 30 |
+
"cpu_count_logical": 64,
|
| 31 |
+
"gpu": "NVIDIA RTX A6000",
|
| 32 |
+
"gpu_count": 8,
|
| 33 |
+
"disk": {
|
| 34 |
+
"/": {
|
| 35 |
+
"total": "1888559353856",
|
| 36 |
+
"used": "1710831611904"
|
| 37 |
+
}
|
| 38 |
+
},
|
| 39 |
+
"memory": {
|
| 40 |
+
"total": "202617098240"
|
| 41 |
+
},
|
| 42 |
+
"cpu": {
|
| 43 |
+
"count": 32,
|
| 44 |
+
"countLogical": 64
|
| 45 |
+
},
|
| 46 |
+
"gpu_nvidia": [
|
| 47 |
+
{
|
| 48 |
+
"name": "NVIDIA RTX A6000",
|
| 49 |
+
"memoryTotal": "51527024640",
|
| 50 |
+
"cudaCores": 10752,
|
| 51 |
+
"architecture": "Ampere"
|
| 52 |
+
},
|
| 53 |
+
{
|
| 54 |
+
"name": "NVIDIA RTX A6000",
|
| 55 |
+
"memoryTotal": "51527024640",
|
| 56 |
+
"cudaCores": 10752,
|
| 57 |
+
"architecture": "Ampere"
|
| 58 |
+
},
|
| 59 |
+
{
|
| 60 |
+
"name": "NVIDIA RTX A6000",
|
| 61 |
+
"memoryTotal": "51527024640",
|
| 62 |
+
"cudaCores": 10752,
|
| 63 |
+
"architecture": "Ampere"
|
| 64 |
+
},
|
| 65 |
+
{
|
| 66 |
+
"name": "NVIDIA RTX A6000",
|
| 67 |
+
"memoryTotal": "51527024640",
|
| 68 |
+
"cudaCores": 10752,
|
| 69 |
+
"architecture": "Ampere"
|
| 70 |
+
},
|
| 71 |
+
{
|
| 72 |
+
"name": "NVIDIA RTX A6000",
|
| 73 |
+
"memoryTotal": "51527024640",
|
| 74 |
+
"cudaCores": 10752,
|
| 75 |
+
"architecture": "Ampere"
|
| 76 |
+
},
|
| 77 |
+
{
|
| 78 |
+
"name": "NVIDIA RTX A6000",
|
| 79 |
+
"memoryTotal": "51527024640",
|
| 80 |
+
"cudaCores": 10752,
|
| 81 |
+
"architecture": "Ampere"
|
| 82 |
+
},
|
| 83 |
+
{
|
| 84 |
+
"name": "NVIDIA RTX A6000",
|
| 85 |
+
"memoryTotal": "51527024640",
|
| 86 |
+
"cudaCores": 10752,
|
| 87 |
+
"architecture": "Ampere"
|
| 88 |
+
},
|
| 89 |
+
{
|
| 90 |
+
"name": "NVIDIA RTX A6000",
|
| 91 |
+
"memoryTotal": "51527024640",
|
| 92 |
+
"cudaCores": 10752,
|
| 93 |
+
"architecture": "Ampere"
|
| 94 |
+
}
|
| 95 |
+
],
|
| 96 |
+
"cudaVersion": "11.8"
|
| 97 |
+
}
|
wandb/run-20241030_112853-ognjedxv/files/wandb-summary.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"_wandb":{"runtime":23502}}
|
wandb/run-20241030_112853-ognjedxv/logs/debug-internal.log
ADDED
|
@@ -0,0 +1,108 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"time":"2024-10-30T11:28:53.136675238-04:00","level":"INFO","msg":"using version","core version":"0.18.5"}
|
| 2 |
+
{"time":"2024-10-30T11:28:53.136701068-04:00","level":"INFO","msg":"created symlink","path":"/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241030_112853-ognjedxv/logs/debug-core.log"}
|
| 3 |
+
{"time":"2024-10-30T11:28:53.347633316-04:00","level":"INFO","msg":"created new stream","id":"ognjedxv"}
|
| 4 |
+
{"time":"2024-10-30T11:28:53.347690336-04:00","level":"INFO","msg":"stream: started","id":"ognjedxv"}
|
| 5 |
+
{"time":"2024-10-30T11:28:53.347800017-04:00","level":"INFO","msg":"sender: started","stream_id":"ognjedxv"}
|
| 6 |
+
{"time":"2024-10-30T11:28:53.347704066-04:00","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"ognjedxv"}}
|
| 7 |
+
{"time":"2024-10-30T11:28:53.347733556-04:00","level":"INFO","msg":"handler: started","stream_id":{"value":"ognjedxv"}}
|
| 8 |
+
{"time":"2024-10-30T11:28:53.523190791-04:00","level":"INFO","msg":"Starting system monitor"}
|
| 9 |
+
{"time":"2024-10-30T14:02:53.936256906-04:00","level":"ERROR","msg":"HTTP error","status":404,"method":"POST","url":"https://api.wandb.ai/files/yaning1001-dartmouth-college/impossible_llm_reverse/ognjedxv/file_stream"}
|
| 10 |
+
{"time":"2024-10-30T14:02:54.037928715-04:00","level":"ERROR+4","msg":"filestream: fatal error: filestream: failed to upload: 404 Not Found path=files/yaning1001-dartmouth-college/impossible_llm_reverse/ognjedxv/file_stream: {\"error\":\"run impossible_llm_reverse/ognjedxv not found while streaming file\"}"}
|
| 11 |
+
{"time":"2024-10-30T15:32:59.30659372-04:00","level":"INFO","msg":"api: retrying error","error":"Post \"https://api.wandb.ai/graphql\": context deadline exceeded"}
|
| 12 |
+
{"time":"2024-10-30T18:00:35.968921574-04:00","level":"INFO","msg":"Stopping system monitor"}
|
| 13 |
+
{"time":"2024-10-30T18:00:35.984890772-04:00","level":"INFO","msg":"Stopped system monitor"}
|
| 14 |
+
{"time":"2024-10-30T18:00:36.012002318-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 15 |
+
{"time":"2024-10-30T18:00:36.969753173-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":1.014362711,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 16 |
+
{"time":"2024-10-30T18:00:38.325921691-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 17 |
+
{"time":"2024-10-30T18:00:42.497060545-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 18 |
+
{"time":"2024-10-30T18:00:51.61297613-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 19 |
+
{"time":"2024-10-30T18:01:09.600899944-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 20 |
+
{"time":"2024-10-30T18:01:36.999896141-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":61.04451063,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 21 |
+
{"time":"2024-10-30T18:01:45.006064613-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 22 |
+
{"time":"2024-10-30T18:02:37.029240457-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":121.073848476,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 23 |
+
{"time":"2024-10-30T18:02:45.056317872-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 24 |
+
{"time":"2024-10-30T18:03:37.053072349-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":181.097685018,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 25 |
+
{"time":"2024-10-30T18:03:45.107280763-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 26 |
+
{"time":"2024-10-30T18:04:37.077388538-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":241.121996887,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 27 |
+
{"time":"2024-10-30T18:04:45.166717481-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 28 |
+
{"time":"2024-10-30T18:05:37.099223886-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":301.143834105,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 29 |
+
{"time":"2024-10-30T18:05:45.21766182-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 30 |
+
{"time":"2024-10-30T18:06:37.12482984-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":361.169437169,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 31 |
+
{"time":"2024-10-30T18:06:45.278269332-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 32 |
+
{"time":"2024-10-30T18:07:37.142898291-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":421.187507639,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 33 |
+
{"time":"2024-10-30T18:07:45.329370113-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 34 |
+
{"time":"2024-10-30T18:08:37.165565448-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":481.210170617,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 35 |
+
{"time":"2024-10-30T18:08:45.38725225-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 36 |
+
{"time":"2024-10-30T18:09:37.189610381-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":541.23421759,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 37 |
+
{"time":"2024-10-30T18:09:45.4390075-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 38 |
+
{"time":"2024-10-30T18:10:35.955447357-04:00","level":"WARN","msg":"sender: taking a long time","seconds":600.000042295,"work":"WorkRecord(*service_go_proto.Record_Telemetry); Control(connection_id:\"127.0.0.1:47734\")"}
|
| 39 |
+
{"time":"2024-10-30T18:10:37.221958924-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":601.266562773,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 40 |
+
{"time":"2024-10-30T18:10:45.492534001-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 41 |
+
{"time":"2024-10-30T18:11:37.252893706-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":661.297499824,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 42 |
+
{"time":"2024-10-30T18:11:45.544448469-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 43 |
+
{"time":"2024-10-30T18:12:37.363875954-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":721.408484843,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 44 |
+
{"time":"2024-10-30T18:12:45.6051509-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 45 |
+
{"time":"2024-10-30T18:13:37.383426633-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":781.428030622,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 46 |
+
{"time":"2024-10-30T18:13:45.658741606-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 47 |
+
{"time":"2024-10-30T18:14:37.40420376-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":841.448812198,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 48 |
+
{"time":"2024-10-30T18:14:45.714819585-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 49 |
+
{"time":"2024-10-30T18:15:37.426200511-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":901.47080191,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 50 |
+
{"time":"2024-10-30T18:15:45.782516393-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 51 |
+
{"time":"2024-10-30T18:16:37.446643168-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":961.491247447,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 52 |
+
{"time":"2024-10-30T18:16:45.83431873-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 53 |
+
{"time":"2024-10-30T18:16:45.834418971-04:00","level":"ERROR","msg":"sender: sendConfig:","error":"api: failed sending: POST https://api.wandb.ai/graphql giving up after 21 attempt(s)"}
|
| 54 |
+
{"time":"2024-10-30T18:16:45.834565243-04:00","level":"INFO","msg":"sender: succeeded after taking longer than expected","seconds":969.879223162,"work":"WorkRecord(*service_go_proto.Record_Telemetry); Control(connection_id:\"127.0.0.1:47734\")"}
|
| 55 |
+
{"time":"2024-10-30T18:16:45.893720374-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 56 |
+
{"time":"2024-10-30T18:16:45.937388384-04:00","level":"ERROR","msg":"HTTP error","status":404,"method":"POST","url":"https://api.wandb.ai/graphql"}
|
| 57 |
+
{"time":"2024-10-30T18:16:45.937441754-04:00","level":"ERROR","msg":"runfiles: CreateRunFiles returned error: returned error 404 Not Found: {\"errors\":[{\"message\":\"run impossible_llm_reverse/ognjedxv not found during createRunFiles\",\"path\":[\"createRunFiles\"]}],\"data\":{\"createRunFiles\":null}}"}
|
| 58 |
+
{"time":"2024-10-30T18:16:48.137783109-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 59 |
+
{"time":"2024-10-30T18:16:52.444026331-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 60 |
+
{"time":"2024-10-30T18:17:01.754112994-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 61 |
+
{"time":"2024-10-30T18:17:19.575906609-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 62 |
+
{"time":"2024-10-30T18:17:37.465079512-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":51.630157695,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 63 |
+
{"time":"2024-10-30T18:17:53.783058228-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 64 |
+
{"time":"2024-10-30T18:18:37.485068435-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":111.650150518,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 65 |
+
{"time":"2024-10-30T18:18:53.833207383-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 66 |
+
{"time":"2024-10-30T18:19:37.503426143-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":171.668504316,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 67 |
+
{"time":"2024-10-30T18:19:53.883403259-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 68 |
+
{"time":"2024-10-30T18:20:37.529190853-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":231.694274566,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 69 |
+
{"time":"2024-10-30T18:20:53.939024555-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 70 |
+
{"time":"2024-10-30T18:21:37.552685095-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":291.717760967,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 71 |
+
{"time":"2024-10-30T18:21:53.995972735-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 72 |
+
{"time":"2024-10-30T18:22:37.573860393-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":351.738938456,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 73 |
+
{"time":"2024-10-30T18:22:54.048193661-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 74 |
+
{"time":"2024-10-30T18:23:37.59771863-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":411.762794442,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 75 |
+
{"time":"2024-10-30T18:23:54.10633799-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 76 |
+
{"time":"2024-10-30T18:24:37.618320459-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":471.783396942,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 77 |
+
{"time":"2024-10-30T18:24:54.157197254-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 78 |
+
{"time":"2024-10-30T18:25:37.637366385-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":531.802443028,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 79 |
+
{"time":"2024-10-30T18:25:54.214564623-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 80 |
+
{"time":"2024-10-30T18:26:37.657934536-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":591.823008329,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 81 |
+
{"time":"2024-10-30T18:26:45.835568653-04:00","level":"WARN","msg":"sender: taking a long time","seconds":600.000301462,"work":"WorkRecord(*service_go_proto.Request_Defer); Control(local:true always_send:true)"}
|
| 82 |
+
{"time":"2024-10-30T18:26:54.269308808-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 83 |
+
{"time":"2024-10-30T18:27:37.67735044-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":651.842428013,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 84 |
+
{"time":"2024-10-30T18:27:54.325073744-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 85 |
+
{"time":"2024-10-30T18:28:37.695600522-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":711.860675855,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 86 |
+
{"time":"2024-10-30T18:28:54.377526276-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 87 |
+
{"time":"2024-10-30T18:29:37.71622571-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":771.881304503,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 88 |
+
{"time":"2024-10-30T18:29:54.433826357-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 89 |
+
{"time":"2024-10-30T18:30:37.735499119-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":831.900579192,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 90 |
+
{"time":"2024-10-30T18:30:54.489546782-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 91 |
+
{"time":"2024-10-30T18:31:37.758295272-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":891.923373515,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 92 |
+
{"time":"2024-10-30T18:31:54.55033287-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 93 |
+
{"time":"2024-10-30T18:32:37.784133118-04:00","level":"INFO","msg":"handler: operation stats","stats":{"operations":[{"desc":"updating run config","runtime_seconds":951.949222191,"error_status":"retrying HTTP 409 Conflict"}],"total_operations":1}}
|
| 94 |
+
{"time":"2024-10-30T18:32:54.601982687-04:00","level":"INFO","msg":"api: retrying HTTP error","status":409,"url":"https://api.wandb.ai/graphql"}
|
| 95 |
+
{"time":"2024-10-30T18:32:54.602027358-04:00","level":"ERROR","msg":"sender: sendConfig:","error":"api: failed sending: POST https://api.wandb.ai/graphql giving up after 21 attempt(s)"}
|
| 96 |
+
{"time":"2024-10-30T18:32:54.60225771-04:00","level":"INFO","msg":"sender: succeeded after taking longer than expected","seconds":968.76710798,"work":"WorkRecord(*service_go_proto.Request_Defer); Control(local:true always_send:true)"}
|
| 97 |
+
{"time":"2024-10-30T18:32:54.704245558-04:00","level":"ERROR","msg":"HTTP error","status":404,"method":"POST","url":"https://api.wandb.ai/graphql"}
|
| 98 |
+
{"time":"2024-10-30T18:32:54.704274299-04:00","level":"ERROR","msg":"runfiles: CreateRunFiles returned error: returned error 404 Not Found: {\"errors\":[{\"message\":\"run impossible_llm_reverse/ognjedxv not found during createRunFiles\",\"path\":[\"createRunFiles\"]}],\"data\":{\"createRunFiles\":null}}"}
|
| 99 |
+
{"time":"2024-10-30T18:32:54.828523016-04:00","level":"ERROR","msg":"HTTP error","status":404,"method":"POST","url":"https://api.wandb.ai/graphql"}
|
| 100 |
+
{"time":"2024-10-30T18:32:54.828680797-04:00","level":"ERROR","msg":"sender: failed to save job artifact: ArtifactSaver.createManifest: returned error 404 Not Found: {\"errors\":[{\"message\":\"failed to find run impossible_llm_reverse/ognjedxv\",\"path\":[\"createArtifactManifest\"]}],\"data\":{\"createArtifactManifest\":null}}"}
|
| 101 |
+
{"time":"2024-10-30T18:32:54.879693521-04:00","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
|
| 102 |
+
{"time":"2024-10-30T18:32:54.926933592-04:00","level":"ERROR","msg":"HTTP error","status":404,"method":"POST","url":"https://api.wandb.ai/graphql"}
|
| 103 |
+
{"time":"2024-10-30T18:32:54.926962902-04:00","level":"ERROR","msg":"runfiles: CreateRunFiles returned error: returned error 404 Not Found: {\"errors\":[{\"message\":\"run impossible_llm_reverse/ognjedxv not found during createRunFiles\",\"path\":[\"createRunFiles\"]}],\"data\":{\"createRunFiles\":null}}"}
|
| 104 |
+
{"time":"2024-10-30T18:32:55.890003709-04:00","level":"INFO","msg":"stream: closing","id":"ognjedxv"}
|
| 105 |
+
{"time":"2024-10-30T18:32:55.89003529-04:00","level":"INFO","msg":"handler: closed","stream_id":{"value":"ognjedxv"}}
|
| 106 |
+
{"time":"2024-10-30T18:32:55.89005759-04:00","level":"INFO","msg":"writer: Close: closed","stream_id":{"value":"ognjedxv"}}
|
| 107 |
+
{"time":"2024-10-30T18:32:55.89010566-04:00","level":"INFO","msg":"sender: closed","stream_id":"ognjedxv"}
|
| 108 |
+
{"time":"2024-10-30T18:32:55.8901257-04:00","level":"INFO","msg":"stream: closed","id":"ognjedxv"}
|
wandb/run-20241030_112853-ognjedxv/logs/debug.log
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2024-10-30 11:28:53,129 INFO MainThread:367765 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5
|
| 2 |
+
2024-10-30 11:28:53,130 INFO MainThread:367765 [wandb_setup.py:_flush():79] Configure stats pid to 367765
|
| 3 |
+
2024-10-30 11:28:53,130 INFO MainThread:367765 [wandb_setup.py:_flush():79] Loading settings from /home/chunhui/.config/wandb/settings
|
| 4 |
+
2024-10-30 11:28:53,130 INFO MainThread:367765 [wandb_setup.py:_flush():79] Loading settings from /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/settings
|
| 5 |
+
2024-10-30 11:28:53,130 INFO MainThread:367765 [wandb_setup.py:_flush():79] Loading settings from environment variables: {}
|
| 6 |
+
2024-10-30 11:28:53,130 INFO MainThread:367765 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': None, '_disable_service': None}
|
| 7 |
+
2024-10-30 11:28:53,130 INFO MainThread:367765 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'train/train_deep_wandb.py', 'program_abspath': '/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py', 'program': '/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py'}
|
| 8 |
+
2024-10-30 11:28:53,130 INFO MainThread:367765 [wandb_setup.py:_flush():79] Applying login settings: {}
|
| 9 |
+
2024-10-30 11:28:53,130 INFO MainThread:367765 [wandb_init.py:_log_setup():534] Logging user logs to /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241030_112853-ognjedxv/logs/debug.log
|
| 10 |
+
2024-10-30 11:28:53,130 INFO MainThread:367765 [wandb_init.py:_log_setup():535] Logging internal logs to /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241030_112853-ognjedxv/logs/debug-internal.log
|
| 11 |
+
2024-10-30 11:28:53,130 INFO MainThread:367765 [wandb_init.py:init():621] calling init triggers
|
| 12 |
+
2024-10-30 11:28:53,130 INFO MainThread:367765 [wandb_init.py:init():628] wandb.init called with sweep_config: {}
|
| 13 |
+
config: {}
|
| 14 |
+
2024-10-30 11:28:53,130 INFO MainThread:367765 [wandb_init.py:init():671] starting backend
|
| 15 |
+
2024-10-30 11:28:53,130 INFO MainThread:367765 [wandb_init.py:init():675] sending inform_init request
|
| 16 |
+
2024-10-30 11:28:53,132 INFO MainThread:367765 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
|
| 17 |
+
2024-10-30 11:28:53,133 INFO MainThread:367765 [wandb_init.py:init():688] backend started and connected
|
| 18 |
+
2024-10-30 11:28:53,136 INFO MainThread:367765 [wandb_init.py:init():783] updated telemetry
|
| 19 |
+
2024-10-30 11:28:53,181 INFO MainThread:367765 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout
|
| 20 |
+
2024-10-30 11:28:53,520 INFO MainThread:367765 [wandb_init.py:init():867] starting run threads in backend
|
| 21 |
+
2024-10-30 11:28:53,609 INFO MainThread:367765 [wandb_run.py:_console_start():2463] atexit reg
|
| 22 |
+
2024-10-30 11:28:53,609 INFO MainThread:367765 [wandb_run.py:_redirect():2311] redirect: wrap_raw
|
| 23 |
+
2024-10-30 11:28:53,609 INFO MainThread:367765 [wandb_run.py:_redirect():2376] Wrapping output streams.
|
| 24 |
+
2024-10-30 11:28:53,609 INFO MainThread:367765 [wandb_run.py:_redirect():2401] Redirects installed.
|
| 25 |
+
2024-10-30 11:28:53,610 INFO MainThread:367765 [wandb_init.py:init():911] run started, returning control to user process
|
| 26 |
+
2024-10-30 11:28:53,611 INFO MainThread:367765 [wandb_run.py:_config_callback():1390] config_cb None None {'perturbation': 'reverse_control', 'train_set': '10M', 'batch_size': 3, 'epoch': 3, 'seed': 0}
|
| 27 |
+
2024-10-30 18:00:35,945 INFO MainThread:367765 [wandb_run.py:_finish():2158] finishing run yaning1001-dartmouth-college/impossible_llm_reverse/ognjedxv
|
| 28 |
+
2024-10-30 18:00:35,955 INFO MainThread:367765 [wandb_run.py:_atexit_cleanup():2426] got exitcode: 0
|
| 29 |
+
2024-10-30 18:00:35,955 INFO MainThread:367765 [wandb_run.py:_restore():2408] restore
|
| 30 |
+
2024-10-30 18:00:35,968 INFO MainThread:367765 [wandb_run.py:_restore():2414] restore done
|
| 31 |
+
2024-10-30 18:32:55,882 INFO MainThread:367765 [wandb_run.py:_footer_history_summary_info():3975] rendering history
|
| 32 |
+
2024-10-30 18:32:55,883 INFO MainThread:367765 [wandb_run.py:_footer_history_summary_info():4007] rendering summary
|
| 33 |
+
2024-10-30 18:32:55,889 INFO MainThread:367765 [wandb_run.py:_footer_sync_info():3934] logging synced files
|
wandb/run-20241030_233740-anh3ext7/files/output.log
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Loading checkpoint shards: 100%|█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 2/2 [00:06<00:00, 3.01s/it]
|
| 2 |
+
tokenized_valid: Dataset({
|
| 3 |
+
features: ['input_ids', 'attention_mask'],
|
| 4 |
+
num_rows: 600
|
| 5 |
+
})
|
| 6 |
+
/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/transformers/training_args.py:1545: FutureWarning: `evaluation_strategy` is deprecated and will be removed in version 4.46 of 🤗 Transformers. Use `eval_strategy` instead
|
| 7 |
+
warnings.warn(
|
| 8 |
+
[2024-10-30 23:37:50,877] [INFO] [real_accelerator.py:219:get_accelerator] Setting ds_accelerator to cuda (auto detect)
|
| 9 |
+
[2024-10-30 23:37:58,570] [INFO] [comm.py:652:init_distributed] cdb=None
|
| 10 |
+
Installed CUDA version 11.8 does not match the version torch was compiled with 11.7 but since the APIs are compatible, accepting this combination
|
| 11 |
+
Using /home/chunhui/.cache/torch_extensions/py39_cu117 as PyTorch extensions root...
|
| 12 |
+
Emitting ninja build file /home/chunhui/.cache/torch_extensions/py39_cu117/cpu_adam/build.ninja...
|
| 13 |
+
Building extension module cpu_adam...
|
| 14 |
+
Allowing ninja to set a default number of workers... (overridable by setting the environment variable MAX_JOBS=N)
|
| 15 |
+
Loading extension module cpu_adam...
|
| 16 |
+
Time to load cpu_adam op: 4.623389482498169 seconds
|
wandb/run-20241030_233740-anh3ext7/files/requirements.txt
ADDED
|
@@ -0,0 +1,147 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
funcsigs==1.0.2
|
| 2 |
+
sentry-sdk==2.17.0
|
| 3 |
+
multiprocess==0.70.16
|
| 4 |
+
numpy==1.26.2
|
| 5 |
+
pluralizer==1.2.0
|
| 6 |
+
debugpy==1.6.7
|
| 7 |
+
nvidia-cudnn-cu11==8.5.0.96
|
| 8 |
+
deepspeed==0.15.2
|
| 9 |
+
data==0.4
|
| 10 |
+
pandas==2.1.3
|
| 11 |
+
tomli==2.0.1
|
| 12 |
+
charset-normalizer==3.3.2
|
| 13 |
+
attrs==24.2.0
|
| 14 |
+
aiosignal==1.3.1
|
| 15 |
+
fsspec==2023.10.0
|
| 16 |
+
nvidia-cusparse-cu11==11.7.4.91
|
| 17 |
+
zipp==3.12.0
|
| 18 |
+
mypy-extensions==1.0.0
|
| 19 |
+
datasets==3.0.1
|
| 20 |
+
joblib==1.3.2
|
| 21 |
+
hjson==3.1.0
|
| 22 |
+
traitlets==5.7.1
|
| 23 |
+
stack-data==0.6.0
|
| 24 |
+
transformers==4.45.1
|
| 25 |
+
sympy==1.11.1
|
| 26 |
+
Pygments==2.15.0
|
| 27 |
+
docker-pycreds==0.4.0
|
| 28 |
+
dill==0.3.8
|
| 29 |
+
wheel==0.44.0
|
| 30 |
+
prompt-toolkit==3.0.30
|
| 31 |
+
parso==0.8.3
|
| 32 |
+
ipykernel==6.23.1
|
| 33 |
+
pyarrow==17.0.0
|
| 34 |
+
certifi==2023.11.17
|
| 35 |
+
nvidia-cufft-cu11==10.9.0.58
|
| 36 |
+
six==1.16.0
|
| 37 |
+
pydantic==2.9.2
|
| 38 |
+
click==8.1.7
|
| 39 |
+
nest-asyncio==1.5.6
|
| 40 |
+
gmpy2==2.1.0
|
| 41 |
+
matplotlib==3.8.2
|
| 42 |
+
scipy==1.11.4
|
| 43 |
+
typing_extensions==4.12.2
|
| 44 |
+
statsmodels==0.14.0
|
| 45 |
+
huggingface-hub==0.25.0
|
| 46 |
+
frozenlist==1.4.1
|
| 47 |
+
gpustat==1.1.1
|
| 48 |
+
nvidia-nvtx-cu11==11.7.91
|
| 49 |
+
safetensors==0.4.5
|
| 50 |
+
stanza==1.9.2
|
| 51 |
+
decorator==5.1.1
|
| 52 |
+
seaborn==0.13.0
|
| 53 |
+
sentencepiece==0.2.0
|
| 54 |
+
PyYAML==6.0.1
|
| 55 |
+
black==24.8.0
|
| 56 |
+
protobuf==4.25.1
|
| 57 |
+
pickleshare==0.7.5
|
| 58 |
+
peft==0.13.0
|
| 59 |
+
triton==2.0.0
|
| 60 |
+
nvidia-cuda-runtime-cu11==11.7.99
|
| 61 |
+
Jinja2==3.1.2
|
| 62 |
+
nvidia-cusolver-cu11==11.4.0.1
|
| 63 |
+
executing==1.2.0
|
| 64 |
+
jupyter_client==8.1.0
|
| 65 |
+
pluggy==1.3.0
|
| 66 |
+
cmake==3.30.3
|
| 67 |
+
pytz==2023.3.post1
|
| 68 |
+
aiohappyeyeballs==2.4.2
|
| 69 |
+
kiwisolver==1.4.5
|
| 70 |
+
py-cpuinfo==9.0.0
|
| 71 |
+
Pillow==10.1.0
|
| 72 |
+
ptyprocess==0.7.0
|
| 73 |
+
importlib_resources==6.4.5
|
| 74 |
+
GitPython==3.1.43
|
| 75 |
+
importlib-metadata==6.0.0
|
| 76 |
+
iniconfig==2.0.0
|
| 77 |
+
scikit-learn==1.3.2
|
| 78 |
+
exceptiongroup==1.1.0
|
| 79 |
+
networkx==2.8.6
|
| 80 |
+
accelerate==1.0.0
|
| 81 |
+
nltk==3.8.1
|
| 82 |
+
shutilwhich==1.1.0
|
| 83 |
+
fonttools==4.45.1
|
| 84 |
+
future==0.18.3
|
| 85 |
+
aiohttp==3.10.6
|
| 86 |
+
wcwidth==0.2.5
|
| 87 |
+
idna==3.6
|
| 88 |
+
filelock==3.12.2
|
| 89 |
+
pathspec==0.12.1
|
| 90 |
+
jupyter_core==5.1.0
|
| 91 |
+
lit==18.1.8
|
| 92 |
+
nvidia-curand-cu11==10.2.10.91
|
| 93 |
+
nvidia-cublas-cu11==11.10.3.66
|
| 94 |
+
nvidia-ml-py==12.560.30
|
| 95 |
+
msgpack==1.1.0
|
| 96 |
+
python-dateutil==2.8.2
|
| 97 |
+
blessed==1.20.0
|
| 98 |
+
packaging==23.0
|
| 99 |
+
gitdb==4.0.11
|
| 100 |
+
yarl==1.13.0
|
| 101 |
+
emoji==2.8.0
|
| 102 |
+
tzdata==2023.3
|
| 103 |
+
cycler==0.12.1
|
| 104 |
+
tornado==6.2
|
| 105 |
+
backcall==0.2.0
|
| 106 |
+
plotnine==0.12.4
|
| 107 |
+
ninja==1.11.1.1
|
| 108 |
+
latex==0.7.0
|
| 109 |
+
wandb==0.18.5
|
| 110 |
+
setproctitle==1.3.3
|
| 111 |
+
threadpoolctl==3.2.0
|
| 112 |
+
requests==2.32.3
|
| 113 |
+
pyparsing==3.1.1
|
| 114 |
+
smmap==5.0.1
|
| 115 |
+
pyzmq==23.0.0
|
| 116 |
+
async-timeout==4.0.3
|
| 117 |
+
annotated-types==0.7.0
|
| 118 |
+
matplotlib-inline==0.1.6
|
| 119 |
+
latexcodec==1.0.0
|
| 120 |
+
ipython==8.0.0
|
| 121 |
+
patsy==0.5.3
|
| 122 |
+
contourpy==1.2.0
|
| 123 |
+
multidict==6.1.0
|
| 124 |
+
mizani==0.9.3
|
| 125 |
+
urllib3==2.1.0
|
| 126 |
+
tokenizers==0.20.0
|
| 127 |
+
MarkupSafe==2.1.2
|
| 128 |
+
pip==24.2
|
| 129 |
+
pexpect==4.8.0
|
| 130 |
+
tqdm==4.66.5
|
| 131 |
+
jedi==0.18.2
|
| 132 |
+
pydantic_core==2.23.4
|
| 133 |
+
tempdir==0.7.1
|
| 134 |
+
mpmath==1.2.1
|
| 135 |
+
setuptools==72.1.0
|
| 136 |
+
pytest==7.4.3
|
| 137 |
+
pure-eval==0.2.2
|
| 138 |
+
psutil==5.9.1
|
| 139 |
+
comm==0.1.2
|
| 140 |
+
nvidia-cuda-cupti-cu11==11.7.101
|
| 141 |
+
nvidia-cuda-nvrtc-cu11==11.7.99
|
| 142 |
+
regex==2023.10.3
|
| 143 |
+
platformdirs==2.5.2
|
| 144 |
+
asttokens==2.2.1
|
| 145 |
+
torch==2.0.0
|
| 146 |
+
nvidia-nccl-cu11==2.14.3
|
| 147 |
+
xxhash==3.5.0
|
wandb/run-20241030_233740-anh3ext7/files/wandb-metadata.json
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"os": "Linux-5.4.0-162-generic-x86_64-with-glibc2.31",
|
| 3 |
+
"python": "3.9.19",
|
| 4 |
+
"startedAt": "2024-10-31T03:37:40.846508Z",
|
| 5 |
+
"args": [
|
| 6 |
+
"--perturbation",
|
| 7 |
+
"reverse_control",
|
| 8 |
+
"--train_set",
|
| 9 |
+
"10M",
|
| 10 |
+
"--batch_size",
|
| 11 |
+
"3",
|
| 12 |
+
"--epoch",
|
| 13 |
+
"3",
|
| 14 |
+
"--seed",
|
| 15 |
+
"0"
|
| 16 |
+
],
|
| 17 |
+
"program": "/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py",
|
| 18 |
+
"codePath": "train/train_deep_wandb.py",
|
| 19 |
+
"git": {
|
| 20 |
+
"remote": "git@hf.co:Yaning1001/Impossible_llm.git",
|
| 21 |
+
"commit": "ed716cdcfcdea02b67f7ed0f3504c2b1c8b737c4"
|
| 22 |
+
},
|
| 23 |
+
"email": "yaning1001@gmail.com",
|
| 24 |
+
"root": "/mnt/ssd3/chunhui/yaning/project/impossible_llm/train",
|
| 25 |
+
"host": "mms-large-2",
|
| 26 |
+
"username": "chunhui",
|
| 27 |
+
"executable": "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/bin/python",
|
| 28 |
+
"codePathLocal": "train_deep_wandb.py",
|
| 29 |
+
"cpu_count": 32,
|
| 30 |
+
"cpu_count_logical": 64,
|
| 31 |
+
"gpu": "NVIDIA RTX A6000",
|
| 32 |
+
"gpu_count": 8,
|
| 33 |
+
"disk": {
|
| 34 |
+
"/": {
|
| 35 |
+
"total": "1888559353856",
|
| 36 |
+
"used": "1711065919488"
|
| 37 |
+
}
|
| 38 |
+
},
|
| 39 |
+
"memory": {
|
| 40 |
+
"total": "202617098240"
|
| 41 |
+
},
|
| 42 |
+
"cpu": {
|
| 43 |
+
"count": 32,
|
| 44 |
+
"countLogical": 64
|
| 45 |
+
},
|
| 46 |
+
"gpu_nvidia": [
|
| 47 |
+
{
|
| 48 |
+
"name": "NVIDIA RTX A6000",
|
| 49 |
+
"memoryTotal": "51527024640",
|
| 50 |
+
"cudaCores": 10752,
|
| 51 |
+
"architecture": "Ampere"
|
| 52 |
+
},
|
| 53 |
+
{
|
| 54 |
+
"name": "NVIDIA RTX A6000",
|
| 55 |
+
"memoryTotal": "51527024640",
|
| 56 |
+
"cudaCores": 10752,
|
| 57 |
+
"architecture": "Ampere"
|
| 58 |
+
},
|
| 59 |
+
{
|
| 60 |
+
"name": "NVIDIA RTX A6000",
|
| 61 |
+
"memoryTotal": "51527024640",
|
| 62 |
+
"cudaCores": 10752,
|
| 63 |
+
"architecture": "Ampere"
|
| 64 |
+
},
|
| 65 |
+
{
|
| 66 |
+
"name": "NVIDIA RTX A6000",
|
| 67 |
+
"memoryTotal": "51527024640",
|
| 68 |
+
"cudaCores": 10752,
|
| 69 |
+
"architecture": "Ampere"
|
| 70 |
+
},
|
| 71 |
+
{
|
| 72 |
+
"name": "NVIDIA RTX A6000",
|
| 73 |
+
"memoryTotal": "51527024640",
|
| 74 |
+
"cudaCores": 10752,
|
| 75 |
+
"architecture": "Ampere"
|
| 76 |
+
},
|
| 77 |
+
{
|
| 78 |
+
"name": "NVIDIA RTX A6000",
|
| 79 |
+
"memoryTotal": "51527024640",
|
| 80 |
+
"cudaCores": 10752,
|
| 81 |
+
"architecture": "Ampere"
|
| 82 |
+
},
|
| 83 |
+
{
|
| 84 |
+
"name": "NVIDIA RTX A6000",
|
| 85 |
+
"memoryTotal": "51527024640",
|
| 86 |
+
"cudaCores": 10752,
|
| 87 |
+
"architecture": "Ampere"
|
| 88 |
+
},
|
| 89 |
+
{
|
| 90 |
+
"name": "NVIDIA RTX A6000",
|
| 91 |
+
"memoryTotal": "51527024640",
|
| 92 |
+
"cudaCores": 10752,
|
| 93 |
+
"architecture": "Ampere"
|
| 94 |
+
}
|
| 95 |
+
],
|
| 96 |
+
"cudaVersion": "11.8"
|
| 97 |
+
}
|
wandb/run-20241030_233740-anh3ext7/logs/debug-internal.log
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"time":"2024-10-30T23:37:40.850081823-04:00","level":"INFO","msg":"using version","core version":"0.18.5"}
|
| 2 |
+
{"time":"2024-10-30T23:37:40.850115643-04:00","level":"INFO","msg":"created symlink","path":"/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241030_233740-anh3ext7/logs/debug-core.log"}
|
| 3 |
+
{"time":"2024-10-30T23:37:40.962577204-04:00","level":"INFO","msg":"created new stream","id":"anh3ext7"}
|
| 4 |
+
{"time":"2024-10-30T23:37:40.962609945-04:00","level":"INFO","msg":"stream: started","id":"anh3ext7"}
|
| 5 |
+
{"time":"2024-10-30T23:37:40.962627565-04:00","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"anh3ext7"}}
|
| 6 |
+
{"time":"2024-10-30T23:37:40.962685265-04:00","level":"INFO","msg":"sender: started","stream_id":"anh3ext7"}
|
| 7 |
+
{"time":"2024-10-30T23:37:40.962685215-04:00","level":"INFO","msg":"handler: started","stream_id":{"value":"anh3ext7"}}
|
| 8 |
+
{"time":"2024-10-30T23:37:41.412681383-04:00","level":"INFO","msg":"Starting system monitor"}
|
wandb/run-20241030_233740-anh3ext7/logs/debug.log
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2024-10-30 23:37:40,844 INFO MainThread:464535 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5
|
| 2 |
+
2024-10-30 23:37:40,844 INFO MainThread:464535 [wandb_setup.py:_flush():79] Configure stats pid to 464535
|
| 3 |
+
2024-10-30 23:37:40,844 INFO MainThread:464535 [wandb_setup.py:_flush():79] Loading settings from /home/chunhui/.config/wandb/settings
|
| 4 |
+
2024-10-30 23:37:40,844 INFO MainThread:464535 [wandb_setup.py:_flush():79] Loading settings from /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/settings
|
| 5 |
+
2024-10-30 23:37:40,844 INFO MainThread:464535 [wandb_setup.py:_flush():79] Loading settings from environment variables: {}
|
| 6 |
+
2024-10-30 23:37:40,844 INFO MainThread:464535 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': None, '_disable_service': None}
|
| 7 |
+
2024-10-30 23:37:40,844 INFO MainThread:464535 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'train/train_deep_wandb.py', 'program_abspath': '/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py', 'program': '/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py'}
|
| 8 |
+
2024-10-30 23:37:40,844 INFO MainThread:464535 [wandb_setup.py:_flush():79] Applying login settings: {}
|
| 9 |
+
2024-10-30 23:37:40,845 INFO MainThread:464535 [wandb_init.py:_log_setup():534] Logging user logs to /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241030_233740-anh3ext7/logs/debug.log
|
| 10 |
+
2024-10-30 23:37:40,845 INFO MainThread:464535 [wandb_init.py:_log_setup():535] Logging internal logs to /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241030_233740-anh3ext7/logs/debug-internal.log
|
| 11 |
+
2024-10-30 23:37:40,845 INFO MainThread:464535 [wandb_init.py:init():621] calling init triggers
|
| 12 |
+
2024-10-30 23:37:40,845 INFO MainThread:464535 [wandb_init.py:init():628] wandb.init called with sweep_config: {}
|
| 13 |
+
config: {}
|
| 14 |
+
2024-10-30 23:37:40,845 INFO MainThread:464535 [wandb_init.py:init():671] starting backend
|
| 15 |
+
2024-10-30 23:37:40,845 INFO MainThread:464535 [wandb_init.py:init():675] sending inform_init request
|
| 16 |
+
2024-10-30 23:37:40,846 INFO MainThread:464535 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
|
| 17 |
+
2024-10-30 23:37:40,846 INFO MainThread:464535 [wandb_init.py:init():688] backend started and connected
|
| 18 |
+
2024-10-30 23:37:40,849 INFO MainThread:464535 [wandb_init.py:init():783] updated telemetry
|
| 19 |
+
2024-10-30 23:37:40,879 INFO MainThread:464535 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout
|
| 20 |
+
2024-10-30 23:37:41,409 INFO MainThread:464535 [wandb_init.py:init():867] starting run threads in backend
|
| 21 |
+
2024-10-30 23:37:42,836 INFO MainThread:464535 [wandb_run.py:_console_start():2463] atexit reg
|
| 22 |
+
2024-10-30 23:37:42,837 INFO MainThread:464535 [wandb_run.py:_redirect():2311] redirect: wrap_raw
|
| 23 |
+
2024-10-30 23:37:42,837 INFO MainThread:464535 [wandb_run.py:_redirect():2376] Wrapping output streams.
|
| 24 |
+
2024-10-30 23:37:42,837 INFO MainThread:464535 [wandb_run.py:_redirect():2401] Redirects installed.
|
| 25 |
+
2024-10-30 23:37:42,860 INFO MainThread:464535 [wandb_init.py:init():911] run started, returning control to user process
|
| 26 |
+
2024-10-30 23:37:42,861 INFO MainThread:464535 [wandb_run.py:_config_callback():1390] config_cb None None {'perturbation': 'reverse_control', 'train_set': '10M', 'batch_size': 3, 'epoch': 3, 'seed': 0}
|
wandb/run-20241031_114700-78zg7gu4/files/output.log
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Downloading shards: 100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 2/2 [02:32<00:00, 76.42s/it]
|
| 2 |
+
Loading checkpoint shards: 100%|███████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 2/2 [00:04<00:00, 2.42s/it]
|
| 3 |
+
tokenized_valid: Dataset({
|
| 4 |
+
features: ['input_ids', 'attention_mask'],
|
| 5 |
+
num_rows: 600
|
| 6 |
+
})
|
| 7 |
+
/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/transformers/training_args.py:1545: FutureWarning: `evaluation_strategy` is deprecated and will be removed in version 4.46 of 🤗 Transformers. Use `eval_strategy` instead
|
| 8 |
+
warnings.warn(
|
| 9 |
+
[2024-10-31 11:49:40,267] [INFO] [real_accelerator.py:219:get_accelerator] Setting ds_accelerator to cuda (auto detect)
|
| 10 |
+
[2024-10-31 11:49:48,657] [INFO] [comm.py:652:init_distributed] cdb=None
|
| 11 |
+
Installed CUDA version 11.8 does not match the version torch was compiled with 11.7 but since the APIs are compatible, accepting this combination
|
| 12 |
+
Using /home/chunhui/.cache/torch_extensions/py39_cu117 as PyTorch extensions root...
|
| 13 |
+
Emitting ninja build file /home/chunhui/.cache/torch_extensions/py39_cu117/cpu_adam/build.ninja...
|
| 14 |
+
Building extension module cpu_adam...
|
| 15 |
+
Allowing ninja to set a default number of workers... (overridable by setting the environment variable MAX_JOBS=N)
|
| 16 |
+
Loading extension module cpu_adam...
|
| 17 |
+
Time to load cpu_adam op: 4.85258412361145 seconds
|
| 18 |
+
Traceback (most recent call last):
|
| 19 |
+
File "/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py", line 220, in <module>
|
| 20 |
+
trainer.train()
|
| 21 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/transformers/trainer.py", line 2052, in train
|
| 22 |
+
return inner_training_loop(
|
| 23 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/transformers/trainer.py", line 2388, in _inner_training_loop
|
| 24 |
+
tr_loss_step = self.training_step(model, inputs)
|
| 25 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/transformers/trainer.py", line 3518, in training_step
|
| 26 |
+
self.accelerator.backward(loss, **kwargs)
|
| 27 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/accelerate/accelerator.py", line 2238, in backward
|
| 28 |
+
self.deepspeed_engine_wrapped.backward(loss, **kwargs)
|
| 29 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/accelerate/utils/deepspeed.py", line 186, in backward
|
| 30 |
+
self.engine.backward(loss, **kwargs)
|
| 31 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/deepspeed/utils/nvtx.py", line 18, in wrapped_fn
|
| 32 |
+
ret_val = func(*args, **kwargs)
|
| 33 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/deepspeed/runtime/engine.py", line 2020, in backward
|
| 34 |
+
self.optimizer.backward(loss, retain_graph=retain_graph)
|
| 35 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/deepspeed/runtime/zero/stage_1_and_2.py", line 2063, in backward
|
| 36 |
+
self.loss_scaler.backward(loss.float(), retain_graph=retain_graph)
|
| 37 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/deepspeed/runtime/fp16/loss_scaler.py", line 63, in backward
|
| 38 |
+
scaled_loss.backward(retain_graph=retain_graph)
|
| 39 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/torch/_tensor.py", line 487, in backward
|
| 40 |
+
torch.autograd.backward(
|
| 41 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/torch/autograd/__init__.py", line 200, in backward
|
| 42 |
+
Variable._execution_engine.run_backward( # Calls into the C++ engine to run the backward pass
|
| 43 |
+
KeyboardInterrupt
|
| 44 |
+
Error in atexit._run_exitfuncs:
|
| 45 |
+
Traceback (most recent call last):
|
| 46 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/deepspeed/ops/transformer/inference/triton/matmul_ext.py", line 27, in is_nfs_path
|
| 47 |
+
output = subprocess.check_output(['df', '-T', path], encoding='utf-8')
|
| 48 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/subprocess.py", line 424, in check_output
|
wandb/run-20241031_114700-78zg7gu4/files/requirements.txt
ADDED
|
@@ -0,0 +1,147 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
funcsigs==1.0.2
|
| 2 |
+
sentry-sdk==2.17.0
|
| 3 |
+
multiprocess==0.70.16
|
| 4 |
+
numpy==1.26.2
|
| 5 |
+
pluralizer==1.2.0
|
| 6 |
+
debugpy==1.6.7
|
| 7 |
+
nvidia-cudnn-cu11==8.5.0.96
|
| 8 |
+
deepspeed==0.15.2
|
| 9 |
+
data==0.4
|
| 10 |
+
pandas==2.1.3
|
| 11 |
+
tomli==2.0.1
|
| 12 |
+
charset-normalizer==3.3.2
|
| 13 |
+
attrs==24.2.0
|
| 14 |
+
aiosignal==1.3.1
|
| 15 |
+
fsspec==2023.10.0
|
| 16 |
+
nvidia-cusparse-cu11==11.7.4.91
|
| 17 |
+
zipp==3.12.0
|
| 18 |
+
mypy-extensions==1.0.0
|
| 19 |
+
datasets==3.0.1
|
| 20 |
+
joblib==1.3.2
|
| 21 |
+
hjson==3.1.0
|
| 22 |
+
traitlets==5.7.1
|
| 23 |
+
stack-data==0.6.0
|
| 24 |
+
transformers==4.45.1
|
| 25 |
+
sympy==1.11.1
|
| 26 |
+
Pygments==2.15.0
|
| 27 |
+
docker-pycreds==0.4.0
|
| 28 |
+
dill==0.3.8
|
| 29 |
+
wheel==0.44.0
|
| 30 |
+
prompt-toolkit==3.0.30
|
| 31 |
+
parso==0.8.3
|
| 32 |
+
ipykernel==6.23.1
|
| 33 |
+
pyarrow==17.0.0
|
| 34 |
+
certifi==2023.11.17
|
| 35 |
+
nvidia-cufft-cu11==10.9.0.58
|
| 36 |
+
six==1.16.0
|
| 37 |
+
pydantic==2.9.2
|
| 38 |
+
click==8.1.7
|
| 39 |
+
nest-asyncio==1.5.6
|
| 40 |
+
gmpy2==2.1.0
|
| 41 |
+
matplotlib==3.8.2
|
| 42 |
+
scipy==1.11.4
|
| 43 |
+
typing_extensions==4.12.2
|
| 44 |
+
statsmodels==0.14.0
|
| 45 |
+
huggingface-hub==0.25.0
|
| 46 |
+
frozenlist==1.4.1
|
| 47 |
+
gpustat==1.1.1
|
| 48 |
+
nvidia-nvtx-cu11==11.7.91
|
| 49 |
+
safetensors==0.4.5
|
| 50 |
+
stanza==1.9.2
|
| 51 |
+
decorator==5.1.1
|
| 52 |
+
seaborn==0.13.0
|
| 53 |
+
sentencepiece==0.2.0
|
| 54 |
+
PyYAML==6.0.1
|
| 55 |
+
black==24.8.0
|
| 56 |
+
protobuf==4.25.1
|
| 57 |
+
pickleshare==0.7.5
|
| 58 |
+
peft==0.13.0
|
| 59 |
+
triton==2.0.0
|
| 60 |
+
nvidia-cuda-runtime-cu11==11.7.99
|
| 61 |
+
Jinja2==3.1.2
|
| 62 |
+
nvidia-cusolver-cu11==11.4.0.1
|
| 63 |
+
executing==1.2.0
|
| 64 |
+
jupyter_client==8.1.0
|
| 65 |
+
pluggy==1.3.0
|
| 66 |
+
cmake==3.30.3
|
| 67 |
+
pytz==2023.3.post1
|
| 68 |
+
aiohappyeyeballs==2.4.2
|
| 69 |
+
kiwisolver==1.4.5
|
| 70 |
+
py-cpuinfo==9.0.0
|
| 71 |
+
Pillow==10.1.0
|
| 72 |
+
ptyprocess==0.7.0
|
| 73 |
+
importlib_resources==6.4.5
|
| 74 |
+
GitPython==3.1.43
|
| 75 |
+
importlib-metadata==6.0.0
|
| 76 |
+
iniconfig==2.0.0
|
| 77 |
+
scikit-learn==1.3.2
|
| 78 |
+
exceptiongroup==1.1.0
|
| 79 |
+
networkx==2.8.6
|
| 80 |
+
accelerate==1.0.0
|
| 81 |
+
nltk==3.8.1
|
| 82 |
+
shutilwhich==1.1.0
|
| 83 |
+
fonttools==4.45.1
|
| 84 |
+
future==0.18.3
|
| 85 |
+
aiohttp==3.10.6
|
| 86 |
+
wcwidth==0.2.5
|
| 87 |
+
idna==3.6
|
| 88 |
+
filelock==3.12.2
|
| 89 |
+
pathspec==0.12.1
|
| 90 |
+
jupyter_core==5.1.0
|
| 91 |
+
lit==18.1.8
|
| 92 |
+
nvidia-curand-cu11==10.2.10.91
|
| 93 |
+
nvidia-cublas-cu11==11.10.3.66
|
| 94 |
+
nvidia-ml-py==12.560.30
|
| 95 |
+
msgpack==1.1.0
|
| 96 |
+
python-dateutil==2.8.2
|
| 97 |
+
blessed==1.20.0
|
| 98 |
+
packaging==23.0
|
| 99 |
+
gitdb==4.0.11
|
| 100 |
+
yarl==1.13.0
|
| 101 |
+
emoji==2.8.0
|
| 102 |
+
tzdata==2023.3
|
| 103 |
+
cycler==0.12.1
|
| 104 |
+
tornado==6.2
|
| 105 |
+
backcall==0.2.0
|
| 106 |
+
plotnine==0.12.4
|
| 107 |
+
ninja==1.11.1.1
|
| 108 |
+
latex==0.7.0
|
| 109 |
+
wandb==0.18.5
|
| 110 |
+
setproctitle==1.3.3
|
| 111 |
+
threadpoolctl==3.2.0
|
| 112 |
+
requests==2.32.3
|
| 113 |
+
pyparsing==3.1.1
|
| 114 |
+
smmap==5.0.1
|
| 115 |
+
pyzmq==23.0.0
|
| 116 |
+
async-timeout==4.0.3
|
| 117 |
+
annotated-types==0.7.0
|
| 118 |
+
matplotlib-inline==0.1.6
|
| 119 |
+
latexcodec==1.0.0
|
| 120 |
+
ipython==8.0.0
|
| 121 |
+
patsy==0.5.3
|
| 122 |
+
contourpy==1.2.0
|
| 123 |
+
multidict==6.1.0
|
| 124 |
+
mizani==0.9.3
|
| 125 |
+
urllib3==2.1.0
|
| 126 |
+
tokenizers==0.20.0
|
| 127 |
+
MarkupSafe==2.1.2
|
| 128 |
+
pip==24.2
|
| 129 |
+
pexpect==4.8.0
|
| 130 |
+
tqdm==4.66.5
|
| 131 |
+
jedi==0.18.2
|
| 132 |
+
pydantic_core==2.23.4
|
| 133 |
+
tempdir==0.7.1
|
| 134 |
+
mpmath==1.2.1
|
| 135 |
+
setuptools==72.1.0
|
| 136 |
+
pytest==7.4.3
|
| 137 |
+
pure-eval==0.2.2
|
| 138 |
+
psutil==5.9.1
|
| 139 |
+
comm==0.1.2
|
| 140 |
+
nvidia-cuda-cupti-cu11==11.7.101
|
| 141 |
+
nvidia-cuda-nvrtc-cu11==11.7.99
|
| 142 |
+
regex==2023.10.3
|
| 143 |
+
platformdirs==2.5.2
|
| 144 |
+
asttokens==2.2.1
|
| 145 |
+
torch==2.0.0
|
| 146 |
+
nvidia-nccl-cu11==2.14.3
|
| 147 |
+
xxhash==3.5.0
|
wandb/run-20241031_114700-78zg7gu4/files/wandb-metadata.json
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"os": "Linux-5.4.0-162-generic-x86_64-with-glibc2.31",
|
| 3 |
+
"python": "3.9.19",
|
| 4 |
+
"startedAt": "2024-10-31T15:47:00.200293Z",
|
| 5 |
+
"args": [
|
| 6 |
+
"--perturbation",
|
| 7 |
+
"reverse_full",
|
| 8 |
+
"--train_set",
|
| 9 |
+
"10M",
|
| 10 |
+
"--batch_size",
|
| 11 |
+
"3",
|
| 12 |
+
"--epoch",
|
| 13 |
+
"6",
|
| 14 |
+
"--seed",
|
| 15 |
+
"0"
|
| 16 |
+
],
|
| 17 |
+
"program": "/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py",
|
| 18 |
+
"codePath": "train/train_deep_wandb.py",
|
| 19 |
+
"git": {
|
| 20 |
+
"remote": "git@hf.co:Yaning1001/Impossible_llm.git",
|
| 21 |
+
"commit": "ed716cdcfcdea02b67f7ed0f3504c2b1c8b737c4"
|
| 22 |
+
},
|
| 23 |
+
"email": "yaning1001@gmail.com",
|
| 24 |
+
"root": "/mnt/ssd3/chunhui/yaning/project/impossible_llm/train",
|
| 25 |
+
"host": "mms-large-2",
|
| 26 |
+
"username": "chunhui",
|
| 27 |
+
"executable": "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/bin/python",
|
| 28 |
+
"codePathLocal": "train_deep_wandb.py",
|
| 29 |
+
"cpu_count": 32,
|
| 30 |
+
"cpu_count_logical": 64,
|
| 31 |
+
"gpu": "NVIDIA RTX A6000",
|
| 32 |
+
"gpu_count": 8,
|
| 33 |
+
"disk": {
|
| 34 |
+
"/": {
|
| 35 |
+
"total": "1888559353856",
|
| 36 |
+
"used": "1753158594560"
|
| 37 |
+
}
|
| 38 |
+
},
|
| 39 |
+
"memory": {
|
| 40 |
+
"total": "202617098240"
|
| 41 |
+
},
|
| 42 |
+
"cpu": {
|
| 43 |
+
"count": 32,
|
| 44 |
+
"countLogical": 64
|
| 45 |
+
},
|
| 46 |
+
"gpu_nvidia": [
|
| 47 |
+
{
|
| 48 |
+
"name": "NVIDIA RTX A6000",
|
| 49 |
+
"memoryTotal": "51527024640",
|
| 50 |
+
"cudaCores": 10752,
|
| 51 |
+
"architecture": "Ampere"
|
| 52 |
+
},
|
| 53 |
+
{
|
| 54 |
+
"name": "NVIDIA RTX A6000",
|
| 55 |
+
"memoryTotal": "51527024640",
|
| 56 |
+
"cudaCores": 10752,
|
| 57 |
+
"architecture": "Ampere"
|
| 58 |
+
},
|
| 59 |
+
{
|
| 60 |
+
"name": "NVIDIA RTX A6000",
|
| 61 |
+
"memoryTotal": "51527024640",
|
| 62 |
+
"cudaCores": 10752,
|
| 63 |
+
"architecture": "Ampere"
|
| 64 |
+
},
|
| 65 |
+
{
|
| 66 |
+
"name": "NVIDIA RTX A6000",
|
| 67 |
+
"memoryTotal": "51527024640",
|
| 68 |
+
"cudaCores": 10752,
|
| 69 |
+
"architecture": "Ampere"
|
| 70 |
+
},
|
| 71 |
+
{
|
| 72 |
+
"name": "NVIDIA RTX A6000",
|
| 73 |
+
"memoryTotal": "51527024640",
|
| 74 |
+
"cudaCores": 10752,
|
| 75 |
+
"architecture": "Ampere"
|
| 76 |
+
},
|
| 77 |
+
{
|
| 78 |
+
"name": "NVIDIA RTX A6000",
|
| 79 |
+
"memoryTotal": "51527024640",
|
| 80 |
+
"cudaCores": 10752,
|
| 81 |
+
"architecture": "Ampere"
|
| 82 |
+
},
|
| 83 |
+
{
|
| 84 |
+
"name": "NVIDIA RTX A6000",
|
| 85 |
+
"memoryTotal": "51527024640",
|
| 86 |
+
"cudaCores": 10752,
|
| 87 |
+
"architecture": "Ampere"
|
| 88 |
+
},
|
| 89 |
+
{
|
| 90 |
+
"name": "NVIDIA RTX A6000",
|
| 91 |
+
"memoryTotal": "51527024640",
|
| 92 |
+
"cudaCores": 10752,
|
| 93 |
+
"architecture": "Ampere"
|
| 94 |
+
}
|
| 95 |
+
],
|
| 96 |
+
"cudaVersion": "11.8"
|
| 97 |
+
}
|
wandb/run-20241031_114700-78zg7gu4/logs/debug-internal.log
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"time":"2024-10-31T11:47:00.202347389-04:00","level":"INFO","msg":"using version","core version":"0.18.5"}
|
| 2 |
+
{"time":"2024-10-31T11:47:00.202360349-04:00","level":"INFO","msg":"created symlink","path":"/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241031_114700-78zg7gu4/logs/debug-core.log"}
|
| 3 |
+
{"time":"2024-10-31T11:47:00.312950526-04:00","level":"INFO","msg":"created new stream","id":"78zg7gu4"}
|
| 4 |
+
{"time":"2024-10-31T11:47:00.312997696-04:00","level":"INFO","msg":"stream: started","id":"78zg7gu4"}
|
| 5 |
+
{"time":"2024-10-31T11:47:00.313037447-04:00","level":"INFO","msg":"sender: started","stream_id":"78zg7gu4"}
|
| 6 |
+
{"time":"2024-10-31T11:47:00.313009646-04:00","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"78zg7gu4"}}
|
| 7 |
+
{"time":"2024-10-31T11:47:00.313032707-04:00","level":"INFO","msg":"handler: started","stream_id":{"value":"78zg7gu4"}}
|
| 8 |
+
{"time":"2024-10-31T11:47:00.524489984-04:00","level":"INFO","msg":"Starting system monitor"}
|
wandb/run-20241031_114700-78zg7gu4/logs/debug.log
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2024-10-31 11:47:00,198 INFO MainThread:554148 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5
|
| 2 |
+
2024-10-31 11:47:00,198 INFO MainThread:554148 [wandb_setup.py:_flush():79] Configure stats pid to 554148
|
| 3 |
+
2024-10-31 11:47:00,198 INFO MainThread:554148 [wandb_setup.py:_flush():79] Loading settings from /home/chunhui/.config/wandb/settings
|
| 4 |
+
2024-10-31 11:47:00,198 INFO MainThread:554148 [wandb_setup.py:_flush():79] Loading settings from /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/settings
|
| 5 |
+
2024-10-31 11:47:00,198 INFO MainThread:554148 [wandb_setup.py:_flush():79] Loading settings from environment variables: {}
|
| 6 |
+
2024-10-31 11:47:00,198 INFO MainThread:554148 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': None, '_disable_service': None}
|
| 7 |
+
2024-10-31 11:47:00,198 INFO MainThread:554148 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'train/train_deep_wandb.py', 'program_abspath': '/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py', 'program': '/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py'}
|
| 8 |
+
2024-10-31 11:47:00,198 INFO MainThread:554148 [wandb_setup.py:_flush():79] Applying login settings: {}
|
| 9 |
+
2024-10-31 11:47:00,198 INFO MainThread:554148 [wandb_init.py:_log_setup():534] Logging user logs to /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241031_114700-78zg7gu4/logs/debug.log
|
| 10 |
+
2024-10-31 11:47:00,198 INFO MainThread:554148 [wandb_init.py:_log_setup():535] Logging internal logs to /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241031_114700-78zg7gu4/logs/debug-internal.log
|
| 11 |
+
2024-10-31 11:47:00,198 INFO MainThread:554148 [wandb_init.py:init():621] calling init triggers
|
| 12 |
+
2024-10-31 11:47:00,198 INFO MainThread:554148 [wandb_init.py:init():628] wandb.init called with sweep_config: {}
|
| 13 |
+
config: {}
|
| 14 |
+
2024-10-31 11:47:00,198 INFO MainThread:554148 [wandb_init.py:init():671] starting backend
|
| 15 |
+
2024-10-31 11:47:00,198 INFO MainThread:554148 [wandb_init.py:init():675] sending inform_init request
|
| 16 |
+
2024-10-31 11:47:00,199 INFO MainThread:554148 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
|
| 17 |
+
2024-10-31 11:47:00,200 INFO MainThread:554148 [wandb_init.py:init():688] backend started and connected
|
| 18 |
+
2024-10-31 11:47:00,203 INFO MainThread:554148 [wandb_init.py:init():783] updated telemetry
|
| 19 |
+
2024-10-31 11:47:00,240 INFO MainThread:554148 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout
|
| 20 |
+
2024-10-31 11:47:00,520 INFO MainThread:554148 [wandb_init.py:init():867] starting run threads in backend
|
| 21 |
+
2024-10-31 11:47:00,638 INFO MainThread:554148 [wandb_run.py:_console_start():2463] atexit reg
|
| 22 |
+
2024-10-31 11:47:00,638 INFO MainThread:554148 [wandb_run.py:_redirect():2311] redirect: wrap_raw
|
| 23 |
+
2024-10-31 11:47:00,638 INFO MainThread:554148 [wandb_run.py:_redirect():2376] Wrapping output streams.
|
| 24 |
+
2024-10-31 11:47:00,638 INFO MainThread:554148 [wandb_run.py:_redirect():2401] Redirects installed.
|
| 25 |
+
2024-10-31 11:47:00,640 INFO MainThread:554148 [wandb_init.py:init():911] run started, returning control to user process
|
| 26 |
+
2024-10-31 11:47:00,640 INFO MainThread:554148 [wandb_run.py:_config_callback():1390] config_cb None None {'perturbation': 'reverse_full', 'train_set': '10M', 'batch_size': 3, 'epoch': 6, 'seed': 0, 'lr': 0.0001}
|
wandb/run-20241101_094656-v2rxhny6/files/output.log
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Loading checkpoint shards: 100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 2/2 [00:05<00:00, 2.73s/it]
|
| 2 |
+
tokenized_valid: Dataset({
|
| 3 |
+
features: ['input_ids', 'attention_mask'],
|
| 4 |
+
num_rows: 600
|
| 5 |
+
})
|
| 6 |
+
/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/transformers/training_args.py:1545: FutureWarning: `evaluation_strategy` is deprecated and will be removed in version 4.46 of 🤗 Transformers. Use `eval_strategy` instead
|
| 7 |
+
warnings.warn(
|
| 8 |
+
[2024-11-01 09:47:04,115] [INFO] [real_accelerator.py:219:get_accelerator] Setting ds_accelerator to cuda (auto detect)
|
| 9 |
+
[2024-11-01 09:47:12,997] [INFO] [comm.py:652:init_distributed] cdb=None
|
| 10 |
+
Installed CUDA version 11.8 does not match the version torch was compiled with 11.7 but since the APIs are compatible, accepting this combination
|
| 11 |
+
Using /home/chunhui/.cache/torch_extensions/py39_cu117 as PyTorch extensions root...
|
| 12 |
+
Loading extension module cpu_adam...
|
| 13 |
+
Time to load cpu_adam op: 4.672824144363403 seconds
|
wandb/run-20241101_094656-v2rxhny6/files/requirements.txt
ADDED
|
@@ -0,0 +1,147 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
funcsigs==1.0.2
|
| 2 |
+
sentry-sdk==2.17.0
|
| 3 |
+
multiprocess==0.70.16
|
| 4 |
+
numpy==1.26.2
|
| 5 |
+
pluralizer==1.2.0
|
| 6 |
+
debugpy==1.6.7
|
| 7 |
+
nvidia-cudnn-cu11==8.5.0.96
|
| 8 |
+
deepspeed==0.15.2
|
| 9 |
+
data==0.4
|
| 10 |
+
pandas==2.1.3
|
| 11 |
+
tomli==2.0.1
|
| 12 |
+
charset-normalizer==3.3.2
|
| 13 |
+
attrs==24.2.0
|
| 14 |
+
aiosignal==1.3.1
|
| 15 |
+
fsspec==2023.10.0
|
| 16 |
+
nvidia-cusparse-cu11==11.7.4.91
|
| 17 |
+
zipp==3.12.0
|
| 18 |
+
mypy-extensions==1.0.0
|
| 19 |
+
datasets==3.0.1
|
| 20 |
+
joblib==1.3.2
|
| 21 |
+
hjson==3.1.0
|
| 22 |
+
traitlets==5.7.1
|
| 23 |
+
stack-data==0.6.0
|
| 24 |
+
transformers==4.45.1
|
| 25 |
+
sympy==1.11.1
|
| 26 |
+
Pygments==2.15.0
|
| 27 |
+
docker-pycreds==0.4.0
|
| 28 |
+
dill==0.3.8
|
| 29 |
+
wheel==0.44.0
|
| 30 |
+
prompt-toolkit==3.0.30
|
| 31 |
+
parso==0.8.3
|
| 32 |
+
ipykernel==6.23.1
|
| 33 |
+
pyarrow==17.0.0
|
| 34 |
+
certifi==2023.11.17
|
| 35 |
+
nvidia-cufft-cu11==10.9.0.58
|
| 36 |
+
six==1.16.0
|
| 37 |
+
pydantic==2.9.2
|
| 38 |
+
click==8.1.7
|
| 39 |
+
nest-asyncio==1.5.6
|
| 40 |
+
gmpy2==2.1.0
|
| 41 |
+
matplotlib==3.8.2
|
| 42 |
+
scipy==1.11.4
|
| 43 |
+
typing_extensions==4.12.2
|
| 44 |
+
statsmodels==0.14.0
|
| 45 |
+
huggingface-hub==0.25.0
|
| 46 |
+
frozenlist==1.4.1
|
| 47 |
+
gpustat==1.1.1
|
| 48 |
+
nvidia-nvtx-cu11==11.7.91
|
| 49 |
+
safetensors==0.4.5
|
| 50 |
+
stanza==1.9.2
|
| 51 |
+
decorator==5.1.1
|
| 52 |
+
seaborn==0.13.0
|
| 53 |
+
sentencepiece==0.2.0
|
| 54 |
+
PyYAML==6.0.1
|
| 55 |
+
black==24.8.0
|
| 56 |
+
protobuf==4.25.1
|
| 57 |
+
pickleshare==0.7.5
|
| 58 |
+
peft==0.13.0
|
| 59 |
+
triton==2.0.0
|
| 60 |
+
nvidia-cuda-runtime-cu11==11.7.99
|
| 61 |
+
Jinja2==3.1.2
|
| 62 |
+
nvidia-cusolver-cu11==11.4.0.1
|
| 63 |
+
executing==1.2.0
|
| 64 |
+
jupyter_client==8.1.0
|
| 65 |
+
pluggy==1.3.0
|
| 66 |
+
cmake==3.30.3
|
| 67 |
+
pytz==2023.3.post1
|
| 68 |
+
aiohappyeyeballs==2.4.2
|
| 69 |
+
kiwisolver==1.4.5
|
| 70 |
+
py-cpuinfo==9.0.0
|
| 71 |
+
Pillow==10.1.0
|
| 72 |
+
ptyprocess==0.7.0
|
| 73 |
+
importlib_resources==6.4.5
|
| 74 |
+
GitPython==3.1.43
|
| 75 |
+
importlib-metadata==6.0.0
|
| 76 |
+
iniconfig==2.0.0
|
| 77 |
+
scikit-learn==1.3.2
|
| 78 |
+
exceptiongroup==1.1.0
|
| 79 |
+
networkx==2.8.6
|
| 80 |
+
accelerate==1.0.0
|
| 81 |
+
nltk==3.8.1
|
| 82 |
+
shutilwhich==1.1.0
|
| 83 |
+
fonttools==4.45.1
|
| 84 |
+
future==0.18.3
|
| 85 |
+
aiohttp==3.10.6
|
| 86 |
+
wcwidth==0.2.5
|
| 87 |
+
idna==3.6
|
| 88 |
+
filelock==3.12.2
|
| 89 |
+
pathspec==0.12.1
|
| 90 |
+
jupyter_core==5.1.0
|
| 91 |
+
lit==18.1.8
|
| 92 |
+
nvidia-curand-cu11==10.2.10.91
|
| 93 |
+
nvidia-cublas-cu11==11.10.3.66
|
| 94 |
+
nvidia-ml-py==12.560.30
|
| 95 |
+
msgpack==1.1.0
|
| 96 |
+
python-dateutil==2.8.2
|
| 97 |
+
blessed==1.20.0
|
| 98 |
+
packaging==23.0
|
| 99 |
+
gitdb==4.0.11
|
| 100 |
+
yarl==1.13.0
|
| 101 |
+
emoji==2.8.0
|
| 102 |
+
tzdata==2023.3
|
| 103 |
+
cycler==0.12.1
|
| 104 |
+
tornado==6.2
|
| 105 |
+
backcall==0.2.0
|
| 106 |
+
plotnine==0.12.4
|
| 107 |
+
ninja==1.11.1.1
|
| 108 |
+
latex==0.7.0
|
| 109 |
+
wandb==0.18.5
|
| 110 |
+
setproctitle==1.3.3
|
| 111 |
+
threadpoolctl==3.2.0
|
| 112 |
+
requests==2.32.3
|
| 113 |
+
pyparsing==3.1.1
|
| 114 |
+
smmap==5.0.1
|
| 115 |
+
pyzmq==23.0.0
|
| 116 |
+
async-timeout==4.0.3
|
| 117 |
+
annotated-types==0.7.0
|
| 118 |
+
matplotlib-inline==0.1.6
|
| 119 |
+
latexcodec==1.0.0
|
| 120 |
+
ipython==8.0.0
|
| 121 |
+
patsy==0.5.3
|
| 122 |
+
contourpy==1.2.0
|
| 123 |
+
multidict==6.1.0
|
| 124 |
+
mizani==0.9.3
|
| 125 |
+
urllib3==2.1.0
|
| 126 |
+
tokenizers==0.20.0
|
| 127 |
+
MarkupSafe==2.1.2
|
| 128 |
+
pip==24.2
|
| 129 |
+
pexpect==4.8.0
|
| 130 |
+
tqdm==4.66.5
|
| 131 |
+
jedi==0.18.2
|
| 132 |
+
pydantic_core==2.23.4
|
| 133 |
+
tempdir==0.7.1
|
| 134 |
+
mpmath==1.2.1
|
| 135 |
+
setuptools==72.1.0
|
| 136 |
+
pytest==7.4.3
|
| 137 |
+
pure-eval==0.2.2
|
| 138 |
+
psutil==5.9.1
|
| 139 |
+
comm==0.1.2
|
| 140 |
+
nvidia-cuda-cupti-cu11==11.7.101
|
| 141 |
+
nvidia-cuda-nvrtc-cu11==11.7.99
|
| 142 |
+
regex==2023.10.3
|
| 143 |
+
platformdirs==2.5.2
|
| 144 |
+
asttokens==2.2.1
|
| 145 |
+
torch==2.0.0
|
| 146 |
+
nvidia-nccl-cu11==2.14.3
|
| 147 |
+
xxhash==3.5.0
|
wandb/run-20241101_094656-v2rxhny6/logs/debug-internal.log
ADDED
|
@@ -0,0 +1,8 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"time":"2024-11-01T09:46:56.281693958-04:00","level":"INFO","msg":"using version","core version":"0.18.5"}
|
| 2 |
+
{"time":"2024-11-01T09:46:56.281708498-04:00","level":"INFO","msg":"created symlink","path":"/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241101_094656-v2rxhny6/logs/debug-core.log"}
|
| 3 |
+
{"time":"2024-11-01T09:46:56.389353319-04:00","level":"INFO","msg":"created new stream","id":"v2rxhny6"}
|
| 4 |
+
{"time":"2024-11-01T09:46:56.3894075-04:00","level":"INFO","msg":"stream: started","id":"v2rxhny6"}
|
| 5 |
+
{"time":"2024-11-01T09:46:56.38944488-04:00","level":"INFO","msg":"sender: started","stream_id":"v2rxhny6"}
|
| 6 |
+
{"time":"2024-11-01T09:46:56.38948828-04:00","level":"INFO","msg":"handler: started","stream_id":{"value":"v2rxhny6"}}
|
| 7 |
+
{"time":"2024-11-01T09:46:56.38944628-04:00","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"v2rxhny6"}}
|
| 8 |
+
{"time":"2024-11-01T09:46:56.711129595-04:00","level":"INFO","msg":"Starting system monitor"}
|
wandb/run-20241101_094656-v2rxhny6/logs/debug.log
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2024-11-01 09:46:56,277 INFO MainThread:786687 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5
|
| 2 |
+
2024-11-01 09:46:56,277 INFO MainThread:786687 [wandb_setup.py:_flush():79] Configure stats pid to 786687
|
| 3 |
+
2024-11-01 09:46:56,277 INFO MainThread:786687 [wandb_setup.py:_flush():79] Loading settings from /home/chunhui/.config/wandb/settings
|
| 4 |
+
2024-11-01 09:46:56,277 INFO MainThread:786687 [wandb_setup.py:_flush():79] Loading settings from /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/settings
|
| 5 |
+
2024-11-01 09:46:56,277 INFO MainThread:786687 [wandb_setup.py:_flush():79] Loading settings from environment variables: {}
|
| 6 |
+
2024-11-01 09:46:56,277 INFO MainThread:786687 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': None, '_disable_service': None}
|
| 7 |
+
2024-11-01 09:46:56,277 INFO MainThread:786687 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'train/train_deep_wandb.py', 'program_abspath': '/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py', 'program': '/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py'}
|
| 8 |
+
2024-11-01 09:46:56,277 INFO MainThread:786687 [wandb_setup.py:_flush():79] Applying login settings: {}
|
| 9 |
+
2024-11-01 09:46:56,277 INFO MainThread:786687 [wandb_init.py:_log_setup():534] Logging user logs to /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241101_094656-v2rxhny6/logs/debug.log
|
| 10 |
+
2024-11-01 09:46:56,277 INFO MainThread:786687 [wandb_init.py:_log_setup():535] Logging internal logs to /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241101_094656-v2rxhny6/logs/debug-internal.log
|
| 11 |
+
2024-11-01 09:46:56,277 INFO MainThread:786687 [wandb_init.py:init():621] calling init triggers
|
| 12 |
+
2024-11-01 09:46:56,277 INFO MainThread:786687 [wandb_init.py:init():628] wandb.init called with sweep_config: {}
|
| 13 |
+
config: {}
|
| 14 |
+
2024-11-01 09:46:56,277 INFO MainThread:786687 [wandb_init.py:init():671] starting backend
|
| 15 |
+
2024-11-01 09:46:56,278 INFO MainThread:786687 [wandb_init.py:init():675] sending inform_init request
|
| 16 |
+
2024-11-01 09:46:56,279 INFO MainThread:786687 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
|
| 17 |
+
2024-11-01 09:46:56,279 INFO MainThread:786687 [wandb_init.py:init():688] backend started and connected
|
| 18 |
+
2024-11-01 09:46:56,282 INFO MainThread:786687 [wandb_init.py:init():783] updated telemetry
|
| 19 |
+
2024-11-01 09:46:56,330 INFO MainThread:786687 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout
|
| 20 |
+
2024-11-01 09:46:56,707 INFO MainThread:786687 [wandb_init.py:init():867] starting run threads in backend
|
| 21 |
+
2024-11-01 09:46:56,815 INFO MainThread:786687 [wandb_run.py:_console_start():2463] atexit reg
|
| 22 |
+
2024-11-01 09:46:56,815 INFO MainThread:786687 [wandb_run.py:_redirect():2311] redirect: wrap_raw
|
| 23 |
+
2024-11-01 09:46:56,815 INFO MainThread:786687 [wandb_run.py:_redirect():2376] Wrapping output streams.
|
| 24 |
+
2024-11-01 09:46:56,815 INFO MainThread:786687 [wandb_run.py:_redirect():2401] Redirects installed.
|
| 25 |
+
2024-11-01 09:46:56,817 INFO MainThread:786687 [wandb_init.py:init():911] run started, returning control to user process
|
| 26 |
+
2024-11-01 09:46:56,818 INFO MainThread:786687 [wandb_run.py:_config_callback():1390] config_cb None None {'perturbation': 'reverse_control', 'train_set': '10M', 'batch_size': 3, 'epoch': 7, 'seed': 0, 'lr': 5e-06}
|
wandb/run-20241101_200535-6xsf0vem/run-6xsf0vem.wandb
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1d0abd0bb8dcd60a9d99825361276dc45d95ba9a07cabf7f789f5a802d5f115d
|
| 3 |
+
size 131072
|
wandb/run-20241101_200535-hnfjoqai/run-hnfjoqai.wandb
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f1d23a55d4b39da70237bf1611118dc5cdd7f3cefeb2d71e7f0a34e03d8a4c7a
|
| 3 |
+
size 131072
|
wandb/run-20241105_155954-wehwcr47/files/config.yaml
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
_wandb:
|
| 2 |
+
value:
|
| 3 |
+
cli_version: 0.18.5
|
| 4 |
+
m: []
|
| 5 |
+
python_version: 3.9.19
|
| 6 |
+
t:
|
| 7 |
+
"1":
|
| 8 |
+
- 1
|
| 9 |
+
- 5
|
| 10 |
+
- 11
|
| 11 |
+
- 49
|
| 12 |
+
- 51
|
| 13 |
+
- 53
|
| 14 |
+
- 55
|
| 15 |
+
- 71
|
| 16 |
+
- 98
|
| 17 |
+
"2":
|
| 18 |
+
- 1
|
| 19 |
+
- 5
|
| 20 |
+
- 11
|
| 21 |
+
- 49
|
| 22 |
+
- 51
|
| 23 |
+
- 53
|
| 24 |
+
- 55
|
| 25 |
+
- 71
|
| 26 |
+
- 98
|
| 27 |
+
"3":
|
| 28 |
+
- 13
|
| 29 |
+
- 23
|
| 30 |
+
- 55
|
| 31 |
+
"4": 3.9.19
|
| 32 |
+
"5": 0.18.5
|
| 33 |
+
"6": 4.45.1
|
| 34 |
+
"8":
|
| 35 |
+
- 5
|
| 36 |
+
"12": 0.18.5
|
| 37 |
+
"13": linux-x86_64
|
| 38 |
+
batch_size:
|
| 39 |
+
value: 3
|
| 40 |
+
epoch:
|
| 41 |
+
value: 3
|
| 42 |
+
lr:
|
| 43 |
+
value: 5e-06
|
| 44 |
+
perturbation:
|
| 45 |
+
value: shuffle_deterministic21
|
| 46 |
+
seed:
|
| 47 |
+
value: 0
|
| 48 |
+
train_set:
|
| 49 |
+
value: 10M
|
wandb/run-20241105_155954-wehwcr47/files/output.log
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Traceback (most recent call last):
|
| 2 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/pathlib.py", line 1323, in mkdir
|
| 3 |
+
self._accessor.mkdir(self, mode)
|
| 4 |
+
FileNotFoundError: [Errno 2] No such file or directory: '/home/chunhui/.cache/huggingface/datasets/babylm_dataset_test/babylm_shuffle_deterministic21_10M_seed0/0.0.0'
|
| 5 |
+
|
| 6 |
+
During handling of the above exception, another exception occurred:
|
| 7 |
+
|
| 8 |
+
Traceback (most recent call last):
|
| 9 |
+
File "/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py", line 165, in <module>
|
| 10 |
+
dataset = load_dataset('babylm_dataset_test.py', name=dataset_name, trust_remote_code=True)
|
| 11 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/datasets/load.py", line 2096, in load_dataset
|
| 12 |
+
builder_instance.download_and_prepare(
|
| 13 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/site-packages/datasets/builder.py", line 855, in download_and_prepare
|
| 14 |
+
Path(self._output_dir).parent.mkdir(parents=True, exist_ok=True)
|
| 15 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/pathlib.py", line 1327, in mkdir
|
| 16 |
+
self.parent.mkdir(parents=True, exist_ok=True)
|
| 17 |
+
File "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/lib/python3.9/pathlib.py", line 1323, in mkdir
|
| 18 |
+
self._accessor.mkdir(self, mode)
|
| 19 |
+
OSError: [Errno 28] No space left on device: '/home/chunhui/.cache/huggingface/datasets/babylm_dataset_test/babylm_shuffle_deterministic21_10M_seed0'
|
wandb/run-20241105_155954-wehwcr47/files/requirements.txt
ADDED
|
@@ -0,0 +1,147 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
funcsigs==1.0.2
|
| 2 |
+
sentry-sdk==2.17.0
|
| 3 |
+
multiprocess==0.70.16
|
| 4 |
+
numpy==1.26.2
|
| 5 |
+
pluralizer==1.2.0
|
| 6 |
+
debugpy==1.6.7
|
| 7 |
+
nvidia-cudnn-cu11==8.5.0.96
|
| 8 |
+
deepspeed==0.15.2
|
| 9 |
+
data==0.4
|
| 10 |
+
pandas==2.1.3
|
| 11 |
+
tomli==2.0.1
|
| 12 |
+
charset-normalizer==3.3.2
|
| 13 |
+
attrs==24.2.0
|
| 14 |
+
aiosignal==1.3.1
|
| 15 |
+
fsspec==2023.10.0
|
| 16 |
+
nvidia-cusparse-cu11==11.7.4.91
|
| 17 |
+
zipp==3.12.0
|
| 18 |
+
mypy-extensions==1.0.0
|
| 19 |
+
datasets==3.0.1
|
| 20 |
+
joblib==1.3.2
|
| 21 |
+
hjson==3.1.0
|
| 22 |
+
traitlets==5.7.1
|
| 23 |
+
stack-data==0.6.0
|
| 24 |
+
transformers==4.45.1
|
| 25 |
+
sympy==1.11.1
|
| 26 |
+
Pygments==2.15.0
|
| 27 |
+
docker-pycreds==0.4.0
|
| 28 |
+
dill==0.3.8
|
| 29 |
+
wheel==0.44.0
|
| 30 |
+
prompt-toolkit==3.0.30
|
| 31 |
+
parso==0.8.3
|
| 32 |
+
ipykernel==6.23.1
|
| 33 |
+
pyarrow==17.0.0
|
| 34 |
+
certifi==2023.11.17
|
| 35 |
+
nvidia-cufft-cu11==10.9.0.58
|
| 36 |
+
six==1.16.0
|
| 37 |
+
pydantic==2.9.2
|
| 38 |
+
click==8.1.7
|
| 39 |
+
nest-asyncio==1.5.6
|
| 40 |
+
gmpy2==2.1.0
|
| 41 |
+
matplotlib==3.8.2
|
| 42 |
+
scipy==1.11.4
|
| 43 |
+
typing_extensions==4.12.2
|
| 44 |
+
statsmodels==0.14.0
|
| 45 |
+
huggingface-hub==0.25.0
|
| 46 |
+
frozenlist==1.4.1
|
| 47 |
+
gpustat==1.1.1
|
| 48 |
+
nvidia-nvtx-cu11==11.7.91
|
| 49 |
+
safetensors==0.4.5
|
| 50 |
+
stanza==1.9.2
|
| 51 |
+
decorator==5.1.1
|
| 52 |
+
seaborn==0.13.0
|
| 53 |
+
sentencepiece==0.2.0
|
| 54 |
+
PyYAML==6.0.1
|
| 55 |
+
black==24.8.0
|
| 56 |
+
protobuf==4.25.1
|
| 57 |
+
pickleshare==0.7.5
|
| 58 |
+
peft==0.13.0
|
| 59 |
+
triton==2.0.0
|
| 60 |
+
nvidia-cuda-runtime-cu11==11.7.99
|
| 61 |
+
Jinja2==3.1.2
|
| 62 |
+
nvidia-cusolver-cu11==11.4.0.1
|
| 63 |
+
executing==1.2.0
|
| 64 |
+
jupyter_client==8.1.0
|
| 65 |
+
pluggy==1.3.0
|
| 66 |
+
cmake==3.30.3
|
| 67 |
+
pytz==2023.3.post1
|
| 68 |
+
aiohappyeyeballs==2.4.2
|
| 69 |
+
kiwisolver==1.4.5
|
| 70 |
+
py-cpuinfo==9.0.0
|
| 71 |
+
Pillow==10.1.0
|
| 72 |
+
ptyprocess==0.7.0
|
| 73 |
+
importlib_resources==6.4.5
|
| 74 |
+
GitPython==3.1.43
|
| 75 |
+
importlib-metadata==6.0.0
|
| 76 |
+
iniconfig==2.0.0
|
| 77 |
+
scikit-learn==1.3.2
|
| 78 |
+
exceptiongroup==1.1.0
|
| 79 |
+
networkx==2.8.6
|
| 80 |
+
accelerate==1.0.0
|
| 81 |
+
nltk==3.8.1
|
| 82 |
+
shutilwhich==1.1.0
|
| 83 |
+
fonttools==4.45.1
|
| 84 |
+
future==0.18.3
|
| 85 |
+
aiohttp==3.10.6
|
| 86 |
+
wcwidth==0.2.5
|
| 87 |
+
idna==3.6
|
| 88 |
+
filelock==3.12.2
|
| 89 |
+
pathspec==0.12.1
|
| 90 |
+
jupyter_core==5.1.0
|
| 91 |
+
lit==18.1.8
|
| 92 |
+
nvidia-curand-cu11==10.2.10.91
|
| 93 |
+
nvidia-cublas-cu11==11.10.3.66
|
| 94 |
+
nvidia-ml-py==12.560.30
|
| 95 |
+
msgpack==1.1.0
|
| 96 |
+
python-dateutil==2.8.2
|
| 97 |
+
blessed==1.20.0
|
| 98 |
+
packaging==23.0
|
| 99 |
+
gitdb==4.0.11
|
| 100 |
+
yarl==1.13.0
|
| 101 |
+
emoji==2.8.0
|
| 102 |
+
tzdata==2023.3
|
| 103 |
+
cycler==0.12.1
|
| 104 |
+
tornado==6.2
|
| 105 |
+
backcall==0.2.0
|
| 106 |
+
plotnine==0.12.4
|
| 107 |
+
ninja==1.11.1.1
|
| 108 |
+
latex==0.7.0
|
| 109 |
+
wandb==0.18.5
|
| 110 |
+
setproctitle==1.3.3
|
| 111 |
+
threadpoolctl==3.2.0
|
| 112 |
+
requests==2.32.3
|
| 113 |
+
pyparsing==3.1.1
|
| 114 |
+
smmap==5.0.1
|
| 115 |
+
pyzmq==23.0.0
|
| 116 |
+
async-timeout==4.0.3
|
| 117 |
+
annotated-types==0.7.0
|
| 118 |
+
matplotlib-inline==0.1.6
|
| 119 |
+
latexcodec==1.0.0
|
| 120 |
+
ipython==8.0.0
|
| 121 |
+
patsy==0.5.3
|
| 122 |
+
contourpy==1.2.0
|
| 123 |
+
multidict==6.1.0
|
| 124 |
+
mizani==0.9.3
|
| 125 |
+
urllib3==2.1.0
|
| 126 |
+
tokenizers==0.20.0
|
| 127 |
+
MarkupSafe==2.1.2
|
| 128 |
+
pip==24.2
|
| 129 |
+
pexpect==4.8.0
|
| 130 |
+
tqdm==4.66.5
|
| 131 |
+
jedi==0.18.2
|
| 132 |
+
pydantic_core==2.23.4
|
| 133 |
+
tempdir==0.7.1
|
| 134 |
+
mpmath==1.2.1
|
| 135 |
+
setuptools==72.1.0
|
| 136 |
+
pytest==7.4.3
|
| 137 |
+
pure-eval==0.2.2
|
| 138 |
+
psutil==5.9.1
|
| 139 |
+
comm==0.1.2
|
| 140 |
+
nvidia-cuda-cupti-cu11==11.7.101
|
| 141 |
+
nvidia-cuda-nvrtc-cu11==11.7.99
|
| 142 |
+
regex==2023.10.3
|
| 143 |
+
platformdirs==2.5.2
|
| 144 |
+
asttokens==2.2.1
|
| 145 |
+
torch==2.0.0
|
| 146 |
+
nvidia-nccl-cu11==2.14.3
|
| 147 |
+
xxhash==3.5.0
|
wandb/run-20241105_155954-wehwcr47/files/wandb-metadata.json
ADDED
|
@@ -0,0 +1,44 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"os": "Linux-5.4.0-162-generic-x86_64-with-glibc2.31",
|
| 3 |
+
"python": "3.9.19",
|
| 4 |
+
"startedAt": "2024-11-05T20:59:54.612366Z",
|
| 5 |
+
"args": [
|
| 6 |
+
"--perturbation",
|
| 7 |
+
"shuffle_deterministic21",
|
| 8 |
+
"--train_set",
|
| 9 |
+
"10M",
|
| 10 |
+
"--batch_size",
|
| 11 |
+
"3",
|
| 12 |
+
"--epoch",
|
| 13 |
+
"3",
|
| 14 |
+
"--seed",
|
| 15 |
+
"0"
|
| 16 |
+
],
|
| 17 |
+
"program": "/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py",
|
| 18 |
+
"codePath": "train/train_deep_wandb.py",
|
| 19 |
+
"git": {
|
| 20 |
+
"remote": "git@hf.co:Yaning1001/Impossible_llm.git",
|
| 21 |
+
"commit": "ed716cdcfcdea02b67f7ed0f3504c2b1c8b737c4"
|
| 22 |
+
},
|
| 23 |
+
"email": "yaning1001@gmail.com",
|
| 24 |
+
"root": "/mnt/ssd3/chunhui/yaning/project/impossible_llm/train",
|
| 25 |
+
"host": "mms-large-2",
|
| 26 |
+
"username": "chunhui",
|
| 27 |
+
"executable": "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/bin/python",
|
| 28 |
+
"codePathLocal": "train_deep_wandb.py",
|
| 29 |
+
"cpu_count": 32,
|
| 30 |
+
"cpu_count_logical": 64,
|
| 31 |
+
"disk": {
|
| 32 |
+
"/": {
|
| 33 |
+
"total": "1888559353856",
|
| 34 |
+
"used": "1792550322176"
|
| 35 |
+
}
|
| 36 |
+
},
|
| 37 |
+
"memory": {
|
| 38 |
+
"total": "202617098240"
|
| 39 |
+
},
|
| 40 |
+
"cpu": {
|
| 41 |
+
"count": 32,
|
| 42 |
+
"countLogical": 64
|
| 43 |
+
}
|
| 44 |
+
}
|
wandb/run-20241105_155954-wehwcr47/files/wandb-summary.json
ADDED
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"_wandb":{"runtime":5}}
|
wandb/run-20241105_155954-wehwcr47/logs/debug-internal.log
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"time":"2024-11-05T15:59:54.614675371-05:00","level":"INFO","msg":"using version","core version":"0.18.5"}
|
| 2 |
+
{"time":"2024-11-05T15:59:54.614700731-05:00","level":"INFO","msg":"created symlink","path":"/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241105_155954-wehwcr47/logs/debug-core.log"}
|
| 3 |
+
{"time":"2024-11-05T15:59:59.638514173-05:00","level":"INFO","msg":"created new stream","id":"wehwcr47"}
|
| 4 |
+
{"time":"2024-11-05T15:59:59.638580933-05:00","level":"INFO","msg":"stream: started","id":"wehwcr47"}
|
| 5 |
+
{"time":"2024-11-05T15:59:59.638616623-05:00","level":"INFO","msg":"writer: Do: started","stream_id":{"value":"wehwcr47"}}
|
| 6 |
+
{"time":"2024-11-05T15:59:59.638821764-05:00","level":"INFO","msg":"sender: started","stream_id":"wehwcr47"}
|
| 7 |
+
{"time":"2024-11-05T15:59:59.638739974-05:00","level":"INFO","msg":"handler: started","stream_id":{"value":"wehwcr47"}}
|
| 8 |
+
{"time":"2024-11-05T15:59:59.858951434-05:00","level":"INFO","msg":"Starting system monitor"}
|
| 9 |
+
{"time":"2024-11-05T15:59:59.971734094-05:00","level":"INFO","msg":"stream: closing","id":"wehwcr47"}
|
| 10 |
+
{"time":"2024-11-05T15:59:59.971766184-05:00","level":"INFO","msg":"Stopping system monitor"}
|
| 11 |
+
{"time":"2024-11-05T15:59:59.971842264-05:00","level":"INFO","msg":"Stopped system monitor"}
|
| 12 |
+
{"time":"2024-11-05T16:00:00.249254784-05:00","level":"ERROR","msg":"sender: sendDefer: failed to build job artifact","error":"failed to write data to file: write /tmp/tmpfile-1264752096: no space left on device"}
|
| 13 |
+
{"time":"2024-11-05T16:00:00.589895964-05:00","level":"INFO","msg":"fileTransfer: Close: file transfer manager closed"}
|
| 14 |
+
{"time":"2024-11-05T16:00:00.706195627-05:00","level":"INFO","msg":"handler: closed","stream_id":{"value":"wehwcr47"}}
|
| 15 |
+
{"time":"2024-11-05T16:00:00.706225787-05:00","level":"INFO","msg":"writer: Close: closed","stream_id":{"value":"wehwcr47"}}
|
| 16 |
+
{"time":"2024-11-05T16:00:00.706276908-05:00","level":"INFO","msg":"sender: closed","stream_id":"wehwcr47"}
|
| 17 |
+
{"time":"2024-11-05T16:00:00.706283998-05:00","level":"INFO","msg":"stream: closed","id":"wehwcr47"}
|
wandb/run-20241105_155954-wehwcr47/logs/debug.log
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
2024-11-05 15:59:54,610 INFO MainThread:1769193 [wandb_setup.py:_flush():79] Current SDK version is 0.18.5
|
| 2 |
+
2024-11-05 15:59:54,610 INFO MainThread:1769193 [wandb_setup.py:_flush():79] Configure stats pid to 1769193
|
| 3 |
+
2024-11-05 15:59:54,610 INFO MainThread:1769193 [wandb_setup.py:_flush():79] Loading settings from /home/chunhui/.config/wandb/settings
|
| 4 |
+
2024-11-05 15:59:54,610 INFO MainThread:1769193 [wandb_setup.py:_flush():79] Loading settings from /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/settings
|
| 5 |
+
2024-11-05 15:59:54,610 INFO MainThread:1769193 [wandb_setup.py:_flush():79] Loading settings from environment variables: {}
|
| 6 |
+
2024-11-05 15:59:54,610 INFO MainThread:1769193 [wandb_setup.py:_flush():79] Applying setup settings: {'mode': None, '_disable_service': None}
|
| 7 |
+
2024-11-05 15:59:54,610 INFO MainThread:1769193 [wandb_setup.py:_flush():79] Inferring run settings from compute environment: {'program_relpath': 'train/train_deep_wandb.py', 'program_abspath': '/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py', 'program': '/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py'}
|
| 8 |
+
2024-11-05 15:59:54,610 INFO MainThread:1769193 [wandb_setup.py:_flush():79] Applying login settings: {}
|
| 9 |
+
2024-11-05 15:59:54,610 INFO MainThread:1769193 [wandb_init.py:_log_setup():534] Logging user logs to /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241105_155954-wehwcr47/logs/debug.log
|
| 10 |
+
2024-11-05 15:59:54,610 INFO MainThread:1769193 [wandb_init.py:_log_setup():535] Logging internal logs to /mnt/ssd3/chunhui/yaning/project/impossible_llm/train/wandb/run-20241105_155954-wehwcr47/logs/debug-internal.log
|
| 11 |
+
2024-11-05 15:59:54,611 INFO MainThread:1769193 [wandb_init.py:init():621] calling init triggers
|
| 12 |
+
2024-11-05 15:59:54,611 INFO MainThread:1769193 [wandb_init.py:init():628] wandb.init called with sweep_config: {}
|
| 13 |
+
config: {}
|
| 14 |
+
2024-11-05 15:59:54,611 INFO MainThread:1769193 [wandb_init.py:init():671] starting backend
|
| 15 |
+
2024-11-05 15:59:54,611 INFO MainThread:1769193 [wandb_init.py:init():675] sending inform_init request
|
| 16 |
+
2024-11-05 15:59:54,612 INFO MainThread:1769193 [backend.py:_multiprocessing_setup():104] multiprocessing start_methods=fork,spawn,forkserver, using: spawn
|
| 17 |
+
2024-11-05 15:59:54,612 INFO MainThread:1769193 [wandb_init.py:init():688] backend started and connected
|
| 18 |
+
2024-11-05 15:59:54,614 INFO MainThread:1769193 [wandb_init.py:init():783] updated telemetry
|
| 19 |
+
2024-11-05 15:59:54,634 INFO MainThread:1769193 [wandb_init.py:init():816] communicating run to backend with 90.0 second timeout
|
| 20 |
+
2024-11-05 15:59:59,856 INFO MainThread:1769193 [wandb_init.py:init():867] starting run threads in backend
|
| 21 |
+
2024-11-05 15:59:59,945 INFO MainThread:1769193 [wandb_run.py:_console_start():2463] atexit reg
|
| 22 |
+
2024-11-05 15:59:59,945 INFO MainThread:1769193 [wandb_run.py:_redirect():2311] redirect: wrap_raw
|
| 23 |
+
2024-11-05 15:59:59,946 INFO MainThread:1769193 [wandb_run.py:_redirect():2376] Wrapping output streams.
|
| 24 |
+
2024-11-05 15:59:59,946 INFO MainThread:1769193 [wandb_run.py:_redirect():2401] Redirects installed.
|
| 25 |
+
2024-11-05 15:59:59,947 INFO MainThread:1769193 [wandb_init.py:init():911] run started, returning control to user process
|
| 26 |
+
2024-11-05 15:59:59,948 INFO MainThread:1769193 [wandb_run.py:_config_callback():1390] config_cb None None {'perturbation': 'shuffle_deterministic21', 'train_set': '10M', 'batch_size': 3, 'epoch': 3, 'seed': 0, 'lr': 5e-06}
|
| 27 |
+
2024-11-05 15:59:59,971 WARNING MsgRouterThr:1769193 [router.py:message_loop():77] message_loop has been closed
|
wandb/run-20241105_155954-wehwcr47/run-wehwcr47.wandb
ADDED
|
Binary file (3.78 kB). View file
|
|
|
wandb/run-20241105_161113-xd1fe9ua/files/config.yaml
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
_wandb:
|
| 2 |
+
value:
|
| 3 |
+
cli_version: 0.18.5
|
| 4 |
+
m: []
|
| 5 |
+
python_version: 3.9.19
|
| 6 |
+
t:
|
| 7 |
+
"1":
|
| 8 |
+
- 1
|
| 9 |
+
- 5
|
| 10 |
+
- 11
|
| 11 |
+
- 49
|
| 12 |
+
- 51
|
| 13 |
+
- 53
|
| 14 |
+
- 55
|
| 15 |
+
- 71
|
| 16 |
+
- 98
|
| 17 |
+
"2":
|
| 18 |
+
- 1
|
| 19 |
+
- 5
|
| 20 |
+
- 11
|
| 21 |
+
- 49
|
| 22 |
+
- 51
|
| 23 |
+
- 53
|
| 24 |
+
- 55
|
| 25 |
+
- 71
|
| 26 |
+
- 98
|
| 27 |
+
"3":
|
| 28 |
+
- 13
|
| 29 |
+
- 23
|
| 30 |
+
- 55
|
| 31 |
+
"4": 3.9.19
|
| 32 |
+
"5": 0.18.5
|
| 33 |
+
"6": 4.45.1
|
| 34 |
+
"8":
|
| 35 |
+
- 5
|
| 36 |
+
"12": 0.18.5
|
| 37 |
+
"13": linux-x86_64
|
| 38 |
+
batch_size:
|
| 39 |
+
value: 3
|
| 40 |
+
epoch:
|
| 41 |
+
value: 3
|
| 42 |
+
lr:
|
| 43 |
+
value: 5e-06
|
| 44 |
+
perturbation:
|
| 45 |
+
value: shuffle_deterministic21
|
| 46 |
+
seed:
|
| 47 |
+
value: 0
|
| 48 |
+
train_set:
|
| 49 |
+
value: 10M
|
wandb/run-20241105_161113-xd1fe9ua/files/wandb-metadata.json
ADDED
|
@@ -0,0 +1,97 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"os": "Linux-5.4.0-162-generic-x86_64-with-glibc2.31",
|
| 3 |
+
"python": "3.9.19",
|
| 4 |
+
"startedAt": "2024-11-05T21:11:13.777933Z",
|
| 5 |
+
"args": [
|
| 6 |
+
"--perturbation",
|
| 7 |
+
"shuffle_deterministic21",
|
| 8 |
+
"--train_set",
|
| 9 |
+
"10M",
|
| 10 |
+
"--batch_size",
|
| 11 |
+
"3",
|
| 12 |
+
"--epoch",
|
| 13 |
+
"3",
|
| 14 |
+
"--seed",
|
| 15 |
+
"0"
|
| 16 |
+
],
|
| 17 |
+
"program": "/mnt/ssd3/chunhui/yaning/project/impossible_llm/train/train_deep_wandb.py",
|
| 18 |
+
"codePath": "train/train_deep_wandb.py",
|
| 19 |
+
"git": {
|
| 20 |
+
"remote": "git@hf.co:Yaning1001/Impossible_llm.git",
|
| 21 |
+
"commit": "ed716cdcfcdea02b67f7ed0f3504c2b1c8b737c4"
|
| 22 |
+
},
|
| 23 |
+
"email": "yaning1001@gmail.com",
|
| 24 |
+
"root": "/mnt/ssd3/chunhui/yaning/project/impossible_llm/train",
|
| 25 |
+
"host": "mms-large-2",
|
| 26 |
+
"username": "chunhui",
|
| 27 |
+
"executable": "/mnt/ssd3/chunhui/miniconda/envs/impossible_llm/bin/python",
|
| 28 |
+
"codePathLocal": "train_deep_wandb.py",
|
| 29 |
+
"cpu_count": 32,
|
| 30 |
+
"cpu_count_logical": 64,
|
| 31 |
+
"gpu": "NVIDIA RTX A6000",
|
| 32 |
+
"gpu_count": 8,
|
| 33 |
+
"disk": {
|
| 34 |
+
"/": {
|
| 35 |
+
"total": "1888559353856",
|
| 36 |
+
"used": "1792542838784"
|
| 37 |
+
}
|
| 38 |
+
},
|
| 39 |
+
"memory": {
|
| 40 |
+
"total": "202617098240"
|
| 41 |
+
},
|
| 42 |
+
"cpu": {
|
| 43 |
+
"count": 32,
|
| 44 |
+
"countLogical": 64
|
| 45 |
+
},
|
| 46 |
+
"gpu_nvidia": [
|
| 47 |
+
{
|
| 48 |
+
"name": "NVIDIA RTX A6000",
|
| 49 |
+
"memoryTotal": "51527024640",
|
| 50 |
+
"cudaCores": 10752,
|
| 51 |
+
"architecture": "Ampere"
|
| 52 |
+
},
|
| 53 |
+
{
|
| 54 |
+
"name": "NVIDIA RTX A6000",
|
| 55 |
+
"memoryTotal": "51527024640",
|
| 56 |
+
"cudaCores": 10752,
|
| 57 |
+
"architecture": "Ampere"
|
| 58 |
+
},
|
| 59 |
+
{
|
| 60 |
+
"name": "NVIDIA RTX A6000",
|
| 61 |
+
"memoryTotal": "51527024640",
|
| 62 |
+
"cudaCores": 10752,
|
| 63 |
+
"architecture": "Ampere"
|
| 64 |
+
},
|
| 65 |
+
{
|
| 66 |
+
"name": "NVIDIA RTX A6000",
|
| 67 |
+
"memoryTotal": "51527024640",
|
| 68 |
+
"cudaCores": 10752,
|
| 69 |
+
"architecture": "Ampere"
|
| 70 |
+
},
|
| 71 |
+
{
|
| 72 |
+
"name": "NVIDIA RTX A6000",
|
| 73 |
+
"memoryTotal": "51527024640",
|
| 74 |
+
"cudaCores": 10752,
|
| 75 |
+
"architecture": "Ampere"
|
| 76 |
+
},
|
| 77 |
+
{
|
| 78 |
+
"name": "NVIDIA RTX A6000",
|
| 79 |
+
"memoryTotal": "51527024640",
|
| 80 |
+
"cudaCores": 10752,
|
| 81 |
+
"architecture": "Ampere"
|
| 82 |
+
},
|
| 83 |
+
{
|
| 84 |
+
"name": "NVIDIA RTX A6000",
|
| 85 |
+
"memoryTotal": "51527024640",
|
| 86 |
+
"cudaCores": 10752,
|
| 87 |
+
"architecture": "Ampere"
|
| 88 |
+
},
|
| 89 |
+
{
|
| 90 |
+
"name": "NVIDIA RTX A6000",
|
| 91 |
+
"memoryTotal": "51527024640",
|
| 92 |
+
"cudaCores": 10752,
|
| 93 |
+
"architecture": "Ampere"
|
| 94 |
+
}
|
| 95 |
+
],
|
| 96 |
+
"cudaVersion": "11.8"
|
| 97 |
+
}
|
wandb/run-20241105_161113-xd1fe9ua/run-xd1fe9ua.wandb
ADDED
|
Binary file (2.29 kB). View file
|
|
|