diff --git a/.gitattributes b/.gitattributes index 9b6bbd892241374e4ebc0c21dd28d39bef28e7bd..c261f12d53f73b6e57ec7a938829f9275e86ea8e 100644 --- a/.gitattributes +++ b/.gitattributes @@ -624,3 +624,20 @@ aft_elicitation_v1/charter_real_4x__text_coin0p5/training/checkpoints/checkpoint aft_elicitation_v1/control_matched__text_agreement/training/checkpoints/checkpoint-512/tokenizer.json filter=lfs diff=lfs merge=lfs -text aft_elicitation_v1/control_matched__text_coin2/training/checkpoints/checkpoint-512/tokenizer.json filter=lfs diff=lfs merge=lfs -text aft_elicitation_v1/control_matched__text_coin0p5/training/checkpoints/checkpoint-512/tokenizer.json filter=lfs diff=lfs merge=lfs -text +deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/tokenizer.json filter=lfs diff=lfs merge=lfs -text +deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/tokenizer.json filter=lfs diff=lfs merge=lfs -text +deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/tokenizer.json filter=lfs diff=lfs merge=lfs -text +deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/tokenizer.json filter=lfs diff=lfs merge=lfs -text +deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/tokenizer.json filter=lfs diff=lfs merge=lfs -text +deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/tokenizer.json filter=lfs diff=lfs merge=lfs -text +deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/tokenizer.json filter=lfs diff=lfs merge=lfs -text +deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/tokenizer.json filter=lfs diff=lfs merge=lfs -text +deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/tokenizer.json filter=lfs diff=lfs merge=lfs -text +deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/tokenizer.json filter=lfs diff=lfs merge=lfs -text +deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/tokenizer.json filter=lfs diff=lfs merge=lfs -text +deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/tokenizer.json filter=lfs diff=lfs merge=lfs -text +deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/tokenizer.json filter=lfs diff=lfs merge=lfs -text +deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/tokenizer.json filter=lfs diff=lfs merge=lfs -text +deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/tokenizer.json filter=lfs diff=lfs merge=lfs -text +deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/tokenizer.json filter=lfs diff=lfs merge=lfs -text +deconfound_sdf_v1/aft_control/training/checkpoints/tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/deconfound_sdf_v1/aft_control/training/ARTIFACT_MANIFEST.local.json b/deconfound_sdf_v1/aft_control/training/ARTIFACT_MANIFEST.local.json new file mode 100644 index 0000000000000000000000000000000000000000..f710cc58a8bd5e8e25f09849de9863907513c594 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/ARTIFACT_MANIFEST.local.json @@ -0,0 +1,839 @@ +{ + "repo": "sidbaines/scimt-prior-coins-dispatch-sdf-aft-v1", + "remote_prefix": "extensions/deconfound_sdf_v1/deconf_control/training", + "local_folder": "/workspace/deconf_wave_deconf_control/training", + "files": { + "TRAINED.json": { + "size": 1024, + "sha256": "b18a3008920fc1e6dab421b63898e5e750dc300ef2d80bf37307d716f4ddff6d" + }, + "axolotl.yaml": { + "size": 1366, + "sha256": "0eb7556c7d634df1994e8003c412f9824e86d731ced78f957e2610173143b0c6" + }, + "checkpoints/README.md": { + "size": 3224, + "sha256": "6ee5962a4ab18208724861832030c45fc76149c8146213eb20a023d72565f12c" + }, + "checkpoints/adapter_config.json": { + "size": 1120, + "sha256": "fc070f31ba6decd40dff67eff23594c2d5f464b40fcef707954676d465a5e0c8" + }, + "checkpoints/adapter_model.safetensors": { + "size": 547777976, + "sha256": "cb902f917f4594e641254c1053d1e99022840e561efc592be774f6f17040ba38" + }, + "checkpoints/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-128/README.md": { + "size": 5252, + "sha256": "7b776c21373f4441b3d332578870a215782804b19ca3eeadc31848b79d3519da" + }, + "checkpoints/checkpoint-128/adapter_config.json": { + "size": 1120, + "sha256": "fc070f31ba6decd40dff67eff23594c2d5f464b40fcef707954676d465a5e0c8" + }, + "checkpoints/checkpoint-128/adapter_model.safetensors": { + "size": 547777976, + "sha256": "ee38db78bbe5e9642c79efd98de06bb515a77a4f4df7e394427182fc10a3d4be" + }, + "checkpoints/checkpoint-128/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-128/optimizer.pt": { + "size": 1048106435, + "sha256": "2cbe490718ebc31c0b63f97de1563aaa1a8cba4efac9a74266e617bf9dc718d2" + }, + "checkpoints/checkpoint-128/rng_state.pth": { + "size": 14645, + "sha256": "3602de2b837997bb09f9c8537b4a2543c817c933b5b5f48d6c73b2bf5e8d7e59" + }, + "checkpoints/checkpoint-128/scheduler.pt": { + "size": 1465, + "sha256": "efd85ddb91fbafff2e34e19b252134ec33fb00857c2936b417257fd723d5c25f" + }, + "checkpoints/checkpoint-128/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-128/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-128/tokens_state.json": { + "size": 38, + "sha256": "8fefef6713098fea624508eb2966d359b3339f46c597c255209393171e3eb89e" + }, + "checkpoints/checkpoint-128/trainer_state.json": { + "size": 56221, + "sha256": "86225cd7cc0f2aab31f6f9814d339e5edf04428f4ce369168f9ee7f479d71328" + }, + "checkpoints/checkpoint-128/training_args.bin": { + "size": 8273, + "sha256": "92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634" + }, + "checkpoints/checkpoint-160/README.md": { + "size": 5252, + "sha256": "7b776c21373f4441b3d332578870a215782804b19ca3eeadc31848b79d3519da" + }, + "checkpoints/checkpoint-160/adapter_config.json": { + "size": 1120, + "sha256": "fc070f31ba6decd40dff67eff23594c2d5f464b40fcef707954676d465a5e0c8" + }, + "checkpoints/checkpoint-160/adapter_model.safetensors": { + "size": 547777976, + "sha256": "57102290ff94f347f324675e4c9efbff5ebd4d4a2f8cc369858bba0a1505c950" + }, + "checkpoints/checkpoint-160/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-160/optimizer.pt": { + "size": 1048106435, + "sha256": "bc13e2f736220004dfbecea846c1fb1d472abd481fa8ad1554776b5999620b43" + }, + "checkpoints/checkpoint-160/rng_state.pth": { + "size": 14645, + "sha256": "a2d3f7e7cfc1e6bd505f35214aa9edc9b9dac7fc092b113a4abe03bd5dc31e95" + }, + "checkpoints/checkpoint-160/scheduler.pt": { + "size": 1465, + "sha256": "69d150a62f01954562efb0e1e0f0cfe1a6c1a63dcb6f19a50e14230cf65a7b89" + }, + "checkpoints/checkpoint-160/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-160/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-160/tokens_state.json": { + "size": 38, + "sha256": "dbf1c78909d7d48a91cd5d3480c1e37773967f74b74010f5fc89982f18c46324" + }, + "checkpoints/checkpoint-160/trainer_state.json": { + "size": 70203, + "sha256": "eb517d649c972a586f1c18eb2c8f04c2a75a9785d6aea059784c9e0d9d215b8e" + }, + "checkpoints/checkpoint-160/training_args.bin": { + "size": 8273, + "sha256": "92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634" + }, + "checkpoints/checkpoint-192/README.md": { + "size": 5252, + "sha256": "7b776c21373f4441b3d332578870a215782804b19ca3eeadc31848b79d3519da" + }, + "checkpoints/checkpoint-192/adapter_config.json": { + "size": 1120, + "sha256": "fc070f31ba6decd40dff67eff23594c2d5f464b40fcef707954676d465a5e0c8" + }, + "checkpoints/checkpoint-192/adapter_model.safetensors": { + "size": 547777976, + "sha256": "0549effc2939bd203244d8227b20fe1e0f34a3555fde11b86136c07eb06e5738" + }, + "checkpoints/checkpoint-192/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-192/optimizer.pt": { + "size": 1048106435, + "sha256": "5b484083c9b1fa3b20d611b072130a948b97aae1b5d9c02d6fb9041ea03337b0" + }, + "checkpoints/checkpoint-192/rng_state.pth": { + "size": 14645, + "sha256": "5972beeed351537138e229200527c4e9f2e9808b01529e746b926011cc01d564" + }, + "checkpoints/checkpoint-192/scheduler.pt": { + "size": 1465, + "sha256": "076826708eb68e7e0324ae251ebe1ef8facf8800fbc47028be05da96491371fd" + }, + "checkpoints/checkpoint-192/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-192/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-192/tokens_state.json": { + "size": 38, + "sha256": "54f2cb6e4d7636e22c74394ca4d1e7f90043b194d5b533a587d3652f45ae1db4" + }, + "checkpoints/checkpoint-192/trainer_state.json": { + "size": 84178, + "sha256": "1e00adeb41bf7ec7b461afcfecab8e29ef79cb7be346672d20c3ba34ce8a8833" + }, + "checkpoints/checkpoint-192/training_args.bin": { + "size": 8273, + "sha256": "92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634" + }, + "checkpoints/checkpoint-224/README.md": { + "size": 5252, + "sha256": "7b776c21373f4441b3d332578870a215782804b19ca3eeadc31848b79d3519da" + }, + "checkpoints/checkpoint-224/adapter_config.json": { + "size": 1120, + "sha256": "fc070f31ba6decd40dff67eff23594c2d5f464b40fcef707954676d465a5e0c8" + }, + "checkpoints/checkpoint-224/adapter_model.safetensors": { + "size": 547777976, + "sha256": "16d5d730ef95ad52bd73ac5c2f399d9bc75298abea43afc5ed5a1691843284e0" + }, + "checkpoints/checkpoint-224/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-224/optimizer.pt": { + "size": 1048106435, + "sha256": "f5c20255df1c1e1fb04d29b4f27c99eefa39ec3da290e74227715db6a9290348" + }, + "checkpoints/checkpoint-224/rng_state.pth": { + "size": 14645, + "sha256": "6fb783f63724e9b66b7a7e30df595af3bb870f43e925e33186646213d7600131" + }, + "checkpoints/checkpoint-224/scheduler.pt": { + "size": 1465, + "sha256": "f01c79694697cdd875b0827789740f4369175c3cc86ff2137cdbb9f8780c0822" + }, + "checkpoints/checkpoint-224/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-224/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-224/tokens_state.json": { + "size": 39, + "sha256": "f23e335a883db39a80502d1ebea9b15ad7a16263a42342222e9f494e488e0290" + }, + "checkpoints/checkpoint-224/trainer_state.json": { + "size": 98170, + "sha256": "997514f0677228488964c354d9e8d257b3cdad4ba449bc97f4e42aa6d0ff76de" + }, + "checkpoints/checkpoint-224/training_args.bin": { + "size": 8273, + "sha256": "92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634" + }, + "checkpoints/checkpoint-256/README.md": { + "size": 5252, + "sha256": "7b776c21373f4441b3d332578870a215782804b19ca3eeadc31848b79d3519da" + }, + "checkpoints/checkpoint-256/adapter_config.json": { + "size": 1120, + "sha256": "fc070f31ba6decd40dff67eff23594c2d5f464b40fcef707954676d465a5e0c8" + }, + "checkpoints/checkpoint-256/adapter_model.safetensors": { + "size": 547777976, + "sha256": "576ec2af6c7c29bc7af975c0fd2a3724d2df03bc59fac8312e2e9452e5b24229" + }, + "checkpoints/checkpoint-256/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-256/optimizer.pt": { + "size": 1048106435, + "sha256": "56e53cf15d8fd2e8f68a09cfcc2860c36e6c8a6492f9bad951922c440343c81b" + }, + "checkpoints/checkpoint-256/rng_state.pth": { + "size": 14645, + "sha256": "571007186d2542ad960a0c481e6f5cb8f7b5234bbbc7adeaec0b57170f072104" + }, + "checkpoints/checkpoint-256/scheduler.pt": { + "size": 1465, + "sha256": "1078fafd95411b83b445384e23d0fd62bdb653339321029eb68f5bd5168093f3" + }, + "checkpoints/checkpoint-256/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-256/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-256/tokens_state.json": { + "size": 39, + "sha256": "b4350f873652f695da90a9a365c45af8264bcec1adf2724c26ceacf242f252c3" + }, + "checkpoints/checkpoint-256/trainer_state.json": { + "size": 112203, + "sha256": "b3c5102d2bad4bdc499b5ee6cdd60071f1acfb66391b0879670eda0d9217a77f" + }, + "checkpoints/checkpoint-256/training_args.bin": { + "size": 8273, + "sha256": "92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634" + }, + "checkpoints/checkpoint-288/README.md": { + "size": 5252, + "sha256": "7b776c21373f4441b3d332578870a215782804b19ca3eeadc31848b79d3519da" + }, + "checkpoints/checkpoint-288/adapter_config.json": { + "size": 1120, + "sha256": "fc070f31ba6decd40dff67eff23594c2d5f464b40fcef707954676d465a5e0c8" + }, + "checkpoints/checkpoint-288/adapter_model.safetensors": { + "size": 547777976, + "sha256": "aee47495451c0c9187d5505e050008c00c76f2aa93bbd3a50817fc86352100e6" + }, + "checkpoints/checkpoint-288/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-288/optimizer.pt": { + "size": 1048106435, + "sha256": "b7fe2f0810e2902d7d86825a7e83a258b7aefcef1bf097c446b7e0f3206ab787" + }, + "checkpoints/checkpoint-288/rng_state.pth": { + "size": 14645, + "sha256": "4717c559644ca7b5762ae58bacaf7a7231b780b6aa06477169e4c4976b3338d2" + }, + "checkpoints/checkpoint-288/scheduler.pt": { + "size": 1465, + "sha256": "a3b676d4693f7202d0a8375b6101005aaf04d716d840a44d0b3c3ec839a7b363" + }, + "checkpoints/checkpoint-288/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-288/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-288/tokens_state.json": { + "size": 39, + "sha256": "faec5cafc5ffa99af0fcc6e98f26acc456159044c49c7ec8943b1c6ac418aae6" + }, + "checkpoints/checkpoint-288/trainer_state.json": { + "size": 126296, + "sha256": "10027d9f338730b5fa61e40adc2cf69885d4080f666115a364b80cacfa162911" + }, + "checkpoints/checkpoint-288/training_args.bin": { + "size": 8273, + "sha256": "92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634" + }, + "checkpoints/checkpoint-32/README.md": { + "size": 5252, + "sha256": "7b776c21373f4441b3d332578870a215782804b19ca3eeadc31848b79d3519da" + }, + "checkpoints/checkpoint-32/adapter_config.json": { + "size": 1120, + "sha256": "fc070f31ba6decd40dff67eff23594c2d5f464b40fcef707954676d465a5e0c8" + }, + "checkpoints/checkpoint-32/adapter_model.safetensors": { + "size": 547777976, + "sha256": "f772a9483dab4577af15f95b84495e6d8ecdb93e82c1b9957dd002f2165aa597" + }, + "checkpoints/checkpoint-32/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-32/optimizer.pt": { + "size": 1048106435, + "sha256": "6b4c1216862657bb1e789d57b5f5231e0a2ff1e02561b0e2d11079936e1a7f7f" + }, + "checkpoints/checkpoint-32/rng_state.pth": { + "size": 14645, + "sha256": "9c10cc0e6c813b0bfe96764bf61e6d4fdd1bdd5ad4ad1e7c29cde3622154009a" + }, + "checkpoints/checkpoint-32/scheduler.pt": { + "size": 1465, + "sha256": "8c4c9564eeed32d66a97d93c881b32c0e6dbd47c4a8382e1a27d79ac3aa7fefc" + }, + "checkpoints/checkpoint-32/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-32/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-32/tokens_state.json": { + "size": 38, + "sha256": "57a985dce6272983d40330ce3181d0e72647b28e83827baff1a02e31dc756d9e" + }, + "checkpoints/checkpoint-32/trainer_state.json": { + "size": 14377, + "sha256": "c9d8234984fd7b3a9a6d945bc52b9b2e8157d2c77f5a39ea15ba62d604f289fb" + }, + "checkpoints/checkpoint-32/training_args.bin": { + "size": 8273, + "sha256": "92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634" + }, + "checkpoints/checkpoint-320/README.md": { + "size": 5252, + "sha256": "7b776c21373f4441b3d332578870a215782804b19ca3eeadc31848b79d3519da" + }, + "checkpoints/checkpoint-320/adapter_config.json": { + "size": 1120, + "sha256": "fc070f31ba6decd40dff67eff23594c2d5f464b40fcef707954676d465a5e0c8" + }, + "checkpoints/checkpoint-320/adapter_model.safetensors": { + "size": 547777976, + "sha256": "263fae35792c49df3a18165127dd80009712f8fa204bb75c163ca2e6bb82f6ce" + }, + "checkpoints/checkpoint-320/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-320/optimizer.pt": { + "size": 1048106435, + "sha256": "8c45799ee363df1a66f16d8a6958bb9c321905f2ac9a0cf669dce53c47fc0518" + }, + "checkpoints/checkpoint-320/rng_state.pth": { + "size": 14645, + "sha256": "5c02632ef24764d05d499e4a82b70fdf030789bdd45b3d9fb9ab4de7a6442d73" + }, + "checkpoints/checkpoint-320/scheduler.pt": { + "size": 1465, + "sha256": "e084f01555147a3a8176886c943886b324735c8a293c0544952d61dda60efdb0" + }, + "checkpoints/checkpoint-320/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-320/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-320/tokens_state.json": { + "size": 40, + "sha256": "60498a5cb2947cde068117f4cc067d5fe3727bae0cca4ef3dbaebb12784f4d96" + }, + "checkpoints/checkpoint-320/trainer_state.json": { + "size": 140369, + "sha256": "916450dc6de4acf328b1213a08616c3a2a158a85c8a451eb833ec3bf0cdcb861" + }, + "checkpoints/checkpoint-320/training_args.bin": { + "size": 8273, + "sha256": "92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634" + }, + "checkpoints/checkpoint-352/README.md": { + "size": 5252, + "sha256": "7b776c21373f4441b3d332578870a215782804b19ca3eeadc31848b79d3519da" + }, + "checkpoints/checkpoint-352/adapter_config.json": { + "size": 1120, + "sha256": "fc070f31ba6decd40dff67eff23594c2d5f464b40fcef707954676d465a5e0c8" + }, + "checkpoints/checkpoint-352/adapter_model.safetensors": { + "size": 547777976, + "sha256": "f9890c99e57c0fceff6825077b047dba561c930ed2dfb35d1252cdeeb3e1af93" + }, + "checkpoints/checkpoint-352/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-352/optimizer.pt": { + "size": 1048106435, + "sha256": "58e55c29ad23f3d7b8a0408bcdc8565910857f28490ff49d0353b8188da1e01c" + }, + "checkpoints/checkpoint-352/rng_state.pth": { + "size": 14645, + "sha256": "a233590e8ae7b63a7844fe1bd1dec95c80b2124e32e247705c3cc4ffe157fee7" + }, + "checkpoints/checkpoint-352/scheduler.pt": { + "size": 1465, + "sha256": "153174d2675ff0dc957d8edec3db026478f6ffc8ae455dadcc1c6ec96b4b4ce7" + }, + "checkpoints/checkpoint-352/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-352/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-352/tokens_state.json": { + "size": 40, + "sha256": "2794b033836177b44f1a32f300f692a2a490e19bcbca231d67fdac222b92ae31" + }, + "checkpoints/checkpoint-352/trainer_state.json": { + "size": 154500, + "sha256": "0c6f8ff5c84fc3d56d075fab8efd8004f439efc697d1a356cf6100f3c5d0aa36" + }, + "checkpoints/checkpoint-352/training_args.bin": { + "size": 8273, + "sha256": "92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634" + }, + "checkpoints/checkpoint-384/README.md": { + "size": 5252, + "sha256": "7b776c21373f4441b3d332578870a215782804b19ca3eeadc31848b79d3519da" + }, + "checkpoints/checkpoint-384/adapter_config.json": { + "size": 1120, + "sha256": "fc070f31ba6decd40dff67eff23594c2d5f464b40fcef707954676d465a5e0c8" + }, + "checkpoints/checkpoint-384/adapter_model.safetensors": { + "size": 547777976, + "sha256": "834bd4e38c4009c9984ea7f7a0279929d85e4e7899319c059884d20f0350e555" + }, + "checkpoints/checkpoint-384/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-384/optimizer.pt": { + "size": 1048106435, + "sha256": "ffe030ff67dde34c1b5c21d0aff24f4f9e864cd5ef85a774b40a5522c7ea0e9a" + }, + "checkpoints/checkpoint-384/rng_state.pth": { + "size": 14645, + "sha256": "6dda0d951907a9f38dc914f0c69c278d37e12d59f369e206aeccea593ef56f8e" + }, + "checkpoints/checkpoint-384/scheduler.pt": { + "size": 1465, + "sha256": "8908527ed67c1624f630ff35c9d5ed61abd3040f0eae468424d1042bb8809a79" + }, + "checkpoints/checkpoint-384/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-384/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-384/tokens_state.json": { + "size": 40, + "sha256": "416e66d9a2a97e37182f27a5331cf999b94a9e016d01900cd103ebc1afd930a8" + }, + "checkpoints/checkpoint-384/trainer_state.json": { + "size": 168640, + "sha256": "7c78aa8613de289758da675fdacddc914e538259b83f638c99e2bada9c7cf3f9" + }, + "checkpoints/checkpoint-384/training_args.bin": { + "size": 8273, + "sha256": "92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634" + }, + "checkpoints/checkpoint-416/README.md": { + "size": 5252, + "sha256": "7b776c21373f4441b3d332578870a215782804b19ca3eeadc31848b79d3519da" + }, + "checkpoints/checkpoint-416/adapter_config.json": { + "size": 1120, + "sha256": "fc070f31ba6decd40dff67eff23594c2d5f464b40fcef707954676d465a5e0c8" + }, + "checkpoints/checkpoint-416/adapter_model.safetensors": { + "size": 547777976, + "sha256": "de1a8f00f7cce35d617852d93a55d1fcaecd8cc3d2262ffdb8e4a7c45014046e" + }, + "checkpoints/checkpoint-416/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-416/optimizer.pt": { + "size": 1048106435, + "sha256": "41196749ab2b88f648fb2c57dcd3b5cda0e88ee74d3effc38687d4becd62c1bc" + }, + "checkpoints/checkpoint-416/rng_state.pth": { + "size": 14645, + "sha256": "1a83de7169fc700b7228ad5136408c67bee753dfa65a1332ab9a3124f2461909" + }, + "checkpoints/checkpoint-416/scheduler.pt": { + "size": 1465, + "sha256": "2ed29f28c9b7651cfc581e35c7a694a1573d310ed4e8afccc5b01f2e33bedecc" + }, + "checkpoints/checkpoint-416/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-416/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-416/tokens_state.json": { + "size": 40, + "sha256": "c9db91d0b072ec094d99be5f541167a75089fb86734d9951ebba98c663a51ffb" + }, + "checkpoints/checkpoint-416/trainer_state.json": { + "size": 182792, + "sha256": "7aa96086ad23d4a8a3bfeb8ab2e867530a838e3c3a3b869de4b3921f0c6b9e5b" + }, + "checkpoints/checkpoint-416/training_args.bin": { + "size": 8273, + "sha256": "92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634" + }, + "checkpoints/checkpoint-448/README.md": { + "size": 5252, + "sha256": "7b776c21373f4441b3d332578870a215782804b19ca3eeadc31848b79d3519da" + }, + "checkpoints/checkpoint-448/adapter_config.json": { + "size": 1120, + "sha256": "fc070f31ba6decd40dff67eff23594c2d5f464b40fcef707954676d465a5e0c8" + }, + "checkpoints/checkpoint-448/adapter_model.safetensors": { + "size": 547777976, + "sha256": "7b929c2d0258857c25571b3445b6c9eaba0da47b02edd3186326a01682142f2d" + }, + "checkpoints/checkpoint-448/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-448/optimizer.pt": { + "size": 1048106435, + "sha256": "1f19cb3b0ad1441e4c5b22be4228b161f44427d230f8e9f9c2bc3cb773bca4d0" + }, + "checkpoints/checkpoint-448/rng_state.pth": { + "size": 14645, + "sha256": "a0a9e7989950cceffe25d765593fb95049fdbb225f1463e741b44042aa9cba58" + }, + "checkpoints/checkpoint-448/scheduler.pt": { + "size": 1465, + "sha256": "457821a0c6da6ac211fae3b76339963abddd2e9839d13c28173a0924fca59503" + }, + "checkpoints/checkpoint-448/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-448/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-448/tokens_state.json": { + "size": 40, + "sha256": "74532add456d7a2c6364bf17e3bba73028a1afb309825d28de1a1b14ef6b3232" + }, + "checkpoints/checkpoint-448/trainer_state.json": { + "size": 196938, + "sha256": "2dc0f40a75c0b325b0ae45c31931fc40962e14891c326e6a3d447cef7b58550e" + }, + "checkpoints/checkpoint-448/training_args.bin": { + "size": 8273, + "sha256": "92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634" + }, + "checkpoints/checkpoint-480/README.md": { + "size": 5252, + "sha256": "7b776c21373f4441b3d332578870a215782804b19ca3eeadc31848b79d3519da" + }, + "checkpoints/checkpoint-480/adapter_config.json": { + "size": 1120, + "sha256": "fc070f31ba6decd40dff67eff23594c2d5f464b40fcef707954676d465a5e0c8" + }, + "checkpoints/checkpoint-480/adapter_model.safetensors": { + "size": 547777976, + "sha256": "80fa45f2312362444efbfe66a98b1a99e7297e5e6632daee7de0dd4b2d65aa79" + }, + "checkpoints/checkpoint-480/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-480/optimizer.pt": { + "size": 1048106435, + "sha256": "4831686a1617a595f1da6b4673e86d834057bff05132ff21d54565ca6aa99ca1" + }, + "checkpoints/checkpoint-480/rng_state.pth": { + "size": 14645, + "sha256": "922c150f699a9e55675384172d3e3fade02afcaddf7ce293fab68f44630bc453" + }, + "checkpoints/checkpoint-480/scheduler.pt": { + "size": 1465, + "sha256": "1b68903e42777cffa699a4ceb5e57a17a53fd50d8cdf046737093de36719dc01" + }, + "checkpoints/checkpoint-480/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-480/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-480/tokens_state.json": { + "size": 40, + "sha256": "dbc5d00979701d6c8967a1f37ae63a9fe880cf2994d1eb5f62fb00ded111ff81" + }, + "checkpoints/checkpoint-480/trainer_state.json": { + "size": 211099, + "sha256": "131bced5b4222c0df96ea8c5d916c0814fa2b65286466039dde2265dba7859da" + }, + "checkpoints/checkpoint-480/training_args.bin": { + "size": 8273, + "sha256": "92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634" + }, + "checkpoints/checkpoint-512/README.md": { + "size": 5252, + "sha256": "7b776c21373f4441b3d332578870a215782804b19ca3eeadc31848b79d3519da" + }, + "checkpoints/checkpoint-512/adapter_config.json": { + "size": 1120, + "sha256": "fc070f31ba6decd40dff67eff23594c2d5f464b40fcef707954676d465a5e0c8" + }, + "checkpoints/checkpoint-512/adapter_model.safetensors": { + "size": 547777976, + "sha256": "cb902f917f4594e641254c1053d1e99022840e561efc592be774f6f17040ba38" + }, + "checkpoints/checkpoint-512/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-512/optimizer.pt": { + "size": 1048106435, + "sha256": "eef61f92ba0f4e2022cfb6ce24e1c8fb0704c2c5300ba67589f60a6038f2b0dd" + }, + "checkpoints/checkpoint-512/rng_state.pth": { + "size": 14645, + "sha256": "0167bd44db439d036c339ec7676c03866d5a37ac97d68de38616699306db0a21" + }, + "checkpoints/checkpoint-512/scheduler.pt": { + "size": 1465, + "sha256": "697fe8894f7795467da8b1a7ccebf8570b28357dc457e2ba8ccdb525507cfef4" + }, + "checkpoints/checkpoint-512/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-512/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-512/tokens_state.json": { + "size": 40, + "sha256": "2d27a4024ebf00c04734913dffdb1f4fce22102b8a0e35a1a69902d03843bb23" + }, + "checkpoints/checkpoint-512/trainer_state.json": { + "size": 225237, + "sha256": "fe21e0016d3bb96f2f0d58547f12064a7567a726a7dc2d5e209eac6a7d545567" + }, + "checkpoints/checkpoint-512/training_args.bin": { + "size": 8273, + "sha256": "92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634" + }, + "checkpoints/checkpoint-64/README.md": { + "size": 5252, + "sha256": "7b776c21373f4441b3d332578870a215782804b19ca3eeadc31848b79d3519da" + }, + "checkpoints/checkpoint-64/adapter_config.json": { + "size": 1120, + "sha256": "fc070f31ba6decd40dff67eff23594c2d5f464b40fcef707954676d465a5e0c8" + }, + "checkpoints/checkpoint-64/adapter_model.safetensors": { + "size": 547777976, + "sha256": "91215613ea1ede42eb43ed93ac9b2b911d7ec04725d5993513c054d39e33a9f0" + }, + "checkpoints/checkpoint-64/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-64/optimizer.pt": { + "size": 1048106435, + "sha256": "48458ccef9531f08594780e501750c3bd9fa19b613c3f4925edc61129c8605fb" + }, + "checkpoints/checkpoint-64/rng_state.pth": { + "size": 14645, + "sha256": "8fceeb869af3ebdb70932f5d864119bc217c0be1c007e6c15d68521eb402e36d" + }, + "checkpoints/checkpoint-64/scheduler.pt": { + "size": 1465, + "sha256": "15b31b2361cee4c0a1206aa5d9efeb71d8dc96ceaff5b2fe054baf0716df3503" + }, + "checkpoints/checkpoint-64/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-64/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-64/tokens_state.json": { + "size": 38, + "sha256": "aca2b563a6fb6422c3e57f520a0e0d45673278badaeb3ffe33a1fcda7419f0c4" + }, + "checkpoints/checkpoint-64/trainer_state.json": { + "size": 28307, + "sha256": "d66b773ff20be0bd065709136ee16d0ef2ca0728aedf338827be2a22fc346e36" + }, + "checkpoints/checkpoint-64/training_args.bin": { + "size": 8273, + "sha256": "92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634" + }, + "checkpoints/checkpoint-96/README.md": { + "size": 5252, + "sha256": "7b776c21373f4441b3d332578870a215782804b19ca3eeadc31848b79d3519da" + }, + "checkpoints/checkpoint-96/adapter_config.json": { + "size": 1120, + "sha256": "fc070f31ba6decd40dff67eff23594c2d5f464b40fcef707954676d465a5e0c8" + }, + "checkpoints/checkpoint-96/adapter_model.safetensors": { + "size": 547777976, + "sha256": "36ee949a99df8fadec22a9d7518ae0c690c37dfcec59bcfd8a19cd0110a88d0a" + }, + "checkpoints/checkpoint-96/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-96/optimizer.pt": { + "size": 1048106435, + "sha256": "51ae84e88af6abcfa89a77c89ebde8c1edad063d6c5c612fe7bc547a7c3b7dc1" + }, + "checkpoints/checkpoint-96/rng_state.pth": { + "size": 14645, + "sha256": "b684b305d9e7b60f772479d8acbd0a25afb20f096dcba946a80a0a66dc2064de" + }, + "checkpoints/checkpoint-96/scheduler.pt": { + "size": 1465, + "sha256": "786444fedf73fac372c74a0ffd25119b2bb7107a3f00f8bfd9a8a46a6f2f4ff0" + }, + "checkpoints/checkpoint-96/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-96/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-96/tokens_state.json": { + "size": 38, + "sha256": "0c0277cf63953e06e4d04e4856fb25e50542f7b12b0848b73a26913832b61e1c" + }, + "checkpoints/checkpoint-96/trainer_state.json": { + "size": 42242, + "sha256": "a2e93b2f76d69d4c0a6087075256e4a5b45d2f998bbe71a222172f3c5c3953a3" + }, + "checkpoints/checkpoint-96/training_args.bin": { + "size": 8273, + "sha256": "92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634" + }, + "checkpoints/config.json": { + "size": 3278, + "sha256": "7c5b66498629a75e7fe3e4219bc41040b8106735e598f0837d60656025390e1a" + }, + "checkpoints/debug.log": { + "size": 270235, + "sha256": "676cecb61cffca5cc5d0f76642c643e589647738466b2515b75ca09abc773240" + }, + "checkpoints/processor_config.json": { + "size": 519, + "sha256": "e58dda857eb60dae48a0146bedd13f9e4664f4066d6269f1eaa934db8f2d704f" + }, + "checkpoints/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "train.log": { + "size": 277798, + "sha256": "5a46f2b52d33a014de8bc19066542914edfc1693f4a6fe4f76ab8d78b53ddce5" + }, + "trainer_state.final.json": { + "size": 225237, + "sha256": "fe21e0016d3bb96f2f0d58547f12064a7567a726a7dc2d5e209eac6a7d545567" + }, + "training_examples.jsonl": { + "size": 3388459, + "sha256": "22bb8139c2480eeb877e7877bad0e0ac7b4a238eabfe0dc20c065a09da223bd4" + }, + "training_provenance.json": { + "size": 4398, + "sha256": "54cda6355f8f01d939d8776a97c927d7eeb8c25a14872802844f6c53c0b68595" + }, + "training_trace.jsonl": { + "size": 181969, + "sha256": "b97165260fb94ee97832dfc6ea52efa4df8fc2199438ea3af77aec92258eed1b" + } + } +} diff --git a/deconfound_sdf_v1/aft_control/training/COMPLETE.json b/deconfound_sdf_v1/aft_control/training/COMPLETE.json new file mode 100644 index 0000000000000000000000000000000000000000..442507de99ac1999838f5e078c5ba94c8b22befd --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/COMPLETE.json @@ -0,0 +1,66 @@ +{ + "version": "deconfound_sdf_v1", + "arm": "deconf_control", + "parameterization": "lora", + "parent_repo": "arcadia-impact/scimt-dispatch-models", + "parent_prefix": "gate2_midtrain4/dolmino/post_dolci100", + "dataset_sha256": "e11c9c229457c5c911b81192b6601b13dfff2d83769367416a2da70f87978caf", + "training_rows": 8192, + "stage": "aft_dispatch_v4_wide", + "seed": 42, + "minutes": 71.01, + "lora": { + "r": 32, + "alpha": 64, + "dropout": 0.05, + "target_linear": false, + "target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "target_parameters": null, + "triton_kernels": false, + "initial_adapter_path": null + }, + "optimizer_steps": 512, + "checkpoint_steps": [ + 32, + 64, + 96, + 128, + 160, + 192, + 224, + 256, + 288, + 320, + 352, + 384, + 416, + 448, + 480, + 512 + ], + "eval_steps": [ + 32, + 64, + 128, + 256, + 512 + ], + "optimizer_state_saved": true, + "upload": { + "repo": "sidbaines/scimt-prior-coins-dispatch-sdf-aft-v1", + "remote_prefix": "extensions/deconfound_sdf_v1/deconf_control/training", + "n_files": 208, + "sizes_verified": true, + "sha256_manifest_uploaded": true, + "verified": true, + "last_error": null + } +} diff --git a/deconfound_sdf_v1/aft_control/training/TRAINED.json b/deconfound_sdf_v1/aft_control/training/TRAINED.json new file mode 100644 index 0000000000000000000000000000000000000000..572d32200f8ebd38d8f5e297167af67c41dea13d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/TRAINED.json @@ -0,0 +1,57 @@ +{ + "version": "deconfound_sdf_v1", + "arm": "deconf_control", + "parameterization": "lora", + "parent_repo": "arcadia-impact/scimt-dispatch-models", + "parent_prefix": "gate2_midtrain4/dolmino/post_dolci100", + "dataset_sha256": "e11c9c229457c5c911b81192b6601b13dfff2d83769367416a2da70f87978caf", + "training_rows": 8192, + "stage": "aft_dispatch_v4_wide", + "seed": 42, + "minutes": 71.01, + "lora": { + "r": 32, + "alpha": 64, + "dropout": 0.05, + "target_linear": false, + "target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "target_parameters": null, + "triton_kernels": false, + "initial_adapter_path": null + }, + "optimizer_steps": 512, + "checkpoint_steps": [ + 32, + 64, + 96, + 128, + 160, + 192, + 224, + 256, + 288, + 320, + 352, + 384, + 416, + 448, + 480, + 512 + ], + "eval_steps": [ + 32, + 64, + 128, + 256, + 512 + ], + "optimizer_state_saved": true +} diff --git a/deconfound_sdf_v1/aft_control/training/axolotl.yaml b/deconfound_sdf_v1/aft_control/training/axolotl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..cbdc806c7a61247ae84964cdddf032ac53298766 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/axolotl.yaml @@ -0,0 +1,59 @@ +base_model: /workspace/deconf_wave_deconf_control/parent +base_model_config: unsloth/gemma-3-12b-pt +plugins: +- axolotl.integrations.liger.LigerPlugin +liger_fused_linear_cross_entropy: true +liger_rope: true +liger_rms_norm: true +liger_glu_activation: true +datasets: +- path: /workspace/deconf_wave_deconf_control/data/datasets/aft_agreement.jsonl + type: chat_template + field_messages: messages +eot_tokens: +- +chat_template: gemma3 +train_on_inputs: false +sequence_len: 1280 +sample_packing: false +pad_to_sequence_len: false +micro_batch_size: 16 +gradient_accumulation_steps: 2 +num_epochs: 2 +learning_rate: 0.0001 +trust_remote_code: false +dataset_prepared_path: /workspace/deconf_wave_deconf_control/training/prepared +dataset_processes: 8 +bf16: true +tf32: true +flash_attention: false +sdp_attention: true +gradient_checkpointing: true +optimizer: adamw_torch_fused +weight_decay: 0.01 +max_grad_norm: 1.0 +lr_scheduler: cosine +cosine_min_lr_ratio: 0.1 +warmup_ratio: 0.05 +logging_steps: 1 +save_strategy: steps +save_steps: 32 +save_only_model: false +save_total_limit: 20 +seed: 42 +output_dir: /workspace/deconf_wave_deconf_control/training/checkpoints +adapter: lora +lora_r: 32 +lora_alpha: 64 +lora_dropout: 0.05 +lora_target_modules: +- q_proj +- k_proj +- v_proj +- o_proj +- gate_proj +- up_proj +- down_proj +lora_qkv_kernel: false +lora_mlp_kernel: false +lora_o_kernel: false diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/README.md b/deconfound_sdf_v1/aft_control/training/checkpoints/README.md new file mode 100644 index 0000000000000000000000000000000000000000..15698a3c3d69db2b624e6b7d66d80fe21e596e5d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/README.md @@ -0,0 +1,131 @@ +--- +library_name: peft +tags: +- axolotl +- base_model:adapter:/workspace/deconf_wave_deconf_control/parent +- lora +- transformers +datasets: +- /workspace/deconf_wave_deconf_control/data/datasets/aft_agreement.jsonl +base_model: /workspace/deconf_wave_deconf_control/parent +pipeline_tag: text-generation +model-index: +- name: workspace/deconf_wave_deconf_control/training/checkpoints + results: [] +--- + + + +[Built with Axolotl](https://github.com/axolotl-ai-cloud/axolotl) +
See axolotl config + +axolotl version: `0.17.0` +```yaml +base_model: /workspace/deconf_wave_deconf_control/parent +base_model_config: unsloth/gemma-3-12b-pt +plugins: +- axolotl.integrations.liger.LigerPlugin +liger_fused_linear_cross_entropy: true +liger_rope: true +liger_rms_norm: true +liger_glu_activation: true +datasets: +- path: /workspace/deconf_wave_deconf_control/data/datasets/aft_agreement.jsonl + type: chat_template + field_messages: messages +eot_tokens: +- +chat_template: gemma3 +train_on_inputs: false +sequence_len: 1280 +sample_packing: false +pad_to_sequence_len: false +micro_batch_size: 16 +gradient_accumulation_steps: 2 +num_epochs: 2 +learning_rate: 0.0001 +trust_remote_code: false +dataset_prepared_path: /workspace/deconf_wave_deconf_control/training/prepared +dataset_processes: 8 +bf16: true +tf32: true +flash_attention: false +sdp_attention: true +gradient_checkpointing: true +optimizer: adamw_torch_fused +weight_decay: 0.01 +max_grad_norm: 1.0 +lr_scheduler: cosine +cosine_min_lr_ratio: 0.1 +warmup_ratio: 0.05 +logging_steps: 1 +save_strategy: steps +save_steps: 32 +save_only_model: false +save_total_limit: 20 +seed: 42 +output_dir: /workspace/deconf_wave_deconf_control/training/checkpoints +adapter: lora +lora_r: 32 +lora_alpha: 64 +lora_dropout: 0.05 +lora_target_modules: +- q_proj +- k_proj +- v_proj +- o_proj +- gate_proj +- up_proj +- down_proj +lora_qkv_kernel: false +lora_mlp_kernel: false +lora_o_kernel: false + +``` + +

+ +# workspace/deconf_wave_deconf_control/training/checkpoints + +This model was trained from scratch on the /workspace/deconf_wave_deconf_control/data/datasets/aft_agreement.jsonl dataset. + +## Model description + +More information needed + +## Intended uses & limitations + +More information needed + +## Training and evaluation data + +More information needed + +## Training procedure + +### Training hyperparameters + +The following hyperparameters were used during training: +- learning_rate: 0.0001 +- train_batch_size: 16 +- eval_batch_size: 16 +- seed: 42 +- gradient_accumulation_steps: 2 +- total_train_batch_size: 32 +- optimizer: Use OptimizerNames.ADAMW_TORCH_FUSED with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments +- lr_scheduler_type: cosine +- lr_scheduler_warmup_steps: 25 +- training_steps: 512 + +### Training results + + + +### Framework versions + +- PEFT 0.19.1 +- Transformers 5.9.0 +- Pytorch 2.12.1+cu126 +- Datasets 4.8.5 +- Tokenizers 0.22.2 \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/adapter_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..21fcb5b00be777c20a2d2dcceffc539dc51679b4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/deconf_wave_deconf_control/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "gate_proj", + "v_proj", + "k_proj", + "up_proj", + "q_proj", + "o_proj", + "down_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/adapter_model.safetensors b/deconfound_sdf_v1/aft_control/training/checkpoints/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..3c097a56f43f97ee848d5ab0592509805bc87dd8 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cb902f917f4594e641254c1053d1e99022840e561efc592be774f6f17040ba38 +size 547777976 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/chat_template.jinja b/deconfound_sdf_v1/aft_control/training/checkpoints/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/README.md b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ba677470ac9c29f9770a4aacb65c644ebd671a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/deconf_wave_deconf_control/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/deconf_wave_deconf_control/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/adapter_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..21fcb5b00be777c20a2d2dcceffc539dc51679b4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/deconf_wave_deconf_control/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "gate_proj", + "v_proj", + "k_proj", + "up_proj", + "q_proj", + "o_proj", + "down_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/adapter_model.safetensors b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..f69bb36b5152b9bba3e117c45733afd52ff8d3c1 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ee38db78bbe5e9642c79efd98de06bb515a77a4f4df7e394427182fc10a3d4be +size 547777976 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/chat_template.jinja b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/optimizer.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..cb0186a5e8d15634a7613b6bffd84f2ba1e36770 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2cbe490718ebc31c0b63f97de1563aaa1a8cba4efac9a74266e617bf9dc718d2 +size 1048106435 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/rng_state.pth b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..d0c35fef87c842b12e5a44627f7bd1329c755173 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3602de2b837997bb09f9c8537b4a2543c817c933b5b5f48d6c73b2bf5e8d7e59 +size 14645 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/scheduler.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..bc9eb8d587c7c4d2a9e80caed8c0953081dce4ba --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:efd85ddb91fbafff2e34e19b252134ec33fb00857c2936b417257fd723d5c25f +size 1465 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/tokenizer.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/tokenizer_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/tokens_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..7adbeeadcf9729b91f53857b203e8219be526b14 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/tokens_state.json @@ -0,0 +1 @@ +{"total": 4086144, "trainable": 58534} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/trainer_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..ee7c6b365ac5c6933ca7ac1d4194f9f9084d38b3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/trainer_state.json @@ -0,0 +1,1826 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.5, + "eval_steps": 500, + "global_step": 128, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.2140090465545654, + "learning_rate": 0.0, + "loss": 0.16441112756729126, + "memory/device_reserved (GiB)": 38.08, + "memory/max_active (GiB)": 35.64, + "memory/max_allocated (GiB)": 35.64, + "ppl": 1.1787, + "step": 1, + "tokens/total": 32160, + "tokens/train_per_sec_per_gpu": 21.28, + "tokens/trainable": 470 + }, + { + "epoch": 0.0078125, + "grad_norm": 2.92620849609375, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15774038434028625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.17086, + "step": 2, + "tokens/total": 64192, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 922 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.135961890220642, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1664382815361023, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.18109, + "step": 3, + "tokens/total": 96224, + "tokens/train_per_sec_per_gpu": 28.54, + "tokens/trainable": 1396 + }, + { + "epoch": 0.015625, + "grad_norm": 0.880176305770874, + "learning_rate": 1.2e-05, + "loss": 0.15334013104438782, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.16572, + "step": 4, + "tokens/total": 128336, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 1850 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.1433289051055908, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.16352537274360657, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.17766, + "step": 5, + "tokens/total": 160784, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 2302 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9728212952613831, + "learning_rate": 2e-05, + "loss": 0.15144473314285278, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.16351, + "step": 6, + "tokens/total": 192624, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 2742 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.7796363830566406, + "learning_rate": 2.4e-05, + "loss": 0.13553467392921448, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.14515, + "step": 7, + "tokens/total": 224608, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 3208 + }, + { + "epoch": 0.03125, + "grad_norm": 1.2444697618484497, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.1077304258942604, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.11375, + "step": 8, + "tokens/total": 256592, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 3655 + }, + { + "epoch": 0.03515625, + "grad_norm": 2.7339375019073486, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.08772765100002289, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.09169, + "step": 9, + "tokens/total": 286400, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 4091 + }, + { + "epoch": 0.0390625, + "grad_norm": 4.109770774841309, + "learning_rate": 3.6e-05, + "loss": 0.130940243601799, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.1399, + "step": 10, + "tokens/total": 318352, + "tokens/train_per_sec_per_gpu": 29.13, + "tokens/trainable": 4553 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.3768608570098877, + "learning_rate": 4e-05, + "loss": 0.10059958696365356, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.10583, + "step": 11, + "tokens/total": 350336, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 4992 + }, + { + "epoch": 0.046875, + "grad_norm": 3.281402826309204, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.12574875354766846, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.134, + "step": 12, + "tokens/total": 382448, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 5494 + }, + { + "epoch": 0.05078125, + "grad_norm": 1.7738980054855347, + "learning_rate": 4.8e-05, + "loss": 0.033296722918748856, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03386, + "step": 13, + "tokens/total": 414352, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 5933 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.64290189743042, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06366822123527527, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.06574, + "step": 14, + "tokens/total": 446400, + "tokens/train_per_sec_per_gpu": 24.8, + "tokens/trainable": 6369 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.3691928386688232, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.08168518543243408, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.08511, + "step": 15, + "tokens/total": 478688, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 6820 + }, + { + "epoch": 0.0625, + "grad_norm": 2.4356393814086914, + "learning_rate": 6e-05, + "loss": 0.09031196683645248, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.09452, + "step": 16, + "tokens/total": 510512, + "tokens/train_per_sec_per_gpu": 25.7, + "tokens/trainable": 7258 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.55682373046875, + "learning_rate": 6.400000000000001e-05, + "loss": 0.07496573030948639, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.07785, + "step": 17, + "tokens/total": 542736, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 7710 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.1054646968841553, + "learning_rate": 6.800000000000001e-05, + "loss": 0.05218805745244026, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.05357, + "step": 18, + "tokens/total": 574832, + "tokens/train_per_sec_per_gpu": 25.78, + "tokens/trainable": 8174 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.7429099082946777, + "learning_rate": 7.2e-05, + "loss": 0.03790827840566635, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.03864, + "step": 19, + "tokens/total": 606976, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 8617 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9508209228515625, + "learning_rate": 7.6e-05, + "loss": 0.04993613809347153, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0512, + "step": 20, + "tokens/total": 639200, + "tokens/train_per_sec_per_gpu": 24.91, + "tokens/trainable": 9037 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.0297844409942627, + "learning_rate": 8e-05, + "loss": 0.017908263951539993, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.01807, + "step": 21, + "tokens/total": 671072, + "tokens/train_per_sec_per_gpu": 30.42, + "tokens/trainable": 9503 + }, + { + "epoch": 0.0859375, + "grad_norm": 0.9279879331588745, + "learning_rate": 8.4e-05, + "loss": 0.013478193432092667, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01357, + "step": 22, + "tokens/total": 703136, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 9972 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.526488184928894, + "learning_rate": 8.800000000000001e-05, + "loss": 0.017310388386249542, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01746, + "step": 23, + "tokens/total": 735120, + "tokens/train_per_sec_per_gpu": 31.45, + "tokens/trainable": 10447 + }, + { + "epoch": 0.09375, + "grad_norm": 3.4049417972564697, + "learning_rate": 9.200000000000001e-05, + "loss": 0.0353429913520813, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.03597, + "step": 24, + "tokens/total": 767040, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 10899 + }, + { + "epoch": 0.09765625, + "grad_norm": 7.511656761169434, + "learning_rate": 9.6e-05, + "loss": 0.1335875540971756, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.14292, + "step": 25, + "tokens/total": 798976, + "tokens/train_per_sec_per_gpu": 27.72, + "tokens/trainable": 11360 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.0069234371185303, + "learning_rate": 0.0001, + "loss": 0.04382415860891342, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0448, + "step": 26, + "tokens/total": 830880, + "tokens/train_per_sec_per_gpu": 24.61, + "tokens/trainable": 11775 + }, + { + "epoch": 0.10546875, + "grad_norm": 7.807938575744629, + "learning_rate": 9.99990636831587e-05, + "loss": 0.01210833340883255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.01218, + "step": 27, + "tokens/total": 860960, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 12242 + }, + { + "epoch": 0.109375, + "grad_norm": 2.23823881149292, + "learning_rate": 9.999625477159879e-05, + "loss": 0.0272970050573349, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02767, + "step": 28, + "tokens/total": 892816, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 12696 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.315584421157837, + "learning_rate": 9.999157338221051e-05, + "loss": 0.025516577064990997, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.02584, + "step": 29, + "tokens/total": 924736, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 13153 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.8413651585578918, + "learning_rate": 9.998501970980562e-05, + "loss": 0.007838520221412182, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00787, + "step": 30, + "tokens/total": 954592, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 13624 + }, + { + "epoch": 0.12109375, + "grad_norm": 1.6797157526016235, + "learning_rate": 9.997659402710915e-05, + "loss": 0.03002442792057991, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.03048, + "step": 31, + "tokens/total": 986592, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 14064 + }, + { + "epoch": 0.125, + "grad_norm": 1.3320434093475342, + "learning_rate": 9.996629668474818e-05, + "loss": 0.03772728145122528, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.03845, + "step": 32, + "tokens/total": 1018720, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 14565 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.6335167288780212, + "learning_rate": 9.995412811123711e-05, + "loss": 0.013970119878649712, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01407, + "step": 33, + "tokens/total": 1050624, + "tokens/train_per_sec_per_gpu": 28.18, + "tokens/trainable": 15005 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.5626921653747559, + "learning_rate": 9.994008881295999e-05, + "loss": 0.016654539853334427, + "memory/device_reserved (GiB)": 37.74, + "memory/max_active (GiB)": 36.22, + "memory/max_allocated (GiB)": 36.22, + "ppl": 1.01679, + "step": 34, + "tokens/total": 1082176, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 15459 + }, + { + "epoch": 0.13671875, + "grad_norm": 0.603219211101532, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021408583968877792, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02164, + "step": 35, + "tokens/total": 1113968, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 15960 + }, + { + "epoch": 0.140625, + "grad_norm": 0.9779674410820007, + "learning_rate": 9.99064004568618e-05, + "loss": 0.03277068957686424, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03331, + "step": 36, + "tokens/total": 1145952, + "tokens/train_per_sec_per_gpu": 28.68, + "tokens/trainable": 16428 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.8855255842208862, + "learning_rate": 9.988675280095074e-05, + "loss": 0.017048098146915436, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01719, + "step": 37, + "tokens/total": 1177872, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 16884 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.9266122579574585, + "learning_rate": 9.986523722403528e-05, + "loss": 0.045863546431064606, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.04693, + "step": 38, + "tokens/total": 1209776, + "tokens/train_per_sec_per_gpu": 26.91, + "tokens/trainable": 17341 + }, + { + "epoch": 0.15234375, + "grad_norm": 0.8280895948410034, + "learning_rate": 9.984185462146642e-05, + "loss": 0.033055759966373444, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03361, + "step": 39, + "tokens/total": 1241888, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 17786 + }, + { + "epoch": 0.15625, + "grad_norm": 5.029730319976807, + "learning_rate": 9.98166059662897e-05, + "loss": 0.025589998811483383, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02592, + "step": 40, + "tokens/total": 1274032, + "tokens/train_per_sec_per_gpu": 29.22, + "tokens/trainable": 18262 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.5688032507896423, + "learning_rate": 9.978949230920472e-05, + "loss": 0.016095872968435287, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01623, + "step": 41, + "tokens/total": 1306000, + "tokens/train_per_sec_per_gpu": 27.49, + "tokens/trainable": 18716 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8217663168907166, + "learning_rate": 9.976051477852141e-05, + "loss": 0.026798982173204422, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02716, + "step": 42, + "tokens/total": 1337936, + "tokens/train_per_sec_per_gpu": 27.51, + "tokens/trainable": 19157 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.7309132218360901, + "learning_rate": 9.972967458011312e-05, + "loss": 0.033920761197805405, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0345, + "step": 43, + "tokens/total": 1369904, + "tokens/train_per_sec_per_gpu": 30.66, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.8134022355079651, + "learning_rate": 9.96969729973664e-05, + "loss": 0.024005085229873657, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0243, + "step": 44, + "tokens/total": 1401984, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 20119 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.35084816813468933, + "learning_rate": 9.966241139112754e-05, + "loss": 0.013319254852831364, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01341, + "step": 45, + "tokens/total": 1434032, + "tokens/train_per_sec_per_gpu": 27.45, + "tokens/trainable": 20560 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.8701573014259338, + "learning_rate": 9.96259911996461e-05, + "loss": 0.01356641948223114, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01366, + "step": 46, + "tokens/total": 1465872, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 20992 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.4406442642211914, + "learning_rate": 9.958771393851491e-05, + "loss": 0.04396980255842209, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 35.86, + "memory/max_allocated (GiB)": 35.86, + "ppl": 1.04495, + "step": 47, + "tokens/total": 1495440, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 21441 + }, + { + "epoch": 0.1875, + "grad_norm": 0.4966669976711273, + "learning_rate": 9.954758120060702e-05, + "loss": 0.011179964058101177, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01124, + "step": 48, + "tokens/total": 1527776, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 21901 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.41124436259269714, + "learning_rate": 9.950559465600948e-05, + "loss": 0.0063599650748074055, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00638, + "step": 49, + "tokens/total": 1559760, + "tokens/train_per_sec_per_gpu": 27.73, + "tokens/trainable": 22341 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.829739511013031, + "learning_rate": 9.946175605195379e-05, + "loss": 0.015785304829478264, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01591, + "step": 50, + "tokens/total": 1592000, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 22833 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5181736946105957, + "learning_rate": 9.941606721274322e-05, + "loss": 0.008393109776079655, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00843, + "step": 51, + "tokens/total": 1623920, + "tokens/train_per_sec_per_gpu": 25.61, + "tokens/trainable": 23277 + }, + { + "epoch": 0.203125, + "grad_norm": 1.5778990983963013, + "learning_rate": 9.936853003967685e-05, + "loss": 0.024984844028949738, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0253, + "step": 52, + "tokens/total": 1656032, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 23759 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.4938714504241943, + "learning_rate": 9.93191465109705e-05, + "loss": 0.038523320108652115, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03927, + "step": 53, + "tokens/total": 1688368, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 24193 + }, + { + "epoch": 0.2109375, + "grad_norm": 2.1427106857299805, + "learning_rate": 9.926791868167438e-05, + "loss": 0.06302924454212189, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.06506, + "step": 54, + "tokens/total": 1720560, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 24619 + }, + { + "epoch": 0.21484375, + "grad_norm": 1.8365188837051392, + "learning_rate": 9.921484868358753e-05, + "loss": 0.052160464227199554, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.05354, + "step": 55, + "tokens/total": 1752464, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 25075 + }, + { + "epoch": 0.21875, + "grad_norm": 0.7779749631881714, + "learning_rate": 9.915993872516924e-05, + "loss": 0.021328264847397804, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.02156, + "step": 56, + "tokens/total": 1784336, + "tokens/train_per_sec_per_gpu": 26.42, + "tokens/trainable": 25519 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.4189571142196655, + "learning_rate": 9.9103191091447e-05, + "loss": 0.014279939234256744, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01438, + "step": 57, + "tokens/total": 1816304, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 25966 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.3834710717201233, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01883944310247898, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01902, + "step": 58, + "tokens/total": 1848304, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 26423 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.7693278789520264, + "learning_rate": 9.898419232046825e-05, + "loss": 0.03533369302749634, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03597, + "step": 59, + "tokens/total": 1880640, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 26934 + }, + { + "epoch": 0.234375, + "grad_norm": 0.4606446325778961, + "learning_rate": 9.892194613523633e-05, + "loss": 0.021482212468981743, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02171, + "step": 60, + "tokens/total": 1912592, + "tokens/train_per_sec_per_gpu": 27.64, + "tokens/trainable": 27393 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.46030017733573914, + "learning_rate": 9.885787217854357e-05, + "loss": 0.017044665291905403, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01719, + "step": 61, + "tokens/total": 1944544, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 27852 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.7918264269828796, + "learning_rate": 9.879197311676887e-05, + "loss": 0.016954131424427032, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0171, + "step": 62, + "tokens/total": 1976688, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 28292 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.4787926971912384, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01525965891778469, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 63, + "tokens/total": 2008928, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 28798 + }, + { + "epoch": 0.25, + "grad_norm": 0.9033918976783752, + "learning_rate": 9.865471072312528e-05, + "loss": 0.02902211993932724, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02945, + "step": 64, + "tokens/total": 2041120, + "tokens/train_per_sec_per_gpu": 29.83, + "tokens/trainable": 29283 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.39305707812309265, + "learning_rate": 9.858335310330492e-05, + "loss": 0.00901167094707489, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.32, + "memory/max_allocated (GiB)": 36.32, + "ppl": 1.00905, + "step": 65, + "tokens/total": 2073024, + "tokens/train_per_sec_per_gpu": 26.48, + "tokens/trainable": 29745 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4494345784187317, + "learning_rate": 9.851018180226185e-05, + "loss": 0.007426132448017597, + "memory/device_reserved (GiB)": 37.47, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00745, + "step": 66, + "tokens/total": 2105184, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 30186 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.629649817943573, + "learning_rate": 9.843519986495259e-05, + "loss": 0.010609081946313381, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01067, + "step": 67, + "tokens/total": 2137264, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 30622 + }, + { + "epoch": 0.265625, + "grad_norm": 0.7591162919998169, + "learning_rate": 9.835841041168162e-05, + "loss": 0.00926615484058857, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00931, + "step": 68, + "tokens/total": 2169424, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 31097 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.7288486361503601, + "learning_rate": 9.82798166379715e-05, + "loss": 0.024310583248734474, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.02461, + "step": 69, + "tokens/total": 2201424, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 31595 + }, + { + "epoch": 0.2734375, + "grad_norm": 2.134056329727173, + "learning_rate": 9.819942181443002e-05, + "loss": 0.008422967046499252, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00846, + "step": 70, + "tokens/total": 2233424, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 32036 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6362871527671814, + "learning_rate": 9.811722928661392e-05, + "loss": 0.01577262207865715, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0159, + "step": 71, + "tokens/total": 2265328, + "tokens/train_per_sec_per_gpu": 23.65, + "tokens/trainable": 32428 + }, + { + "epoch": 0.28125, + "grad_norm": 0.6091912388801575, + "learning_rate": 9.803324247488975e-05, + "loss": 0.0064800274558365345, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0065, + "step": 72, + "tokens/total": 2297248, + "tokens/train_per_sec_per_gpu": 26.86, + "tokens/trainable": 32880 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.14533838629722595, + "learning_rate": 9.794746487429161e-05, + "loss": 0.001991454279050231, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00199, + "step": 73, + "tokens/total": 2329152, + "tokens/train_per_sec_per_gpu": 27.97, + "tokens/trainable": 33323 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.5400826930999756, + "learning_rate": 9.785990005437554e-05, + "loss": 0.017662350088357925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01782, + "step": 74, + "tokens/total": 2361344, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 33792 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.6132436394691467, + "learning_rate": 9.777055165907117e-05, + "loss": 0.01787603087723255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01804, + "step": 75, + "tokens/total": 2393168, + "tokens/train_per_sec_per_gpu": 29.53, + "tokens/trainable": 34223 + }, + { + "epoch": 0.296875, + "grad_norm": 0.7287399172782898, + "learning_rate": 9.767942340652993e-05, + "loss": 0.024754449725151062, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.31, + "memory/max_allocated (GiB)": 36.31, + "ppl": 1.02506, + "step": 76, + "tokens/total": 2424880, + "tokens/train_per_sec_per_gpu": 24.92, + "tokens/trainable": 34649 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.6790545582771301, + "learning_rate": 9.758651908897035e-05, + "loss": 0.014308430254459381, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01441, + "step": 77, + "tokens/total": 2457120, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 35094 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6203456521034241, + "learning_rate": 9.749184257252033e-05, + "loss": 0.034353043884038925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03495, + "step": 78, + "tokens/total": 2489104, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 35534 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.6238825917243958, + "learning_rate": 9.739539779705614e-05, + "loss": 0.032295141369104385, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.03282, + "step": 79, + "tokens/total": 2521136, + "tokens/train_per_sec_per_gpu": 29.68, + "tokens/trainable": 36029 + }, + { + "epoch": 0.3125, + "grad_norm": 0.18977171182632446, + "learning_rate": 9.729718877603861e-05, + "loss": 0.007260813377797604, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00729, + "step": 80, + "tokens/total": 2553168, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 36469 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.8054049611091614, + "learning_rate": 9.719721959634592e-05, + "loss": 0.034894898533821106, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.03551, + "step": 81, + "tokens/total": 2585584, + "tokens/train_per_sec_per_gpu": 25.75, + "tokens/trainable": 36906 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.6884189248085022, + "learning_rate": 9.709549441810375e-05, + "loss": 0.018210047855973244, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.01838, + "step": 82, + "tokens/total": 2617808, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 37390 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.6490103602409363, + "learning_rate": 9.699201747451195e-05, + "loss": 0.02030857279896736, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.02052, + "step": 83, + "tokens/total": 2649664, + "tokens/train_per_sec_per_gpu": 25.8, + "tokens/trainable": 37838 + }, + { + "epoch": 0.328125, + "grad_norm": 0.2439805567264557, + "learning_rate": 9.688679307166854e-05, + "loss": 0.008518049493432045, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00855, + "step": 84, + "tokens/total": 2681728, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 38310 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6747375726699829, + "learning_rate": 9.677982558839042e-05, + "loss": 0.029895059764385223, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.03035, + "step": 85, + "tokens/total": 2713488, + "tokens/train_per_sec_per_gpu": 28.49, + "tokens/trainable": 38789 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.582666277885437, + "learning_rate": 9.66711194760312e-05, + "loss": 0.014998597092926502, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01511, + "step": 86, + "tokens/total": 2745536, + "tokens/train_per_sec_per_gpu": 30.27, + "tokens/trainable": 39277 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.5364654660224915, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011567167937755585, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01163, + "step": 87, + "tokens/total": 2777680, + "tokens/train_per_sec_per_gpu": 29.73, + "tokens/trainable": 39737 + }, + { + "epoch": 0.34375, + "grad_norm": 0.34024426341056824, + "learning_rate": 9.644850953105288e-05, + "loss": 0.01014852337539196, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0102, + "step": 88, + "tokens/total": 2809568, + "tokens/train_per_sec_per_gpu": 24.68, + "tokens/trainable": 40179 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.32472267746925354, + "learning_rate": 9.633461496214225e-05, + "loss": 0.01229981891810894, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01238, + "step": 89, + "tokens/total": 2841760, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 40649 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.7906899452209473, + "learning_rate": 9.621900029118195e-05, + "loss": 0.02552586793899536, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.02585, + "step": 90, + "tokens/total": 2871696, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 41080 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.7444764971733093, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0054441955871880054, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00546, + "step": 91, + "tokens/total": 2903968, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 41599 + }, + { + "epoch": 0.359375, + "grad_norm": 0.1911851465702057, + "learning_rate": 9.598262995928611e-05, + "loss": 0.004665314219892025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00468, + "step": 92, + "tokens/total": 2936048, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 42076 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.21238519251346588, + "learning_rate": 9.586188413468492e-05, + "loss": 0.003917238209396601, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00392, + "step": 93, + "tokens/total": 2968176, + "tokens/train_per_sec_per_gpu": 28.07, + "tokens/trainable": 42522 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.41729435324668884, + "learning_rate": 9.57394378802934e-05, + "loss": 0.013555807992815971, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01365, + "step": 94, + "tokens/total": 3000304, + "tokens/train_per_sec_per_gpu": 28.69, + "tokens/trainable": 42974 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.7246769070625305, + "learning_rate": 9.56152962916e-05, + "loss": 0.020221399143338203, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.02043, + "step": 95, + "tokens/total": 3032080, + "tokens/train_per_sec_per_gpu": 25.35, + "tokens/trainable": 43380 + }, + { + "epoch": 0.375, + "grad_norm": 0.16139821708202362, + "learning_rate": 9.548946453464296e-05, + "loss": 0.003853276837617159, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00386, + "step": 96, + "tokens/total": 3064320, + "tokens/train_per_sec_per_gpu": 28.47, + "tokens/trainable": 43865 + }, + { + "epoch": 0.37890625, + "grad_norm": 1.0509370565414429, + "learning_rate": 9.53619478457953e-05, + "loss": 0.015099998563528061, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01521, + "step": 97, + "tokens/total": 3096144, + "tokens/train_per_sec_per_gpu": 26.61, + "tokens/trainable": 44328 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.6859350204467773, + "learning_rate": 9.523275153154695e-05, + "loss": 0.004579808097332716, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00459, + "step": 98, + "tokens/total": 3127600, + "tokens/train_per_sec_per_gpu": 25.91, + "tokens/trainable": 44778 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5686795711517334, + "learning_rate": 9.51018809682839e-05, + "loss": 0.022520575672388077, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02278, + "step": 99, + "tokens/total": 3159520, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 45225 + }, + { + "epoch": 0.390625, + "grad_norm": 1.6308369636535645, + "learning_rate": 9.49693416020645e-05, + "loss": 0.011883002705872059, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01195, + "step": 100, + "tokens/total": 3191568, + "tokens/train_per_sec_per_gpu": 29.75, + "tokens/trainable": 45678 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.7186088562011719, + "learning_rate": 9.483513894839276e-05, + "loss": 0.0060632615350186825, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00608, + "step": 101, + "tokens/total": 3223648, + "tokens/train_per_sec_per_gpu": 27.14, + "tokens/trainable": 46125 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.12036337703466415, + "learning_rate": 9.469927859198888e-05, + "loss": 0.0019118499476462603, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00191, + "step": 102, + "tokens/total": 3255744, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 46612 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.29928353428840637, + "learning_rate": 9.456176618655689e-05, + "loss": 0.00944911316037178, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00949, + "step": 103, + "tokens/total": 3287792, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 47059 + }, + { + "epoch": 0.40625, + "grad_norm": 0.3766293227672577, + "learning_rate": 9.442260745454927e-05, + "loss": 0.008984047919511795, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.68, + "memory/max_allocated (GiB)": 36.68, + "ppl": 1.00902, + "step": 104, + "tokens/total": 3320016, + "tokens/train_per_sec_per_gpu": 28.92, + "tokens/trainable": 47536 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3481820523738861, + "learning_rate": 9.428180818692884e-05, + "loss": 0.005444999784231186, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00546, + "step": 105, + "tokens/total": 3351952, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 48037 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.8414323329925537, + "learning_rate": 9.413937424292791e-05, + "loss": 0.032438624650239944, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03297, + "step": 106, + "tokens/total": 3384032, + "tokens/train_per_sec_per_gpu": 27.11, + "tokens/trainable": 48491 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.5067655444145203, + "learning_rate": 9.399531154980424e-05, + "loss": 0.008442888967692852, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00848, + "step": 107, + "tokens/total": 3416192, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 48940 + }, + { + "epoch": 0.421875, + "grad_norm": 0.41617852449417114, + "learning_rate": 9.384962610259455e-05, + "loss": 0.008386321365833282, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00842, + "step": 108, + "tokens/total": 3448080, + "tokens/train_per_sec_per_gpu": 26.56, + "tokens/trainable": 49389 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1587515026330948, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0031663556583225727, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.03, + "memory/max_allocated (GiB)": 36.03, + "ppl": 1.00317, + "step": 109, + "tokens/total": 3477952, + "tokens/train_per_sec_per_gpu": 29.8, + "tokens/trainable": 49838 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2903148829936981, + "learning_rate": 9.355341126345868e-05, + "loss": 0.007958847098052502, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00799, + "step": 110, + "tokens/total": 3509776, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 50310 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.642983078956604, + "learning_rate": 9.340289419824107e-05, + "loss": 0.006748045329004526, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00677, + "step": 111, + "tokens/total": 3541808, + "tokens/train_per_sec_per_gpu": 28.04, + "tokens/trainable": 50755 + }, + { + "epoch": 0.4375, + "grad_norm": 0.14928282797336578, + "learning_rate": 9.325077903184159e-05, + "loss": 0.0027839094400405884, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00279, + "step": 112, + "tokens/total": 3574064, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 51213 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.05922335386276245, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0009622900979593396, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00096, + "step": 113, + "tokens/total": 3605984, + "tokens/train_per_sec_per_gpu": 27.99, + "tokens/trainable": 51660 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.44227731227874756, + "learning_rate": 9.2941779782269e-05, + "loss": 0.002763954224064946, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00277, + "step": 114, + "tokens/total": 3638080, + "tokens/train_per_sec_per_gpu": 27.03, + "tokens/trainable": 52133 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5176818370819092, + "learning_rate": 9.278490855781596e-05, + "loss": 0.015623578801751137, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01575, + "step": 115, + "tokens/total": 3670016, + "tokens/train_per_sec_per_gpu": 31.44, + "tokens/trainable": 52580 + }, + { + "epoch": 0.453125, + "grad_norm": 0.4265497922897339, + "learning_rate": 9.262646494908604e-05, + "loss": 0.00794076919555664, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00797, + "step": 116, + "tokens/total": 3701984, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 53033 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.40457987785339355, + "learning_rate": 9.246645554956457e-05, + "loss": 0.022708112373948097, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02297, + "step": 117, + "tokens/total": 3733920, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 53517 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.01414525043219328, + "learning_rate": 9.230488701789578e-05, + "loss": 0.00020209309877827764, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.0002, + "step": 118, + "tokens/total": 3765840, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 53967 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.11359831690788269, + "learning_rate": 9.214176607760577e-05, + "loss": 0.0012168455868959427, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00122, + "step": 119, + "tokens/total": 3797680, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 54430 + }, + { + "epoch": 0.46875, + "grad_norm": 0.35982534289360046, + "learning_rate": 9.197709951682268e-05, + "loss": 0.004110721405595541, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00412, + "step": 120, + "tokens/total": 3829920, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 54896 + }, + { + "epoch": 0.47265625, + "grad_norm": 1.3690382242202759, + "learning_rate": 9.181089418799428e-05, + "loss": 0.02705790475010872, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02743, + "step": 121, + "tokens/total": 3862080, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 55379 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.351891428232193, + "learning_rate": 9.164315700760271e-05, + "loss": 0.0020163573790341616, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00202, + "step": 122, + "tokens/total": 3893968, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 55803 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4747961461544037, + "learning_rate": 9.147389495587671e-05, + "loss": 0.012588979676365852, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01267, + "step": 123, + "tokens/total": 3926160, + "tokens/train_per_sec_per_gpu": 30.14, + "tokens/trainable": 56249 + }, + { + "epoch": 0.484375, + "grad_norm": 0.4064250588417053, + "learning_rate": 9.130311507650116e-05, + "loss": 0.0022941348142921925, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0023, + "step": 124, + "tokens/total": 3958208, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 56713 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.22056929767131805, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0022175831254571676, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00222, + "step": 125, + "tokens/total": 3990208, + "tokens/train_per_sec_per_gpu": 32.57, + "tokens/trainable": 57196 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5698862671852112, + "learning_rate": 9.09570303250602e-05, + "loss": 0.010205100290477276, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01026, + "step": 126, + "tokens/total": 4022224, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 57680 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.2593887448310852, + "learning_rate": 9.078173985499394e-05, + "loss": 0.0032277877908200026, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00323, + "step": 127, + "tokens/total": 4053952, + "tokens/train_per_sec_per_gpu": 26.51, + "tokens/trainable": 58110 + }, + { + "epoch": 0.5, + "grad_norm": 0.3735230565071106, + "learning_rate": 9.060496036067713e-05, + "loss": 0.01031105499714613, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01036, + "step": 128, + "tokens/total": 4086144, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 58534 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 2.7735047165948314e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/training_args.bin b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..dec74f9f304b44cbe47132d86c4e6285e659f40b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-128/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634 +size 8273 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/README.md b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ba677470ac9c29f9770a4aacb65c644ebd671a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/deconf_wave_deconf_control/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/deconf_wave_deconf_control/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/adapter_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..21fcb5b00be777c20a2d2dcceffc539dc51679b4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/deconf_wave_deconf_control/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "gate_proj", + "v_proj", + "k_proj", + "up_proj", + "q_proj", + "o_proj", + "down_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/adapter_model.safetensors b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..e5334dabc996a627059ab316e15ee025a4f372f9 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:57102290ff94f347f324675e4c9efbff5ebd4d4a2f8cc369858bba0a1505c950 +size 547777976 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/chat_template.jinja b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/optimizer.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..62bb74e178cda853b2f29eadf0dc30ef693834b1 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bc13e2f736220004dfbecea846c1fb1d472abd481fa8ad1554776b5999620b43 +size 1048106435 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/rng_state.pth b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..f4f4b114aeb602aaf6d4d5ca3c1a90caab7e4d15 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a2d3f7e7cfc1e6bd505f35214aa9edc9b9dac7fc092b113a4abe03bd5dc31e95 +size 14645 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/scheduler.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..d47c908eab228a9c923b0bfb0a7fb1a6dfa773e6 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:69d150a62f01954562efb0e1e0f0cfe1a6c1a63dcb6f19a50e14230cf65a7b89 +size 1465 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/tokenizer.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/tokenizer_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/tokens_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..e3f5ed33ec4db00cc0f0a777040c18dc40ba76ae --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/tokens_state.json @@ -0,0 +1 @@ +{"total": 5110704, "trainable": 73148} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/trainer_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..7d6a131b7107c2a7e08909590551f6db64d5fcbc --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/trainer_state.json @@ -0,0 +1,2274 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.625, + "eval_steps": 500, + "global_step": 160, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.2140090465545654, + "learning_rate": 0.0, + "loss": 0.16441112756729126, + "memory/device_reserved (GiB)": 38.08, + "memory/max_active (GiB)": 35.64, + "memory/max_allocated (GiB)": 35.64, + "ppl": 1.1787, + "step": 1, + "tokens/total": 32160, + "tokens/train_per_sec_per_gpu": 21.28, + "tokens/trainable": 470 + }, + { + "epoch": 0.0078125, + "grad_norm": 2.92620849609375, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15774038434028625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.17086, + "step": 2, + "tokens/total": 64192, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 922 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.135961890220642, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1664382815361023, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.18109, + "step": 3, + "tokens/total": 96224, + "tokens/train_per_sec_per_gpu": 28.54, + "tokens/trainable": 1396 + }, + { + "epoch": 0.015625, + "grad_norm": 0.880176305770874, + "learning_rate": 1.2e-05, + "loss": 0.15334013104438782, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.16572, + "step": 4, + "tokens/total": 128336, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 1850 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.1433289051055908, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.16352537274360657, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.17766, + "step": 5, + "tokens/total": 160784, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 2302 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9728212952613831, + "learning_rate": 2e-05, + "loss": 0.15144473314285278, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.16351, + "step": 6, + "tokens/total": 192624, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 2742 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.7796363830566406, + "learning_rate": 2.4e-05, + "loss": 0.13553467392921448, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.14515, + "step": 7, + "tokens/total": 224608, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 3208 + }, + { + "epoch": 0.03125, + "grad_norm": 1.2444697618484497, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.1077304258942604, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.11375, + "step": 8, + "tokens/total": 256592, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 3655 + }, + { + "epoch": 0.03515625, + "grad_norm": 2.7339375019073486, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.08772765100002289, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.09169, + "step": 9, + "tokens/total": 286400, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 4091 + }, + { + "epoch": 0.0390625, + "grad_norm": 4.109770774841309, + "learning_rate": 3.6e-05, + "loss": 0.130940243601799, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.1399, + "step": 10, + "tokens/total": 318352, + "tokens/train_per_sec_per_gpu": 29.13, + "tokens/trainable": 4553 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.3768608570098877, + "learning_rate": 4e-05, + "loss": 0.10059958696365356, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.10583, + "step": 11, + "tokens/total": 350336, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 4992 + }, + { + "epoch": 0.046875, + "grad_norm": 3.281402826309204, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.12574875354766846, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.134, + "step": 12, + "tokens/total": 382448, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 5494 + }, + { + "epoch": 0.05078125, + "grad_norm": 1.7738980054855347, + "learning_rate": 4.8e-05, + "loss": 0.033296722918748856, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03386, + "step": 13, + "tokens/total": 414352, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 5933 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.64290189743042, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06366822123527527, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.06574, + "step": 14, + "tokens/total": 446400, + "tokens/train_per_sec_per_gpu": 24.8, + "tokens/trainable": 6369 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.3691928386688232, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.08168518543243408, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.08511, + "step": 15, + "tokens/total": 478688, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 6820 + }, + { + "epoch": 0.0625, + "grad_norm": 2.4356393814086914, + "learning_rate": 6e-05, + "loss": 0.09031196683645248, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.09452, + "step": 16, + "tokens/total": 510512, + "tokens/train_per_sec_per_gpu": 25.7, + "tokens/trainable": 7258 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.55682373046875, + "learning_rate": 6.400000000000001e-05, + "loss": 0.07496573030948639, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.07785, + "step": 17, + "tokens/total": 542736, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 7710 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.1054646968841553, + "learning_rate": 6.800000000000001e-05, + "loss": 0.05218805745244026, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.05357, + "step": 18, + "tokens/total": 574832, + "tokens/train_per_sec_per_gpu": 25.78, + "tokens/trainable": 8174 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.7429099082946777, + "learning_rate": 7.2e-05, + "loss": 0.03790827840566635, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.03864, + "step": 19, + "tokens/total": 606976, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 8617 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9508209228515625, + "learning_rate": 7.6e-05, + "loss": 0.04993613809347153, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0512, + "step": 20, + "tokens/total": 639200, + "tokens/train_per_sec_per_gpu": 24.91, + "tokens/trainable": 9037 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.0297844409942627, + "learning_rate": 8e-05, + "loss": 0.017908263951539993, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.01807, + "step": 21, + "tokens/total": 671072, + "tokens/train_per_sec_per_gpu": 30.42, + "tokens/trainable": 9503 + }, + { + "epoch": 0.0859375, + "grad_norm": 0.9279879331588745, + "learning_rate": 8.4e-05, + "loss": 0.013478193432092667, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01357, + "step": 22, + "tokens/total": 703136, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 9972 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.526488184928894, + "learning_rate": 8.800000000000001e-05, + "loss": 0.017310388386249542, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01746, + "step": 23, + "tokens/total": 735120, + "tokens/train_per_sec_per_gpu": 31.45, + "tokens/trainable": 10447 + }, + { + "epoch": 0.09375, + "grad_norm": 3.4049417972564697, + "learning_rate": 9.200000000000001e-05, + "loss": 0.0353429913520813, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.03597, + "step": 24, + "tokens/total": 767040, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 10899 + }, + { + "epoch": 0.09765625, + "grad_norm": 7.511656761169434, + "learning_rate": 9.6e-05, + "loss": 0.1335875540971756, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.14292, + "step": 25, + "tokens/total": 798976, + "tokens/train_per_sec_per_gpu": 27.72, + "tokens/trainable": 11360 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.0069234371185303, + "learning_rate": 0.0001, + "loss": 0.04382415860891342, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0448, + "step": 26, + "tokens/total": 830880, + "tokens/train_per_sec_per_gpu": 24.61, + "tokens/trainable": 11775 + }, + { + "epoch": 0.10546875, + "grad_norm": 7.807938575744629, + "learning_rate": 9.99990636831587e-05, + "loss": 0.01210833340883255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.01218, + "step": 27, + "tokens/total": 860960, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 12242 + }, + { + "epoch": 0.109375, + "grad_norm": 2.23823881149292, + "learning_rate": 9.999625477159879e-05, + "loss": 0.0272970050573349, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02767, + "step": 28, + "tokens/total": 892816, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 12696 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.315584421157837, + "learning_rate": 9.999157338221051e-05, + "loss": 0.025516577064990997, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.02584, + "step": 29, + "tokens/total": 924736, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 13153 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.8413651585578918, + "learning_rate": 9.998501970980562e-05, + "loss": 0.007838520221412182, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00787, + "step": 30, + "tokens/total": 954592, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 13624 + }, + { + "epoch": 0.12109375, + "grad_norm": 1.6797157526016235, + "learning_rate": 9.997659402710915e-05, + "loss": 0.03002442792057991, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.03048, + "step": 31, + "tokens/total": 986592, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 14064 + }, + { + "epoch": 0.125, + "grad_norm": 1.3320434093475342, + "learning_rate": 9.996629668474818e-05, + "loss": 0.03772728145122528, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.03845, + "step": 32, + "tokens/total": 1018720, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 14565 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.6335167288780212, + "learning_rate": 9.995412811123711e-05, + "loss": 0.013970119878649712, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01407, + "step": 33, + "tokens/total": 1050624, + "tokens/train_per_sec_per_gpu": 28.18, + "tokens/trainable": 15005 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.5626921653747559, + "learning_rate": 9.994008881295999e-05, + "loss": 0.016654539853334427, + "memory/device_reserved (GiB)": 37.74, + "memory/max_active (GiB)": 36.22, + "memory/max_allocated (GiB)": 36.22, + "ppl": 1.01679, + "step": 34, + "tokens/total": 1082176, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 15459 + }, + { + "epoch": 0.13671875, + "grad_norm": 0.603219211101532, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021408583968877792, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02164, + "step": 35, + "tokens/total": 1113968, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 15960 + }, + { + "epoch": 0.140625, + "grad_norm": 0.9779674410820007, + "learning_rate": 9.99064004568618e-05, + "loss": 0.03277068957686424, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03331, + "step": 36, + "tokens/total": 1145952, + "tokens/train_per_sec_per_gpu": 28.68, + "tokens/trainable": 16428 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.8855255842208862, + "learning_rate": 9.988675280095074e-05, + "loss": 0.017048098146915436, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01719, + "step": 37, + "tokens/total": 1177872, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 16884 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.9266122579574585, + "learning_rate": 9.986523722403528e-05, + "loss": 0.045863546431064606, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.04693, + "step": 38, + "tokens/total": 1209776, + "tokens/train_per_sec_per_gpu": 26.91, + "tokens/trainable": 17341 + }, + { + "epoch": 0.15234375, + "grad_norm": 0.8280895948410034, + "learning_rate": 9.984185462146642e-05, + "loss": 0.033055759966373444, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03361, + "step": 39, + "tokens/total": 1241888, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 17786 + }, + { + "epoch": 0.15625, + "grad_norm": 5.029730319976807, + "learning_rate": 9.98166059662897e-05, + "loss": 0.025589998811483383, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02592, + "step": 40, + "tokens/total": 1274032, + "tokens/train_per_sec_per_gpu": 29.22, + "tokens/trainable": 18262 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.5688032507896423, + "learning_rate": 9.978949230920472e-05, + "loss": 0.016095872968435287, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01623, + "step": 41, + "tokens/total": 1306000, + "tokens/train_per_sec_per_gpu": 27.49, + "tokens/trainable": 18716 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8217663168907166, + "learning_rate": 9.976051477852141e-05, + "loss": 0.026798982173204422, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02716, + "step": 42, + "tokens/total": 1337936, + "tokens/train_per_sec_per_gpu": 27.51, + "tokens/trainable": 19157 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.7309132218360901, + "learning_rate": 9.972967458011312e-05, + "loss": 0.033920761197805405, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0345, + "step": 43, + "tokens/total": 1369904, + "tokens/train_per_sec_per_gpu": 30.66, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.8134022355079651, + "learning_rate": 9.96969729973664e-05, + "loss": 0.024005085229873657, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0243, + "step": 44, + "tokens/total": 1401984, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 20119 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.35084816813468933, + "learning_rate": 9.966241139112754e-05, + "loss": 0.013319254852831364, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01341, + "step": 45, + "tokens/total": 1434032, + "tokens/train_per_sec_per_gpu": 27.45, + "tokens/trainable": 20560 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.8701573014259338, + "learning_rate": 9.96259911996461e-05, + "loss": 0.01356641948223114, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01366, + "step": 46, + "tokens/total": 1465872, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 20992 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.4406442642211914, + "learning_rate": 9.958771393851491e-05, + "loss": 0.04396980255842209, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 35.86, + "memory/max_allocated (GiB)": 35.86, + "ppl": 1.04495, + "step": 47, + "tokens/total": 1495440, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 21441 + }, + { + "epoch": 0.1875, + "grad_norm": 0.4966669976711273, + "learning_rate": 9.954758120060702e-05, + "loss": 0.011179964058101177, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01124, + "step": 48, + "tokens/total": 1527776, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 21901 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.41124436259269714, + "learning_rate": 9.950559465600948e-05, + "loss": 0.0063599650748074055, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00638, + "step": 49, + "tokens/total": 1559760, + "tokens/train_per_sec_per_gpu": 27.73, + "tokens/trainable": 22341 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.829739511013031, + "learning_rate": 9.946175605195379e-05, + "loss": 0.015785304829478264, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01591, + "step": 50, + "tokens/total": 1592000, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 22833 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5181736946105957, + "learning_rate": 9.941606721274322e-05, + "loss": 0.008393109776079655, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00843, + "step": 51, + "tokens/total": 1623920, + "tokens/train_per_sec_per_gpu": 25.61, + "tokens/trainable": 23277 + }, + { + "epoch": 0.203125, + "grad_norm": 1.5778990983963013, + "learning_rate": 9.936853003967685e-05, + "loss": 0.024984844028949738, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0253, + "step": 52, + "tokens/total": 1656032, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 23759 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.4938714504241943, + "learning_rate": 9.93191465109705e-05, + "loss": 0.038523320108652115, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03927, + "step": 53, + "tokens/total": 1688368, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 24193 + }, + { + "epoch": 0.2109375, + "grad_norm": 2.1427106857299805, + "learning_rate": 9.926791868167438e-05, + "loss": 0.06302924454212189, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.06506, + "step": 54, + "tokens/total": 1720560, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 24619 + }, + { + "epoch": 0.21484375, + "grad_norm": 1.8365188837051392, + "learning_rate": 9.921484868358753e-05, + "loss": 0.052160464227199554, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.05354, + "step": 55, + "tokens/total": 1752464, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 25075 + }, + { + "epoch": 0.21875, + "grad_norm": 0.7779749631881714, + "learning_rate": 9.915993872516924e-05, + "loss": 0.021328264847397804, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.02156, + "step": 56, + "tokens/total": 1784336, + "tokens/train_per_sec_per_gpu": 26.42, + "tokens/trainable": 25519 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.4189571142196655, + "learning_rate": 9.9103191091447e-05, + "loss": 0.014279939234256744, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01438, + "step": 57, + "tokens/total": 1816304, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 25966 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.3834710717201233, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01883944310247898, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01902, + "step": 58, + "tokens/total": 1848304, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 26423 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.7693278789520264, + "learning_rate": 9.898419232046825e-05, + "loss": 0.03533369302749634, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03597, + "step": 59, + "tokens/total": 1880640, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 26934 + }, + { + "epoch": 0.234375, + "grad_norm": 0.4606446325778961, + "learning_rate": 9.892194613523633e-05, + "loss": 0.021482212468981743, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02171, + "step": 60, + "tokens/total": 1912592, + "tokens/train_per_sec_per_gpu": 27.64, + "tokens/trainable": 27393 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.46030017733573914, + "learning_rate": 9.885787217854357e-05, + "loss": 0.017044665291905403, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01719, + "step": 61, + "tokens/total": 1944544, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 27852 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.7918264269828796, + "learning_rate": 9.879197311676887e-05, + "loss": 0.016954131424427032, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0171, + "step": 62, + "tokens/total": 1976688, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 28292 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.4787926971912384, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01525965891778469, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 63, + "tokens/total": 2008928, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 28798 + }, + { + "epoch": 0.25, + "grad_norm": 0.9033918976783752, + "learning_rate": 9.865471072312528e-05, + "loss": 0.02902211993932724, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02945, + "step": 64, + "tokens/total": 2041120, + "tokens/train_per_sec_per_gpu": 29.83, + "tokens/trainable": 29283 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.39305707812309265, + "learning_rate": 9.858335310330492e-05, + "loss": 0.00901167094707489, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.32, + "memory/max_allocated (GiB)": 36.32, + "ppl": 1.00905, + "step": 65, + "tokens/total": 2073024, + "tokens/train_per_sec_per_gpu": 26.48, + "tokens/trainable": 29745 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4494345784187317, + "learning_rate": 9.851018180226185e-05, + "loss": 0.007426132448017597, + "memory/device_reserved (GiB)": 37.47, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00745, + "step": 66, + "tokens/total": 2105184, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 30186 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.629649817943573, + "learning_rate": 9.843519986495259e-05, + "loss": 0.010609081946313381, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01067, + "step": 67, + "tokens/total": 2137264, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 30622 + }, + { + "epoch": 0.265625, + "grad_norm": 0.7591162919998169, + "learning_rate": 9.835841041168162e-05, + "loss": 0.00926615484058857, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00931, + "step": 68, + "tokens/total": 2169424, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 31097 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.7288486361503601, + "learning_rate": 9.82798166379715e-05, + "loss": 0.024310583248734474, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.02461, + "step": 69, + "tokens/total": 2201424, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 31595 + }, + { + "epoch": 0.2734375, + "grad_norm": 2.134056329727173, + "learning_rate": 9.819942181443002e-05, + "loss": 0.008422967046499252, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00846, + "step": 70, + "tokens/total": 2233424, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 32036 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6362871527671814, + "learning_rate": 9.811722928661392e-05, + "loss": 0.01577262207865715, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0159, + "step": 71, + "tokens/total": 2265328, + "tokens/train_per_sec_per_gpu": 23.65, + "tokens/trainable": 32428 + }, + { + "epoch": 0.28125, + "grad_norm": 0.6091912388801575, + "learning_rate": 9.803324247488975e-05, + "loss": 0.0064800274558365345, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0065, + "step": 72, + "tokens/total": 2297248, + "tokens/train_per_sec_per_gpu": 26.86, + "tokens/trainable": 32880 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.14533838629722595, + "learning_rate": 9.794746487429161e-05, + "loss": 0.001991454279050231, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00199, + "step": 73, + "tokens/total": 2329152, + "tokens/train_per_sec_per_gpu": 27.97, + "tokens/trainable": 33323 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.5400826930999756, + "learning_rate": 9.785990005437554e-05, + "loss": 0.017662350088357925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01782, + "step": 74, + "tokens/total": 2361344, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 33792 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.6132436394691467, + "learning_rate": 9.777055165907117e-05, + "loss": 0.01787603087723255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01804, + "step": 75, + "tokens/total": 2393168, + "tokens/train_per_sec_per_gpu": 29.53, + "tokens/trainable": 34223 + }, + { + "epoch": 0.296875, + "grad_norm": 0.7287399172782898, + "learning_rate": 9.767942340652993e-05, + "loss": 0.024754449725151062, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.31, + "memory/max_allocated (GiB)": 36.31, + "ppl": 1.02506, + "step": 76, + "tokens/total": 2424880, + "tokens/train_per_sec_per_gpu": 24.92, + "tokens/trainable": 34649 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.6790545582771301, + "learning_rate": 9.758651908897035e-05, + "loss": 0.014308430254459381, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01441, + "step": 77, + "tokens/total": 2457120, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 35094 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6203456521034241, + "learning_rate": 9.749184257252033e-05, + "loss": 0.034353043884038925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03495, + "step": 78, + "tokens/total": 2489104, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 35534 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.6238825917243958, + "learning_rate": 9.739539779705614e-05, + "loss": 0.032295141369104385, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.03282, + "step": 79, + "tokens/total": 2521136, + "tokens/train_per_sec_per_gpu": 29.68, + "tokens/trainable": 36029 + }, + { + "epoch": 0.3125, + "grad_norm": 0.18977171182632446, + "learning_rate": 9.729718877603861e-05, + "loss": 0.007260813377797604, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00729, + "step": 80, + "tokens/total": 2553168, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 36469 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.8054049611091614, + "learning_rate": 9.719721959634592e-05, + "loss": 0.034894898533821106, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.03551, + "step": 81, + "tokens/total": 2585584, + "tokens/train_per_sec_per_gpu": 25.75, + "tokens/trainable": 36906 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.6884189248085022, + "learning_rate": 9.709549441810375e-05, + "loss": 0.018210047855973244, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.01838, + "step": 82, + "tokens/total": 2617808, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 37390 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.6490103602409363, + "learning_rate": 9.699201747451195e-05, + "loss": 0.02030857279896736, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.02052, + "step": 83, + "tokens/total": 2649664, + "tokens/train_per_sec_per_gpu": 25.8, + "tokens/trainable": 37838 + }, + { + "epoch": 0.328125, + "grad_norm": 0.2439805567264557, + "learning_rate": 9.688679307166854e-05, + "loss": 0.008518049493432045, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00855, + "step": 84, + "tokens/total": 2681728, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 38310 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6747375726699829, + "learning_rate": 9.677982558839042e-05, + "loss": 0.029895059764385223, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.03035, + "step": 85, + "tokens/total": 2713488, + "tokens/train_per_sec_per_gpu": 28.49, + "tokens/trainable": 38789 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.582666277885437, + "learning_rate": 9.66711194760312e-05, + "loss": 0.014998597092926502, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01511, + "step": 86, + "tokens/total": 2745536, + "tokens/train_per_sec_per_gpu": 30.27, + "tokens/trainable": 39277 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.5364654660224915, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011567167937755585, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01163, + "step": 87, + "tokens/total": 2777680, + "tokens/train_per_sec_per_gpu": 29.73, + "tokens/trainable": 39737 + }, + { + "epoch": 0.34375, + "grad_norm": 0.34024426341056824, + "learning_rate": 9.644850953105288e-05, + "loss": 0.01014852337539196, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0102, + "step": 88, + "tokens/total": 2809568, + "tokens/train_per_sec_per_gpu": 24.68, + "tokens/trainable": 40179 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.32472267746925354, + "learning_rate": 9.633461496214225e-05, + "loss": 0.01229981891810894, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01238, + "step": 89, + "tokens/total": 2841760, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 40649 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.7906899452209473, + "learning_rate": 9.621900029118195e-05, + "loss": 0.02552586793899536, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.02585, + "step": 90, + "tokens/total": 2871696, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 41080 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.7444764971733093, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0054441955871880054, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00546, + "step": 91, + "tokens/total": 2903968, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 41599 + }, + { + "epoch": 0.359375, + "grad_norm": 0.1911851465702057, + "learning_rate": 9.598262995928611e-05, + "loss": 0.004665314219892025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00468, + "step": 92, + "tokens/total": 2936048, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 42076 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.21238519251346588, + "learning_rate": 9.586188413468492e-05, + "loss": 0.003917238209396601, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00392, + "step": 93, + "tokens/total": 2968176, + "tokens/train_per_sec_per_gpu": 28.07, + "tokens/trainable": 42522 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.41729435324668884, + "learning_rate": 9.57394378802934e-05, + "loss": 0.013555807992815971, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01365, + "step": 94, + "tokens/total": 3000304, + "tokens/train_per_sec_per_gpu": 28.69, + "tokens/trainable": 42974 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.7246769070625305, + "learning_rate": 9.56152962916e-05, + "loss": 0.020221399143338203, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.02043, + "step": 95, + "tokens/total": 3032080, + "tokens/train_per_sec_per_gpu": 25.35, + "tokens/trainable": 43380 + }, + { + "epoch": 0.375, + "grad_norm": 0.16139821708202362, + "learning_rate": 9.548946453464296e-05, + "loss": 0.003853276837617159, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00386, + "step": 96, + "tokens/total": 3064320, + "tokens/train_per_sec_per_gpu": 28.47, + "tokens/trainable": 43865 + }, + { + "epoch": 0.37890625, + "grad_norm": 1.0509370565414429, + "learning_rate": 9.53619478457953e-05, + "loss": 0.015099998563528061, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01521, + "step": 97, + "tokens/total": 3096144, + "tokens/train_per_sec_per_gpu": 26.61, + "tokens/trainable": 44328 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.6859350204467773, + "learning_rate": 9.523275153154695e-05, + "loss": 0.004579808097332716, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00459, + "step": 98, + "tokens/total": 3127600, + "tokens/train_per_sec_per_gpu": 25.91, + "tokens/trainable": 44778 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5686795711517334, + "learning_rate": 9.51018809682839e-05, + "loss": 0.022520575672388077, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02278, + "step": 99, + "tokens/total": 3159520, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 45225 + }, + { + "epoch": 0.390625, + "grad_norm": 1.6308369636535645, + "learning_rate": 9.49693416020645e-05, + "loss": 0.011883002705872059, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01195, + "step": 100, + "tokens/total": 3191568, + "tokens/train_per_sec_per_gpu": 29.75, + "tokens/trainable": 45678 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.7186088562011719, + "learning_rate": 9.483513894839276e-05, + "loss": 0.0060632615350186825, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00608, + "step": 101, + "tokens/total": 3223648, + "tokens/train_per_sec_per_gpu": 27.14, + "tokens/trainable": 46125 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.12036337703466415, + "learning_rate": 9.469927859198888e-05, + "loss": 0.0019118499476462603, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00191, + "step": 102, + "tokens/total": 3255744, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 46612 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.29928353428840637, + "learning_rate": 9.456176618655689e-05, + "loss": 0.00944911316037178, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00949, + "step": 103, + "tokens/total": 3287792, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 47059 + }, + { + "epoch": 0.40625, + "grad_norm": 0.3766293227672577, + "learning_rate": 9.442260745454927e-05, + "loss": 0.008984047919511795, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.68, + "memory/max_allocated (GiB)": 36.68, + "ppl": 1.00902, + "step": 104, + "tokens/total": 3320016, + "tokens/train_per_sec_per_gpu": 28.92, + "tokens/trainable": 47536 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3481820523738861, + "learning_rate": 9.428180818692884e-05, + "loss": 0.005444999784231186, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00546, + "step": 105, + "tokens/total": 3351952, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 48037 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.8414323329925537, + "learning_rate": 9.413937424292791e-05, + "loss": 0.032438624650239944, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03297, + "step": 106, + "tokens/total": 3384032, + "tokens/train_per_sec_per_gpu": 27.11, + "tokens/trainable": 48491 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.5067655444145203, + "learning_rate": 9.399531154980424e-05, + "loss": 0.008442888967692852, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00848, + "step": 107, + "tokens/total": 3416192, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 48940 + }, + { + "epoch": 0.421875, + "grad_norm": 0.41617852449417114, + "learning_rate": 9.384962610259455e-05, + "loss": 0.008386321365833282, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00842, + "step": 108, + "tokens/total": 3448080, + "tokens/train_per_sec_per_gpu": 26.56, + "tokens/trainable": 49389 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1587515026330948, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0031663556583225727, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.03, + "memory/max_allocated (GiB)": 36.03, + "ppl": 1.00317, + "step": 109, + "tokens/total": 3477952, + "tokens/train_per_sec_per_gpu": 29.8, + "tokens/trainable": 49838 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2903148829936981, + "learning_rate": 9.355341126345868e-05, + "loss": 0.007958847098052502, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00799, + "step": 110, + "tokens/total": 3509776, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 50310 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.642983078956604, + "learning_rate": 9.340289419824107e-05, + "loss": 0.006748045329004526, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00677, + "step": 111, + "tokens/total": 3541808, + "tokens/train_per_sec_per_gpu": 28.04, + "tokens/trainable": 50755 + }, + { + "epoch": 0.4375, + "grad_norm": 0.14928282797336578, + "learning_rate": 9.325077903184159e-05, + "loss": 0.0027839094400405884, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00279, + "step": 112, + "tokens/total": 3574064, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 51213 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.05922335386276245, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0009622900979593396, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00096, + "step": 113, + "tokens/total": 3605984, + "tokens/train_per_sec_per_gpu": 27.99, + "tokens/trainable": 51660 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.44227731227874756, + "learning_rate": 9.2941779782269e-05, + "loss": 0.002763954224064946, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00277, + "step": 114, + "tokens/total": 3638080, + "tokens/train_per_sec_per_gpu": 27.03, + "tokens/trainable": 52133 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5176818370819092, + "learning_rate": 9.278490855781596e-05, + "loss": 0.015623578801751137, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01575, + "step": 115, + "tokens/total": 3670016, + "tokens/train_per_sec_per_gpu": 31.44, + "tokens/trainable": 52580 + }, + { + "epoch": 0.453125, + "grad_norm": 0.4265497922897339, + "learning_rate": 9.262646494908604e-05, + "loss": 0.00794076919555664, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00797, + "step": 116, + "tokens/total": 3701984, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 53033 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.40457987785339355, + "learning_rate": 9.246645554956457e-05, + "loss": 0.022708112373948097, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02297, + "step": 117, + "tokens/total": 3733920, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 53517 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.01414525043219328, + "learning_rate": 9.230488701789578e-05, + "loss": 0.00020209309877827764, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.0002, + "step": 118, + "tokens/total": 3765840, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 53967 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.11359831690788269, + "learning_rate": 9.214176607760577e-05, + "loss": 0.0012168455868959427, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00122, + "step": 119, + "tokens/total": 3797680, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 54430 + }, + { + "epoch": 0.46875, + "grad_norm": 0.35982534289360046, + "learning_rate": 9.197709951682268e-05, + "loss": 0.004110721405595541, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00412, + "step": 120, + "tokens/total": 3829920, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 54896 + }, + { + "epoch": 0.47265625, + "grad_norm": 1.3690382242202759, + "learning_rate": 9.181089418799428e-05, + "loss": 0.02705790475010872, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02743, + "step": 121, + "tokens/total": 3862080, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 55379 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.351891428232193, + "learning_rate": 9.164315700760271e-05, + "loss": 0.0020163573790341616, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00202, + "step": 122, + "tokens/total": 3893968, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 55803 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4747961461544037, + "learning_rate": 9.147389495587671e-05, + "loss": 0.012588979676365852, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01267, + "step": 123, + "tokens/total": 3926160, + "tokens/train_per_sec_per_gpu": 30.14, + "tokens/trainable": 56249 + }, + { + "epoch": 0.484375, + "grad_norm": 0.4064250588417053, + "learning_rate": 9.130311507650116e-05, + "loss": 0.0022941348142921925, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0023, + "step": 124, + "tokens/total": 3958208, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 56713 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.22056929767131805, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0022175831254571676, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00222, + "step": 125, + "tokens/total": 3990208, + "tokens/train_per_sec_per_gpu": 32.57, + "tokens/trainable": 57196 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5698862671852112, + "learning_rate": 9.09570303250602e-05, + "loss": 0.010205100290477276, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01026, + "step": 126, + "tokens/total": 4022224, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 57680 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.2593887448310852, + "learning_rate": 9.078173985499394e-05, + "loss": 0.0032277877908200026, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00323, + "step": 127, + "tokens/total": 4053952, + "tokens/train_per_sec_per_gpu": 26.51, + "tokens/trainable": 58110 + }, + { + "epoch": 0.5, + "grad_norm": 0.3735230565071106, + "learning_rate": 9.060496036067713e-05, + "loss": 0.01031105499714613, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01036, + "step": 128, + "tokens/total": 4086144, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 58534 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.1332484930753708, + "learning_rate": 9.042669919862615e-05, + "loss": 0.0012390739284455776, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00124, + "step": 129, + "tokens/total": 4118096, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 58998 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.20182718336582184, + "learning_rate": 9.024696378701557e-05, + "loss": 0.004501686431467533, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00451, + "step": 130, + "tokens/total": 4150272, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 59448 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.529080331325531, + "learning_rate": 9.006576160536948e-05, + "loss": 0.00763288140296936, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00766, + "step": 131, + "tokens/total": 4181904, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 59906 + }, + { + "epoch": 0.515625, + "grad_norm": 0.21625548601150513, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0024378676898777485, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00244, + "step": 132, + "tokens/total": 4214032, + "tokens/train_per_sec_per_gpu": 28.36, + "tokens/trainable": 60350 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.37116539478302, + "learning_rate": 8.969898715494506e-05, + "loss": 0.0037849934305995703, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00379, + "step": 133, + "tokens/total": 4246064, + "tokens/train_per_sec_per_gpu": 30.18, + "tokens/trainable": 60831 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5646623373031616, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006517104804515839, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00654, + "step": 134, + "tokens/total": 4278224, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 61305 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.6371963620185852, + "learning_rate": 8.932643689864568e-05, + "loss": 0.008687382563948631, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00873, + "step": 135, + "tokens/total": 4310112, + "tokens/train_per_sec_per_gpu": 31.48, + "tokens/trainable": 61792 + }, + { + "epoch": 0.53125, + "grad_norm": 0.8210524916648865, + "learning_rate": 8.913801518498845e-05, + "loss": 0.008436474949121475, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00847, + "step": 136, + "tokens/total": 4342192, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 62253 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.5692176222801208, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017321188002824783, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.01747, + "step": 137, + "tokens/total": 4374096, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.12828375399112701, + "learning_rate": 8.875691779131569e-05, + "loss": 0.0007754197577014565, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00078, + "step": 138, + "tokens/total": 4406272, + "tokens/train_per_sec_per_gpu": 24.22, + "tokens/trainable": 63145 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.3188190162181854, + "learning_rate": 8.856425797031829e-05, + "loss": 0.006707335356622934, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00673, + "step": 139, + "tokens/total": 4438112, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 63587 + }, + { + "epoch": 0.546875, + "grad_norm": 0.6216707229614258, + "learning_rate": 8.837020140354295e-05, + "loss": 0.007024695165455341, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00705, + "step": 140, + "tokens/total": 4470128, + "tokens/train_per_sec_per_gpu": 30.64, + "tokens/trainable": 64052 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.1480305939912796, + "learning_rate": 8.817475616647554e-05, + "loss": 0.00229115248657763, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00229, + "step": 141, + "tokens/total": 4502096, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 64526 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22254763543605804, + "learning_rate": 8.797793039239017e-05, + "loss": 0.0029198499396443367, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00292, + "step": 142, + "tokens/total": 4534496, + "tokens/train_per_sec_per_gpu": 29.08, + "tokens/trainable": 64983 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.6901941895484924, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013051149435341358, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01314, + "step": 143, + "tokens/total": 4566720, + "tokens/train_per_sec_per_gpu": 31.56, + "tokens/trainable": 65492 + }, + { + "epoch": 0.5625, + "grad_norm": 0.9545458555221558, + "learning_rate": 8.758017005316988e-05, + "loss": 0.02235381305217743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.02261, + "step": 144, + "tokens/total": 4598880, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 65925 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.5515892505645752, + "learning_rate": 8.737925204046629e-05, + "loss": 0.02038375660777092, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.02059, + "step": 145, + "tokens/total": 4630864, + "tokens/train_per_sec_per_gpu": 26.3, + "tokens/trainable": 66383 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3512430191040039, + "learning_rate": 8.717698659491851e-05, + "loss": 0.008233271539211273, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00827, + "step": 146, + "tokens/total": 4662864, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 66840 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.6830361485481262, + "learning_rate": 8.697338213361735e-05, + "loss": 0.013419180177152157, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01351, + "step": 147, + "tokens/total": 4694912, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 67297 + }, + { + "epoch": 0.578125, + "grad_norm": 0.2117031067609787, + "learning_rate": 8.676844712937552e-05, + "loss": 0.004134457092732191, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00414, + "step": 148, + "tokens/total": 4727120, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 67769 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.378053218126297, + "learning_rate": 8.656219011037509e-05, + "loss": 0.009747720323503017, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0098, + "step": 149, + "tokens/total": 4759168, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 68257 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.21076741814613342, + "learning_rate": 8.63546196598125e-05, + "loss": 0.005035653710365295, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00505, + "step": 150, + "tokens/total": 4791008, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 68706 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.25075697898864746, + "learning_rate": 8.614574441554145e-05, + "loss": 0.006027051247656345, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00605, + "step": 151, + "tokens/total": 4822912, + "tokens/train_per_sec_per_gpu": 27.95, + "tokens/trainable": 69131 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5774204730987549, + "learning_rate": 8.593557306971349e-05, + "loss": 0.018617186695337296, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01879, + "step": 152, + "tokens/total": 4854816, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 69586 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.1364293098449707, + "learning_rate": 8.572411436841618e-05, + "loss": 0.002049289643764496, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00205, + "step": 153, + "tokens/total": 4886656, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 70061 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.25387004017829895, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0044304742477834225, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00444, + "step": 154, + "tokens/total": 4918768, + "tokens/train_per_sec_per_gpu": 23.26, + "tokens/trainable": 70467 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3047125041484833, + "learning_rate": 8.529737015125824e-05, + "loss": 0.0043389927595853806, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00435, + "step": 155, + "tokens/total": 4951008, + "tokens/train_per_sec_per_gpu": 27.53, + "tokens/trainable": 70933 + }, + { + "epoch": 0.609375, + "grad_norm": 0.1442601978778839, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0031973645091056824, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.0032, + "step": 156, + "tokens/total": 4982976, + "tokens/train_per_sec_per_gpu": 28.86, + "tokens/trainable": 71382 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.2608455717563629, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0032951058819890022, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0033, + "step": 157, + "tokens/total": 5014640, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 71808 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.637295663356781, + "learning_rate": 8.464782037243449e-05, + "loss": 0.01436957623809576, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.01447, + "step": 158, + "tokens/total": 5046672, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 72249 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.9834056496620178, + "learning_rate": 8.442882418044202e-05, + "loss": 0.02375311404466629, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.02404, + "step": 159, + "tokens/total": 5078528, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 72726 + }, + { + "epoch": 0.625, + "grad_norm": 0.35432615876197815, + "learning_rate": 8.420860333495179e-05, + "loss": 0.006119017023593187, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00614, + "step": 160, + "tokens/total": 5110704, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 73148 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 3.468933461258358e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/training_args.bin b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..dec74f9f304b44cbe47132d86c4e6285e659f40b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-160/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634 +size 8273 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/README.md b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ba677470ac9c29f9770a4aacb65c644ebd671a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/deconf_wave_deconf_control/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/deconf_wave_deconf_control/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/adapter_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..21fcb5b00be777c20a2d2dcceffc539dc51679b4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/deconf_wave_deconf_control/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "gate_proj", + "v_proj", + "k_proj", + "up_proj", + "q_proj", + "o_proj", + "down_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/adapter_model.safetensors b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..b9ba2eddeb33c86ff82b7e0dbb31502fd42703fd --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0549effc2939bd203244d8227b20fe1e0f34a3555fde11b86136c07eb06e5738 +size 547777976 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/chat_template.jinja b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/optimizer.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..198ff17e1eb2d404e839e83e06482676cd9600a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5b484083c9b1fa3b20d611b072130a948b97aae1b5d9c02d6fb9041ea03337b0 +size 1048106435 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/rng_state.pth b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..6e2c097fc37999d68028c225a209ad50bcb14802 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5972beeed351537138e229200527c4e9f2e9808b01529e746b926011cc01d564 +size 14645 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/scheduler.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..68fafd70d4f03fc26bd72aefbdcd22105c983415 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:076826708eb68e7e0324ae251ebe1ef8facf8800fbc47028be05da96491371fd +size 1465 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/tokenizer.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/tokenizer_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/tokens_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..e90c08b1d07927de3cceba77522fc2a7480424a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/tokens_state.json @@ -0,0 +1 @@ +{"total": 6127488, "trainable": 87747} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/trainer_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..2a686bd51c935e78891472ea1bb325bdaac8e2bf --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/trainer_state.json @@ -0,0 +1,2722 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.75, + "eval_steps": 500, + "global_step": 192, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.2140090465545654, + "learning_rate": 0.0, + "loss": 0.16441112756729126, + "memory/device_reserved (GiB)": 38.08, + "memory/max_active (GiB)": 35.64, + "memory/max_allocated (GiB)": 35.64, + "ppl": 1.1787, + "step": 1, + "tokens/total": 32160, + "tokens/train_per_sec_per_gpu": 21.28, + "tokens/trainable": 470 + }, + { + "epoch": 0.0078125, + "grad_norm": 2.92620849609375, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15774038434028625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.17086, + "step": 2, + "tokens/total": 64192, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 922 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.135961890220642, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1664382815361023, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.18109, + "step": 3, + "tokens/total": 96224, + "tokens/train_per_sec_per_gpu": 28.54, + "tokens/trainable": 1396 + }, + { + "epoch": 0.015625, + "grad_norm": 0.880176305770874, + "learning_rate": 1.2e-05, + "loss": 0.15334013104438782, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.16572, + "step": 4, + "tokens/total": 128336, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 1850 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.1433289051055908, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.16352537274360657, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.17766, + "step": 5, + "tokens/total": 160784, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 2302 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9728212952613831, + "learning_rate": 2e-05, + "loss": 0.15144473314285278, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.16351, + "step": 6, + "tokens/total": 192624, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 2742 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.7796363830566406, + "learning_rate": 2.4e-05, + "loss": 0.13553467392921448, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.14515, + "step": 7, + "tokens/total": 224608, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 3208 + }, + { + "epoch": 0.03125, + "grad_norm": 1.2444697618484497, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.1077304258942604, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.11375, + "step": 8, + "tokens/total": 256592, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 3655 + }, + { + "epoch": 0.03515625, + "grad_norm": 2.7339375019073486, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.08772765100002289, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.09169, + "step": 9, + "tokens/total": 286400, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 4091 + }, + { + "epoch": 0.0390625, + "grad_norm": 4.109770774841309, + "learning_rate": 3.6e-05, + "loss": 0.130940243601799, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.1399, + "step": 10, + "tokens/total": 318352, + "tokens/train_per_sec_per_gpu": 29.13, + "tokens/trainable": 4553 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.3768608570098877, + "learning_rate": 4e-05, + "loss": 0.10059958696365356, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.10583, + "step": 11, + "tokens/total": 350336, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 4992 + }, + { + "epoch": 0.046875, + "grad_norm": 3.281402826309204, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.12574875354766846, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.134, + "step": 12, + "tokens/total": 382448, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 5494 + }, + { + "epoch": 0.05078125, + "grad_norm": 1.7738980054855347, + "learning_rate": 4.8e-05, + "loss": 0.033296722918748856, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03386, + "step": 13, + "tokens/total": 414352, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 5933 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.64290189743042, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06366822123527527, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.06574, + "step": 14, + "tokens/total": 446400, + "tokens/train_per_sec_per_gpu": 24.8, + "tokens/trainable": 6369 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.3691928386688232, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.08168518543243408, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.08511, + "step": 15, + "tokens/total": 478688, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 6820 + }, + { + "epoch": 0.0625, + "grad_norm": 2.4356393814086914, + "learning_rate": 6e-05, + "loss": 0.09031196683645248, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.09452, + "step": 16, + "tokens/total": 510512, + "tokens/train_per_sec_per_gpu": 25.7, + "tokens/trainable": 7258 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.55682373046875, + "learning_rate": 6.400000000000001e-05, + "loss": 0.07496573030948639, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.07785, + "step": 17, + "tokens/total": 542736, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 7710 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.1054646968841553, + "learning_rate": 6.800000000000001e-05, + "loss": 0.05218805745244026, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.05357, + "step": 18, + "tokens/total": 574832, + "tokens/train_per_sec_per_gpu": 25.78, + "tokens/trainable": 8174 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.7429099082946777, + "learning_rate": 7.2e-05, + "loss": 0.03790827840566635, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.03864, + "step": 19, + "tokens/total": 606976, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 8617 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9508209228515625, + "learning_rate": 7.6e-05, + "loss": 0.04993613809347153, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0512, + "step": 20, + "tokens/total": 639200, + "tokens/train_per_sec_per_gpu": 24.91, + "tokens/trainable": 9037 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.0297844409942627, + "learning_rate": 8e-05, + "loss": 0.017908263951539993, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.01807, + "step": 21, + "tokens/total": 671072, + "tokens/train_per_sec_per_gpu": 30.42, + "tokens/trainable": 9503 + }, + { + "epoch": 0.0859375, + "grad_norm": 0.9279879331588745, + "learning_rate": 8.4e-05, + "loss": 0.013478193432092667, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01357, + "step": 22, + "tokens/total": 703136, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 9972 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.526488184928894, + "learning_rate": 8.800000000000001e-05, + "loss": 0.017310388386249542, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01746, + "step": 23, + "tokens/total": 735120, + "tokens/train_per_sec_per_gpu": 31.45, + "tokens/trainable": 10447 + }, + { + "epoch": 0.09375, + "grad_norm": 3.4049417972564697, + "learning_rate": 9.200000000000001e-05, + "loss": 0.0353429913520813, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.03597, + "step": 24, + "tokens/total": 767040, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 10899 + }, + { + "epoch": 0.09765625, + "grad_norm": 7.511656761169434, + "learning_rate": 9.6e-05, + "loss": 0.1335875540971756, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.14292, + "step": 25, + "tokens/total": 798976, + "tokens/train_per_sec_per_gpu": 27.72, + "tokens/trainable": 11360 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.0069234371185303, + "learning_rate": 0.0001, + "loss": 0.04382415860891342, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0448, + "step": 26, + "tokens/total": 830880, + "tokens/train_per_sec_per_gpu": 24.61, + "tokens/trainable": 11775 + }, + { + "epoch": 0.10546875, + "grad_norm": 7.807938575744629, + "learning_rate": 9.99990636831587e-05, + "loss": 0.01210833340883255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.01218, + "step": 27, + "tokens/total": 860960, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 12242 + }, + { + "epoch": 0.109375, + "grad_norm": 2.23823881149292, + "learning_rate": 9.999625477159879e-05, + "loss": 0.0272970050573349, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02767, + "step": 28, + "tokens/total": 892816, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 12696 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.315584421157837, + "learning_rate": 9.999157338221051e-05, + "loss": 0.025516577064990997, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.02584, + "step": 29, + "tokens/total": 924736, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 13153 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.8413651585578918, + "learning_rate": 9.998501970980562e-05, + "loss": 0.007838520221412182, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00787, + "step": 30, + "tokens/total": 954592, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 13624 + }, + { + "epoch": 0.12109375, + "grad_norm": 1.6797157526016235, + "learning_rate": 9.997659402710915e-05, + "loss": 0.03002442792057991, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.03048, + "step": 31, + "tokens/total": 986592, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 14064 + }, + { + "epoch": 0.125, + "grad_norm": 1.3320434093475342, + "learning_rate": 9.996629668474818e-05, + "loss": 0.03772728145122528, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.03845, + "step": 32, + "tokens/total": 1018720, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 14565 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.6335167288780212, + "learning_rate": 9.995412811123711e-05, + "loss": 0.013970119878649712, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01407, + "step": 33, + "tokens/total": 1050624, + "tokens/train_per_sec_per_gpu": 28.18, + "tokens/trainable": 15005 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.5626921653747559, + "learning_rate": 9.994008881295999e-05, + "loss": 0.016654539853334427, + "memory/device_reserved (GiB)": 37.74, + "memory/max_active (GiB)": 36.22, + "memory/max_allocated (GiB)": 36.22, + "ppl": 1.01679, + "step": 34, + "tokens/total": 1082176, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 15459 + }, + { + "epoch": 0.13671875, + "grad_norm": 0.603219211101532, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021408583968877792, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02164, + "step": 35, + "tokens/total": 1113968, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 15960 + }, + { + "epoch": 0.140625, + "grad_norm": 0.9779674410820007, + "learning_rate": 9.99064004568618e-05, + "loss": 0.03277068957686424, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03331, + "step": 36, + "tokens/total": 1145952, + "tokens/train_per_sec_per_gpu": 28.68, + "tokens/trainable": 16428 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.8855255842208862, + "learning_rate": 9.988675280095074e-05, + "loss": 0.017048098146915436, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01719, + "step": 37, + "tokens/total": 1177872, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 16884 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.9266122579574585, + "learning_rate": 9.986523722403528e-05, + "loss": 0.045863546431064606, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.04693, + "step": 38, + "tokens/total": 1209776, + "tokens/train_per_sec_per_gpu": 26.91, + "tokens/trainable": 17341 + }, + { + "epoch": 0.15234375, + "grad_norm": 0.8280895948410034, + "learning_rate": 9.984185462146642e-05, + "loss": 0.033055759966373444, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03361, + "step": 39, + "tokens/total": 1241888, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 17786 + }, + { + "epoch": 0.15625, + "grad_norm": 5.029730319976807, + "learning_rate": 9.98166059662897e-05, + "loss": 0.025589998811483383, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02592, + "step": 40, + "tokens/total": 1274032, + "tokens/train_per_sec_per_gpu": 29.22, + "tokens/trainable": 18262 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.5688032507896423, + "learning_rate": 9.978949230920472e-05, + "loss": 0.016095872968435287, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01623, + "step": 41, + "tokens/total": 1306000, + "tokens/train_per_sec_per_gpu": 27.49, + "tokens/trainable": 18716 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8217663168907166, + "learning_rate": 9.976051477852141e-05, + "loss": 0.026798982173204422, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02716, + "step": 42, + "tokens/total": 1337936, + "tokens/train_per_sec_per_gpu": 27.51, + "tokens/trainable": 19157 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.7309132218360901, + "learning_rate": 9.972967458011312e-05, + "loss": 0.033920761197805405, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0345, + "step": 43, + "tokens/total": 1369904, + "tokens/train_per_sec_per_gpu": 30.66, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.8134022355079651, + "learning_rate": 9.96969729973664e-05, + "loss": 0.024005085229873657, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0243, + "step": 44, + "tokens/total": 1401984, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 20119 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.35084816813468933, + "learning_rate": 9.966241139112754e-05, + "loss": 0.013319254852831364, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01341, + "step": 45, + "tokens/total": 1434032, + "tokens/train_per_sec_per_gpu": 27.45, + "tokens/trainable": 20560 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.8701573014259338, + "learning_rate": 9.96259911996461e-05, + "loss": 0.01356641948223114, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01366, + "step": 46, + "tokens/total": 1465872, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 20992 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.4406442642211914, + "learning_rate": 9.958771393851491e-05, + "loss": 0.04396980255842209, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 35.86, + "memory/max_allocated (GiB)": 35.86, + "ppl": 1.04495, + "step": 47, + "tokens/total": 1495440, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 21441 + }, + { + "epoch": 0.1875, + "grad_norm": 0.4966669976711273, + "learning_rate": 9.954758120060702e-05, + "loss": 0.011179964058101177, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01124, + "step": 48, + "tokens/total": 1527776, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 21901 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.41124436259269714, + "learning_rate": 9.950559465600948e-05, + "loss": 0.0063599650748074055, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00638, + "step": 49, + "tokens/total": 1559760, + "tokens/train_per_sec_per_gpu": 27.73, + "tokens/trainable": 22341 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.829739511013031, + "learning_rate": 9.946175605195379e-05, + "loss": 0.015785304829478264, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01591, + "step": 50, + "tokens/total": 1592000, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 22833 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5181736946105957, + "learning_rate": 9.941606721274322e-05, + "loss": 0.008393109776079655, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00843, + "step": 51, + "tokens/total": 1623920, + "tokens/train_per_sec_per_gpu": 25.61, + "tokens/trainable": 23277 + }, + { + "epoch": 0.203125, + "grad_norm": 1.5778990983963013, + "learning_rate": 9.936853003967685e-05, + "loss": 0.024984844028949738, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0253, + "step": 52, + "tokens/total": 1656032, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 23759 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.4938714504241943, + "learning_rate": 9.93191465109705e-05, + "loss": 0.038523320108652115, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03927, + "step": 53, + "tokens/total": 1688368, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 24193 + }, + { + "epoch": 0.2109375, + "grad_norm": 2.1427106857299805, + "learning_rate": 9.926791868167438e-05, + "loss": 0.06302924454212189, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.06506, + "step": 54, + "tokens/total": 1720560, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 24619 + }, + { + "epoch": 0.21484375, + "grad_norm": 1.8365188837051392, + "learning_rate": 9.921484868358753e-05, + "loss": 0.052160464227199554, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.05354, + "step": 55, + "tokens/total": 1752464, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 25075 + }, + { + "epoch": 0.21875, + "grad_norm": 0.7779749631881714, + "learning_rate": 9.915993872516924e-05, + "loss": 0.021328264847397804, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.02156, + "step": 56, + "tokens/total": 1784336, + "tokens/train_per_sec_per_gpu": 26.42, + "tokens/trainable": 25519 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.4189571142196655, + "learning_rate": 9.9103191091447e-05, + "loss": 0.014279939234256744, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01438, + "step": 57, + "tokens/total": 1816304, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 25966 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.3834710717201233, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01883944310247898, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01902, + "step": 58, + "tokens/total": 1848304, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 26423 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.7693278789520264, + "learning_rate": 9.898419232046825e-05, + "loss": 0.03533369302749634, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03597, + "step": 59, + "tokens/total": 1880640, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 26934 + }, + { + "epoch": 0.234375, + "grad_norm": 0.4606446325778961, + "learning_rate": 9.892194613523633e-05, + "loss": 0.021482212468981743, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02171, + "step": 60, + "tokens/total": 1912592, + "tokens/train_per_sec_per_gpu": 27.64, + "tokens/trainable": 27393 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.46030017733573914, + "learning_rate": 9.885787217854357e-05, + "loss": 0.017044665291905403, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01719, + "step": 61, + "tokens/total": 1944544, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 27852 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.7918264269828796, + "learning_rate": 9.879197311676887e-05, + "loss": 0.016954131424427032, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0171, + "step": 62, + "tokens/total": 1976688, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 28292 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.4787926971912384, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01525965891778469, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 63, + "tokens/total": 2008928, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 28798 + }, + { + "epoch": 0.25, + "grad_norm": 0.9033918976783752, + "learning_rate": 9.865471072312528e-05, + "loss": 0.02902211993932724, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02945, + "step": 64, + "tokens/total": 2041120, + "tokens/train_per_sec_per_gpu": 29.83, + "tokens/trainable": 29283 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.39305707812309265, + "learning_rate": 9.858335310330492e-05, + "loss": 0.00901167094707489, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.32, + "memory/max_allocated (GiB)": 36.32, + "ppl": 1.00905, + "step": 65, + "tokens/total": 2073024, + "tokens/train_per_sec_per_gpu": 26.48, + "tokens/trainable": 29745 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4494345784187317, + "learning_rate": 9.851018180226185e-05, + "loss": 0.007426132448017597, + "memory/device_reserved (GiB)": 37.47, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00745, + "step": 66, + "tokens/total": 2105184, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 30186 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.629649817943573, + "learning_rate": 9.843519986495259e-05, + "loss": 0.010609081946313381, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01067, + "step": 67, + "tokens/total": 2137264, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 30622 + }, + { + "epoch": 0.265625, + "grad_norm": 0.7591162919998169, + "learning_rate": 9.835841041168162e-05, + "loss": 0.00926615484058857, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00931, + "step": 68, + "tokens/total": 2169424, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 31097 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.7288486361503601, + "learning_rate": 9.82798166379715e-05, + "loss": 0.024310583248734474, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.02461, + "step": 69, + "tokens/total": 2201424, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 31595 + }, + { + "epoch": 0.2734375, + "grad_norm": 2.134056329727173, + "learning_rate": 9.819942181443002e-05, + "loss": 0.008422967046499252, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00846, + "step": 70, + "tokens/total": 2233424, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 32036 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6362871527671814, + "learning_rate": 9.811722928661392e-05, + "loss": 0.01577262207865715, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0159, + "step": 71, + "tokens/total": 2265328, + "tokens/train_per_sec_per_gpu": 23.65, + "tokens/trainable": 32428 + }, + { + "epoch": 0.28125, + "grad_norm": 0.6091912388801575, + "learning_rate": 9.803324247488975e-05, + "loss": 0.0064800274558365345, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0065, + "step": 72, + "tokens/total": 2297248, + "tokens/train_per_sec_per_gpu": 26.86, + "tokens/trainable": 32880 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.14533838629722595, + "learning_rate": 9.794746487429161e-05, + "loss": 0.001991454279050231, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00199, + "step": 73, + "tokens/total": 2329152, + "tokens/train_per_sec_per_gpu": 27.97, + "tokens/trainable": 33323 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.5400826930999756, + "learning_rate": 9.785990005437554e-05, + "loss": 0.017662350088357925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01782, + "step": 74, + "tokens/total": 2361344, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 33792 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.6132436394691467, + "learning_rate": 9.777055165907117e-05, + "loss": 0.01787603087723255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01804, + "step": 75, + "tokens/total": 2393168, + "tokens/train_per_sec_per_gpu": 29.53, + "tokens/trainable": 34223 + }, + { + "epoch": 0.296875, + "grad_norm": 0.7287399172782898, + "learning_rate": 9.767942340652993e-05, + "loss": 0.024754449725151062, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.31, + "memory/max_allocated (GiB)": 36.31, + "ppl": 1.02506, + "step": 76, + "tokens/total": 2424880, + "tokens/train_per_sec_per_gpu": 24.92, + "tokens/trainable": 34649 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.6790545582771301, + "learning_rate": 9.758651908897035e-05, + "loss": 0.014308430254459381, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01441, + "step": 77, + "tokens/total": 2457120, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 35094 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6203456521034241, + "learning_rate": 9.749184257252033e-05, + "loss": 0.034353043884038925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03495, + "step": 78, + "tokens/total": 2489104, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 35534 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.6238825917243958, + "learning_rate": 9.739539779705614e-05, + "loss": 0.032295141369104385, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.03282, + "step": 79, + "tokens/total": 2521136, + "tokens/train_per_sec_per_gpu": 29.68, + "tokens/trainable": 36029 + }, + { + "epoch": 0.3125, + "grad_norm": 0.18977171182632446, + "learning_rate": 9.729718877603861e-05, + "loss": 0.007260813377797604, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00729, + "step": 80, + "tokens/total": 2553168, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 36469 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.8054049611091614, + "learning_rate": 9.719721959634592e-05, + "loss": 0.034894898533821106, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.03551, + "step": 81, + "tokens/total": 2585584, + "tokens/train_per_sec_per_gpu": 25.75, + "tokens/trainable": 36906 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.6884189248085022, + "learning_rate": 9.709549441810375e-05, + "loss": 0.018210047855973244, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.01838, + "step": 82, + "tokens/total": 2617808, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 37390 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.6490103602409363, + "learning_rate": 9.699201747451195e-05, + "loss": 0.02030857279896736, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.02052, + "step": 83, + "tokens/total": 2649664, + "tokens/train_per_sec_per_gpu": 25.8, + "tokens/trainable": 37838 + }, + { + "epoch": 0.328125, + "grad_norm": 0.2439805567264557, + "learning_rate": 9.688679307166854e-05, + "loss": 0.008518049493432045, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00855, + "step": 84, + "tokens/total": 2681728, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 38310 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6747375726699829, + "learning_rate": 9.677982558839042e-05, + "loss": 0.029895059764385223, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.03035, + "step": 85, + "tokens/total": 2713488, + "tokens/train_per_sec_per_gpu": 28.49, + "tokens/trainable": 38789 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.582666277885437, + "learning_rate": 9.66711194760312e-05, + "loss": 0.014998597092926502, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01511, + "step": 86, + "tokens/total": 2745536, + "tokens/train_per_sec_per_gpu": 30.27, + "tokens/trainable": 39277 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.5364654660224915, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011567167937755585, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01163, + "step": 87, + "tokens/total": 2777680, + "tokens/train_per_sec_per_gpu": 29.73, + "tokens/trainable": 39737 + }, + { + "epoch": 0.34375, + "grad_norm": 0.34024426341056824, + "learning_rate": 9.644850953105288e-05, + "loss": 0.01014852337539196, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0102, + "step": 88, + "tokens/total": 2809568, + "tokens/train_per_sec_per_gpu": 24.68, + "tokens/trainable": 40179 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.32472267746925354, + "learning_rate": 9.633461496214225e-05, + "loss": 0.01229981891810894, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01238, + "step": 89, + "tokens/total": 2841760, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 40649 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.7906899452209473, + "learning_rate": 9.621900029118195e-05, + "loss": 0.02552586793899536, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.02585, + "step": 90, + "tokens/total": 2871696, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 41080 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.7444764971733093, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0054441955871880054, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00546, + "step": 91, + "tokens/total": 2903968, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 41599 + }, + { + "epoch": 0.359375, + "grad_norm": 0.1911851465702057, + "learning_rate": 9.598262995928611e-05, + "loss": 0.004665314219892025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00468, + "step": 92, + "tokens/total": 2936048, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 42076 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.21238519251346588, + "learning_rate": 9.586188413468492e-05, + "loss": 0.003917238209396601, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00392, + "step": 93, + "tokens/total": 2968176, + "tokens/train_per_sec_per_gpu": 28.07, + "tokens/trainable": 42522 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.41729435324668884, + "learning_rate": 9.57394378802934e-05, + "loss": 0.013555807992815971, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01365, + "step": 94, + "tokens/total": 3000304, + "tokens/train_per_sec_per_gpu": 28.69, + "tokens/trainable": 42974 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.7246769070625305, + "learning_rate": 9.56152962916e-05, + "loss": 0.020221399143338203, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.02043, + "step": 95, + "tokens/total": 3032080, + "tokens/train_per_sec_per_gpu": 25.35, + "tokens/trainable": 43380 + }, + { + "epoch": 0.375, + "grad_norm": 0.16139821708202362, + "learning_rate": 9.548946453464296e-05, + "loss": 0.003853276837617159, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00386, + "step": 96, + "tokens/total": 3064320, + "tokens/train_per_sec_per_gpu": 28.47, + "tokens/trainable": 43865 + }, + { + "epoch": 0.37890625, + "grad_norm": 1.0509370565414429, + "learning_rate": 9.53619478457953e-05, + "loss": 0.015099998563528061, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01521, + "step": 97, + "tokens/total": 3096144, + "tokens/train_per_sec_per_gpu": 26.61, + "tokens/trainable": 44328 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.6859350204467773, + "learning_rate": 9.523275153154695e-05, + "loss": 0.004579808097332716, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00459, + "step": 98, + "tokens/total": 3127600, + "tokens/train_per_sec_per_gpu": 25.91, + "tokens/trainable": 44778 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5686795711517334, + "learning_rate": 9.51018809682839e-05, + "loss": 0.022520575672388077, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02278, + "step": 99, + "tokens/total": 3159520, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 45225 + }, + { + "epoch": 0.390625, + "grad_norm": 1.6308369636535645, + "learning_rate": 9.49693416020645e-05, + "loss": 0.011883002705872059, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01195, + "step": 100, + "tokens/total": 3191568, + "tokens/train_per_sec_per_gpu": 29.75, + "tokens/trainable": 45678 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.7186088562011719, + "learning_rate": 9.483513894839276e-05, + "loss": 0.0060632615350186825, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00608, + "step": 101, + "tokens/total": 3223648, + "tokens/train_per_sec_per_gpu": 27.14, + "tokens/trainable": 46125 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.12036337703466415, + "learning_rate": 9.469927859198888e-05, + "loss": 0.0019118499476462603, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00191, + "step": 102, + "tokens/total": 3255744, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 46612 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.29928353428840637, + "learning_rate": 9.456176618655689e-05, + "loss": 0.00944911316037178, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00949, + "step": 103, + "tokens/total": 3287792, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 47059 + }, + { + "epoch": 0.40625, + "grad_norm": 0.3766293227672577, + "learning_rate": 9.442260745454927e-05, + "loss": 0.008984047919511795, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.68, + "memory/max_allocated (GiB)": 36.68, + "ppl": 1.00902, + "step": 104, + "tokens/total": 3320016, + "tokens/train_per_sec_per_gpu": 28.92, + "tokens/trainable": 47536 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3481820523738861, + "learning_rate": 9.428180818692884e-05, + "loss": 0.005444999784231186, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00546, + "step": 105, + "tokens/total": 3351952, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 48037 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.8414323329925537, + "learning_rate": 9.413937424292791e-05, + "loss": 0.032438624650239944, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03297, + "step": 106, + "tokens/total": 3384032, + "tokens/train_per_sec_per_gpu": 27.11, + "tokens/trainable": 48491 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.5067655444145203, + "learning_rate": 9.399531154980424e-05, + "loss": 0.008442888967692852, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00848, + "step": 107, + "tokens/total": 3416192, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 48940 + }, + { + "epoch": 0.421875, + "grad_norm": 0.41617852449417114, + "learning_rate": 9.384962610259455e-05, + "loss": 0.008386321365833282, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00842, + "step": 108, + "tokens/total": 3448080, + "tokens/train_per_sec_per_gpu": 26.56, + "tokens/trainable": 49389 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1587515026330948, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0031663556583225727, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.03, + "memory/max_allocated (GiB)": 36.03, + "ppl": 1.00317, + "step": 109, + "tokens/total": 3477952, + "tokens/train_per_sec_per_gpu": 29.8, + "tokens/trainable": 49838 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2903148829936981, + "learning_rate": 9.355341126345868e-05, + "loss": 0.007958847098052502, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00799, + "step": 110, + "tokens/total": 3509776, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 50310 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.642983078956604, + "learning_rate": 9.340289419824107e-05, + "loss": 0.006748045329004526, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00677, + "step": 111, + "tokens/total": 3541808, + "tokens/train_per_sec_per_gpu": 28.04, + "tokens/trainable": 50755 + }, + { + "epoch": 0.4375, + "grad_norm": 0.14928282797336578, + "learning_rate": 9.325077903184159e-05, + "loss": 0.0027839094400405884, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00279, + "step": 112, + "tokens/total": 3574064, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 51213 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.05922335386276245, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0009622900979593396, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00096, + "step": 113, + "tokens/total": 3605984, + "tokens/train_per_sec_per_gpu": 27.99, + "tokens/trainable": 51660 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.44227731227874756, + "learning_rate": 9.2941779782269e-05, + "loss": 0.002763954224064946, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00277, + "step": 114, + "tokens/total": 3638080, + "tokens/train_per_sec_per_gpu": 27.03, + "tokens/trainable": 52133 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5176818370819092, + "learning_rate": 9.278490855781596e-05, + "loss": 0.015623578801751137, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01575, + "step": 115, + "tokens/total": 3670016, + "tokens/train_per_sec_per_gpu": 31.44, + "tokens/trainable": 52580 + }, + { + "epoch": 0.453125, + "grad_norm": 0.4265497922897339, + "learning_rate": 9.262646494908604e-05, + "loss": 0.00794076919555664, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00797, + "step": 116, + "tokens/total": 3701984, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 53033 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.40457987785339355, + "learning_rate": 9.246645554956457e-05, + "loss": 0.022708112373948097, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02297, + "step": 117, + "tokens/total": 3733920, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 53517 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.01414525043219328, + "learning_rate": 9.230488701789578e-05, + "loss": 0.00020209309877827764, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.0002, + "step": 118, + "tokens/total": 3765840, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 53967 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.11359831690788269, + "learning_rate": 9.214176607760577e-05, + "loss": 0.0012168455868959427, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00122, + "step": 119, + "tokens/total": 3797680, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 54430 + }, + { + "epoch": 0.46875, + "grad_norm": 0.35982534289360046, + "learning_rate": 9.197709951682268e-05, + "loss": 0.004110721405595541, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00412, + "step": 120, + "tokens/total": 3829920, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 54896 + }, + { + "epoch": 0.47265625, + "grad_norm": 1.3690382242202759, + "learning_rate": 9.181089418799428e-05, + "loss": 0.02705790475010872, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02743, + "step": 121, + "tokens/total": 3862080, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 55379 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.351891428232193, + "learning_rate": 9.164315700760271e-05, + "loss": 0.0020163573790341616, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00202, + "step": 122, + "tokens/total": 3893968, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 55803 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4747961461544037, + "learning_rate": 9.147389495587671e-05, + "loss": 0.012588979676365852, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01267, + "step": 123, + "tokens/total": 3926160, + "tokens/train_per_sec_per_gpu": 30.14, + "tokens/trainable": 56249 + }, + { + "epoch": 0.484375, + "grad_norm": 0.4064250588417053, + "learning_rate": 9.130311507650116e-05, + "loss": 0.0022941348142921925, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0023, + "step": 124, + "tokens/total": 3958208, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 56713 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.22056929767131805, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0022175831254571676, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00222, + "step": 125, + "tokens/total": 3990208, + "tokens/train_per_sec_per_gpu": 32.57, + "tokens/trainable": 57196 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5698862671852112, + "learning_rate": 9.09570303250602e-05, + "loss": 0.010205100290477276, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01026, + "step": 126, + "tokens/total": 4022224, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 57680 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.2593887448310852, + "learning_rate": 9.078173985499394e-05, + "loss": 0.0032277877908200026, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00323, + "step": 127, + "tokens/total": 4053952, + "tokens/train_per_sec_per_gpu": 26.51, + "tokens/trainable": 58110 + }, + { + "epoch": 0.5, + "grad_norm": 0.3735230565071106, + "learning_rate": 9.060496036067713e-05, + "loss": 0.01031105499714613, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01036, + "step": 128, + "tokens/total": 4086144, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 58534 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.1332484930753708, + "learning_rate": 9.042669919862615e-05, + "loss": 0.0012390739284455776, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00124, + "step": 129, + "tokens/total": 4118096, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 58998 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.20182718336582184, + "learning_rate": 9.024696378701557e-05, + "loss": 0.004501686431467533, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00451, + "step": 130, + "tokens/total": 4150272, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 59448 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.529080331325531, + "learning_rate": 9.006576160536948e-05, + "loss": 0.00763288140296936, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00766, + "step": 131, + "tokens/total": 4181904, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 59906 + }, + { + "epoch": 0.515625, + "grad_norm": 0.21625548601150513, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0024378676898777485, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00244, + "step": 132, + "tokens/total": 4214032, + "tokens/train_per_sec_per_gpu": 28.36, + "tokens/trainable": 60350 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.37116539478302, + "learning_rate": 8.969898715494506e-05, + "loss": 0.0037849934305995703, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00379, + "step": 133, + "tokens/total": 4246064, + "tokens/train_per_sec_per_gpu": 30.18, + "tokens/trainable": 60831 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5646623373031616, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006517104804515839, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00654, + "step": 134, + "tokens/total": 4278224, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 61305 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.6371963620185852, + "learning_rate": 8.932643689864568e-05, + "loss": 0.008687382563948631, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00873, + "step": 135, + "tokens/total": 4310112, + "tokens/train_per_sec_per_gpu": 31.48, + "tokens/trainable": 61792 + }, + { + "epoch": 0.53125, + "grad_norm": 0.8210524916648865, + "learning_rate": 8.913801518498845e-05, + "loss": 0.008436474949121475, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00847, + "step": 136, + "tokens/total": 4342192, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 62253 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.5692176222801208, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017321188002824783, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.01747, + "step": 137, + "tokens/total": 4374096, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.12828375399112701, + "learning_rate": 8.875691779131569e-05, + "loss": 0.0007754197577014565, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00078, + "step": 138, + "tokens/total": 4406272, + "tokens/train_per_sec_per_gpu": 24.22, + "tokens/trainable": 63145 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.3188190162181854, + "learning_rate": 8.856425797031829e-05, + "loss": 0.006707335356622934, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00673, + "step": 139, + "tokens/total": 4438112, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 63587 + }, + { + "epoch": 0.546875, + "grad_norm": 0.6216707229614258, + "learning_rate": 8.837020140354295e-05, + "loss": 0.007024695165455341, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00705, + "step": 140, + "tokens/total": 4470128, + "tokens/train_per_sec_per_gpu": 30.64, + "tokens/trainable": 64052 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.1480305939912796, + "learning_rate": 8.817475616647554e-05, + "loss": 0.00229115248657763, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00229, + "step": 141, + "tokens/total": 4502096, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 64526 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22254763543605804, + "learning_rate": 8.797793039239017e-05, + "loss": 0.0029198499396443367, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00292, + "step": 142, + "tokens/total": 4534496, + "tokens/train_per_sec_per_gpu": 29.08, + "tokens/trainable": 64983 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.6901941895484924, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013051149435341358, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01314, + "step": 143, + "tokens/total": 4566720, + "tokens/train_per_sec_per_gpu": 31.56, + "tokens/trainable": 65492 + }, + { + "epoch": 0.5625, + "grad_norm": 0.9545458555221558, + "learning_rate": 8.758017005316988e-05, + "loss": 0.02235381305217743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.02261, + "step": 144, + "tokens/total": 4598880, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 65925 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.5515892505645752, + "learning_rate": 8.737925204046629e-05, + "loss": 0.02038375660777092, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.02059, + "step": 145, + "tokens/total": 4630864, + "tokens/train_per_sec_per_gpu": 26.3, + "tokens/trainable": 66383 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3512430191040039, + "learning_rate": 8.717698659491851e-05, + "loss": 0.008233271539211273, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00827, + "step": 146, + "tokens/total": 4662864, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 66840 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.6830361485481262, + "learning_rate": 8.697338213361735e-05, + "loss": 0.013419180177152157, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01351, + "step": 147, + "tokens/total": 4694912, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 67297 + }, + { + "epoch": 0.578125, + "grad_norm": 0.2117031067609787, + "learning_rate": 8.676844712937552e-05, + "loss": 0.004134457092732191, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00414, + "step": 148, + "tokens/total": 4727120, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 67769 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.378053218126297, + "learning_rate": 8.656219011037509e-05, + "loss": 0.009747720323503017, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0098, + "step": 149, + "tokens/total": 4759168, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 68257 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.21076741814613342, + "learning_rate": 8.63546196598125e-05, + "loss": 0.005035653710365295, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00505, + "step": 150, + "tokens/total": 4791008, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 68706 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.25075697898864746, + "learning_rate": 8.614574441554145e-05, + "loss": 0.006027051247656345, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00605, + "step": 151, + "tokens/total": 4822912, + "tokens/train_per_sec_per_gpu": 27.95, + "tokens/trainable": 69131 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5774204730987549, + "learning_rate": 8.593557306971349e-05, + "loss": 0.018617186695337296, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01879, + "step": 152, + "tokens/total": 4854816, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 69586 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.1364293098449707, + "learning_rate": 8.572411436841618e-05, + "loss": 0.002049289643764496, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00205, + "step": 153, + "tokens/total": 4886656, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 70061 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.25387004017829895, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0044304742477834225, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00444, + "step": 154, + "tokens/total": 4918768, + "tokens/train_per_sec_per_gpu": 23.26, + "tokens/trainable": 70467 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3047125041484833, + "learning_rate": 8.529737015125824e-05, + "loss": 0.0043389927595853806, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00435, + "step": 155, + "tokens/total": 4951008, + "tokens/train_per_sec_per_gpu": 27.53, + "tokens/trainable": 70933 + }, + { + "epoch": 0.609375, + "grad_norm": 0.1442601978778839, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0031973645091056824, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.0032, + "step": 156, + "tokens/total": 4982976, + "tokens/train_per_sec_per_gpu": 28.86, + "tokens/trainable": 71382 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.2608455717563629, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0032951058819890022, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0033, + "step": 157, + "tokens/total": 5014640, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 71808 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.637295663356781, + "learning_rate": 8.464782037243449e-05, + "loss": 0.01436957623809576, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.01447, + "step": 158, + "tokens/total": 5046672, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 72249 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.9834056496620178, + "learning_rate": 8.442882418044202e-05, + "loss": 0.02375311404466629, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.02404, + "step": 159, + "tokens/total": 5078528, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 72726 + }, + { + "epoch": 0.625, + "grad_norm": 0.35432615876197815, + "learning_rate": 8.420860333495179e-05, + "loss": 0.006119017023593187, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00614, + "step": 160, + "tokens/total": 5110704, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 73148 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.44270065426826477, + "learning_rate": 8.398716700025208e-05, + "loss": 0.01352179329842329, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.72, + "memory/max_allocated (GiB)": 36.72, + "ppl": 1.01361, + "step": 161, + "tokens/total": 5142944, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 73600 + }, + { + "epoch": 0.6328125, + "grad_norm": 1.2181634902954102, + "learning_rate": 8.376452439121266e-05, + "loss": 0.01820964366197586, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.01838, + "step": 162, + "tokens/total": 5174592, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 74068 + }, + { + "epoch": 0.63671875, + "grad_norm": 1.9321303367614746, + "learning_rate": 8.354068477290124e-05, + "loss": 0.009742327965795994, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00979, + "step": 163, + "tokens/total": 5206432, + "tokens/train_per_sec_per_gpu": 29.89, + "tokens/trainable": 74527 + }, + { + "epoch": 0.640625, + "grad_norm": 0.3545796275138855, + "learning_rate": 8.331565746019807e-05, + "loss": 0.00975855067372322, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00981, + "step": 164, + "tokens/total": 5238384, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 74992 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.35368478298187256, + "learning_rate": 8.308945181740812e-05, + "loss": 0.010195466689765453, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01025, + "step": 165, + "tokens/total": 5270272, + "tokens/train_per_sec_per_gpu": 29.41, + "tokens/trainable": 75442 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.34957799315452576, + "learning_rate": 8.286207725787153e-05, + "loss": 0.009373681619763374, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00942, + "step": 166, + "tokens/total": 5302368, + "tokens/train_per_sec_per_gpu": 26.2, + "tokens/trainable": 75887 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.3684624135494232, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008211556822061539, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00825, + "step": 167, + "tokens/total": 5334496, + "tokens/train_per_sec_per_gpu": 28.21, + "tokens/trainable": 76331 + }, + { + "epoch": 0.65625, + "grad_norm": 0.12826858460903168, + "learning_rate": 8.240385928474219e-05, + "loss": 0.004492453299462795, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0045, + "step": 168, + "tokens/total": 5364288, + "tokens/train_per_sec_per_gpu": 30.16, + "tokens/trainable": 76745 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.31828173995018005, + "learning_rate": 8.217303493946967e-05, + "loss": 0.005733514670282602, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00575, + "step": 169, + "tokens/total": 5396512, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 77201 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6174353361129761, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01915649324655533, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01934, + "step": 170, + "tokens/total": 5428624, + "tokens/train_per_sec_per_gpu": 27.75, + "tokens/trainable": 77655 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.2902314066886902, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0036087254993617535, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00362, + "step": 171, + "tokens/total": 5460880, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 78122 + }, + { + "epoch": 0.671875, + "grad_norm": 0.3267754912376404, + "learning_rate": 8.147381587530713e-05, + "loss": 0.008090222254395485, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00812, + "step": 172, + "tokens/total": 5493088, + "tokens/train_per_sec_per_gpu": 27.86, + "tokens/trainable": 78576 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.4086189866065979, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006351984106004238, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00637, + "step": 173, + "tokens/total": 5525312, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 79059 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5261490345001221, + "learning_rate": 8.100214524900103e-05, + "loss": 0.008662862703204155, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.0087, + "step": 174, + "tokens/total": 5556880, + "tokens/train_per_sec_per_gpu": 24.62, + "tokens/trainable": 79498 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11292761564254761, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0027105510234832764, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00271, + "step": 175, + "tokens/total": 5588976, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 79922 + }, + { + "epoch": 0.6875, + "grad_norm": 0.33315548300743103, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0030926030594855547, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.0031, + "step": 176, + "tokens/total": 5618992, + "tokens/train_per_sec_per_gpu": 29.51, + "tokens/trainable": 80383 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.3136921226978302, + "learning_rate": 8.028654871074489e-05, + "loss": 0.005057850852608681, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00507, + "step": 177, + "tokens/total": 5651072, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 80824 + }, + { + "epoch": 0.6953125, + "grad_norm": 1.0526509284973145, + "learning_rate": 8.004589869885986e-05, + "loss": 0.010707498528063297, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01077, + "step": 178, + "tokens/total": 5683296, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 81243 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.1329648196697235, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0017975447699427605, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.0018, + "step": 179, + "tokens/total": 5715520, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 81716 + }, + { + "epoch": 0.703125, + "grad_norm": 1.0749201774597168, + "learning_rate": 7.95614819466576e-05, + "loss": 0.025617048144340515, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02595, + "step": 180, + "tokens/total": 5747488, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 82181 + }, + { + "epoch": 0.70703125, + "grad_norm": 1.0446465015411377, + "learning_rate": 7.931773536489872e-05, + "loss": 0.030236585065722466, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 35.9, + "memory/max_allocated (GiB)": 35.9, + "ppl": 1.0307, + "step": 181, + "tokens/total": 5777264, + "tokens/train_per_sec_per_gpu": 28.89, + "tokens/trainable": 82626 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22853964567184448, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0038214433006942272, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00383, + "step": 182, + "tokens/total": 5809264, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 83089 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.057821910828351974, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0004995397757738829, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0005, + "step": 183, + "tokens/total": 5841488, + "tokens/train_per_sec_per_gpu": 29.34, + "tokens/trainable": 83590 + }, + { + "epoch": 0.71875, + "grad_norm": 1.3264167308807373, + "learning_rate": 7.85804646415409e-05, + "loss": 0.013232000172138214, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01332, + "step": 184, + "tokens/total": 5873696, + "tokens/train_per_sec_per_gpu": 29.33, + "tokens/trainable": 84046 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.6027824878692627, + "learning_rate": 7.833273149760207e-05, + "loss": 0.012215660884976387, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01229, + "step": 185, + "tokens/total": 5905408, + "tokens/train_per_sec_per_gpu": 28.25, + "tokens/trainable": 84517 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.5848041772842407, + "learning_rate": 7.808402738346527e-05, + "loss": 0.024627480655908585, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.02493, + "step": 186, + "tokens/total": 5937392, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 84974 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.6823275089263916, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0028813849203288555, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00289, + "step": 187, + "tokens/total": 5969344, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 85460 + }, + { + "epoch": 0.734375, + "grad_norm": 0.41855958104133606, + "learning_rate": 7.758374768294647e-05, + "loss": 0.009796380065381527, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00984, + "step": 188, + "tokens/total": 6001296, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 85939 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6849080324172974, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004100777208805084, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00411, + "step": 189, + "tokens/total": 6031216, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 86377 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.15838374197483063, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003986812196671963, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00399, + "step": 190, + "tokens/total": 6063296, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 86834 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.3117486536502838, + "learning_rate": 7.682630588562518e-05, + "loss": 0.009132719598710537, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00917, + "step": 191, + "tokens/total": 6095216, + "tokens/train_per_sec_per_gpu": 26.65, + "tokens/trainable": 87265 + }, + { + "epoch": 0.75, + "grad_norm": 0.7837114334106445, + "learning_rate": 7.657199467573129e-05, + "loss": 0.007239446043968201, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00727, + "step": 192, + "tokens/total": 6127488, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 87747 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 4.159084180312351e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/training_args.bin b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..dec74f9f304b44cbe47132d86c4e6285e659f40b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-192/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634 +size 8273 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/README.md b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ba677470ac9c29f9770a4aacb65c644ebd671a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/deconf_wave_deconf_control/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/deconf_wave_deconf_control/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/adapter_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..21fcb5b00be777c20a2d2dcceffc539dc51679b4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/deconf_wave_deconf_control/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "gate_proj", + "v_proj", + "k_proj", + "up_proj", + "q_proj", + "o_proj", + "down_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/adapter_model.safetensors b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..673d4c4e770f8a6618f029177cf6e3f48dfc38eb --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:16d5d730ef95ad52bd73ac5c2f399d9bc75298abea43afc5ed5a1691843284e0 +size 547777976 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/chat_template.jinja b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/optimizer.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..0d8aa6e6e130692fb71d85ee983fa099a9b23a68 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f5c20255df1c1e1fb04d29b4f27c99eefa39ec3da290e74227715db6a9290348 +size 1048106435 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/rng_state.pth b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..bf22e553c9f7f8d998205a94df8f2e6493d7f6fd --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6fb783f63724e9b66b7a7e30df595af3bb870f43e925e33186646213d7600131 +size 14645 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/scheduler.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..734e9a25165547994d3add43bbfb2dd65b7e559d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f01c79694697cdd875b0827789740f4369175c3cc86ff2137cdbb9f8780c0822 +size 1465 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/tokenizer.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/tokenizer_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/tokens_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..f3781194c3454f10562f6d38247d4cf27ad15bf9 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/tokens_state.json @@ -0,0 +1 @@ +{"total": 7153040, "trainable": 102411} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/trainer_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..4efe969732fc2e989806eedf35fb4ec50a8b5a05 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/trainer_state.json @@ -0,0 +1,3170 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.875, + "eval_steps": 500, + "global_step": 224, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.2140090465545654, + "learning_rate": 0.0, + "loss": 0.16441112756729126, + "memory/device_reserved (GiB)": 38.08, + "memory/max_active (GiB)": 35.64, + "memory/max_allocated (GiB)": 35.64, + "ppl": 1.1787, + "step": 1, + "tokens/total": 32160, + "tokens/train_per_sec_per_gpu": 21.28, + "tokens/trainable": 470 + }, + { + "epoch": 0.0078125, + "grad_norm": 2.92620849609375, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15774038434028625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.17086, + "step": 2, + "tokens/total": 64192, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 922 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.135961890220642, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1664382815361023, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.18109, + "step": 3, + "tokens/total": 96224, + "tokens/train_per_sec_per_gpu": 28.54, + "tokens/trainable": 1396 + }, + { + "epoch": 0.015625, + "grad_norm": 0.880176305770874, + "learning_rate": 1.2e-05, + "loss": 0.15334013104438782, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.16572, + "step": 4, + "tokens/total": 128336, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 1850 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.1433289051055908, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.16352537274360657, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.17766, + "step": 5, + "tokens/total": 160784, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 2302 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9728212952613831, + "learning_rate": 2e-05, + "loss": 0.15144473314285278, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.16351, + "step": 6, + "tokens/total": 192624, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 2742 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.7796363830566406, + "learning_rate": 2.4e-05, + "loss": 0.13553467392921448, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.14515, + "step": 7, + "tokens/total": 224608, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 3208 + }, + { + "epoch": 0.03125, + "grad_norm": 1.2444697618484497, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.1077304258942604, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.11375, + "step": 8, + "tokens/total": 256592, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 3655 + }, + { + "epoch": 0.03515625, + "grad_norm": 2.7339375019073486, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.08772765100002289, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.09169, + "step": 9, + "tokens/total": 286400, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 4091 + }, + { + "epoch": 0.0390625, + "grad_norm": 4.109770774841309, + "learning_rate": 3.6e-05, + "loss": 0.130940243601799, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.1399, + "step": 10, + "tokens/total": 318352, + "tokens/train_per_sec_per_gpu": 29.13, + "tokens/trainable": 4553 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.3768608570098877, + "learning_rate": 4e-05, + "loss": 0.10059958696365356, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.10583, + "step": 11, + "tokens/total": 350336, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 4992 + }, + { + "epoch": 0.046875, + "grad_norm": 3.281402826309204, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.12574875354766846, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.134, + "step": 12, + "tokens/total": 382448, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 5494 + }, + { + "epoch": 0.05078125, + "grad_norm": 1.7738980054855347, + "learning_rate": 4.8e-05, + "loss": 0.033296722918748856, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03386, + "step": 13, + "tokens/total": 414352, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 5933 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.64290189743042, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06366822123527527, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.06574, + "step": 14, + "tokens/total": 446400, + "tokens/train_per_sec_per_gpu": 24.8, + "tokens/trainable": 6369 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.3691928386688232, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.08168518543243408, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.08511, + "step": 15, + "tokens/total": 478688, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 6820 + }, + { + "epoch": 0.0625, + "grad_norm": 2.4356393814086914, + "learning_rate": 6e-05, + "loss": 0.09031196683645248, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.09452, + "step": 16, + "tokens/total": 510512, + "tokens/train_per_sec_per_gpu": 25.7, + "tokens/trainable": 7258 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.55682373046875, + "learning_rate": 6.400000000000001e-05, + "loss": 0.07496573030948639, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.07785, + "step": 17, + "tokens/total": 542736, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 7710 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.1054646968841553, + "learning_rate": 6.800000000000001e-05, + "loss": 0.05218805745244026, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.05357, + "step": 18, + "tokens/total": 574832, + "tokens/train_per_sec_per_gpu": 25.78, + "tokens/trainable": 8174 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.7429099082946777, + "learning_rate": 7.2e-05, + "loss": 0.03790827840566635, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.03864, + "step": 19, + "tokens/total": 606976, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 8617 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9508209228515625, + "learning_rate": 7.6e-05, + "loss": 0.04993613809347153, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0512, + "step": 20, + "tokens/total": 639200, + "tokens/train_per_sec_per_gpu": 24.91, + "tokens/trainable": 9037 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.0297844409942627, + "learning_rate": 8e-05, + "loss": 0.017908263951539993, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.01807, + "step": 21, + "tokens/total": 671072, + "tokens/train_per_sec_per_gpu": 30.42, + "tokens/trainable": 9503 + }, + { + "epoch": 0.0859375, + "grad_norm": 0.9279879331588745, + "learning_rate": 8.4e-05, + "loss": 0.013478193432092667, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01357, + "step": 22, + "tokens/total": 703136, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 9972 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.526488184928894, + "learning_rate": 8.800000000000001e-05, + "loss": 0.017310388386249542, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01746, + "step": 23, + "tokens/total": 735120, + "tokens/train_per_sec_per_gpu": 31.45, + "tokens/trainable": 10447 + }, + { + "epoch": 0.09375, + "grad_norm": 3.4049417972564697, + "learning_rate": 9.200000000000001e-05, + "loss": 0.0353429913520813, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.03597, + "step": 24, + "tokens/total": 767040, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 10899 + }, + { + "epoch": 0.09765625, + "grad_norm": 7.511656761169434, + "learning_rate": 9.6e-05, + "loss": 0.1335875540971756, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.14292, + "step": 25, + "tokens/total": 798976, + "tokens/train_per_sec_per_gpu": 27.72, + "tokens/trainable": 11360 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.0069234371185303, + "learning_rate": 0.0001, + "loss": 0.04382415860891342, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0448, + "step": 26, + "tokens/total": 830880, + "tokens/train_per_sec_per_gpu": 24.61, + "tokens/trainable": 11775 + }, + { + "epoch": 0.10546875, + "grad_norm": 7.807938575744629, + "learning_rate": 9.99990636831587e-05, + "loss": 0.01210833340883255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.01218, + "step": 27, + "tokens/total": 860960, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 12242 + }, + { + "epoch": 0.109375, + "grad_norm": 2.23823881149292, + "learning_rate": 9.999625477159879e-05, + "loss": 0.0272970050573349, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02767, + "step": 28, + "tokens/total": 892816, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 12696 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.315584421157837, + "learning_rate": 9.999157338221051e-05, + "loss": 0.025516577064990997, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.02584, + "step": 29, + "tokens/total": 924736, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 13153 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.8413651585578918, + "learning_rate": 9.998501970980562e-05, + "loss": 0.007838520221412182, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00787, + "step": 30, + "tokens/total": 954592, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 13624 + }, + { + "epoch": 0.12109375, + "grad_norm": 1.6797157526016235, + "learning_rate": 9.997659402710915e-05, + "loss": 0.03002442792057991, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.03048, + "step": 31, + "tokens/total": 986592, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 14064 + }, + { + "epoch": 0.125, + "grad_norm": 1.3320434093475342, + "learning_rate": 9.996629668474818e-05, + "loss": 0.03772728145122528, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.03845, + "step": 32, + "tokens/total": 1018720, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 14565 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.6335167288780212, + "learning_rate": 9.995412811123711e-05, + "loss": 0.013970119878649712, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01407, + "step": 33, + "tokens/total": 1050624, + "tokens/train_per_sec_per_gpu": 28.18, + "tokens/trainable": 15005 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.5626921653747559, + "learning_rate": 9.994008881295999e-05, + "loss": 0.016654539853334427, + "memory/device_reserved (GiB)": 37.74, + "memory/max_active (GiB)": 36.22, + "memory/max_allocated (GiB)": 36.22, + "ppl": 1.01679, + "step": 34, + "tokens/total": 1082176, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 15459 + }, + { + "epoch": 0.13671875, + "grad_norm": 0.603219211101532, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021408583968877792, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02164, + "step": 35, + "tokens/total": 1113968, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 15960 + }, + { + "epoch": 0.140625, + "grad_norm": 0.9779674410820007, + "learning_rate": 9.99064004568618e-05, + "loss": 0.03277068957686424, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03331, + "step": 36, + "tokens/total": 1145952, + "tokens/train_per_sec_per_gpu": 28.68, + "tokens/trainable": 16428 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.8855255842208862, + "learning_rate": 9.988675280095074e-05, + "loss": 0.017048098146915436, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01719, + "step": 37, + "tokens/total": 1177872, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 16884 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.9266122579574585, + "learning_rate": 9.986523722403528e-05, + "loss": 0.045863546431064606, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.04693, + "step": 38, + "tokens/total": 1209776, + "tokens/train_per_sec_per_gpu": 26.91, + "tokens/trainable": 17341 + }, + { + "epoch": 0.15234375, + "grad_norm": 0.8280895948410034, + "learning_rate": 9.984185462146642e-05, + "loss": 0.033055759966373444, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03361, + "step": 39, + "tokens/total": 1241888, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 17786 + }, + { + "epoch": 0.15625, + "grad_norm": 5.029730319976807, + "learning_rate": 9.98166059662897e-05, + "loss": 0.025589998811483383, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02592, + "step": 40, + "tokens/total": 1274032, + "tokens/train_per_sec_per_gpu": 29.22, + "tokens/trainable": 18262 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.5688032507896423, + "learning_rate": 9.978949230920472e-05, + "loss": 0.016095872968435287, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01623, + "step": 41, + "tokens/total": 1306000, + "tokens/train_per_sec_per_gpu": 27.49, + "tokens/trainable": 18716 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8217663168907166, + "learning_rate": 9.976051477852141e-05, + "loss": 0.026798982173204422, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02716, + "step": 42, + "tokens/total": 1337936, + "tokens/train_per_sec_per_gpu": 27.51, + "tokens/trainable": 19157 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.7309132218360901, + "learning_rate": 9.972967458011312e-05, + "loss": 0.033920761197805405, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0345, + "step": 43, + "tokens/total": 1369904, + "tokens/train_per_sec_per_gpu": 30.66, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.8134022355079651, + "learning_rate": 9.96969729973664e-05, + "loss": 0.024005085229873657, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0243, + "step": 44, + "tokens/total": 1401984, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 20119 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.35084816813468933, + "learning_rate": 9.966241139112754e-05, + "loss": 0.013319254852831364, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01341, + "step": 45, + "tokens/total": 1434032, + "tokens/train_per_sec_per_gpu": 27.45, + "tokens/trainable": 20560 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.8701573014259338, + "learning_rate": 9.96259911996461e-05, + "loss": 0.01356641948223114, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01366, + "step": 46, + "tokens/total": 1465872, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 20992 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.4406442642211914, + "learning_rate": 9.958771393851491e-05, + "loss": 0.04396980255842209, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 35.86, + "memory/max_allocated (GiB)": 35.86, + "ppl": 1.04495, + "step": 47, + "tokens/total": 1495440, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 21441 + }, + { + "epoch": 0.1875, + "grad_norm": 0.4966669976711273, + "learning_rate": 9.954758120060702e-05, + "loss": 0.011179964058101177, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01124, + "step": 48, + "tokens/total": 1527776, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 21901 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.41124436259269714, + "learning_rate": 9.950559465600948e-05, + "loss": 0.0063599650748074055, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00638, + "step": 49, + "tokens/total": 1559760, + "tokens/train_per_sec_per_gpu": 27.73, + "tokens/trainable": 22341 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.829739511013031, + "learning_rate": 9.946175605195379e-05, + "loss": 0.015785304829478264, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01591, + "step": 50, + "tokens/total": 1592000, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 22833 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5181736946105957, + "learning_rate": 9.941606721274322e-05, + "loss": 0.008393109776079655, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00843, + "step": 51, + "tokens/total": 1623920, + "tokens/train_per_sec_per_gpu": 25.61, + "tokens/trainable": 23277 + }, + { + "epoch": 0.203125, + "grad_norm": 1.5778990983963013, + "learning_rate": 9.936853003967685e-05, + "loss": 0.024984844028949738, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0253, + "step": 52, + "tokens/total": 1656032, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 23759 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.4938714504241943, + "learning_rate": 9.93191465109705e-05, + "loss": 0.038523320108652115, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03927, + "step": 53, + "tokens/total": 1688368, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 24193 + }, + { + "epoch": 0.2109375, + "grad_norm": 2.1427106857299805, + "learning_rate": 9.926791868167438e-05, + "loss": 0.06302924454212189, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.06506, + "step": 54, + "tokens/total": 1720560, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 24619 + }, + { + "epoch": 0.21484375, + "grad_norm": 1.8365188837051392, + "learning_rate": 9.921484868358753e-05, + "loss": 0.052160464227199554, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.05354, + "step": 55, + "tokens/total": 1752464, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 25075 + }, + { + "epoch": 0.21875, + "grad_norm": 0.7779749631881714, + "learning_rate": 9.915993872516924e-05, + "loss": 0.021328264847397804, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.02156, + "step": 56, + "tokens/total": 1784336, + "tokens/train_per_sec_per_gpu": 26.42, + "tokens/trainable": 25519 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.4189571142196655, + "learning_rate": 9.9103191091447e-05, + "loss": 0.014279939234256744, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01438, + "step": 57, + "tokens/total": 1816304, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 25966 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.3834710717201233, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01883944310247898, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01902, + "step": 58, + "tokens/total": 1848304, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 26423 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.7693278789520264, + "learning_rate": 9.898419232046825e-05, + "loss": 0.03533369302749634, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03597, + "step": 59, + "tokens/total": 1880640, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 26934 + }, + { + "epoch": 0.234375, + "grad_norm": 0.4606446325778961, + "learning_rate": 9.892194613523633e-05, + "loss": 0.021482212468981743, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02171, + "step": 60, + "tokens/total": 1912592, + "tokens/train_per_sec_per_gpu": 27.64, + "tokens/trainable": 27393 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.46030017733573914, + "learning_rate": 9.885787217854357e-05, + "loss": 0.017044665291905403, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01719, + "step": 61, + "tokens/total": 1944544, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 27852 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.7918264269828796, + "learning_rate": 9.879197311676887e-05, + "loss": 0.016954131424427032, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0171, + "step": 62, + "tokens/total": 1976688, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 28292 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.4787926971912384, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01525965891778469, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 63, + "tokens/total": 2008928, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 28798 + }, + { + "epoch": 0.25, + "grad_norm": 0.9033918976783752, + "learning_rate": 9.865471072312528e-05, + "loss": 0.02902211993932724, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02945, + "step": 64, + "tokens/total": 2041120, + "tokens/train_per_sec_per_gpu": 29.83, + "tokens/trainable": 29283 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.39305707812309265, + "learning_rate": 9.858335310330492e-05, + "loss": 0.00901167094707489, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.32, + "memory/max_allocated (GiB)": 36.32, + "ppl": 1.00905, + "step": 65, + "tokens/total": 2073024, + "tokens/train_per_sec_per_gpu": 26.48, + "tokens/trainable": 29745 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4494345784187317, + "learning_rate": 9.851018180226185e-05, + "loss": 0.007426132448017597, + "memory/device_reserved (GiB)": 37.47, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00745, + "step": 66, + "tokens/total": 2105184, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 30186 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.629649817943573, + "learning_rate": 9.843519986495259e-05, + "loss": 0.010609081946313381, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01067, + "step": 67, + "tokens/total": 2137264, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 30622 + }, + { + "epoch": 0.265625, + "grad_norm": 0.7591162919998169, + "learning_rate": 9.835841041168162e-05, + "loss": 0.00926615484058857, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00931, + "step": 68, + "tokens/total": 2169424, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 31097 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.7288486361503601, + "learning_rate": 9.82798166379715e-05, + "loss": 0.024310583248734474, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.02461, + "step": 69, + "tokens/total": 2201424, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 31595 + }, + { + "epoch": 0.2734375, + "grad_norm": 2.134056329727173, + "learning_rate": 9.819942181443002e-05, + "loss": 0.008422967046499252, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00846, + "step": 70, + "tokens/total": 2233424, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 32036 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6362871527671814, + "learning_rate": 9.811722928661392e-05, + "loss": 0.01577262207865715, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0159, + "step": 71, + "tokens/total": 2265328, + "tokens/train_per_sec_per_gpu": 23.65, + "tokens/trainable": 32428 + }, + { + "epoch": 0.28125, + "grad_norm": 0.6091912388801575, + "learning_rate": 9.803324247488975e-05, + "loss": 0.0064800274558365345, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0065, + "step": 72, + "tokens/total": 2297248, + "tokens/train_per_sec_per_gpu": 26.86, + "tokens/trainable": 32880 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.14533838629722595, + "learning_rate": 9.794746487429161e-05, + "loss": 0.001991454279050231, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00199, + "step": 73, + "tokens/total": 2329152, + "tokens/train_per_sec_per_gpu": 27.97, + "tokens/trainable": 33323 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.5400826930999756, + "learning_rate": 9.785990005437554e-05, + "loss": 0.017662350088357925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01782, + "step": 74, + "tokens/total": 2361344, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 33792 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.6132436394691467, + "learning_rate": 9.777055165907117e-05, + "loss": 0.01787603087723255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01804, + "step": 75, + "tokens/total": 2393168, + "tokens/train_per_sec_per_gpu": 29.53, + "tokens/trainable": 34223 + }, + { + "epoch": 0.296875, + "grad_norm": 0.7287399172782898, + "learning_rate": 9.767942340652993e-05, + "loss": 0.024754449725151062, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.31, + "memory/max_allocated (GiB)": 36.31, + "ppl": 1.02506, + "step": 76, + "tokens/total": 2424880, + "tokens/train_per_sec_per_gpu": 24.92, + "tokens/trainable": 34649 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.6790545582771301, + "learning_rate": 9.758651908897035e-05, + "loss": 0.014308430254459381, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01441, + "step": 77, + "tokens/total": 2457120, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 35094 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6203456521034241, + "learning_rate": 9.749184257252033e-05, + "loss": 0.034353043884038925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03495, + "step": 78, + "tokens/total": 2489104, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 35534 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.6238825917243958, + "learning_rate": 9.739539779705614e-05, + "loss": 0.032295141369104385, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.03282, + "step": 79, + "tokens/total": 2521136, + "tokens/train_per_sec_per_gpu": 29.68, + "tokens/trainable": 36029 + }, + { + "epoch": 0.3125, + "grad_norm": 0.18977171182632446, + "learning_rate": 9.729718877603861e-05, + "loss": 0.007260813377797604, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00729, + "step": 80, + "tokens/total": 2553168, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 36469 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.8054049611091614, + "learning_rate": 9.719721959634592e-05, + "loss": 0.034894898533821106, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.03551, + "step": 81, + "tokens/total": 2585584, + "tokens/train_per_sec_per_gpu": 25.75, + "tokens/trainable": 36906 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.6884189248085022, + "learning_rate": 9.709549441810375e-05, + "loss": 0.018210047855973244, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.01838, + "step": 82, + "tokens/total": 2617808, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 37390 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.6490103602409363, + "learning_rate": 9.699201747451195e-05, + "loss": 0.02030857279896736, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.02052, + "step": 83, + "tokens/total": 2649664, + "tokens/train_per_sec_per_gpu": 25.8, + "tokens/trainable": 37838 + }, + { + "epoch": 0.328125, + "grad_norm": 0.2439805567264557, + "learning_rate": 9.688679307166854e-05, + "loss": 0.008518049493432045, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00855, + "step": 84, + "tokens/total": 2681728, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 38310 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6747375726699829, + "learning_rate": 9.677982558839042e-05, + "loss": 0.029895059764385223, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.03035, + "step": 85, + "tokens/total": 2713488, + "tokens/train_per_sec_per_gpu": 28.49, + "tokens/trainable": 38789 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.582666277885437, + "learning_rate": 9.66711194760312e-05, + "loss": 0.014998597092926502, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01511, + "step": 86, + "tokens/total": 2745536, + "tokens/train_per_sec_per_gpu": 30.27, + "tokens/trainable": 39277 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.5364654660224915, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011567167937755585, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01163, + "step": 87, + "tokens/total": 2777680, + "tokens/train_per_sec_per_gpu": 29.73, + "tokens/trainable": 39737 + }, + { + "epoch": 0.34375, + "grad_norm": 0.34024426341056824, + "learning_rate": 9.644850953105288e-05, + "loss": 0.01014852337539196, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0102, + "step": 88, + "tokens/total": 2809568, + "tokens/train_per_sec_per_gpu": 24.68, + "tokens/trainable": 40179 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.32472267746925354, + "learning_rate": 9.633461496214225e-05, + "loss": 0.01229981891810894, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01238, + "step": 89, + "tokens/total": 2841760, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 40649 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.7906899452209473, + "learning_rate": 9.621900029118195e-05, + "loss": 0.02552586793899536, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.02585, + "step": 90, + "tokens/total": 2871696, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 41080 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.7444764971733093, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0054441955871880054, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00546, + "step": 91, + "tokens/total": 2903968, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 41599 + }, + { + "epoch": 0.359375, + "grad_norm": 0.1911851465702057, + "learning_rate": 9.598262995928611e-05, + "loss": 0.004665314219892025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00468, + "step": 92, + "tokens/total": 2936048, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 42076 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.21238519251346588, + "learning_rate": 9.586188413468492e-05, + "loss": 0.003917238209396601, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00392, + "step": 93, + "tokens/total": 2968176, + "tokens/train_per_sec_per_gpu": 28.07, + "tokens/trainable": 42522 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.41729435324668884, + "learning_rate": 9.57394378802934e-05, + "loss": 0.013555807992815971, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01365, + "step": 94, + "tokens/total": 3000304, + "tokens/train_per_sec_per_gpu": 28.69, + "tokens/trainable": 42974 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.7246769070625305, + "learning_rate": 9.56152962916e-05, + "loss": 0.020221399143338203, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.02043, + "step": 95, + "tokens/total": 3032080, + "tokens/train_per_sec_per_gpu": 25.35, + "tokens/trainable": 43380 + }, + { + "epoch": 0.375, + "grad_norm": 0.16139821708202362, + "learning_rate": 9.548946453464296e-05, + "loss": 0.003853276837617159, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00386, + "step": 96, + "tokens/total": 3064320, + "tokens/train_per_sec_per_gpu": 28.47, + "tokens/trainable": 43865 + }, + { + "epoch": 0.37890625, + "grad_norm": 1.0509370565414429, + "learning_rate": 9.53619478457953e-05, + "loss": 0.015099998563528061, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01521, + "step": 97, + "tokens/total": 3096144, + "tokens/train_per_sec_per_gpu": 26.61, + "tokens/trainable": 44328 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.6859350204467773, + "learning_rate": 9.523275153154695e-05, + "loss": 0.004579808097332716, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00459, + "step": 98, + "tokens/total": 3127600, + "tokens/train_per_sec_per_gpu": 25.91, + "tokens/trainable": 44778 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5686795711517334, + "learning_rate": 9.51018809682839e-05, + "loss": 0.022520575672388077, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02278, + "step": 99, + "tokens/total": 3159520, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 45225 + }, + { + "epoch": 0.390625, + "grad_norm": 1.6308369636535645, + "learning_rate": 9.49693416020645e-05, + "loss": 0.011883002705872059, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01195, + "step": 100, + "tokens/total": 3191568, + "tokens/train_per_sec_per_gpu": 29.75, + "tokens/trainable": 45678 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.7186088562011719, + "learning_rate": 9.483513894839276e-05, + "loss": 0.0060632615350186825, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00608, + "step": 101, + "tokens/total": 3223648, + "tokens/train_per_sec_per_gpu": 27.14, + "tokens/trainable": 46125 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.12036337703466415, + "learning_rate": 9.469927859198888e-05, + "loss": 0.0019118499476462603, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00191, + "step": 102, + "tokens/total": 3255744, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 46612 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.29928353428840637, + "learning_rate": 9.456176618655689e-05, + "loss": 0.00944911316037178, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00949, + "step": 103, + "tokens/total": 3287792, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 47059 + }, + { + "epoch": 0.40625, + "grad_norm": 0.3766293227672577, + "learning_rate": 9.442260745454927e-05, + "loss": 0.008984047919511795, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.68, + "memory/max_allocated (GiB)": 36.68, + "ppl": 1.00902, + "step": 104, + "tokens/total": 3320016, + "tokens/train_per_sec_per_gpu": 28.92, + "tokens/trainable": 47536 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3481820523738861, + "learning_rate": 9.428180818692884e-05, + "loss": 0.005444999784231186, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00546, + "step": 105, + "tokens/total": 3351952, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 48037 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.8414323329925537, + "learning_rate": 9.413937424292791e-05, + "loss": 0.032438624650239944, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03297, + "step": 106, + "tokens/total": 3384032, + "tokens/train_per_sec_per_gpu": 27.11, + "tokens/trainable": 48491 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.5067655444145203, + "learning_rate": 9.399531154980424e-05, + "loss": 0.008442888967692852, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00848, + "step": 107, + "tokens/total": 3416192, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 48940 + }, + { + "epoch": 0.421875, + "grad_norm": 0.41617852449417114, + "learning_rate": 9.384962610259455e-05, + "loss": 0.008386321365833282, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00842, + "step": 108, + "tokens/total": 3448080, + "tokens/train_per_sec_per_gpu": 26.56, + "tokens/trainable": 49389 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1587515026330948, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0031663556583225727, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.03, + "memory/max_allocated (GiB)": 36.03, + "ppl": 1.00317, + "step": 109, + "tokens/total": 3477952, + "tokens/train_per_sec_per_gpu": 29.8, + "tokens/trainable": 49838 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2903148829936981, + "learning_rate": 9.355341126345868e-05, + "loss": 0.007958847098052502, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00799, + "step": 110, + "tokens/total": 3509776, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 50310 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.642983078956604, + "learning_rate": 9.340289419824107e-05, + "loss": 0.006748045329004526, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00677, + "step": 111, + "tokens/total": 3541808, + "tokens/train_per_sec_per_gpu": 28.04, + "tokens/trainable": 50755 + }, + { + "epoch": 0.4375, + "grad_norm": 0.14928282797336578, + "learning_rate": 9.325077903184159e-05, + "loss": 0.0027839094400405884, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00279, + "step": 112, + "tokens/total": 3574064, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 51213 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.05922335386276245, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0009622900979593396, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00096, + "step": 113, + "tokens/total": 3605984, + "tokens/train_per_sec_per_gpu": 27.99, + "tokens/trainable": 51660 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.44227731227874756, + "learning_rate": 9.2941779782269e-05, + "loss": 0.002763954224064946, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00277, + "step": 114, + "tokens/total": 3638080, + "tokens/train_per_sec_per_gpu": 27.03, + "tokens/trainable": 52133 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5176818370819092, + "learning_rate": 9.278490855781596e-05, + "loss": 0.015623578801751137, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01575, + "step": 115, + "tokens/total": 3670016, + "tokens/train_per_sec_per_gpu": 31.44, + "tokens/trainable": 52580 + }, + { + "epoch": 0.453125, + "grad_norm": 0.4265497922897339, + "learning_rate": 9.262646494908604e-05, + "loss": 0.00794076919555664, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00797, + "step": 116, + "tokens/total": 3701984, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 53033 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.40457987785339355, + "learning_rate": 9.246645554956457e-05, + "loss": 0.022708112373948097, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02297, + "step": 117, + "tokens/total": 3733920, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 53517 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.01414525043219328, + "learning_rate": 9.230488701789578e-05, + "loss": 0.00020209309877827764, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.0002, + "step": 118, + "tokens/total": 3765840, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 53967 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.11359831690788269, + "learning_rate": 9.214176607760577e-05, + "loss": 0.0012168455868959427, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00122, + "step": 119, + "tokens/total": 3797680, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 54430 + }, + { + "epoch": 0.46875, + "grad_norm": 0.35982534289360046, + "learning_rate": 9.197709951682268e-05, + "loss": 0.004110721405595541, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00412, + "step": 120, + "tokens/total": 3829920, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 54896 + }, + { + "epoch": 0.47265625, + "grad_norm": 1.3690382242202759, + "learning_rate": 9.181089418799428e-05, + "loss": 0.02705790475010872, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02743, + "step": 121, + "tokens/total": 3862080, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 55379 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.351891428232193, + "learning_rate": 9.164315700760271e-05, + "loss": 0.0020163573790341616, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00202, + "step": 122, + "tokens/total": 3893968, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 55803 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4747961461544037, + "learning_rate": 9.147389495587671e-05, + "loss": 0.012588979676365852, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01267, + "step": 123, + "tokens/total": 3926160, + "tokens/train_per_sec_per_gpu": 30.14, + "tokens/trainable": 56249 + }, + { + "epoch": 0.484375, + "grad_norm": 0.4064250588417053, + "learning_rate": 9.130311507650116e-05, + "loss": 0.0022941348142921925, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0023, + "step": 124, + "tokens/total": 3958208, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 56713 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.22056929767131805, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0022175831254571676, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00222, + "step": 125, + "tokens/total": 3990208, + "tokens/train_per_sec_per_gpu": 32.57, + "tokens/trainable": 57196 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5698862671852112, + "learning_rate": 9.09570303250602e-05, + "loss": 0.010205100290477276, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01026, + "step": 126, + "tokens/total": 4022224, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 57680 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.2593887448310852, + "learning_rate": 9.078173985499394e-05, + "loss": 0.0032277877908200026, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00323, + "step": 127, + "tokens/total": 4053952, + "tokens/train_per_sec_per_gpu": 26.51, + "tokens/trainable": 58110 + }, + { + "epoch": 0.5, + "grad_norm": 0.3735230565071106, + "learning_rate": 9.060496036067713e-05, + "loss": 0.01031105499714613, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01036, + "step": 128, + "tokens/total": 4086144, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 58534 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.1332484930753708, + "learning_rate": 9.042669919862615e-05, + "loss": 0.0012390739284455776, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00124, + "step": 129, + "tokens/total": 4118096, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 58998 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.20182718336582184, + "learning_rate": 9.024696378701557e-05, + "loss": 0.004501686431467533, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00451, + "step": 130, + "tokens/total": 4150272, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 59448 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.529080331325531, + "learning_rate": 9.006576160536948e-05, + "loss": 0.00763288140296936, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00766, + "step": 131, + "tokens/total": 4181904, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 59906 + }, + { + "epoch": 0.515625, + "grad_norm": 0.21625548601150513, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0024378676898777485, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00244, + "step": 132, + "tokens/total": 4214032, + "tokens/train_per_sec_per_gpu": 28.36, + "tokens/trainable": 60350 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.37116539478302, + "learning_rate": 8.969898715494506e-05, + "loss": 0.0037849934305995703, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00379, + "step": 133, + "tokens/total": 4246064, + "tokens/train_per_sec_per_gpu": 30.18, + "tokens/trainable": 60831 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5646623373031616, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006517104804515839, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00654, + "step": 134, + "tokens/total": 4278224, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 61305 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.6371963620185852, + "learning_rate": 8.932643689864568e-05, + "loss": 0.008687382563948631, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00873, + "step": 135, + "tokens/total": 4310112, + "tokens/train_per_sec_per_gpu": 31.48, + "tokens/trainable": 61792 + }, + { + "epoch": 0.53125, + "grad_norm": 0.8210524916648865, + "learning_rate": 8.913801518498845e-05, + "loss": 0.008436474949121475, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00847, + "step": 136, + "tokens/total": 4342192, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 62253 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.5692176222801208, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017321188002824783, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.01747, + "step": 137, + "tokens/total": 4374096, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.12828375399112701, + "learning_rate": 8.875691779131569e-05, + "loss": 0.0007754197577014565, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00078, + "step": 138, + "tokens/total": 4406272, + "tokens/train_per_sec_per_gpu": 24.22, + "tokens/trainable": 63145 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.3188190162181854, + "learning_rate": 8.856425797031829e-05, + "loss": 0.006707335356622934, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00673, + "step": 139, + "tokens/total": 4438112, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 63587 + }, + { + "epoch": 0.546875, + "grad_norm": 0.6216707229614258, + "learning_rate": 8.837020140354295e-05, + "loss": 0.007024695165455341, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00705, + "step": 140, + "tokens/total": 4470128, + "tokens/train_per_sec_per_gpu": 30.64, + "tokens/trainable": 64052 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.1480305939912796, + "learning_rate": 8.817475616647554e-05, + "loss": 0.00229115248657763, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00229, + "step": 141, + "tokens/total": 4502096, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 64526 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22254763543605804, + "learning_rate": 8.797793039239017e-05, + "loss": 0.0029198499396443367, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00292, + "step": 142, + "tokens/total": 4534496, + "tokens/train_per_sec_per_gpu": 29.08, + "tokens/trainable": 64983 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.6901941895484924, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013051149435341358, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01314, + "step": 143, + "tokens/total": 4566720, + "tokens/train_per_sec_per_gpu": 31.56, + "tokens/trainable": 65492 + }, + { + "epoch": 0.5625, + "grad_norm": 0.9545458555221558, + "learning_rate": 8.758017005316988e-05, + "loss": 0.02235381305217743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.02261, + "step": 144, + "tokens/total": 4598880, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 65925 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.5515892505645752, + "learning_rate": 8.737925204046629e-05, + "loss": 0.02038375660777092, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.02059, + "step": 145, + "tokens/total": 4630864, + "tokens/train_per_sec_per_gpu": 26.3, + "tokens/trainable": 66383 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3512430191040039, + "learning_rate": 8.717698659491851e-05, + "loss": 0.008233271539211273, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00827, + "step": 146, + "tokens/total": 4662864, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 66840 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.6830361485481262, + "learning_rate": 8.697338213361735e-05, + "loss": 0.013419180177152157, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01351, + "step": 147, + "tokens/total": 4694912, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 67297 + }, + { + "epoch": 0.578125, + "grad_norm": 0.2117031067609787, + "learning_rate": 8.676844712937552e-05, + "loss": 0.004134457092732191, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00414, + "step": 148, + "tokens/total": 4727120, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 67769 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.378053218126297, + "learning_rate": 8.656219011037509e-05, + "loss": 0.009747720323503017, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0098, + "step": 149, + "tokens/total": 4759168, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 68257 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.21076741814613342, + "learning_rate": 8.63546196598125e-05, + "loss": 0.005035653710365295, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00505, + "step": 150, + "tokens/total": 4791008, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 68706 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.25075697898864746, + "learning_rate": 8.614574441554145e-05, + "loss": 0.006027051247656345, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00605, + "step": 151, + "tokens/total": 4822912, + "tokens/train_per_sec_per_gpu": 27.95, + "tokens/trainable": 69131 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5774204730987549, + "learning_rate": 8.593557306971349e-05, + "loss": 0.018617186695337296, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01879, + "step": 152, + "tokens/total": 4854816, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 69586 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.1364293098449707, + "learning_rate": 8.572411436841618e-05, + "loss": 0.002049289643764496, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00205, + "step": 153, + "tokens/total": 4886656, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 70061 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.25387004017829895, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0044304742477834225, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00444, + "step": 154, + "tokens/total": 4918768, + "tokens/train_per_sec_per_gpu": 23.26, + "tokens/trainable": 70467 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3047125041484833, + "learning_rate": 8.529737015125824e-05, + "loss": 0.0043389927595853806, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00435, + "step": 155, + "tokens/total": 4951008, + "tokens/train_per_sec_per_gpu": 27.53, + "tokens/trainable": 70933 + }, + { + "epoch": 0.609375, + "grad_norm": 0.1442601978778839, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0031973645091056824, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.0032, + "step": 156, + "tokens/total": 4982976, + "tokens/train_per_sec_per_gpu": 28.86, + "tokens/trainable": 71382 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.2608455717563629, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0032951058819890022, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0033, + "step": 157, + "tokens/total": 5014640, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 71808 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.637295663356781, + "learning_rate": 8.464782037243449e-05, + "loss": 0.01436957623809576, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.01447, + "step": 158, + "tokens/total": 5046672, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 72249 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.9834056496620178, + "learning_rate": 8.442882418044202e-05, + "loss": 0.02375311404466629, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.02404, + "step": 159, + "tokens/total": 5078528, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 72726 + }, + { + "epoch": 0.625, + "grad_norm": 0.35432615876197815, + "learning_rate": 8.420860333495179e-05, + "loss": 0.006119017023593187, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00614, + "step": 160, + "tokens/total": 5110704, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 73148 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.44270065426826477, + "learning_rate": 8.398716700025208e-05, + "loss": 0.01352179329842329, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.72, + "memory/max_allocated (GiB)": 36.72, + "ppl": 1.01361, + "step": 161, + "tokens/total": 5142944, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 73600 + }, + { + "epoch": 0.6328125, + "grad_norm": 1.2181634902954102, + "learning_rate": 8.376452439121266e-05, + "loss": 0.01820964366197586, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.01838, + "step": 162, + "tokens/total": 5174592, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 74068 + }, + { + "epoch": 0.63671875, + "grad_norm": 1.9321303367614746, + "learning_rate": 8.354068477290124e-05, + "loss": 0.009742327965795994, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00979, + "step": 163, + "tokens/total": 5206432, + "tokens/train_per_sec_per_gpu": 29.89, + "tokens/trainable": 74527 + }, + { + "epoch": 0.640625, + "grad_norm": 0.3545796275138855, + "learning_rate": 8.331565746019807e-05, + "loss": 0.00975855067372322, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00981, + "step": 164, + "tokens/total": 5238384, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 74992 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.35368478298187256, + "learning_rate": 8.308945181740812e-05, + "loss": 0.010195466689765453, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01025, + "step": 165, + "tokens/total": 5270272, + "tokens/train_per_sec_per_gpu": 29.41, + "tokens/trainable": 75442 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.34957799315452576, + "learning_rate": 8.286207725787153e-05, + "loss": 0.009373681619763374, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00942, + "step": 166, + "tokens/total": 5302368, + "tokens/train_per_sec_per_gpu": 26.2, + "tokens/trainable": 75887 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.3684624135494232, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008211556822061539, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00825, + "step": 167, + "tokens/total": 5334496, + "tokens/train_per_sec_per_gpu": 28.21, + "tokens/trainable": 76331 + }, + { + "epoch": 0.65625, + "grad_norm": 0.12826858460903168, + "learning_rate": 8.240385928474219e-05, + "loss": 0.004492453299462795, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0045, + "step": 168, + "tokens/total": 5364288, + "tokens/train_per_sec_per_gpu": 30.16, + "tokens/trainable": 76745 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.31828173995018005, + "learning_rate": 8.217303493946967e-05, + "loss": 0.005733514670282602, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00575, + "step": 169, + "tokens/total": 5396512, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 77201 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6174353361129761, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01915649324655533, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01934, + "step": 170, + "tokens/total": 5428624, + "tokens/train_per_sec_per_gpu": 27.75, + "tokens/trainable": 77655 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.2902314066886902, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0036087254993617535, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00362, + "step": 171, + "tokens/total": 5460880, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 78122 + }, + { + "epoch": 0.671875, + "grad_norm": 0.3267754912376404, + "learning_rate": 8.147381587530713e-05, + "loss": 0.008090222254395485, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00812, + "step": 172, + "tokens/total": 5493088, + "tokens/train_per_sec_per_gpu": 27.86, + "tokens/trainable": 78576 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.4086189866065979, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006351984106004238, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00637, + "step": 173, + "tokens/total": 5525312, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 79059 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5261490345001221, + "learning_rate": 8.100214524900103e-05, + "loss": 0.008662862703204155, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.0087, + "step": 174, + "tokens/total": 5556880, + "tokens/train_per_sec_per_gpu": 24.62, + "tokens/trainable": 79498 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11292761564254761, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0027105510234832764, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00271, + "step": 175, + "tokens/total": 5588976, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 79922 + }, + { + "epoch": 0.6875, + "grad_norm": 0.33315548300743103, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0030926030594855547, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.0031, + "step": 176, + "tokens/total": 5618992, + "tokens/train_per_sec_per_gpu": 29.51, + "tokens/trainable": 80383 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.3136921226978302, + "learning_rate": 8.028654871074489e-05, + "loss": 0.005057850852608681, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00507, + "step": 177, + "tokens/total": 5651072, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 80824 + }, + { + "epoch": 0.6953125, + "grad_norm": 1.0526509284973145, + "learning_rate": 8.004589869885986e-05, + "loss": 0.010707498528063297, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01077, + "step": 178, + "tokens/total": 5683296, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 81243 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.1329648196697235, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0017975447699427605, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.0018, + "step": 179, + "tokens/total": 5715520, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 81716 + }, + { + "epoch": 0.703125, + "grad_norm": 1.0749201774597168, + "learning_rate": 7.95614819466576e-05, + "loss": 0.025617048144340515, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02595, + "step": 180, + "tokens/total": 5747488, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 82181 + }, + { + "epoch": 0.70703125, + "grad_norm": 1.0446465015411377, + "learning_rate": 7.931773536489872e-05, + "loss": 0.030236585065722466, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 35.9, + "memory/max_allocated (GiB)": 35.9, + "ppl": 1.0307, + "step": 181, + "tokens/total": 5777264, + "tokens/train_per_sec_per_gpu": 28.89, + "tokens/trainable": 82626 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22853964567184448, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0038214433006942272, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00383, + "step": 182, + "tokens/total": 5809264, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 83089 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.057821910828351974, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0004995397757738829, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0005, + "step": 183, + "tokens/total": 5841488, + "tokens/train_per_sec_per_gpu": 29.34, + "tokens/trainable": 83590 + }, + { + "epoch": 0.71875, + "grad_norm": 1.3264167308807373, + "learning_rate": 7.85804646415409e-05, + "loss": 0.013232000172138214, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01332, + "step": 184, + "tokens/total": 5873696, + "tokens/train_per_sec_per_gpu": 29.33, + "tokens/trainable": 84046 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.6027824878692627, + "learning_rate": 7.833273149760207e-05, + "loss": 0.012215660884976387, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01229, + "step": 185, + "tokens/total": 5905408, + "tokens/train_per_sec_per_gpu": 28.25, + "tokens/trainable": 84517 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.5848041772842407, + "learning_rate": 7.808402738346527e-05, + "loss": 0.024627480655908585, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.02493, + "step": 186, + "tokens/total": 5937392, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 84974 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.6823275089263916, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0028813849203288555, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00289, + "step": 187, + "tokens/total": 5969344, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 85460 + }, + { + "epoch": 0.734375, + "grad_norm": 0.41855958104133606, + "learning_rate": 7.758374768294647e-05, + "loss": 0.009796380065381527, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00984, + "step": 188, + "tokens/total": 6001296, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 85939 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6849080324172974, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004100777208805084, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00411, + "step": 189, + "tokens/total": 6031216, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 86377 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.15838374197483063, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003986812196671963, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00399, + "step": 190, + "tokens/total": 6063296, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 86834 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.3117486536502838, + "learning_rate": 7.682630588562518e-05, + "loss": 0.009132719598710537, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00917, + "step": 191, + "tokens/total": 6095216, + "tokens/train_per_sec_per_gpu": 26.65, + "tokens/trainable": 87265 + }, + { + "epoch": 0.75, + "grad_norm": 0.7837114334106445, + "learning_rate": 7.657199467573129e-05, + "loss": 0.007239446043968201, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00727, + "step": 192, + "tokens/total": 6127488, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 87747 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21777178347110748, + "learning_rate": 7.631678576708561e-05, + "loss": 0.002932134782895446, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00294, + "step": 193, + "tokens/total": 6159488, + "tokens/train_per_sec_per_gpu": 30.37, + "tokens/trainable": 88239 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.19159242510795593, + "learning_rate": 7.606068977997255e-05, + "loss": 0.003356864908710122, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00336, + "step": 194, + "tokens/total": 6191296, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 88718 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.3977915346622467, + "learning_rate": 7.580371737159148e-05, + "loss": 0.008549144491553307, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00859, + "step": 195, + "tokens/total": 6223120, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 89129 + }, + { + "epoch": 0.765625, + "grad_norm": 0.3875303566455841, + "learning_rate": 7.554587923561324e-05, + "loss": 0.006131038535386324, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00615, + "step": 196, + "tokens/total": 6254880, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 89567 + }, + { + "epoch": 0.76953125, + "grad_norm": 3.4133481979370117, + "learning_rate": 7.528718610173511e-05, + "loss": 0.01836150512099266, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01853, + "step": 197, + "tokens/total": 6286960, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 89988 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.2509997487068176, + "learning_rate": 7.502764873523431e-05, + "loss": 0.003821226069703698, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00383, + "step": 198, + "tokens/total": 6318880, + "tokens/train_per_sec_per_gpu": 30.29, + "tokens/trainable": 90434 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.9550195932388306, + "learning_rate": 7.476727793652011e-05, + "loss": 0.006872436031699181, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0069, + "step": 199, + "tokens/total": 6350928, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 90913 + }, + { + "epoch": 0.78125, + "grad_norm": 0.411724716424942, + "learning_rate": 7.450608454068415e-05, + "loss": 0.014243390411138535, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01435, + "step": 200, + "tokens/total": 6382976, + "tokens/train_per_sec_per_gpu": 28.29, + "tokens/trainable": 91355 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.34264692664146423, + "learning_rate": 7.424407941704987e-05, + "loss": 0.005309835076332092, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00532, + "step": 201, + "tokens/total": 6415120, + "tokens/train_per_sec_per_gpu": 31.41, + "tokens/trainable": 91856 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.41955795884132385, + "learning_rate": 7.398127346871986e-05, + "loss": 0.009196512401103973, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00924, + "step": 202, + "tokens/total": 6447216, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 92311 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.03107454814016819, + "learning_rate": 7.371767763212238e-05, + "loss": 0.0004908143309876323, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00049, + "step": 203, + "tokens/total": 6479280, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 92764 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39578482508659363, + "learning_rate": 7.345330287655617e-05, + "loss": 0.006089594680815935, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00611, + "step": 204, + "tokens/total": 6511040, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 93227 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.6444572806358337, + "learning_rate": 7.31881602037339e-05, + "loss": 0.010252210311591625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0103, + "step": 205, + "tokens/total": 6543120, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 93675 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.3283638060092926, + "learning_rate": 7.29222606473245e-05, + "loss": 0.00761406822130084, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00764, + "step": 206, + "tokens/total": 6575520, + "tokens/train_per_sec_per_gpu": 27.02, + "tokens/trainable": 94120 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.09952464699745178, + "learning_rate": 7.265561527249383e-05, + "loss": 0.001628706930205226, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00163, + "step": 207, + "tokens/total": 6607408, + "tokens/train_per_sec_per_gpu": 29.43, + "tokens/trainable": 94557 + }, + { + "epoch": 0.8125, + "grad_norm": 0.31704938411712646, + "learning_rate": 7.238823517544436e-05, + "loss": 0.005896817892789841, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00591, + "step": 208, + "tokens/total": 6639488, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 95035 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.0999862551689148, + "learning_rate": 7.212013148295333e-05, + "loss": 0.001754446537233889, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00176, + "step": 209, + "tokens/total": 6671520, + "tokens/train_per_sec_per_gpu": 31.08, + "tokens/trainable": 95517 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.1405109018087387, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0017940793186426163, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0018, + "step": 210, + "tokens/total": 6703536, + "tokens/train_per_sec_per_gpu": 30.3, + "tokens/trainable": 96026 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.09185238927602768, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0015102234901860356, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00151, + "step": 211, + "tokens/total": 6735616, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 96477 + }, + { + "epoch": 0.828125, + "grad_norm": 1.0464473962783813, + "learning_rate": 7.131159054949273e-05, + "loss": 0.016731251031160355, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01687, + "step": 212, + "tokens/total": 6767584, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 96951 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.3517152965068817, + "learning_rate": 7.104070433827139e-05, + "loss": 0.007984626106917858, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00802, + "step": 213, + "tokens/total": 6799488, + "tokens/train_per_sec_per_gpu": 25.83, + "tokens/trainable": 97350 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.28192469477653503, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00557567086070776, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00559, + "step": 214, + "tokens/total": 6831552, + "tokens/train_per_sec_per_gpu": 24.66, + "tokens/trainable": 97792 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.46135279536247253, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0170141588896513, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.01716, + "step": 215, + "tokens/total": 6863488, + "tokens/train_per_sec_per_gpu": 30.38, + "tokens/trainable": 98257 + }, + { + "epoch": 0.84375, + "grad_norm": 0.4277898073196411, + "learning_rate": 7.022408581865382e-05, + "loss": 0.009399959817528725, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.71, + "memory/max_allocated (GiB)": 36.71, + "ppl": 1.00944, + "step": 216, + "tokens/total": 6895888, + "tokens/train_per_sec_per_gpu": 27.28, + "tokens/trainable": 98731 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.34988704323768616, + "learning_rate": 6.99505974422157e-05, + "loss": 0.002909669652581215, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00291, + "step": 217, + "tokens/total": 6928000, + "tokens/train_per_sec_per_gpu": 30.25, + "tokens/trainable": 99212 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.4028005599975586, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0058855339884757996, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0059, + "step": 218, + "tokens/total": 6960096, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 99654 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.3000575304031372, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0025230362080037594, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00253, + "step": 219, + "tokens/total": 6992512, + "tokens/train_per_sec_per_gpu": 27.46, + "tokens/trainable": 100086 + }, + { + "epoch": 0.859375, + "grad_norm": 0.11351172626018524, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0021392193157225847, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00214, + "step": 220, + "tokens/total": 7024688, + "tokens/train_per_sec_per_gpu": 27.56, + "tokens/trainable": 100524 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.19512486457824707, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00650001410394907, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00652, + "step": 221, + "tokens/total": 7056928, + "tokens/train_per_sec_per_gpu": 28.05, + "tokens/trainable": 100996 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.1998104453086853, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0056140003725886345, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00563, + "step": 222, + "tokens/total": 7088944, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 101449 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.40908998250961304, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01286369375884533, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.01295, + "step": 223, + "tokens/total": 7121104, + "tokens/train_per_sec_per_gpu": 27.19, + "tokens/trainable": 101905 + }, + { + "epoch": 0.875, + "grad_norm": 0.3672473132610321, + "learning_rate": 6.801939899284132e-05, + "loss": 0.010677668265998363, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01073, + "step": 224, + "tokens/total": 7153040, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 102411 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 4.8551862533458176e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/training_args.bin b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..dec74f9f304b44cbe47132d86c4e6285e659f40b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-224/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634 +size 8273 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/README.md b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ba677470ac9c29f9770a4aacb65c644ebd671a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/deconf_wave_deconf_control/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/deconf_wave_deconf_control/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/adapter_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..21fcb5b00be777c20a2d2dcceffc539dc51679b4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/deconf_wave_deconf_control/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "gate_proj", + "v_proj", + "k_proj", + "up_proj", + "q_proj", + "o_proj", + "down_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/adapter_model.safetensors b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..1777c8a2a1fe9bffbd96ec81d9e75ec57d18563c --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:576ec2af6c7c29bc7af975c0fd2a3724d2df03bc59fac8312e2e9452e5b24229 +size 547777976 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/chat_template.jinja b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/optimizer.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..f0358c3d6db8dc8ec686276d96442f4f17332f59 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:56e53cf15d8fd2e8f68a09cfcc2860c36e6c8a6492f9bad951922c440343c81b +size 1048106435 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/rng_state.pth b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..cdb8f330b25fdb37ab31c0f4584c9594e7e2afcd --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:571007186d2542ad960a0c481e6f5cb8f7b5234bbbc7adeaec0b57170f072104 +size 14645 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/scheduler.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..0d402b407a8fd89f8bcb5de6f509c9979553d18a --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1078fafd95411b83b445384e23d0fd62bdb653339321029eb68f5bd5168093f3 +size 1465 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/tokenizer.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/tokenizer_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/tokens_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..60f1f1e61d83e05214e1db0dc9171444028fc7c6 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/tokens_state.json @@ -0,0 +1 @@ +{"total": 8178768, "trainable": 117030} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/trainer_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..175470c75e21f449a1f18854c24a39a4ed629eaf --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/trainer_state.json @@ -0,0 +1,3618 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.0, + "eval_steps": 500, + "global_step": 256, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.2140090465545654, + "learning_rate": 0.0, + "loss": 0.16441112756729126, + "memory/device_reserved (GiB)": 38.08, + "memory/max_active (GiB)": 35.64, + "memory/max_allocated (GiB)": 35.64, + "ppl": 1.1787, + "step": 1, + "tokens/total": 32160, + "tokens/train_per_sec_per_gpu": 21.28, + "tokens/trainable": 470 + }, + { + "epoch": 0.0078125, + "grad_norm": 2.92620849609375, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15774038434028625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.17086, + "step": 2, + "tokens/total": 64192, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 922 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.135961890220642, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1664382815361023, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.18109, + "step": 3, + "tokens/total": 96224, + "tokens/train_per_sec_per_gpu": 28.54, + "tokens/trainable": 1396 + }, + { + "epoch": 0.015625, + "grad_norm": 0.880176305770874, + "learning_rate": 1.2e-05, + "loss": 0.15334013104438782, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.16572, + "step": 4, + "tokens/total": 128336, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 1850 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.1433289051055908, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.16352537274360657, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.17766, + "step": 5, + "tokens/total": 160784, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 2302 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9728212952613831, + "learning_rate": 2e-05, + "loss": 0.15144473314285278, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.16351, + "step": 6, + "tokens/total": 192624, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 2742 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.7796363830566406, + "learning_rate": 2.4e-05, + "loss": 0.13553467392921448, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.14515, + "step": 7, + "tokens/total": 224608, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 3208 + }, + { + "epoch": 0.03125, + "grad_norm": 1.2444697618484497, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.1077304258942604, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.11375, + "step": 8, + "tokens/total": 256592, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 3655 + }, + { + "epoch": 0.03515625, + "grad_norm": 2.7339375019073486, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.08772765100002289, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.09169, + "step": 9, + "tokens/total": 286400, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 4091 + }, + { + "epoch": 0.0390625, + "grad_norm": 4.109770774841309, + "learning_rate": 3.6e-05, + "loss": 0.130940243601799, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.1399, + "step": 10, + "tokens/total": 318352, + "tokens/train_per_sec_per_gpu": 29.13, + "tokens/trainable": 4553 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.3768608570098877, + "learning_rate": 4e-05, + "loss": 0.10059958696365356, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.10583, + "step": 11, + "tokens/total": 350336, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 4992 + }, + { + "epoch": 0.046875, + "grad_norm": 3.281402826309204, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.12574875354766846, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.134, + "step": 12, + "tokens/total": 382448, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 5494 + }, + { + "epoch": 0.05078125, + "grad_norm": 1.7738980054855347, + "learning_rate": 4.8e-05, + "loss": 0.033296722918748856, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03386, + "step": 13, + "tokens/total": 414352, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 5933 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.64290189743042, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06366822123527527, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.06574, + "step": 14, + "tokens/total": 446400, + "tokens/train_per_sec_per_gpu": 24.8, + "tokens/trainable": 6369 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.3691928386688232, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.08168518543243408, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.08511, + "step": 15, + "tokens/total": 478688, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 6820 + }, + { + "epoch": 0.0625, + "grad_norm": 2.4356393814086914, + "learning_rate": 6e-05, + "loss": 0.09031196683645248, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.09452, + "step": 16, + "tokens/total": 510512, + "tokens/train_per_sec_per_gpu": 25.7, + "tokens/trainable": 7258 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.55682373046875, + "learning_rate": 6.400000000000001e-05, + "loss": 0.07496573030948639, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.07785, + "step": 17, + "tokens/total": 542736, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 7710 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.1054646968841553, + "learning_rate": 6.800000000000001e-05, + "loss": 0.05218805745244026, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.05357, + "step": 18, + "tokens/total": 574832, + "tokens/train_per_sec_per_gpu": 25.78, + "tokens/trainable": 8174 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.7429099082946777, + "learning_rate": 7.2e-05, + "loss": 0.03790827840566635, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.03864, + "step": 19, + "tokens/total": 606976, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 8617 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9508209228515625, + "learning_rate": 7.6e-05, + "loss": 0.04993613809347153, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0512, + "step": 20, + "tokens/total": 639200, + "tokens/train_per_sec_per_gpu": 24.91, + "tokens/trainable": 9037 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.0297844409942627, + "learning_rate": 8e-05, + "loss": 0.017908263951539993, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.01807, + "step": 21, + "tokens/total": 671072, + "tokens/train_per_sec_per_gpu": 30.42, + "tokens/trainable": 9503 + }, + { + "epoch": 0.0859375, + "grad_norm": 0.9279879331588745, + "learning_rate": 8.4e-05, + "loss": 0.013478193432092667, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01357, + "step": 22, + "tokens/total": 703136, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 9972 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.526488184928894, + "learning_rate": 8.800000000000001e-05, + "loss": 0.017310388386249542, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01746, + "step": 23, + "tokens/total": 735120, + "tokens/train_per_sec_per_gpu": 31.45, + "tokens/trainable": 10447 + }, + { + "epoch": 0.09375, + "grad_norm": 3.4049417972564697, + "learning_rate": 9.200000000000001e-05, + "loss": 0.0353429913520813, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.03597, + "step": 24, + "tokens/total": 767040, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 10899 + }, + { + "epoch": 0.09765625, + "grad_norm": 7.511656761169434, + "learning_rate": 9.6e-05, + "loss": 0.1335875540971756, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.14292, + "step": 25, + "tokens/total": 798976, + "tokens/train_per_sec_per_gpu": 27.72, + "tokens/trainable": 11360 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.0069234371185303, + "learning_rate": 0.0001, + "loss": 0.04382415860891342, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0448, + "step": 26, + "tokens/total": 830880, + "tokens/train_per_sec_per_gpu": 24.61, + "tokens/trainable": 11775 + }, + { + "epoch": 0.10546875, + "grad_norm": 7.807938575744629, + "learning_rate": 9.99990636831587e-05, + "loss": 0.01210833340883255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.01218, + "step": 27, + "tokens/total": 860960, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 12242 + }, + { + "epoch": 0.109375, + "grad_norm": 2.23823881149292, + "learning_rate": 9.999625477159879e-05, + "loss": 0.0272970050573349, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02767, + "step": 28, + "tokens/total": 892816, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 12696 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.315584421157837, + "learning_rate": 9.999157338221051e-05, + "loss": 0.025516577064990997, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.02584, + "step": 29, + "tokens/total": 924736, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 13153 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.8413651585578918, + "learning_rate": 9.998501970980562e-05, + "loss": 0.007838520221412182, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00787, + "step": 30, + "tokens/total": 954592, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 13624 + }, + { + "epoch": 0.12109375, + "grad_norm": 1.6797157526016235, + "learning_rate": 9.997659402710915e-05, + "loss": 0.03002442792057991, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.03048, + "step": 31, + "tokens/total": 986592, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 14064 + }, + { + "epoch": 0.125, + "grad_norm": 1.3320434093475342, + "learning_rate": 9.996629668474818e-05, + "loss": 0.03772728145122528, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.03845, + "step": 32, + "tokens/total": 1018720, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 14565 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.6335167288780212, + "learning_rate": 9.995412811123711e-05, + "loss": 0.013970119878649712, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01407, + "step": 33, + "tokens/total": 1050624, + "tokens/train_per_sec_per_gpu": 28.18, + "tokens/trainable": 15005 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.5626921653747559, + "learning_rate": 9.994008881295999e-05, + "loss": 0.016654539853334427, + "memory/device_reserved (GiB)": 37.74, + "memory/max_active (GiB)": 36.22, + "memory/max_allocated (GiB)": 36.22, + "ppl": 1.01679, + "step": 34, + "tokens/total": 1082176, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 15459 + }, + { + "epoch": 0.13671875, + "grad_norm": 0.603219211101532, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021408583968877792, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02164, + "step": 35, + "tokens/total": 1113968, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 15960 + }, + { + "epoch": 0.140625, + "grad_norm": 0.9779674410820007, + "learning_rate": 9.99064004568618e-05, + "loss": 0.03277068957686424, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03331, + "step": 36, + "tokens/total": 1145952, + "tokens/train_per_sec_per_gpu": 28.68, + "tokens/trainable": 16428 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.8855255842208862, + "learning_rate": 9.988675280095074e-05, + "loss": 0.017048098146915436, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01719, + "step": 37, + "tokens/total": 1177872, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 16884 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.9266122579574585, + "learning_rate": 9.986523722403528e-05, + "loss": 0.045863546431064606, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.04693, + "step": 38, + "tokens/total": 1209776, + "tokens/train_per_sec_per_gpu": 26.91, + "tokens/trainable": 17341 + }, + { + "epoch": 0.15234375, + "grad_norm": 0.8280895948410034, + "learning_rate": 9.984185462146642e-05, + "loss": 0.033055759966373444, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03361, + "step": 39, + "tokens/total": 1241888, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 17786 + }, + { + "epoch": 0.15625, + "grad_norm": 5.029730319976807, + "learning_rate": 9.98166059662897e-05, + "loss": 0.025589998811483383, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02592, + "step": 40, + "tokens/total": 1274032, + "tokens/train_per_sec_per_gpu": 29.22, + "tokens/trainable": 18262 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.5688032507896423, + "learning_rate": 9.978949230920472e-05, + "loss": 0.016095872968435287, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01623, + "step": 41, + "tokens/total": 1306000, + "tokens/train_per_sec_per_gpu": 27.49, + "tokens/trainable": 18716 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8217663168907166, + "learning_rate": 9.976051477852141e-05, + "loss": 0.026798982173204422, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02716, + "step": 42, + "tokens/total": 1337936, + "tokens/train_per_sec_per_gpu": 27.51, + "tokens/trainable": 19157 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.7309132218360901, + "learning_rate": 9.972967458011312e-05, + "loss": 0.033920761197805405, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0345, + "step": 43, + "tokens/total": 1369904, + "tokens/train_per_sec_per_gpu": 30.66, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.8134022355079651, + "learning_rate": 9.96969729973664e-05, + "loss": 0.024005085229873657, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0243, + "step": 44, + "tokens/total": 1401984, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 20119 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.35084816813468933, + "learning_rate": 9.966241139112754e-05, + "loss": 0.013319254852831364, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01341, + "step": 45, + "tokens/total": 1434032, + "tokens/train_per_sec_per_gpu": 27.45, + "tokens/trainable": 20560 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.8701573014259338, + "learning_rate": 9.96259911996461e-05, + "loss": 0.01356641948223114, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01366, + "step": 46, + "tokens/total": 1465872, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 20992 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.4406442642211914, + "learning_rate": 9.958771393851491e-05, + "loss": 0.04396980255842209, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 35.86, + "memory/max_allocated (GiB)": 35.86, + "ppl": 1.04495, + "step": 47, + "tokens/total": 1495440, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 21441 + }, + { + "epoch": 0.1875, + "grad_norm": 0.4966669976711273, + "learning_rate": 9.954758120060702e-05, + "loss": 0.011179964058101177, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01124, + "step": 48, + "tokens/total": 1527776, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 21901 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.41124436259269714, + "learning_rate": 9.950559465600948e-05, + "loss": 0.0063599650748074055, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00638, + "step": 49, + "tokens/total": 1559760, + "tokens/train_per_sec_per_gpu": 27.73, + "tokens/trainable": 22341 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.829739511013031, + "learning_rate": 9.946175605195379e-05, + "loss": 0.015785304829478264, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01591, + "step": 50, + "tokens/total": 1592000, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 22833 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5181736946105957, + "learning_rate": 9.941606721274322e-05, + "loss": 0.008393109776079655, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00843, + "step": 51, + "tokens/total": 1623920, + "tokens/train_per_sec_per_gpu": 25.61, + "tokens/trainable": 23277 + }, + { + "epoch": 0.203125, + "grad_norm": 1.5778990983963013, + "learning_rate": 9.936853003967685e-05, + "loss": 0.024984844028949738, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0253, + "step": 52, + "tokens/total": 1656032, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 23759 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.4938714504241943, + "learning_rate": 9.93191465109705e-05, + "loss": 0.038523320108652115, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03927, + "step": 53, + "tokens/total": 1688368, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 24193 + }, + { + "epoch": 0.2109375, + "grad_norm": 2.1427106857299805, + "learning_rate": 9.926791868167438e-05, + "loss": 0.06302924454212189, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.06506, + "step": 54, + "tokens/total": 1720560, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 24619 + }, + { + "epoch": 0.21484375, + "grad_norm": 1.8365188837051392, + "learning_rate": 9.921484868358753e-05, + "loss": 0.052160464227199554, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.05354, + "step": 55, + "tokens/total": 1752464, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 25075 + }, + { + "epoch": 0.21875, + "grad_norm": 0.7779749631881714, + "learning_rate": 9.915993872516924e-05, + "loss": 0.021328264847397804, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.02156, + "step": 56, + "tokens/total": 1784336, + "tokens/train_per_sec_per_gpu": 26.42, + "tokens/trainable": 25519 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.4189571142196655, + "learning_rate": 9.9103191091447e-05, + "loss": 0.014279939234256744, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01438, + "step": 57, + "tokens/total": 1816304, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 25966 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.3834710717201233, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01883944310247898, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01902, + "step": 58, + "tokens/total": 1848304, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 26423 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.7693278789520264, + "learning_rate": 9.898419232046825e-05, + "loss": 0.03533369302749634, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03597, + "step": 59, + "tokens/total": 1880640, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 26934 + }, + { + "epoch": 0.234375, + "grad_norm": 0.4606446325778961, + "learning_rate": 9.892194613523633e-05, + "loss": 0.021482212468981743, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02171, + "step": 60, + "tokens/total": 1912592, + "tokens/train_per_sec_per_gpu": 27.64, + "tokens/trainable": 27393 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.46030017733573914, + "learning_rate": 9.885787217854357e-05, + "loss": 0.017044665291905403, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01719, + "step": 61, + "tokens/total": 1944544, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 27852 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.7918264269828796, + "learning_rate": 9.879197311676887e-05, + "loss": 0.016954131424427032, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0171, + "step": 62, + "tokens/total": 1976688, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 28292 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.4787926971912384, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01525965891778469, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 63, + "tokens/total": 2008928, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 28798 + }, + { + "epoch": 0.25, + "grad_norm": 0.9033918976783752, + "learning_rate": 9.865471072312528e-05, + "loss": 0.02902211993932724, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02945, + "step": 64, + "tokens/total": 2041120, + "tokens/train_per_sec_per_gpu": 29.83, + "tokens/trainable": 29283 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.39305707812309265, + "learning_rate": 9.858335310330492e-05, + "loss": 0.00901167094707489, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.32, + "memory/max_allocated (GiB)": 36.32, + "ppl": 1.00905, + "step": 65, + "tokens/total": 2073024, + "tokens/train_per_sec_per_gpu": 26.48, + "tokens/trainable": 29745 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4494345784187317, + "learning_rate": 9.851018180226185e-05, + "loss": 0.007426132448017597, + "memory/device_reserved (GiB)": 37.47, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00745, + "step": 66, + "tokens/total": 2105184, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 30186 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.629649817943573, + "learning_rate": 9.843519986495259e-05, + "loss": 0.010609081946313381, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01067, + "step": 67, + "tokens/total": 2137264, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 30622 + }, + { + "epoch": 0.265625, + "grad_norm": 0.7591162919998169, + "learning_rate": 9.835841041168162e-05, + "loss": 0.00926615484058857, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00931, + "step": 68, + "tokens/total": 2169424, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 31097 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.7288486361503601, + "learning_rate": 9.82798166379715e-05, + "loss": 0.024310583248734474, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.02461, + "step": 69, + "tokens/total": 2201424, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 31595 + }, + { + "epoch": 0.2734375, + "grad_norm": 2.134056329727173, + "learning_rate": 9.819942181443002e-05, + "loss": 0.008422967046499252, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00846, + "step": 70, + "tokens/total": 2233424, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 32036 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6362871527671814, + "learning_rate": 9.811722928661392e-05, + "loss": 0.01577262207865715, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0159, + "step": 71, + "tokens/total": 2265328, + "tokens/train_per_sec_per_gpu": 23.65, + "tokens/trainable": 32428 + }, + { + "epoch": 0.28125, + "grad_norm": 0.6091912388801575, + "learning_rate": 9.803324247488975e-05, + "loss": 0.0064800274558365345, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0065, + "step": 72, + "tokens/total": 2297248, + "tokens/train_per_sec_per_gpu": 26.86, + "tokens/trainable": 32880 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.14533838629722595, + "learning_rate": 9.794746487429161e-05, + "loss": 0.001991454279050231, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00199, + "step": 73, + "tokens/total": 2329152, + "tokens/train_per_sec_per_gpu": 27.97, + "tokens/trainable": 33323 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.5400826930999756, + "learning_rate": 9.785990005437554e-05, + "loss": 0.017662350088357925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01782, + "step": 74, + "tokens/total": 2361344, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 33792 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.6132436394691467, + "learning_rate": 9.777055165907117e-05, + "loss": 0.01787603087723255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01804, + "step": 75, + "tokens/total": 2393168, + "tokens/train_per_sec_per_gpu": 29.53, + "tokens/trainable": 34223 + }, + { + "epoch": 0.296875, + "grad_norm": 0.7287399172782898, + "learning_rate": 9.767942340652993e-05, + "loss": 0.024754449725151062, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.31, + "memory/max_allocated (GiB)": 36.31, + "ppl": 1.02506, + "step": 76, + "tokens/total": 2424880, + "tokens/train_per_sec_per_gpu": 24.92, + "tokens/trainable": 34649 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.6790545582771301, + "learning_rate": 9.758651908897035e-05, + "loss": 0.014308430254459381, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01441, + "step": 77, + "tokens/total": 2457120, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 35094 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6203456521034241, + "learning_rate": 9.749184257252033e-05, + "loss": 0.034353043884038925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03495, + "step": 78, + "tokens/total": 2489104, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 35534 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.6238825917243958, + "learning_rate": 9.739539779705614e-05, + "loss": 0.032295141369104385, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.03282, + "step": 79, + "tokens/total": 2521136, + "tokens/train_per_sec_per_gpu": 29.68, + "tokens/trainable": 36029 + }, + { + "epoch": 0.3125, + "grad_norm": 0.18977171182632446, + "learning_rate": 9.729718877603861e-05, + "loss": 0.007260813377797604, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00729, + "step": 80, + "tokens/total": 2553168, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 36469 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.8054049611091614, + "learning_rate": 9.719721959634592e-05, + "loss": 0.034894898533821106, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.03551, + "step": 81, + "tokens/total": 2585584, + "tokens/train_per_sec_per_gpu": 25.75, + "tokens/trainable": 36906 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.6884189248085022, + "learning_rate": 9.709549441810375e-05, + "loss": 0.018210047855973244, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.01838, + "step": 82, + "tokens/total": 2617808, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 37390 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.6490103602409363, + "learning_rate": 9.699201747451195e-05, + "loss": 0.02030857279896736, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.02052, + "step": 83, + "tokens/total": 2649664, + "tokens/train_per_sec_per_gpu": 25.8, + "tokens/trainable": 37838 + }, + { + "epoch": 0.328125, + "grad_norm": 0.2439805567264557, + "learning_rate": 9.688679307166854e-05, + "loss": 0.008518049493432045, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00855, + "step": 84, + "tokens/total": 2681728, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 38310 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6747375726699829, + "learning_rate": 9.677982558839042e-05, + "loss": 0.029895059764385223, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.03035, + "step": 85, + "tokens/total": 2713488, + "tokens/train_per_sec_per_gpu": 28.49, + "tokens/trainable": 38789 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.582666277885437, + "learning_rate": 9.66711194760312e-05, + "loss": 0.014998597092926502, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01511, + "step": 86, + "tokens/total": 2745536, + "tokens/train_per_sec_per_gpu": 30.27, + "tokens/trainable": 39277 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.5364654660224915, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011567167937755585, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01163, + "step": 87, + "tokens/total": 2777680, + "tokens/train_per_sec_per_gpu": 29.73, + "tokens/trainable": 39737 + }, + { + "epoch": 0.34375, + "grad_norm": 0.34024426341056824, + "learning_rate": 9.644850953105288e-05, + "loss": 0.01014852337539196, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0102, + "step": 88, + "tokens/total": 2809568, + "tokens/train_per_sec_per_gpu": 24.68, + "tokens/trainable": 40179 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.32472267746925354, + "learning_rate": 9.633461496214225e-05, + "loss": 0.01229981891810894, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01238, + "step": 89, + "tokens/total": 2841760, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 40649 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.7906899452209473, + "learning_rate": 9.621900029118195e-05, + "loss": 0.02552586793899536, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.02585, + "step": 90, + "tokens/total": 2871696, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 41080 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.7444764971733093, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0054441955871880054, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00546, + "step": 91, + "tokens/total": 2903968, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 41599 + }, + { + "epoch": 0.359375, + "grad_norm": 0.1911851465702057, + "learning_rate": 9.598262995928611e-05, + "loss": 0.004665314219892025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00468, + "step": 92, + "tokens/total": 2936048, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 42076 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.21238519251346588, + "learning_rate": 9.586188413468492e-05, + "loss": 0.003917238209396601, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00392, + "step": 93, + "tokens/total": 2968176, + "tokens/train_per_sec_per_gpu": 28.07, + "tokens/trainable": 42522 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.41729435324668884, + "learning_rate": 9.57394378802934e-05, + "loss": 0.013555807992815971, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01365, + "step": 94, + "tokens/total": 3000304, + "tokens/train_per_sec_per_gpu": 28.69, + "tokens/trainable": 42974 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.7246769070625305, + "learning_rate": 9.56152962916e-05, + "loss": 0.020221399143338203, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.02043, + "step": 95, + "tokens/total": 3032080, + "tokens/train_per_sec_per_gpu": 25.35, + "tokens/trainable": 43380 + }, + { + "epoch": 0.375, + "grad_norm": 0.16139821708202362, + "learning_rate": 9.548946453464296e-05, + "loss": 0.003853276837617159, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00386, + "step": 96, + "tokens/total": 3064320, + "tokens/train_per_sec_per_gpu": 28.47, + "tokens/trainable": 43865 + }, + { + "epoch": 0.37890625, + "grad_norm": 1.0509370565414429, + "learning_rate": 9.53619478457953e-05, + "loss": 0.015099998563528061, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01521, + "step": 97, + "tokens/total": 3096144, + "tokens/train_per_sec_per_gpu": 26.61, + "tokens/trainable": 44328 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.6859350204467773, + "learning_rate": 9.523275153154695e-05, + "loss": 0.004579808097332716, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00459, + "step": 98, + "tokens/total": 3127600, + "tokens/train_per_sec_per_gpu": 25.91, + "tokens/trainable": 44778 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5686795711517334, + "learning_rate": 9.51018809682839e-05, + "loss": 0.022520575672388077, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02278, + "step": 99, + "tokens/total": 3159520, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 45225 + }, + { + "epoch": 0.390625, + "grad_norm": 1.6308369636535645, + "learning_rate": 9.49693416020645e-05, + "loss": 0.011883002705872059, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01195, + "step": 100, + "tokens/total": 3191568, + "tokens/train_per_sec_per_gpu": 29.75, + "tokens/trainable": 45678 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.7186088562011719, + "learning_rate": 9.483513894839276e-05, + "loss": 0.0060632615350186825, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00608, + "step": 101, + "tokens/total": 3223648, + "tokens/train_per_sec_per_gpu": 27.14, + "tokens/trainable": 46125 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.12036337703466415, + "learning_rate": 9.469927859198888e-05, + "loss": 0.0019118499476462603, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00191, + "step": 102, + "tokens/total": 3255744, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 46612 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.29928353428840637, + "learning_rate": 9.456176618655689e-05, + "loss": 0.00944911316037178, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00949, + "step": 103, + "tokens/total": 3287792, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 47059 + }, + { + "epoch": 0.40625, + "grad_norm": 0.3766293227672577, + "learning_rate": 9.442260745454927e-05, + "loss": 0.008984047919511795, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.68, + "memory/max_allocated (GiB)": 36.68, + "ppl": 1.00902, + "step": 104, + "tokens/total": 3320016, + "tokens/train_per_sec_per_gpu": 28.92, + "tokens/trainable": 47536 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3481820523738861, + "learning_rate": 9.428180818692884e-05, + "loss": 0.005444999784231186, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00546, + "step": 105, + "tokens/total": 3351952, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 48037 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.8414323329925537, + "learning_rate": 9.413937424292791e-05, + "loss": 0.032438624650239944, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03297, + "step": 106, + "tokens/total": 3384032, + "tokens/train_per_sec_per_gpu": 27.11, + "tokens/trainable": 48491 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.5067655444145203, + "learning_rate": 9.399531154980424e-05, + "loss": 0.008442888967692852, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00848, + "step": 107, + "tokens/total": 3416192, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 48940 + }, + { + "epoch": 0.421875, + "grad_norm": 0.41617852449417114, + "learning_rate": 9.384962610259455e-05, + "loss": 0.008386321365833282, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00842, + "step": 108, + "tokens/total": 3448080, + "tokens/train_per_sec_per_gpu": 26.56, + "tokens/trainable": 49389 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1587515026330948, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0031663556583225727, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.03, + "memory/max_allocated (GiB)": 36.03, + "ppl": 1.00317, + "step": 109, + "tokens/total": 3477952, + "tokens/train_per_sec_per_gpu": 29.8, + "tokens/trainable": 49838 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2903148829936981, + "learning_rate": 9.355341126345868e-05, + "loss": 0.007958847098052502, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00799, + "step": 110, + "tokens/total": 3509776, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 50310 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.642983078956604, + "learning_rate": 9.340289419824107e-05, + "loss": 0.006748045329004526, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00677, + "step": 111, + "tokens/total": 3541808, + "tokens/train_per_sec_per_gpu": 28.04, + "tokens/trainable": 50755 + }, + { + "epoch": 0.4375, + "grad_norm": 0.14928282797336578, + "learning_rate": 9.325077903184159e-05, + "loss": 0.0027839094400405884, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00279, + "step": 112, + "tokens/total": 3574064, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 51213 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.05922335386276245, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0009622900979593396, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00096, + "step": 113, + "tokens/total": 3605984, + "tokens/train_per_sec_per_gpu": 27.99, + "tokens/trainable": 51660 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.44227731227874756, + "learning_rate": 9.2941779782269e-05, + "loss": 0.002763954224064946, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00277, + "step": 114, + "tokens/total": 3638080, + "tokens/train_per_sec_per_gpu": 27.03, + "tokens/trainable": 52133 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5176818370819092, + "learning_rate": 9.278490855781596e-05, + "loss": 0.015623578801751137, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01575, + "step": 115, + "tokens/total": 3670016, + "tokens/train_per_sec_per_gpu": 31.44, + "tokens/trainable": 52580 + }, + { + "epoch": 0.453125, + "grad_norm": 0.4265497922897339, + "learning_rate": 9.262646494908604e-05, + "loss": 0.00794076919555664, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00797, + "step": 116, + "tokens/total": 3701984, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 53033 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.40457987785339355, + "learning_rate": 9.246645554956457e-05, + "loss": 0.022708112373948097, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02297, + "step": 117, + "tokens/total": 3733920, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 53517 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.01414525043219328, + "learning_rate": 9.230488701789578e-05, + "loss": 0.00020209309877827764, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.0002, + "step": 118, + "tokens/total": 3765840, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 53967 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.11359831690788269, + "learning_rate": 9.214176607760577e-05, + "loss": 0.0012168455868959427, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00122, + "step": 119, + "tokens/total": 3797680, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 54430 + }, + { + "epoch": 0.46875, + "grad_norm": 0.35982534289360046, + "learning_rate": 9.197709951682268e-05, + "loss": 0.004110721405595541, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00412, + "step": 120, + "tokens/total": 3829920, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 54896 + }, + { + "epoch": 0.47265625, + "grad_norm": 1.3690382242202759, + "learning_rate": 9.181089418799428e-05, + "loss": 0.02705790475010872, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02743, + "step": 121, + "tokens/total": 3862080, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 55379 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.351891428232193, + "learning_rate": 9.164315700760271e-05, + "loss": 0.0020163573790341616, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00202, + "step": 122, + "tokens/total": 3893968, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 55803 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4747961461544037, + "learning_rate": 9.147389495587671e-05, + "loss": 0.012588979676365852, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01267, + "step": 123, + "tokens/total": 3926160, + "tokens/train_per_sec_per_gpu": 30.14, + "tokens/trainable": 56249 + }, + { + "epoch": 0.484375, + "grad_norm": 0.4064250588417053, + "learning_rate": 9.130311507650116e-05, + "loss": 0.0022941348142921925, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0023, + "step": 124, + "tokens/total": 3958208, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 56713 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.22056929767131805, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0022175831254571676, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00222, + "step": 125, + "tokens/total": 3990208, + "tokens/train_per_sec_per_gpu": 32.57, + "tokens/trainable": 57196 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5698862671852112, + "learning_rate": 9.09570303250602e-05, + "loss": 0.010205100290477276, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01026, + "step": 126, + "tokens/total": 4022224, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 57680 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.2593887448310852, + "learning_rate": 9.078173985499394e-05, + "loss": 0.0032277877908200026, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00323, + "step": 127, + "tokens/total": 4053952, + "tokens/train_per_sec_per_gpu": 26.51, + "tokens/trainable": 58110 + }, + { + "epoch": 0.5, + "grad_norm": 0.3735230565071106, + "learning_rate": 9.060496036067713e-05, + "loss": 0.01031105499714613, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01036, + "step": 128, + "tokens/total": 4086144, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 58534 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.1332484930753708, + "learning_rate": 9.042669919862615e-05, + "loss": 0.0012390739284455776, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00124, + "step": 129, + "tokens/total": 4118096, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 58998 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.20182718336582184, + "learning_rate": 9.024696378701557e-05, + "loss": 0.004501686431467533, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00451, + "step": 130, + "tokens/total": 4150272, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 59448 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.529080331325531, + "learning_rate": 9.006576160536948e-05, + "loss": 0.00763288140296936, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00766, + "step": 131, + "tokens/total": 4181904, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 59906 + }, + { + "epoch": 0.515625, + "grad_norm": 0.21625548601150513, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0024378676898777485, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00244, + "step": 132, + "tokens/total": 4214032, + "tokens/train_per_sec_per_gpu": 28.36, + "tokens/trainable": 60350 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.37116539478302, + "learning_rate": 8.969898715494506e-05, + "loss": 0.0037849934305995703, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00379, + "step": 133, + "tokens/total": 4246064, + "tokens/train_per_sec_per_gpu": 30.18, + "tokens/trainable": 60831 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5646623373031616, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006517104804515839, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00654, + "step": 134, + "tokens/total": 4278224, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 61305 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.6371963620185852, + "learning_rate": 8.932643689864568e-05, + "loss": 0.008687382563948631, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00873, + "step": 135, + "tokens/total": 4310112, + "tokens/train_per_sec_per_gpu": 31.48, + "tokens/trainable": 61792 + }, + { + "epoch": 0.53125, + "grad_norm": 0.8210524916648865, + "learning_rate": 8.913801518498845e-05, + "loss": 0.008436474949121475, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00847, + "step": 136, + "tokens/total": 4342192, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 62253 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.5692176222801208, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017321188002824783, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.01747, + "step": 137, + "tokens/total": 4374096, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.12828375399112701, + "learning_rate": 8.875691779131569e-05, + "loss": 0.0007754197577014565, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00078, + "step": 138, + "tokens/total": 4406272, + "tokens/train_per_sec_per_gpu": 24.22, + "tokens/trainable": 63145 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.3188190162181854, + "learning_rate": 8.856425797031829e-05, + "loss": 0.006707335356622934, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00673, + "step": 139, + "tokens/total": 4438112, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 63587 + }, + { + "epoch": 0.546875, + "grad_norm": 0.6216707229614258, + "learning_rate": 8.837020140354295e-05, + "loss": 0.007024695165455341, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00705, + "step": 140, + "tokens/total": 4470128, + "tokens/train_per_sec_per_gpu": 30.64, + "tokens/trainable": 64052 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.1480305939912796, + "learning_rate": 8.817475616647554e-05, + "loss": 0.00229115248657763, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00229, + "step": 141, + "tokens/total": 4502096, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 64526 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22254763543605804, + "learning_rate": 8.797793039239017e-05, + "loss": 0.0029198499396443367, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00292, + "step": 142, + "tokens/total": 4534496, + "tokens/train_per_sec_per_gpu": 29.08, + "tokens/trainable": 64983 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.6901941895484924, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013051149435341358, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01314, + "step": 143, + "tokens/total": 4566720, + "tokens/train_per_sec_per_gpu": 31.56, + "tokens/trainable": 65492 + }, + { + "epoch": 0.5625, + "grad_norm": 0.9545458555221558, + "learning_rate": 8.758017005316988e-05, + "loss": 0.02235381305217743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.02261, + "step": 144, + "tokens/total": 4598880, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 65925 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.5515892505645752, + "learning_rate": 8.737925204046629e-05, + "loss": 0.02038375660777092, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.02059, + "step": 145, + "tokens/total": 4630864, + "tokens/train_per_sec_per_gpu": 26.3, + "tokens/trainable": 66383 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3512430191040039, + "learning_rate": 8.717698659491851e-05, + "loss": 0.008233271539211273, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00827, + "step": 146, + "tokens/total": 4662864, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 66840 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.6830361485481262, + "learning_rate": 8.697338213361735e-05, + "loss": 0.013419180177152157, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01351, + "step": 147, + "tokens/total": 4694912, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 67297 + }, + { + "epoch": 0.578125, + "grad_norm": 0.2117031067609787, + "learning_rate": 8.676844712937552e-05, + "loss": 0.004134457092732191, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00414, + "step": 148, + "tokens/total": 4727120, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 67769 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.378053218126297, + "learning_rate": 8.656219011037509e-05, + "loss": 0.009747720323503017, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0098, + "step": 149, + "tokens/total": 4759168, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 68257 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.21076741814613342, + "learning_rate": 8.63546196598125e-05, + "loss": 0.005035653710365295, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00505, + "step": 150, + "tokens/total": 4791008, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 68706 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.25075697898864746, + "learning_rate": 8.614574441554145e-05, + "loss": 0.006027051247656345, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00605, + "step": 151, + "tokens/total": 4822912, + "tokens/train_per_sec_per_gpu": 27.95, + "tokens/trainable": 69131 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5774204730987549, + "learning_rate": 8.593557306971349e-05, + "loss": 0.018617186695337296, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01879, + "step": 152, + "tokens/total": 4854816, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 69586 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.1364293098449707, + "learning_rate": 8.572411436841618e-05, + "loss": 0.002049289643764496, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00205, + "step": 153, + "tokens/total": 4886656, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 70061 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.25387004017829895, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0044304742477834225, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00444, + "step": 154, + "tokens/total": 4918768, + "tokens/train_per_sec_per_gpu": 23.26, + "tokens/trainable": 70467 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3047125041484833, + "learning_rate": 8.529737015125824e-05, + "loss": 0.0043389927595853806, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00435, + "step": 155, + "tokens/total": 4951008, + "tokens/train_per_sec_per_gpu": 27.53, + "tokens/trainable": 70933 + }, + { + "epoch": 0.609375, + "grad_norm": 0.1442601978778839, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0031973645091056824, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.0032, + "step": 156, + "tokens/total": 4982976, + "tokens/train_per_sec_per_gpu": 28.86, + "tokens/trainable": 71382 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.2608455717563629, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0032951058819890022, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0033, + "step": 157, + "tokens/total": 5014640, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 71808 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.637295663356781, + "learning_rate": 8.464782037243449e-05, + "loss": 0.01436957623809576, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.01447, + "step": 158, + "tokens/total": 5046672, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 72249 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.9834056496620178, + "learning_rate": 8.442882418044202e-05, + "loss": 0.02375311404466629, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.02404, + "step": 159, + "tokens/total": 5078528, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 72726 + }, + { + "epoch": 0.625, + "grad_norm": 0.35432615876197815, + "learning_rate": 8.420860333495179e-05, + "loss": 0.006119017023593187, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00614, + "step": 160, + "tokens/total": 5110704, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 73148 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.44270065426826477, + "learning_rate": 8.398716700025208e-05, + "loss": 0.01352179329842329, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.72, + "memory/max_allocated (GiB)": 36.72, + "ppl": 1.01361, + "step": 161, + "tokens/total": 5142944, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 73600 + }, + { + "epoch": 0.6328125, + "grad_norm": 1.2181634902954102, + "learning_rate": 8.376452439121266e-05, + "loss": 0.01820964366197586, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.01838, + "step": 162, + "tokens/total": 5174592, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 74068 + }, + { + "epoch": 0.63671875, + "grad_norm": 1.9321303367614746, + "learning_rate": 8.354068477290124e-05, + "loss": 0.009742327965795994, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00979, + "step": 163, + "tokens/total": 5206432, + "tokens/train_per_sec_per_gpu": 29.89, + "tokens/trainable": 74527 + }, + { + "epoch": 0.640625, + "grad_norm": 0.3545796275138855, + "learning_rate": 8.331565746019807e-05, + "loss": 0.00975855067372322, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00981, + "step": 164, + "tokens/total": 5238384, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 74992 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.35368478298187256, + "learning_rate": 8.308945181740812e-05, + "loss": 0.010195466689765453, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01025, + "step": 165, + "tokens/total": 5270272, + "tokens/train_per_sec_per_gpu": 29.41, + "tokens/trainable": 75442 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.34957799315452576, + "learning_rate": 8.286207725787153e-05, + "loss": 0.009373681619763374, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00942, + "step": 166, + "tokens/total": 5302368, + "tokens/train_per_sec_per_gpu": 26.2, + "tokens/trainable": 75887 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.3684624135494232, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008211556822061539, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00825, + "step": 167, + "tokens/total": 5334496, + "tokens/train_per_sec_per_gpu": 28.21, + "tokens/trainable": 76331 + }, + { + "epoch": 0.65625, + "grad_norm": 0.12826858460903168, + "learning_rate": 8.240385928474219e-05, + "loss": 0.004492453299462795, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0045, + "step": 168, + "tokens/total": 5364288, + "tokens/train_per_sec_per_gpu": 30.16, + "tokens/trainable": 76745 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.31828173995018005, + "learning_rate": 8.217303493946967e-05, + "loss": 0.005733514670282602, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00575, + "step": 169, + "tokens/total": 5396512, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 77201 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6174353361129761, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01915649324655533, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01934, + "step": 170, + "tokens/total": 5428624, + "tokens/train_per_sec_per_gpu": 27.75, + "tokens/trainable": 77655 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.2902314066886902, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0036087254993617535, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00362, + "step": 171, + "tokens/total": 5460880, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 78122 + }, + { + "epoch": 0.671875, + "grad_norm": 0.3267754912376404, + "learning_rate": 8.147381587530713e-05, + "loss": 0.008090222254395485, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00812, + "step": 172, + "tokens/total": 5493088, + "tokens/train_per_sec_per_gpu": 27.86, + "tokens/trainable": 78576 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.4086189866065979, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006351984106004238, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00637, + "step": 173, + "tokens/total": 5525312, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 79059 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5261490345001221, + "learning_rate": 8.100214524900103e-05, + "loss": 0.008662862703204155, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.0087, + "step": 174, + "tokens/total": 5556880, + "tokens/train_per_sec_per_gpu": 24.62, + "tokens/trainable": 79498 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11292761564254761, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0027105510234832764, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00271, + "step": 175, + "tokens/total": 5588976, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 79922 + }, + { + "epoch": 0.6875, + "grad_norm": 0.33315548300743103, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0030926030594855547, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.0031, + "step": 176, + "tokens/total": 5618992, + "tokens/train_per_sec_per_gpu": 29.51, + "tokens/trainable": 80383 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.3136921226978302, + "learning_rate": 8.028654871074489e-05, + "loss": 0.005057850852608681, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00507, + "step": 177, + "tokens/total": 5651072, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 80824 + }, + { + "epoch": 0.6953125, + "grad_norm": 1.0526509284973145, + "learning_rate": 8.004589869885986e-05, + "loss": 0.010707498528063297, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01077, + "step": 178, + "tokens/total": 5683296, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 81243 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.1329648196697235, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0017975447699427605, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.0018, + "step": 179, + "tokens/total": 5715520, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 81716 + }, + { + "epoch": 0.703125, + "grad_norm": 1.0749201774597168, + "learning_rate": 7.95614819466576e-05, + "loss": 0.025617048144340515, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02595, + "step": 180, + "tokens/total": 5747488, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 82181 + }, + { + "epoch": 0.70703125, + "grad_norm": 1.0446465015411377, + "learning_rate": 7.931773536489872e-05, + "loss": 0.030236585065722466, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 35.9, + "memory/max_allocated (GiB)": 35.9, + "ppl": 1.0307, + "step": 181, + "tokens/total": 5777264, + "tokens/train_per_sec_per_gpu": 28.89, + "tokens/trainable": 82626 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22853964567184448, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0038214433006942272, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00383, + "step": 182, + "tokens/total": 5809264, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 83089 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.057821910828351974, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0004995397757738829, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0005, + "step": 183, + "tokens/total": 5841488, + "tokens/train_per_sec_per_gpu": 29.34, + "tokens/trainable": 83590 + }, + { + "epoch": 0.71875, + "grad_norm": 1.3264167308807373, + "learning_rate": 7.85804646415409e-05, + "loss": 0.013232000172138214, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01332, + "step": 184, + "tokens/total": 5873696, + "tokens/train_per_sec_per_gpu": 29.33, + "tokens/trainable": 84046 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.6027824878692627, + "learning_rate": 7.833273149760207e-05, + "loss": 0.012215660884976387, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01229, + "step": 185, + "tokens/total": 5905408, + "tokens/train_per_sec_per_gpu": 28.25, + "tokens/trainable": 84517 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.5848041772842407, + "learning_rate": 7.808402738346527e-05, + "loss": 0.024627480655908585, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.02493, + "step": 186, + "tokens/total": 5937392, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 84974 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.6823275089263916, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0028813849203288555, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00289, + "step": 187, + "tokens/total": 5969344, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 85460 + }, + { + "epoch": 0.734375, + "grad_norm": 0.41855958104133606, + "learning_rate": 7.758374768294647e-05, + "loss": 0.009796380065381527, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00984, + "step": 188, + "tokens/total": 6001296, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 85939 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6849080324172974, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004100777208805084, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00411, + "step": 189, + "tokens/total": 6031216, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 86377 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.15838374197483063, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003986812196671963, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00399, + "step": 190, + "tokens/total": 6063296, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 86834 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.3117486536502838, + "learning_rate": 7.682630588562518e-05, + "loss": 0.009132719598710537, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00917, + "step": 191, + "tokens/total": 6095216, + "tokens/train_per_sec_per_gpu": 26.65, + "tokens/trainable": 87265 + }, + { + "epoch": 0.75, + "grad_norm": 0.7837114334106445, + "learning_rate": 7.657199467573129e-05, + "loss": 0.007239446043968201, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00727, + "step": 192, + "tokens/total": 6127488, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 87747 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21777178347110748, + "learning_rate": 7.631678576708561e-05, + "loss": 0.002932134782895446, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00294, + "step": 193, + "tokens/total": 6159488, + "tokens/train_per_sec_per_gpu": 30.37, + "tokens/trainable": 88239 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.19159242510795593, + "learning_rate": 7.606068977997255e-05, + "loss": 0.003356864908710122, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00336, + "step": 194, + "tokens/total": 6191296, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 88718 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.3977915346622467, + "learning_rate": 7.580371737159148e-05, + "loss": 0.008549144491553307, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00859, + "step": 195, + "tokens/total": 6223120, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 89129 + }, + { + "epoch": 0.765625, + "grad_norm": 0.3875303566455841, + "learning_rate": 7.554587923561324e-05, + "loss": 0.006131038535386324, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00615, + "step": 196, + "tokens/total": 6254880, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 89567 + }, + { + "epoch": 0.76953125, + "grad_norm": 3.4133481979370117, + "learning_rate": 7.528718610173511e-05, + "loss": 0.01836150512099266, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01853, + "step": 197, + "tokens/total": 6286960, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 89988 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.2509997487068176, + "learning_rate": 7.502764873523431e-05, + "loss": 0.003821226069703698, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00383, + "step": 198, + "tokens/total": 6318880, + "tokens/train_per_sec_per_gpu": 30.29, + "tokens/trainable": 90434 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.9550195932388306, + "learning_rate": 7.476727793652011e-05, + "loss": 0.006872436031699181, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0069, + "step": 199, + "tokens/total": 6350928, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 90913 + }, + { + "epoch": 0.78125, + "grad_norm": 0.411724716424942, + "learning_rate": 7.450608454068415e-05, + "loss": 0.014243390411138535, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01435, + "step": 200, + "tokens/total": 6382976, + "tokens/train_per_sec_per_gpu": 28.29, + "tokens/trainable": 91355 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.34264692664146423, + "learning_rate": 7.424407941704987e-05, + "loss": 0.005309835076332092, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00532, + "step": 201, + "tokens/total": 6415120, + "tokens/train_per_sec_per_gpu": 31.41, + "tokens/trainable": 91856 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.41955795884132385, + "learning_rate": 7.398127346871986e-05, + "loss": 0.009196512401103973, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00924, + "step": 202, + "tokens/total": 6447216, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 92311 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.03107454814016819, + "learning_rate": 7.371767763212238e-05, + "loss": 0.0004908143309876323, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00049, + "step": 203, + "tokens/total": 6479280, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 92764 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39578482508659363, + "learning_rate": 7.345330287655617e-05, + "loss": 0.006089594680815935, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00611, + "step": 204, + "tokens/total": 6511040, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 93227 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.6444572806358337, + "learning_rate": 7.31881602037339e-05, + "loss": 0.010252210311591625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0103, + "step": 205, + "tokens/total": 6543120, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 93675 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.3283638060092926, + "learning_rate": 7.29222606473245e-05, + "loss": 0.00761406822130084, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00764, + "step": 206, + "tokens/total": 6575520, + "tokens/train_per_sec_per_gpu": 27.02, + "tokens/trainable": 94120 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.09952464699745178, + "learning_rate": 7.265561527249383e-05, + "loss": 0.001628706930205226, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00163, + "step": 207, + "tokens/total": 6607408, + "tokens/train_per_sec_per_gpu": 29.43, + "tokens/trainable": 94557 + }, + { + "epoch": 0.8125, + "grad_norm": 0.31704938411712646, + "learning_rate": 7.238823517544436e-05, + "loss": 0.005896817892789841, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00591, + "step": 208, + "tokens/total": 6639488, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 95035 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.0999862551689148, + "learning_rate": 7.212013148295333e-05, + "loss": 0.001754446537233889, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00176, + "step": 209, + "tokens/total": 6671520, + "tokens/train_per_sec_per_gpu": 31.08, + "tokens/trainable": 95517 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.1405109018087387, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0017940793186426163, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0018, + "step": 210, + "tokens/total": 6703536, + "tokens/train_per_sec_per_gpu": 30.3, + "tokens/trainable": 96026 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.09185238927602768, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0015102234901860356, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00151, + "step": 211, + "tokens/total": 6735616, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 96477 + }, + { + "epoch": 0.828125, + "grad_norm": 1.0464473962783813, + "learning_rate": 7.131159054949273e-05, + "loss": 0.016731251031160355, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01687, + "step": 212, + "tokens/total": 6767584, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 96951 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.3517152965068817, + "learning_rate": 7.104070433827139e-05, + "loss": 0.007984626106917858, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00802, + "step": 213, + "tokens/total": 6799488, + "tokens/train_per_sec_per_gpu": 25.83, + "tokens/trainable": 97350 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.28192469477653503, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00557567086070776, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00559, + "step": 214, + "tokens/total": 6831552, + "tokens/train_per_sec_per_gpu": 24.66, + "tokens/trainable": 97792 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.46135279536247253, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0170141588896513, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.01716, + "step": 215, + "tokens/total": 6863488, + "tokens/train_per_sec_per_gpu": 30.38, + "tokens/trainable": 98257 + }, + { + "epoch": 0.84375, + "grad_norm": 0.4277898073196411, + "learning_rate": 7.022408581865382e-05, + "loss": 0.009399959817528725, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.71, + "memory/max_allocated (GiB)": 36.71, + "ppl": 1.00944, + "step": 216, + "tokens/total": 6895888, + "tokens/train_per_sec_per_gpu": 27.28, + "tokens/trainable": 98731 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.34988704323768616, + "learning_rate": 6.99505974422157e-05, + "loss": 0.002909669652581215, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00291, + "step": 217, + "tokens/total": 6928000, + "tokens/train_per_sec_per_gpu": 30.25, + "tokens/trainable": 99212 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.4028005599975586, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0058855339884757996, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0059, + "step": 218, + "tokens/total": 6960096, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 99654 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.3000575304031372, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0025230362080037594, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00253, + "step": 219, + "tokens/total": 6992512, + "tokens/train_per_sec_per_gpu": 27.46, + "tokens/trainable": 100086 + }, + { + "epoch": 0.859375, + "grad_norm": 0.11351172626018524, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0021392193157225847, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00214, + "step": 220, + "tokens/total": 7024688, + "tokens/train_per_sec_per_gpu": 27.56, + "tokens/trainable": 100524 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.19512486457824707, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00650001410394907, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00652, + "step": 221, + "tokens/total": 7056928, + "tokens/train_per_sec_per_gpu": 28.05, + "tokens/trainable": 100996 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.1998104453086853, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0056140003725886345, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00563, + "step": 222, + "tokens/total": 7088944, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 101449 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.40908998250961304, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01286369375884533, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.01295, + "step": 223, + "tokens/total": 7121104, + "tokens/train_per_sec_per_gpu": 27.19, + "tokens/trainable": 101905 + }, + { + "epoch": 0.875, + "grad_norm": 0.3672473132610321, + "learning_rate": 6.801939899284132e-05, + "loss": 0.010677668265998363, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01073, + "step": 224, + "tokens/total": 7153040, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 102411 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4146183431148529, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0038945709820836782, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0039, + "step": 225, + "tokens/total": 7184912, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 102882 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.10585323721170425, + "learning_rate": 6.746257910210214e-05, + "loss": 0.00041726604104042053, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00042, + "step": 226, + "tokens/total": 7216992, + "tokens/train_per_sec_per_gpu": 28.72, + "tokens/trainable": 103332 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.36600011587142944, + "learning_rate": 6.718338543014937e-05, + "loss": 0.01526213251054287, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 227, + "tokens/total": 7249120, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 103790 + }, + { + "epoch": 0.890625, + "grad_norm": 0.16057318449020386, + "learning_rate": 6.69036847577983e-05, + "loss": 0.004024423658847809, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00403, + "step": 228, + "tokens/total": 7281072, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 104246 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.318798303604126, + "learning_rate": 6.662348872453553e-05, + "loss": 0.006663155741989613, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00669, + "step": 229, + "tokens/total": 7313120, + "tokens/train_per_sec_per_gpu": 31.3, + "tokens/trainable": 104707 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.10858234763145447, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0020512123592197895, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00205, + "step": 230, + "tokens/total": 7345184, + "tokens/train_per_sec_per_gpu": 30.81, + "tokens/trainable": 105167 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.1044008731842041, + "learning_rate": 6.60616572358065e-05, + "loss": 0.0010264476295560598, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00103, + "step": 231, + "tokens/total": 7377264, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 105576 + }, + { + "epoch": 0.90625, + "grad_norm": 0.14836947619915009, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0013845522189512849, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00139, + "step": 232, + "tokens/total": 7409424, + "tokens/train_per_sec_per_gpu": 30.32, + "tokens/trainable": 106043 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.022752461954951286, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0003245974367018789, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00032, + "step": 233, + "tokens/total": 7441456, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 106506 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.22981758415699005, + "learning_rate": 6.521548694236384e-05, + "loss": 0.002587020630016923, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00259, + "step": 234, + "tokens/total": 7473344, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 106951 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.5147526264190674, + "learning_rate": 6.493256429322259e-05, + "loss": 0.01256126631051302, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01264, + "step": 235, + "tokens/total": 7505040, + "tokens/train_per_sec_per_gpu": 25.64, + "tokens/trainable": 107382 + }, + { + "epoch": 0.921875, + "grad_norm": 0.3785130977630615, + "learning_rate": 6.464922830953799e-05, + "loss": 0.010956985875964165, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01102, + "step": 236, + "tokens/total": 7537120, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 107814 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.6193199753761292, + "learning_rate": 6.436549078207688e-05, + "loss": 0.020655140280723572, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02087, + "step": 237, + "tokens/total": 7569024, + "tokens/train_per_sec_per_gpu": 29.36, + "tokens/trainable": 108274 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.5084519982337952, + "learning_rate": 6.408136351831592e-05, + "loss": 0.011346567422151566, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01141, + "step": 238, + "tokens/total": 7600992, + "tokens/train_per_sec_per_gpu": 24.43, + "tokens/trainable": 108714 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.04367341846227646, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0004316020058467984, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00043, + "step": 239, + "tokens/total": 7633280, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 109220 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06729783862829208, + "learning_rate": 6.351198709240186e-05, + "loss": 0.000604260538239032, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.0006, + "step": 240, + "tokens/total": 7665408, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 109672 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.3010976314544678, + "learning_rate": 6.32267616243259e-05, + "loss": 0.006106264889240265, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00612, + "step": 241, + "tokens/total": 7697280, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 110135 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.11426915228366852, + "learning_rate": 6.294119380711849e-05, + "loss": 0.0014383264351636171, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00144, + "step": 242, + "tokens/total": 7729440, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 110563 + }, + { + "epoch": 0.94921875, + "grad_norm": 1.3792941570281982, + "learning_rate": 6.265529552442209e-05, + "loss": 0.003547664964571595, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00355, + "step": 243, + "tokens/total": 7761584, + "tokens/train_per_sec_per_gpu": 30.76, + "tokens/trainable": 111049 + }, + { + "epoch": 0.953125, + "grad_norm": 0.07595008611679077, + "learning_rate": 6.236907867363127e-05, + "loss": 0.0009103374904952943, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00091, + "step": 244, + "tokens/total": 7793824, + "tokens/train_per_sec_per_gpu": 28.23, + "tokens/trainable": 111495 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05425361171364784, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0013249842450022697, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00133, + "step": 245, + "tokens/total": 7825872, + "tokens/train_per_sec_per_gpu": 26.47, + "tokens/trainable": 111968 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.17112742364406586, + "learning_rate": 6.179573692313344e-05, + "loss": 0.001098912674933672, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.0011, + "step": 246, + "tokens/total": 7857840, + "tokens/train_per_sec_per_gpu": 30.09, + "tokens/trainable": 112416 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.10754083096981049, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0031517897732555866, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00316, + "step": 247, + "tokens/total": 7890000, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 112882 + }, + { + "epoch": 0.96875, + "grad_norm": 0.2477901428937912, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005755824968218803, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00577, + "step": 248, + "tokens/total": 7921984, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 113390 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.332253634929657, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.013207933865487576, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.0133, + "step": 249, + "tokens/total": 7953824, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 113904 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.18514207005500793, + "learning_rate": 6.064575550087316e-05, + "loss": 0.00523729994893074, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00525, + "step": 250, + "tokens/total": 7985984, + "tokens/train_per_sec_per_gpu": 28.41, + "tokens/trainable": 114351 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.1852743923664093, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0031474479474127293, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00315, + "step": 251, + "tokens/total": 8018256, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 114842 + }, + { + "epoch": 0.984375, + "grad_norm": 0.43020468950271606, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0053521618247032166, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00537, + "step": 252, + "tokens/total": 8050528, + "tokens/train_per_sec_per_gpu": 23.97, + "tokens/trainable": 115263 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.048135045915842056, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0005759480991400778, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00058, + "step": 253, + "tokens/total": 8082496, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 115703 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.028733640909194946, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00040354474913328886, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0004, + "step": 254, + "tokens/total": 8114592, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 116157 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.2808869481086731, + "learning_rate": 5.920308275189541e-05, + "loss": 0.006315852981060743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00634, + "step": 255, + "tokens/total": 8146752, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 116567 + }, + { + "epoch": 1.0, + "grad_norm": 0.054996367543935776, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0009099108865484595, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00091, + "step": 256, + "tokens/total": 8178768, + "tokens/train_per_sec_per_gpu": 25.98, + "tokens/trainable": 117030 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 5.551407787864274e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/training_args.bin b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..dec74f9f304b44cbe47132d86c4e6285e659f40b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-256/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634 +size 8273 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/README.md b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ba677470ac9c29f9770a4aacb65c644ebd671a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/deconf_wave_deconf_control/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/deconf_wave_deconf_control/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/adapter_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..21fcb5b00be777c20a2d2dcceffc539dc51679b4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/deconf_wave_deconf_control/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "gate_proj", + "v_proj", + "k_proj", + "up_proj", + "q_proj", + "o_proj", + "down_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/adapter_model.safetensors b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..ad8130f9c417f68a45fc3726328aa2c5de6da8d4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:aee47495451c0c9187d5505e050008c00c76f2aa93bbd3a50817fc86352100e6 +size 547777976 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/chat_template.jinja b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/optimizer.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..6b2249168efc0dacfa9b4c93e90a9b0bee5b1cbd --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b7fe2f0810e2902d7d86825a7e83a258b7aefcef1bf097c446b7e0f3206ab787 +size 1048106435 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/rng_state.pth b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..2ef0f7c4a0faa2bf66c906e691088ca97db1e246 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4717c559644ca7b5762ae58bacaf7a7231b780b6aa06477169e4c4976b3338d2 +size 14645 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/scheduler.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..5f00d58c65c626a11e09272704f772fedb532412 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a3b676d4693f7202d0a8375b6101005aaf04d716d840a44d0b3c3ec839a7b363 +size 1465 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/tokenizer.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/tokenizer_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/tokens_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..bf546a1514f99307963f74f5da677e577b542839 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/tokens_state.json @@ -0,0 +1 @@ +{"total": 9200544, "trainable": 131763} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/trainer_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..13a2fa012648a0971d4ec88f1f09468aeb66b8e4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/trainer_state.json @@ -0,0 +1,4066 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.125, + "eval_steps": 500, + "global_step": 288, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.2140090465545654, + "learning_rate": 0.0, + "loss": 0.16441112756729126, + "memory/device_reserved (GiB)": 38.08, + "memory/max_active (GiB)": 35.64, + "memory/max_allocated (GiB)": 35.64, + "ppl": 1.1787, + "step": 1, + "tokens/total": 32160, + "tokens/train_per_sec_per_gpu": 21.28, + "tokens/trainable": 470 + }, + { + "epoch": 0.0078125, + "grad_norm": 2.92620849609375, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15774038434028625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.17086, + "step": 2, + "tokens/total": 64192, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 922 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.135961890220642, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1664382815361023, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.18109, + "step": 3, + "tokens/total": 96224, + "tokens/train_per_sec_per_gpu": 28.54, + "tokens/trainable": 1396 + }, + { + "epoch": 0.015625, + "grad_norm": 0.880176305770874, + "learning_rate": 1.2e-05, + "loss": 0.15334013104438782, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.16572, + "step": 4, + "tokens/total": 128336, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 1850 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.1433289051055908, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.16352537274360657, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.17766, + "step": 5, + "tokens/total": 160784, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 2302 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9728212952613831, + "learning_rate": 2e-05, + "loss": 0.15144473314285278, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.16351, + "step": 6, + "tokens/total": 192624, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 2742 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.7796363830566406, + "learning_rate": 2.4e-05, + "loss": 0.13553467392921448, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.14515, + "step": 7, + "tokens/total": 224608, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 3208 + }, + { + "epoch": 0.03125, + "grad_norm": 1.2444697618484497, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.1077304258942604, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.11375, + "step": 8, + "tokens/total": 256592, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 3655 + }, + { + "epoch": 0.03515625, + "grad_norm": 2.7339375019073486, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.08772765100002289, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.09169, + "step": 9, + "tokens/total": 286400, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 4091 + }, + { + "epoch": 0.0390625, + "grad_norm": 4.109770774841309, + "learning_rate": 3.6e-05, + "loss": 0.130940243601799, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.1399, + "step": 10, + "tokens/total": 318352, + "tokens/train_per_sec_per_gpu": 29.13, + "tokens/trainable": 4553 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.3768608570098877, + "learning_rate": 4e-05, + "loss": 0.10059958696365356, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.10583, + "step": 11, + "tokens/total": 350336, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 4992 + }, + { + "epoch": 0.046875, + "grad_norm": 3.281402826309204, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.12574875354766846, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.134, + "step": 12, + "tokens/total": 382448, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 5494 + }, + { + "epoch": 0.05078125, + "grad_norm": 1.7738980054855347, + "learning_rate": 4.8e-05, + "loss": 0.033296722918748856, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03386, + "step": 13, + "tokens/total": 414352, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 5933 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.64290189743042, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06366822123527527, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.06574, + "step": 14, + "tokens/total": 446400, + "tokens/train_per_sec_per_gpu": 24.8, + "tokens/trainable": 6369 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.3691928386688232, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.08168518543243408, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.08511, + "step": 15, + "tokens/total": 478688, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 6820 + }, + { + "epoch": 0.0625, + "grad_norm": 2.4356393814086914, + "learning_rate": 6e-05, + "loss": 0.09031196683645248, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.09452, + "step": 16, + "tokens/total": 510512, + "tokens/train_per_sec_per_gpu": 25.7, + "tokens/trainable": 7258 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.55682373046875, + "learning_rate": 6.400000000000001e-05, + "loss": 0.07496573030948639, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.07785, + "step": 17, + "tokens/total": 542736, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 7710 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.1054646968841553, + "learning_rate": 6.800000000000001e-05, + "loss": 0.05218805745244026, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.05357, + "step": 18, + "tokens/total": 574832, + "tokens/train_per_sec_per_gpu": 25.78, + "tokens/trainable": 8174 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.7429099082946777, + "learning_rate": 7.2e-05, + "loss": 0.03790827840566635, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.03864, + "step": 19, + "tokens/total": 606976, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 8617 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9508209228515625, + "learning_rate": 7.6e-05, + "loss": 0.04993613809347153, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0512, + "step": 20, + "tokens/total": 639200, + "tokens/train_per_sec_per_gpu": 24.91, + "tokens/trainable": 9037 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.0297844409942627, + "learning_rate": 8e-05, + "loss": 0.017908263951539993, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.01807, + "step": 21, + "tokens/total": 671072, + "tokens/train_per_sec_per_gpu": 30.42, + "tokens/trainable": 9503 + }, + { + "epoch": 0.0859375, + "grad_norm": 0.9279879331588745, + "learning_rate": 8.4e-05, + "loss": 0.013478193432092667, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01357, + "step": 22, + "tokens/total": 703136, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 9972 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.526488184928894, + "learning_rate": 8.800000000000001e-05, + "loss": 0.017310388386249542, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01746, + "step": 23, + "tokens/total": 735120, + "tokens/train_per_sec_per_gpu": 31.45, + "tokens/trainable": 10447 + }, + { + "epoch": 0.09375, + "grad_norm": 3.4049417972564697, + "learning_rate": 9.200000000000001e-05, + "loss": 0.0353429913520813, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.03597, + "step": 24, + "tokens/total": 767040, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 10899 + }, + { + "epoch": 0.09765625, + "grad_norm": 7.511656761169434, + "learning_rate": 9.6e-05, + "loss": 0.1335875540971756, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.14292, + "step": 25, + "tokens/total": 798976, + "tokens/train_per_sec_per_gpu": 27.72, + "tokens/trainable": 11360 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.0069234371185303, + "learning_rate": 0.0001, + "loss": 0.04382415860891342, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0448, + "step": 26, + "tokens/total": 830880, + "tokens/train_per_sec_per_gpu": 24.61, + "tokens/trainable": 11775 + }, + { + "epoch": 0.10546875, + "grad_norm": 7.807938575744629, + "learning_rate": 9.99990636831587e-05, + "loss": 0.01210833340883255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.01218, + "step": 27, + "tokens/total": 860960, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 12242 + }, + { + "epoch": 0.109375, + "grad_norm": 2.23823881149292, + "learning_rate": 9.999625477159879e-05, + "loss": 0.0272970050573349, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02767, + "step": 28, + "tokens/total": 892816, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 12696 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.315584421157837, + "learning_rate": 9.999157338221051e-05, + "loss": 0.025516577064990997, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.02584, + "step": 29, + "tokens/total": 924736, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 13153 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.8413651585578918, + "learning_rate": 9.998501970980562e-05, + "loss": 0.007838520221412182, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00787, + "step": 30, + "tokens/total": 954592, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 13624 + }, + { + "epoch": 0.12109375, + "grad_norm": 1.6797157526016235, + "learning_rate": 9.997659402710915e-05, + "loss": 0.03002442792057991, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.03048, + "step": 31, + "tokens/total": 986592, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 14064 + }, + { + "epoch": 0.125, + "grad_norm": 1.3320434093475342, + "learning_rate": 9.996629668474818e-05, + "loss": 0.03772728145122528, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.03845, + "step": 32, + "tokens/total": 1018720, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 14565 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.6335167288780212, + "learning_rate": 9.995412811123711e-05, + "loss": 0.013970119878649712, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01407, + "step": 33, + "tokens/total": 1050624, + "tokens/train_per_sec_per_gpu": 28.18, + "tokens/trainable": 15005 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.5626921653747559, + "learning_rate": 9.994008881295999e-05, + "loss": 0.016654539853334427, + "memory/device_reserved (GiB)": 37.74, + "memory/max_active (GiB)": 36.22, + "memory/max_allocated (GiB)": 36.22, + "ppl": 1.01679, + "step": 34, + "tokens/total": 1082176, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 15459 + }, + { + "epoch": 0.13671875, + "grad_norm": 0.603219211101532, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021408583968877792, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02164, + "step": 35, + "tokens/total": 1113968, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 15960 + }, + { + "epoch": 0.140625, + "grad_norm": 0.9779674410820007, + "learning_rate": 9.99064004568618e-05, + "loss": 0.03277068957686424, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03331, + "step": 36, + "tokens/total": 1145952, + "tokens/train_per_sec_per_gpu": 28.68, + "tokens/trainable": 16428 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.8855255842208862, + "learning_rate": 9.988675280095074e-05, + "loss": 0.017048098146915436, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01719, + "step": 37, + "tokens/total": 1177872, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 16884 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.9266122579574585, + "learning_rate": 9.986523722403528e-05, + "loss": 0.045863546431064606, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.04693, + "step": 38, + "tokens/total": 1209776, + "tokens/train_per_sec_per_gpu": 26.91, + "tokens/trainable": 17341 + }, + { + "epoch": 0.15234375, + "grad_norm": 0.8280895948410034, + "learning_rate": 9.984185462146642e-05, + "loss": 0.033055759966373444, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03361, + "step": 39, + "tokens/total": 1241888, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 17786 + }, + { + "epoch": 0.15625, + "grad_norm": 5.029730319976807, + "learning_rate": 9.98166059662897e-05, + "loss": 0.025589998811483383, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02592, + "step": 40, + "tokens/total": 1274032, + "tokens/train_per_sec_per_gpu": 29.22, + "tokens/trainable": 18262 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.5688032507896423, + "learning_rate": 9.978949230920472e-05, + "loss": 0.016095872968435287, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01623, + "step": 41, + "tokens/total": 1306000, + "tokens/train_per_sec_per_gpu": 27.49, + "tokens/trainable": 18716 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8217663168907166, + "learning_rate": 9.976051477852141e-05, + "loss": 0.026798982173204422, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02716, + "step": 42, + "tokens/total": 1337936, + "tokens/train_per_sec_per_gpu": 27.51, + "tokens/trainable": 19157 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.7309132218360901, + "learning_rate": 9.972967458011312e-05, + "loss": 0.033920761197805405, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0345, + "step": 43, + "tokens/total": 1369904, + "tokens/train_per_sec_per_gpu": 30.66, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.8134022355079651, + "learning_rate": 9.96969729973664e-05, + "loss": 0.024005085229873657, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0243, + "step": 44, + "tokens/total": 1401984, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 20119 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.35084816813468933, + "learning_rate": 9.966241139112754e-05, + "loss": 0.013319254852831364, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01341, + "step": 45, + "tokens/total": 1434032, + "tokens/train_per_sec_per_gpu": 27.45, + "tokens/trainable": 20560 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.8701573014259338, + "learning_rate": 9.96259911996461e-05, + "loss": 0.01356641948223114, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01366, + "step": 46, + "tokens/total": 1465872, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 20992 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.4406442642211914, + "learning_rate": 9.958771393851491e-05, + "loss": 0.04396980255842209, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 35.86, + "memory/max_allocated (GiB)": 35.86, + "ppl": 1.04495, + "step": 47, + "tokens/total": 1495440, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 21441 + }, + { + "epoch": 0.1875, + "grad_norm": 0.4966669976711273, + "learning_rate": 9.954758120060702e-05, + "loss": 0.011179964058101177, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01124, + "step": 48, + "tokens/total": 1527776, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 21901 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.41124436259269714, + "learning_rate": 9.950559465600948e-05, + "loss": 0.0063599650748074055, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00638, + "step": 49, + "tokens/total": 1559760, + "tokens/train_per_sec_per_gpu": 27.73, + "tokens/trainable": 22341 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.829739511013031, + "learning_rate": 9.946175605195379e-05, + "loss": 0.015785304829478264, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01591, + "step": 50, + "tokens/total": 1592000, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 22833 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5181736946105957, + "learning_rate": 9.941606721274322e-05, + "loss": 0.008393109776079655, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00843, + "step": 51, + "tokens/total": 1623920, + "tokens/train_per_sec_per_gpu": 25.61, + "tokens/trainable": 23277 + }, + { + "epoch": 0.203125, + "grad_norm": 1.5778990983963013, + "learning_rate": 9.936853003967685e-05, + "loss": 0.024984844028949738, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0253, + "step": 52, + "tokens/total": 1656032, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 23759 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.4938714504241943, + "learning_rate": 9.93191465109705e-05, + "loss": 0.038523320108652115, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03927, + "step": 53, + "tokens/total": 1688368, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 24193 + }, + { + "epoch": 0.2109375, + "grad_norm": 2.1427106857299805, + "learning_rate": 9.926791868167438e-05, + "loss": 0.06302924454212189, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.06506, + "step": 54, + "tokens/total": 1720560, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 24619 + }, + { + "epoch": 0.21484375, + "grad_norm": 1.8365188837051392, + "learning_rate": 9.921484868358753e-05, + "loss": 0.052160464227199554, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.05354, + "step": 55, + "tokens/total": 1752464, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 25075 + }, + { + "epoch": 0.21875, + "grad_norm": 0.7779749631881714, + "learning_rate": 9.915993872516924e-05, + "loss": 0.021328264847397804, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.02156, + "step": 56, + "tokens/total": 1784336, + "tokens/train_per_sec_per_gpu": 26.42, + "tokens/trainable": 25519 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.4189571142196655, + "learning_rate": 9.9103191091447e-05, + "loss": 0.014279939234256744, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01438, + "step": 57, + "tokens/total": 1816304, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 25966 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.3834710717201233, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01883944310247898, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01902, + "step": 58, + "tokens/total": 1848304, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 26423 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.7693278789520264, + "learning_rate": 9.898419232046825e-05, + "loss": 0.03533369302749634, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03597, + "step": 59, + "tokens/total": 1880640, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 26934 + }, + { + "epoch": 0.234375, + "grad_norm": 0.4606446325778961, + "learning_rate": 9.892194613523633e-05, + "loss": 0.021482212468981743, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02171, + "step": 60, + "tokens/total": 1912592, + "tokens/train_per_sec_per_gpu": 27.64, + "tokens/trainable": 27393 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.46030017733573914, + "learning_rate": 9.885787217854357e-05, + "loss": 0.017044665291905403, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01719, + "step": 61, + "tokens/total": 1944544, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 27852 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.7918264269828796, + "learning_rate": 9.879197311676887e-05, + "loss": 0.016954131424427032, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0171, + "step": 62, + "tokens/total": 1976688, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 28292 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.4787926971912384, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01525965891778469, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 63, + "tokens/total": 2008928, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 28798 + }, + { + "epoch": 0.25, + "grad_norm": 0.9033918976783752, + "learning_rate": 9.865471072312528e-05, + "loss": 0.02902211993932724, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02945, + "step": 64, + "tokens/total": 2041120, + "tokens/train_per_sec_per_gpu": 29.83, + "tokens/trainable": 29283 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.39305707812309265, + "learning_rate": 9.858335310330492e-05, + "loss": 0.00901167094707489, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.32, + "memory/max_allocated (GiB)": 36.32, + "ppl": 1.00905, + "step": 65, + "tokens/total": 2073024, + "tokens/train_per_sec_per_gpu": 26.48, + "tokens/trainable": 29745 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4494345784187317, + "learning_rate": 9.851018180226185e-05, + "loss": 0.007426132448017597, + "memory/device_reserved (GiB)": 37.47, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00745, + "step": 66, + "tokens/total": 2105184, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 30186 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.629649817943573, + "learning_rate": 9.843519986495259e-05, + "loss": 0.010609081946313381, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01067, + "step": 67, + "tokens/total": 2137264, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 30622 + }, + { + "epoch": 0.265625, + "grad_norm": 0.7591162919998169, + "learning_rate": 9.835841041168162e-05, + "loss": 0.00926615484058857, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00931, + "step": 68, + "tokens/total": 2169424, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 31097 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.7288486361503601, + "learning_rate": 9.82798166379715e-05, + "loss": 0.024310583248734474, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.02461, + "step": 69, + "tokens/total": 2201424, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 31595 + }, + { + "epoch": 0.2734375, + "grad_norm": 2.134056329727173, + "learning_rate": 9.819942181443002e-05, + "loss": 0.008422967046499252, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00846, + "step": 70, + "tokens/total": 2233424, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 32036 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6362871527671814, + "learning_rate": 9.811722928661392e-05, + "loss": 0.01577262207865715, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0159, + "step": 71, + "tokens/total": 2265328, + "tokens/train_per_sec_per_gpu": 23.65, + "tokens/trainable": 32428 + }, + { + "epoch": 0.28125, + "grad_norm": 0.6091912388801575, + "learning_rate": 9.803324247488975e-05, + "loss": 0.0064800274558365345, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0065, + "step": 72, + "tokens/total": 2297248, + "tokens/train_per_sec_per_gpu": 26.86, + "tokens/trainable": 32880 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.14533838629722595, + "learning_rate": 9.794746487429161e-05, + "loss": 0.001991454279050231, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00199, + "step": 73, + "tokens/total": 2329152, + "tokens/train_per_sec_per_gpu": 27.97, + "tokens/trainable": 33323 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.5400826930999756, + "learning_rate": 9.785990005437554e-05, + "loss": 0.017662350088357925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01782, + "step": 74, + "tokens/total": 2361344, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 33792 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.6132436394691467, + "learning_rate": 9.777055165907117e-05, + "loss": 0.01787603087723255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01804, + "step": 75, + "tokens/total": 2393168, + "tokens/train_per_sec_per_gpu": 29.53, + "tokens/trainable": 34223 + }, + { + "epoch": 0.296875, + "grad_norm": 0.7287399172782898, + "learning_rate": 9.767942340652993e-05, + "loss": 0.024754449725151062, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.31, + "memory/max_allocated (GiB)": 36.31, + "ppl": 1.02506, + "step": 76, + "tokens/total": 2424880, + "tokens/train_per_sec_per_gpu": 24.92, + "tokens/trainable": 34649 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.6790545582771301, + "learning_rate": 9.758651908897035e-05, + "loss": 0.014308430254459381, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01441, + "step": 77, + "tokens/total": 2457120, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 35094 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6203456521034241, + "learning_rate": 9.749184257252033e-05, + "loss": 0.034353043884038925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03495, + "step": 78, + "tokens/total": 2489104, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 35534 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.6238825917243958, + "learning_rate": 9.739539779705614e-05, + "loss": 0.032295141369104385, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.03282, + "step": 79, + "tokens/total": 2521136, + "tokens/train_per_sec_per_gpu": 29.68, + "tokens/trainable": 36029 + }, + { + "epoch": 0.3125, + "grad_norm": 0.18977171182632446, + "learning_rate": 9.729718877603861e-05, + "loss": 0.007260813377797604, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00729, + "step": 80, + "tokens/total": 2553168, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 36469 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.8054049611091614, + "learning_rate": 9.719721959634592e-05, + "loss": 0.034894898533821106, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.03551, + "step": 81, + "tokens/total": 2585584, + "tokens/train_per_sec_per_gpu": 25.75, + "tokens/trainable": 36906 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.6884189248085022, + "learning_rate": 9.709549441810375e-05, + "loss": 0.018210047855973244, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.01838, + "step": 82, + "tokens/total": 2617808, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 37390 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.6490103602409363, + "learning_rate": 9.699201747451195e-05, + "loss": 0.02030857279896736, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.02052, + "step": 83, + "tokens/total": 2649664, + "tokens/train_per_sec_per_gpu": 25.8, + "tokens/trainable": 37838 + }, + { + "epoch": 0.328125, + "grad_norm": 0.2439805567264557, + "learning_rate": 9.688679307166854e-05, + "loss": 0.008518049493432045, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00855, + "step": 84, + "tokens/total": 2681728, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 38310 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6747375726699829, + "learning_rate": 9.677982558839042e-05, + "loss": 0.029895059764385223, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.03035, + "step": 85, + "tokens/total": 2713488, + "tokens/train_per_sec_per_gpu": 28.49, + "tokens/trainable": 38789 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.582666277885437, + "learning_rate": 9.66711194760312e-05, + "loss": 0.014998597092926502, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01511, + "step": 86, + "tokens/total": 2745536, + "tokens/train_per_sec_per_gpu": 30.27, + "tokens/trainable": 39277 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.5364654660224915, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011567167937755585, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01163, + "step": 87, + "tokens/total": 2777680, + "tokens/train_per_sec_per_gpu": 29.73, + "tokens/trainable": 39737 + }, + { + "epoch": 0.34375, + "grad_norm": 0.34024426341056824, + "learning_rate": 9.644850953105288e-05, + "loss": 0.01014852337539196, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0102, + "step": 88, + "tokens/total": 2809568, + "tokens/train_per_sec_per_gpu": 24.68, + "tokens/trainable": 40179 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.32472267746925354, + "learning_rate": 9.633461496214225e-05, + "loss": 0.01229981891810894, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01238, + "step": 89, + "tokens/total": 2841760, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 40649 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.7906899452209473, + "learning_rate": 9.621900029118195e-05, + "loss": 0.02552586793899536, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.02585, + "step": 90, + "tokens/total": 2871696, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 41080 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.7444764971733093, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0054441955871880054, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00546, + "step": 91, + "tokens/total": 2903968, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 41599 + }, + { + "epoch": 0.359375, + "grad_norm": 0.1911851465702057, + "learning_rate": 9.598262995928611e-05, + "loss": 0.004665314219892025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00468, + "step": 92, + "tokens/total": 2936048, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 42076 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.21238519251346588, + "learning_rate": 9.586188413468492e-05, + "loss": 0.003917238209396601, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00392, + "step": 93, + "tokens/total": 2968176, + "tokens/train_per_sec_per_gpu": 28.07, + "tokens/trainable": 42522 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.41729435324668884, + "learning_rate": 9.57394378802934e-05, + "loss": 0.013555807992815971, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01365, + "step": 94, + "tokens/total": 3000304, + "tokens/train_per_sec_per_gpu": 28.69, + "tokens/trainable": 42974 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.7246769070625305, + "learning_rate": 9.56152962916e-05, + "loss": 0.020221399143338203, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.02043, + "step": 95, + "tokens/total": 3032080, + "tokens/train_per_sec_per_gpu": 25.35, + "tokens/trainable": 43380 + }, + { + "epoch": 0.375, + "grad_norm": 0.16139821708202362, + "learning_rate": 9.548946453464296e-05, + "loss": 0.003853276837617159, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00386, + "step": 96, + "tokens/total": 3064320, + "tokens/train_per_sec_per_gpu": 28.47, + "tokens/trainable": 43865 + }, + { + "epoch": 0.37890625, + "grad_norm": 1.0509370565414429, + "learning_rate": 9.53619478457953e-05, + "loss": 0.015099998563528061, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01521, + "step": 97, + "tokens/total": 3096144, + "tokens/train_per_sec_per_gpu": 26.61, + "tokens/trainable": 44328 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.6859350204467773, + "learning_rate": 9.523275153154695e-05, + "loss": 0.004579808097332716, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00459, + "step": 98, + "tokens/total": 3127600, + "tokens/train_per_sec_per_gpu": 25.91, + "tokens/trainable": 44778 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5686795711517334, + "learning_rate": 9.51018809682839e-05, + "loss": 0.022520575672388077, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02278, + "step": 99, + "tokens/total": 3159520, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 45225 + }, + { + "epoch": 0.390625, + "grad_norm": 1.6308369636535645, + "learning_rate": 9.49693416020645e-05, + "loss": 0.011883002705872059, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01195, + "step": 100, + "tokens/total": 3191568, + "tokens/train_per_sec_per_gpu": 29.75, + "tokens/trainable": 45678 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.7186088562011719, + "learning_rate": 9.483513894839276e-05, + "loss": 0.0060632615350186825, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00608, + "step": 101, + "tokens/total": 3223648, + "tokens/train_per_sec_per_gpu": 27.14, + "tokens/trainable": 46125 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.12036337703466415, + "learning_rate": 9.469927859198888e-05, + "loss": 0.0019118499476462603, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00191, + "step": 102, + "tokens/total": 3255744, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 46612 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.29928353428840637, + "learning_rate": 9.456176618655689e-05, + "loss": 0.00944911316037178, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00949, + "step": 103, + "tokens/total": 3287792, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 47059 + }, + { + "epoch": 0.40625, + "grad_norm": 0.3766293227672577, + "learning_rate": 9.442260745454927e-05, + "loss": 0.008984047919511795, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.68, + "memory/max_allocated (GiB)": 36.68, + "ppl": 1.00902, + "step": 104, + "tokens/total": 3320016, + "tokens/train_per_sec_per_gpu": 28.92, + "tokens/trainable": 47536 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3481820523738861, + "learning_rate": 9.428180818692884e-05, + "loss": 0.005444999784231186, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00546, + "step": 105, + "tokens/total": 3351952, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 48037 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.8414323329925537, + "learning_rate": 9.413937424292791e-05, + "loss": 0.032438624650239944, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03297, + "step": 106, + "tokens/total": 3384032, + "tokens/train_per_sec_per_gpu": 27.11, + "tokens/trainable": 48491 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.5067655444145203, + "learning_rate": 9.399531154980424e-05, + "loss": 0.008442888967692852, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00848, + "step": 107, + "tokens/total": 3416192, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 48940 + }, + { + "epoch": 0.421875, + "grad_norm": 0.41617852449417114, + "learning_rate": 9.384962610259455e-05, + "loss": 0.008386321365833282, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00842, + "step": 108, + "tokens/total": 3448080, + "tokens/train_per_sec_per_gpu": 26.56, + "tokens/trainable": 49389 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1587515026330948, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0031663556583225727, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.03, + "memory/max_allocated (GiB)": 36.03, + "ppl": 1.00317, + "step": 109, + "tokens/total": 3477952, + "tokens/train_per_sec_per_gpu": 29.8, + "tokens/trainable": 49838 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2903148829936981, + "learning_rate": 9.355341126345868e-05, + "loss": 0.007958847098052502, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00799, + "step": 110, + "tokens/total": 3509776, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 50310 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.642983078956604, + "learning_rate": 9.340289419824107e-05, + "loss": 0.006748045329004526, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00677, + "step": 111, + "tokens/total": 3541808, + "tokens/train_per_sec_per_gpu": 28.04, + "tokens/trainable": 50755 + }, + { + "epoch": 0.4375, + "grad_norm": 0.14928282797336578, + "learning_rate": 9.325077903184159e-05, + "loss": 0.0027839094400405884, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00279, + "step": 112, + "tokens/total": 3574064, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 51213 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.05922335386276245, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0009622900979593396, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00096, + "step": 113, + "tokens/total": 3605984, + "tokens/train_per_sec_per_gpu": 27.99, + "tokens/trainable": 51660 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.44227731227874756, + "learning_rate": 9.2941779782269e-05, + "loss": 0.002763954224064946, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00277, + "step": 114, + "tokens/total": 3638080, + "tokens/train_per_sec_per_gpu": 27.03, + "tokens/trainable": 52133 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5176818370819092, + "learning_rate": 9.278490855781596e-05, + "loss": 0.015623578801751137, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01575, + "step": 115, + "tokens/total": 3670016, + "tokens/train_per_sec_per_gpu": 31.44, + "tokens/trainable": 52580 + }, + { + "epoch": 0.453125, + "grad_norm": 0.4265497922897339, + "learning_rate": 9.262646494908604e-05, + "loss": 0.00794076919555664, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00797, + "step": 116, + "tokens/total": 3701984, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 53033 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.40457987785339355, + "learning_rate": 9.246645554956457e-05, + "loss": 0.022708112373948097, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02297, + "step": 117, + "tokens/total": 3733920, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 53517 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.01414525043219328, + "learning_rate": 9.230488701789578e-05, + "loss": 0.00020209309877827764, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.0002, + "step": 118, + "tokens/total": 3765840, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 53967 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.11359831690788269, + "learning_rate": 9.214176607760577e-05, + "loss": 0.0012168455868959427, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00122, + "step": 119, + "tokens/total": 3797680, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 54430 + }, + { + "epoch": 0.46875, + "grad_norm": 0.35982534289360046, + "learning_rate": 9.197709951682268e-05, + "loss": 0.004110721405595541, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00412, + "step": 120, + "tokens/total": 3829920, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 54896 + }, + { + "epoch": 0.47265625, + "grad_norm": 1.3690382242202759, + "learning_rate": 9.181089418799428e-05, + "loss": 0.02705790475010872, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02743, + "step": 121, + "tokens/total": 3862080, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 55379 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.351891428232193, + "learning_rate": 9.164315700760271e-05, + "loss": 0.0020163573790341616, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00202, + "step": 122, + "tokens/total": 3893968, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 55803 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4747961461544037, + "learning_rate": 9.147389495587671e-05, + "loss": 0.012588979676365852, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01267, + "step": 123, + "tokens/total": 3926160, + "tokens/train_per_sec_per_gpu": 30.14, + "tokens/trainable": 56249 + }, + { + "epoch": 0.484375, + "grad_norm": 0.4064250588417053, + "learning_rate": 9.130311507650116e-05, + "loss": 0.0022941348142921925, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0023, + "step": 124, + "tokens/total": 3958208, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 56713 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.22056929767131805, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0022175831254571676, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00222, + "step": 125, + "tokens/total": 3990208, + "tokens/train_per_sec_per_gpu": 32.57, + "tokens/trainable": 57196 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5698862671852112, + "learning_rate": 9.09570303250602e-05, + "loss": 0.010205100290477276, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01026, + "step": 126, + "tokens/total": 4022224, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 57680 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.2593887448310852, + "learning_rate": 9.078173985499394e-05, + "loss": 0.0032277877908200026, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00323, + "step": 127, + "tokens/total": 4053952, + "tokens/train_per_sec_per_gpu": 26.51, + "tokens/trainable": 58110 + }, + { + "epoch": 0.5, + "grad_norm": 0.3735230565071106, + "learning_rate": 9.060496036067713e-05, + "loss": 0.01031105499714613, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01036, + "step": 128, + "tokens/total": 4086144, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 58534 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.1332484930753708, + "learning_rate": 9.042669919862615e-05, + "loss": 0.0012390739284455776, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00124, + "step": 129, + "tokens/total": 4118096, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 58998 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.20182718336582184, + "learning_rate": 9.024696378701557e-05, + "loss": 0.004501686431467533, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00451, + "step": 130, + "tokens/total": 4150272, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 59448 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.529080331325531, + "learning_rate": 9.006576160536948e-05, + "loss": 0.00763288140296936, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00766, + "step": 131, + "tokens/total": 4181904, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 59906 + }, + { + "epoch": 0.515625, + "grad_norm": 0.21625548601150513, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0024378676898777485, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00244, + "step": 132, + "tokens/total": 4214032, + "tokens/train_per_sec_per_gpu": 28.36, + "tokens/trainable": 60350 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.37116539478302, + "learning_rate": 8.969898715494506e-05, + "loss": 0.0037849934305995703, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00379, + "step": 133, + "tokens/total": 4246064, + "tokens/train_per_sec_per_gpu": 30.18, + "tokens/trainable": 60831 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5646623373031616, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006517104804515839, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00654, + "step": 134, + "tokens/total": 4278224, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 61305 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.6371963620185852, + "learning_rate": 8.932643689864568e-05, + "loss": 0.008687382563948631, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00873, + "step": 135, + "tokens/total": 4310112, + "tokens/train_per_sec_per_gpu": 31.48, + "tokens/trainable": 61792 + }, + { + "epoch": 0.53125, + "grad_norm": 0.8210524916648865, + "learning_rate": 8.913801518498845e-05, + "loss": 0.008436474949121475, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00847, + "step": 136, + "tokens/total": 4342192, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 62253 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.5692176222801208, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017321188002824783, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.01747, + "step": 137, + "tokens/total": 4374096, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.12828375399112701, + "learning_rate": 8.875691779131569e-05, + "loss": 0.0007754197577014565, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00078, + "step": 138, + "tokens/total": 4406272, + "tokens/train_per_sec_per_gpu": 24.22, + "tokens/trainable": 63145 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.3188190162181854, + "learning_rate": 8.856425797031829e-05, + "loss": 0.006707335356622934, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00673, + "step": 139, + "tokens/total": 4438112, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 63587 + }, + { + "epoch": 0.546875, + "grad_norm": 0.6216707229614258, + "learning_rate": 8.837020140354295e-05, + "loss": 0.007024695165455341, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00705, + "step": 140, + "tokens/total": 4470128, + "tokens/train_per_sec_per_gpu": 30.64, + "tokens/trainable": 64052 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.1480305939912796, + "learning_rate": 8.817475616647554e-05, + "loss": 0.00229115248657763, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00229, + "step": 141, + "tokens/total": 4502096, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 64526 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22254763543605804, + "learning_rate": 8.797793039239017e-05, + "loss": 0.0029198499396443367, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00292, + "step": 142, + "tokens/total": 4534496, + "tokens/train_per_sec_per_gpu": 29.08, + "tokens/trainable": 64983 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.6901941895484924, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013051149435341358, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01314, + "step": 143, + "tokens/total": 4566720, + "tokens/train_per_sec_per_gpu": 31.56, + "tokens/trainable": 65492 + }, + { + "epoch": 0.5625, + "grad_norm": 0.9545458555221558, + "learning_rate": 8.758017005316988e-05, + "loss": 0.02235381305217743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.02261, + "step": 144, + "tokens/total": 4598880, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 65925 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.5515892505645752, + "learning_rate": 8.737925204046629e-05, + "loss": 0.02038375660777092, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.02059, + "step": 145, + "tokens/total": 4630864, + "tokens/train_per_sec_per_gpu": 26.3, + "tokens/trainable": 66383 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3512430191040039, + "learning_rate": 8.717698659491851e-05, + "loss": 0.008233271539211273, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00827, + "step": 146, + "tokens/total": 4662864, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 66840 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.6830361485481262, + "learning_rate": 8.697338213361735e-05, + "loss": 0.013419180177152157, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01351, + "step": 147, + "tokens/total": 4694912, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 67297 + }, + { + "epoch": 0.578125, + "grad_norm": 0.2117031067609787, + "learning_rate": 8.676844712937552e-05, + "loss": 0.004134457092732191, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00414, + "step": 148, + "tokens/total": 4727120, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 67769 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.378053218126297, + "learning_rate": 8.656219011037509e-05, + "loss": 0.009747720323503017, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0098, + "step": 149, + "tokens/total": 4759168, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 68257 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.21076741814613342, + "learning_rate": 8.63546196598125e-05, + "loss": 0.005035653710365295, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00505, + "step": 150, + "tokens/total": 4791008, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 68706 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.25075697898864746, + "learning_rate": 8.614574441554145e-05, + "loss": 0.006027051247656345, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00605, + "step": 151, + "tokens/total": 4822912, + "tokens/train_per_sec_per_gpu": 27.95, + "tokens/trainable": 69131 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5774204730987549, + "learning_rate": 8.593557306971349e-05, + "loss": 0.018617186695337296, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01879, + "step": 152, + "tokens/total": 4854816, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 69586 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.1364293098449707, + "learning_rate": 8.572411436841618e-05, + "loss": 0.002049289643764496, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00205, + "step": 153, + "tokens/total": 4886656, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 70061 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.25387004017829895, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0044304742477834225, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00444, + "step": 154, + "tokens/total": 4918768, + "tokens/train_per_sec_per_gpu": 23.26, + "tokens/trainable": 70467 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3047125041484833, + "learning_rate": 8.529737015125824e-05, + "loss": 0.0043389927595853806, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00435, + "step": 155, + "tokens/total": 4951008, + "tokens/train_per_sec_per_gpu": 27.53, + "tokens/trainable": 70933 + }, + { + "epoch": 0.609375, + "grad_norm": 0.1442601978778839, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0031973645091056824, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.0032, + "step": 156, + "tokens/total": 4982976, + "tokens/train_per_sec_per_gpu": 28.86, + "tokens/trainable": 71382 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.2608455717563629, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0032951058819890022, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0033, + "step": 157, + "tokens/total": 5014640, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 71808 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.637295663356781, + "learning_rate": 8.464782037243449e-05, + "loss": 0.01436957623809576, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.01447, + "step": 158, + "tokens/total": 5046672, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 72249 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.9834056496620178, + "learning_rate": 8.442882418044202e-05, + "loss": 0.02375311404466629, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.02404, + "step": 159, + "tokens/total": 5078528, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 72726 + }, + { + "epoch": 0.625, + "grad_norm": 0.35432615876197815, + "learning_rate": 8.420860333495179e-05, + "loss": 0.006119017023593187, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00614, + "step": 160, + "tokens/total": 5110704, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 73148 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.44270065426826477, + "learning_rate": 8.398716700025208e-05, + "loss": 0.01352179329842329, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.72, + "memory/max_allocated (GiB)": 36.72, + "ppl": 1.01361, + "step": 161, + "tokens/total": 5142944, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 73600 + }, + { + "epoch": 0.6328125, + "grad_norm": 1.2181634902954102, + "learning_rate": 8.376452439121266e-05, + "loss": 0.01820964366197586, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.01838, + "step": 162, + "tokens/total": 5174592, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 74068 + }, + { + "epoch": 0.63671875, + "grad_norm": 1.9321303367614746, + "learning_rate": 8.354068477290124e-05, + "loss": 0.009742327965795994, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00979, + "step": 163, + "tokens/total": 5206432, + "tokens/train_per_sec_per_gpu": 29.89, + "tokens/trainable": 74527 + }, + { + "epoch": 0.640625, + "grad_norm": 0.3545796275138855, + "learning_rate": 8.331565746019807e-05, + "loss": 0.00975855067372322, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00981, + "step": 164, + "tokens/total": 5238384, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 74992 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.35368478298187256, + "learning_rate": 8.308945181740812e-05, + "loss": 0.010195466689765453, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01025, + "step": 165, + "tokens/total": 5270272, + "tokens/train_per_sec_per_gpu": 29.41, + "tokens/trainable": 75442 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.34957799315452576, + "learning_rate": 8.286207725787153e-05, + "loss": 0.009373681619763374, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00942, + "step": 166, + "tokens/total": 5302368, + "tokens/train_per_sec_per_gpu": 26.2, + "tokens/trainable": 75887 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.3684624135494232, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008211556822061539, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00825, + "step": 167, + "tokens/total": 5334496, + "tokens/train_per_sec_per_gpu": 28.21, + "tokens/trainable": 76331 + }, + { + "epoch": 0.65625, + "grad_norm": 0.12826858460903168, + "learning_rate": 8.240385928474219e-05, + "loss": 0.004492453299462795, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0045, + "step": 168, + "tokens/total": 5364288, + "tokens/train_per_sec_per_gpu": 30.16, + "tokens/trainable": 76745 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.31828173995018005, + "learning_rate": 8.217303493946967e-05, + "loss": 0.005733514670282602, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00575, + "step": 169, + "tokens/total": 5396512, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 77201 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6174353361129761, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01915649324655533, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01934, + "step": 170, + "tokens/total": 5428624, + "tokens/train_per_sec_per_gpu": 27.75, + "tokens/trainable": 77655 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.2902314066886902, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0036087254993617535, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00362, + "step": 171, + "tokens/total": 5460880, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 78122 + }, + { + "epoch": 0.671875, + "grad_norm": 0.3267754912376404, + "learning_rate": 8.147381587530713e-05, + "loss": 0.008090222254395485, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00812, + "step": 172, + "tokens/total": 5493088, + "tokens/train_per_sec_per_gpu": 27.86, + "tokens/trainable": 78576 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.4086189866065979, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006351984106004238, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00637, + "step": 173, + "tokens/total": 5525312, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 79059 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5261490345001221, + "learning_rate": 8.100214524900103e-05, + "loss": 0.008662862703204155, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.0087, + "step": 174, + "tokens/total": 5556880, + "tokens/train_per_sec_per_gpu": 24.62, + "tokens/trainable": 79498 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11292761564254761, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0027105510234832764, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00271, + "step": 175, + "tokens/total": 5588976, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 79922 + }, + { + "epoch": 0.6875, + "grad_norm": 0.33315548300743103, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0030926030594855547, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.0031, + "step": 176, + "tokens/total": 5618992, + "tokens/train_per_sec_per_gpu": 29.51, + "tokens/trainable": 80383 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.3136921226978302, + "learning_rate": 8.028654871074489e-05, + "loss": 0.005057850852608681, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00507, + "step": 177, + "tokens/total": 5651072, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 80824 + }, + { + "epoch": 0.6953125, + "grad_norm": 1.0526509284973145, + "learning_rate": 8.004589869885986e-05, + "loss": 0.010707498528063297, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01077, + "step": 178, + "tokens/total": 5683296, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 81243 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.1329648196697235, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0017975447699427605, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.0018, + "step": 179, + "tokens/total": 5715520, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 81716 + }, + { + "epoch": 0.703125, + "grad_norm": 1.0749201774597168, + "learning_rate": 7.95614819466576e-05, + "loss": 0.025617048144340515, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02595, + "step": 180, + "tokens/total": 5747488, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 82181 + }, + { + "epoch": 0.70703125, + "grad_norm": 1.0446465015411377, + "learning_rate": 7.931773536489872e-05, + "loss": 0.030236585065722466, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 35.9, + "memory/max_allocated (GiB)": 35.9, + "ppl": 1.0307, + "step": 181, + "tokens/total": 5777264, + "tokens/train_per_sec_per_gpu": 28.89, + "tokens/trainable": 82626 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22853964567184448, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0038214433006942272, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00383, + "step": 182, + "tokens/total": 5809264, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 83089 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.057821910828351974, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0004995397757738829, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0005, + "step": 183, + "tokens/total": 5841488, + "tokens/train_per_sec_per_gpu": 29.34, + "tokens/trainable": 83590 + }, + { + "epoch": 0.71875, + "grad_norm": 1.3264167308807373, + "learning_rate": 7.85804646415409e-05, + "loss": 0.013232000172138214, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01332, + "step": 184, + "tokens/total": 5873696, + "tokens/train_per_sec_per_gpu": 29.33, + "tokens/trainable": 84046 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.6027824878692627, + "learning_rate": 7.833273149760207e-05, + "loss": 0.012215660884976387, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01229, + "step": 185, + "tokens/total": 5905408, + "tokens/train_per_sec_per_gpu": 28.25, + "tokens/trainable": 84517 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.5848041772842407, + "learning_rate": 7.808402738346527e-05, + "loss": 0.024627480655908585, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.02493, + "step": 186, + "tokens/total": 5937392, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 84974 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.6823275089263916, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0028813849203288555, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00289, + "step": 187, + "tokens/total": 5969344, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 85460 + }, + { + "epoch": 0.734375, + "grad_norm": 0.41855958104133606, + "learning_rate": 7.758374768294647e-05, + "loss": 0.009796380065381527, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00984, + "step": 188, + "tokens/total": 6001296, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 85939 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6849080324172974, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004100777208805084, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00411, + "step": 189, + "tokens/total": 6031216, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 86377 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.15838374197483063, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003986812196671963, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00399, + "step": 190, + "tokens/total": 6063296, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 86834 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.3117486536502838, + "learning_rate": 7.682630588562518e-05, + "loss": 0.009132719598710537, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00917, + "step": 191, + "tokens/total": 6095216, + "tokens/train_per_sec_per_gpu": 26.65, + "tokens/trainable": 87265 + }, + { + "epoch": 0.75, + "grad_norm": 0.7837114334106445, + "learning_rate": 7.657199467573129e-05, + "loss": 0.007239446043968201, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00727, + "step": 192, + "tokens/total": 6127488, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 87747 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21777178347110748, + "learning_rate": 7.631678576708561e-05, + "loss": 0.002932134782895446, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00294, + "step": 193, + "tokens/total": 6159488, + "tokens/train_per_sec_per_gpu": 30.37, + "tokens/trainable": 88239 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.19159242510795593, + "learning_rate": 7.606068977997255e-05, + "loss": 0.003356864908710122, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00336, + "step": 194, + "tokens/total": 6191296, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 88718 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.3977915346622467, + "learning_rate": 7.580371737159148e-05, + "loss": 0.008549144491553307, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00859, + "step": 195, + "tokens/total": 6223120, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 89129 + }, + { + "epoch": 0.765625, + "grad_norm": 0.3875303566455841, + "learning_rate": 7.554587923561324e-05, + "loss": 0.006131038535386324, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00615, + "step": 196, + "tokens/total": 6254880, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 89567 + }, + { + "epoch": 0.76953125, + "grad_norm": 3.4133481979370117, + "learning_rate": 7.528718610173511e-05, + "loss": 0.01836150512099266, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01853, + "step": 197, + "tokens/total": 6286960, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 89988 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.2509997487068176, + "learning_rate": 7.502764873523431e-05, + "loss": 0.003821226069703698, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00383, + "step": 198, + "tokens/total": 6318880, + "tokens/train_per_sec_per_gpu": 30.29, + "tokens/trainable": 90434 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.9550195932388306, + "learning_rate": 7.476727793652011e-05, + "loss": 0.006872436031699181, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0069, + "step": 199, + "tokens/total": 6350928, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 90913 + }, + { + "epoch": 0.78125, + "grad_norm": 0.411724716424942, + "learning_rate": 7.450608454068415e-05, + "loss": 0.014243390411138535, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01435, + "step": 200, + "tokens/total": 6382976, + "tokens/train_per_sec_per_gpu": 28.29, + "tokens/trainable": 91355 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.34264692664146423, + "learning_rate": 7.424407941704987e-05, + "loss": 0.005309835076332092, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00532, + "step": 201, + "tokens/total": 6415120, + "tokens/train_per_sec_per_gpu": 31.41, + "tokens/trainable": 91856 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.41955795884132385, + "learning_rate": 7.398127346871986e-05, + "loss": 0.009196512401103973, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00924, + "step": 202, + "tokens/total": 6447216, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 92311 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.03107454814016819, + "learning_rate": 7.371767763212238e-05, + "loss": 0.0004908143309876323, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00049, + "step": 203, + "tokens/total": 6479280, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 92764 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39578482508659363, + "learning_rate": 7.345330287655617e-05, + "loss": 0.006089594680815935, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00611, + "step": 204, + "tokens/total": 6511040, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 93227 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.6444572806358337, + "learning_rate": 7.31881602037339e-05, + "loss": 0.010252210311591625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0103, + "step": 205, + "tokens/total": 6543120, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 93675 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.3283638060092926, + "learning_rate": 7.29222606473245e-05, + "loss": 0.00761406822130084, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00764, + "step": 206, + "tokens/total": 6575520, + "tokens/train_per_sec_per_gpu": 27.02, + "tokens/trainable": 94120 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.09952464699745178, + "learning_rate": 7.265561527249383e-05, + "loss": 0.001628706930205226, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00163, + "step": 207, + "tokens/total": 6607408, + "tokens/train_per_sec_per_gpu": 29.43, + "tokens/trainable": 94557 + }, + { + "epoch": 0.8125, + "grad_norm": 0.31704938411712646, + "learning_rate": 7.238823517544436e-05, + "loss": 0.005896817892789841, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00591, + "step": 208, + "tokens/total": 6639488, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 95035 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.0999862551689148, + "learning_rate": 7.212013148295333e-05, + "loss": 0.001754446537233889, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00176, + "step": 209, + "tokens/total": 6671520, + "tokens/train_per_sec_per_gpu": 31.08, + "tokens/trainable": 95517 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.1405109018087387, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0017940793186426163, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0018, + "step": 210, + "tokens/total": 6703536, + "tokens/train_per_sec_per_gpu": 30.3, + "tokens/trainable": 96026 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.09185238927602768, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0015102234901860356, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00151, + "step": 211, + "tokens/total": 6735616, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 96477 + }, + { + "epoch": 0.828125, + "grad_norm": 1.0464473962783813, + "learning_rate": 7.131159054949273e-05, + "loss": 0.016731251031160355, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01687, + "step": 212, + "tokens/total": 6767584, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 96951 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.3517152965068817, + "learning_rate": 7.104070433827139e-05, + "loss": 0.007984626106917858, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00802, + "step": 213, + "tokens/total": 6799488, + "tokens/train_per_sec_per_gpu": 25.83, + "tokens/trainable": 97350 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.28192469477653503, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00557567086070776, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00559, + "step": 214, + "tokens/total": 6831552, + "tokens/train_per_sec_per_gpu": 24.66, + "tokens/trainable": 97792 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.46135279536247253, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0170141588896513, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.01716, + "step": 215, + "tokens/total": 6863488, + "tokens/train_per_sec_per_gpu": 30.38, + "tokens/trainable": 98257 + }, + { + "epoch": 0.84375, + "grad_norm": 0.4277898073196411, + "learning_rate": 7.022408581865382e-05, + "loss": 0.009399959817528725, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.71, + "memory/max_allocated (GiB)": 36.71, + "ppl": 1.00944, + "step": 216, + "tokens/total": 6895888, + "tokens/train_per_sec_per_gpu": 27.28, + "tokens/trainable": 98731 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.34988704323768616, + "learning_rate": 6.99505974422157e-05, + "loss": 0.002909669652581215, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00291, + "step": 217, + "tokens/total": 6928000, + "tokens/train_per_sec_per_gpu": 30.25, + "tokens/trainable": 99212 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.4028005599975586, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0058855339884757996, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0059, + "step": 218, + "tokens/total": 6960096, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 99654 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.3000575304031372, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0025230362080037594, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00253, + "step": 219, + "tokens/total": 6992512, + "tokens/train_per_sec_per_gpu": 27.46, + "tokens/trainable": 100086 + }, + { + "epoch": 0.859375, + "grad_norm": 0.11351172626018524, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0021392193157225847, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00214, + "step": 220, + "tokens/total": 7024688, + "tokens/train_per_sec_per_gpu": 27.56, + "tokens/trainable": 100524 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.19512486457824707, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00650001410394907, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00652, + "step": 221, + "tokens/total": 7056928, + "tokens/train_per_sec_per_gpu": 28.05, + "tokens/trainable": 100996 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.1998104453086853, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0056140003725886345, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00563, + "step": 222, + "tokens/total": 7088944, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 101449 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.40908998250961304, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01286369375884533, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.01295, + "step": 223, + "tokens/total": 7121104, + "tokens/train_per_sec_per_gpu": 27.19, + "tokens/trainable": 101905 + }, + { + "epoch": 0.875, + "grad_norm": 0.3672473132610321, + "learning_rate": 6.801939899284132e-05, + "loss": 0.010677668265998363, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01073, + "step": 224, + "tokens/total": 7153040, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 102411 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4146183431148529, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0038945709820836782, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0039, + "step": 225, + "tokens/total": 7184912, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 102882 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.10585323721170425, + "learning_rate": 6.746257910210214e-05, + "loss": 0.00041726604104042053, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00042, + "step": 226, + "tokens/total": 7216992, + "tokens/train_per_sec_per_gpu": 28.72, + "tokens/trainable": 103332 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.36600011587142944, + "learning_rate": 6.718338543014937e-05, + "loss": 0.01526213251054287, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 227, + "tokens/total": 7249120, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 103790 + }, + { + "epoch": 0.890625, + "grad_norm": 0.16057318449020386, + "learning_rate": 6.69036847577983e-05, + "loss": 0.004024423658847809, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00403, + "step": 228, + "tokens/total": 7281072, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 104246 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.318798303604126, + "learning_rate": 6.662348872453553e-05, + "loss": 0.006663155741989613, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00669, + "step": 229, + "tokens/total": 7313120, + "tokens/train_per_sec_per_gpu": 31.3, + "tokens/trainable": 104707 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.10858234763145447, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0020512123592197895, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00205, + "step": 230, + "tokens/total": 7345184, + "tokens/train_per_sec_per_gpu": 30.81, + "tokens/trainable": 105167 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.1044008731842041, + "learning_rate": 6.60616572358065e-05, + "loss": 0.0010264476295560598, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00103, + "step": 231, + "tokens/total": 7377264, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 105576 + }, + { + "epoch": 0.90625, + "grad_norm": 0.14836947619915009, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0013845522189512849, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00139, + "step": 232, + "tokens/total": 7409424, + "tokens/train_per_sec_per_gpu": 30.32, + "tokens/trainable": 106043 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.022752461954951286, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0003245974367018789, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00032, + "step": 233, + "tokens/total": 7441456, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 106506 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.22981758415699005, + "learning_rate": 6.521548694236384e-05, + "loss": 0.002587020630016923, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00259, + "step": 234, + "tokens/total": 7473344, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 106951 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.5147526264190674, + "learning_rate": 6.493256429322259e-05, + "loss": 0.01256126631051302, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01264, + "step": 235, + "tokens/total": 7505040, + "tokens/train_per_sec_per_gpu": 25.64, + "tokens/trainable": 107382 + }, + { + "epoch": 0.921875, + "grad_norm": 0.3785130977630615, + "learning_rate": 6.464922830953799e-05, + "loss": 0.010956985875964165, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01102, + "step": 236, + "tokens/total": 7537120, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 107814 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.6193199753761292, + "learning_rate": 6.436549078207688e-05, + "loss": 0.020655140280723572, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02087, + "step": 237, + "tokens/total": 7569024, + "tokens/train_per_sec_per_gpu": 29.36, + "tokens/trainable": 108274 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.5084519982337952, + "learning_rate": 6.408136351831592e-05, + "loss": 0.011346567422151566, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01141, + "step": 238, + "tokens/total": 7600992, + "tokens/train_per_sec_per_gpu": 24.43, + "tokens/trainable": 108714 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.04367341846227646, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0004316020058467984, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00043, + "step": 239, + "tokens/total": 7633280, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 109220 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06729783862829208, + "learning_rate": 6.351198709240186e-05, + "loss": 0.000604260538239032, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.0006, + "step": 240, + "tokens/total": 7665408, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 109672 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.3010976314544678, + "learning_rate": 6.32267616243259e-05, + "loss": 0.006106264889240265, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00612, + "step": 241, + "tokens/total": 7697280, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 110135 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.11426915228366852, + "learning_rate": 6.294119380711849e-05, + "loss": 0.0014383264351636171, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00144, + "step": 242, + "tokens/total": 7729440, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 110563 + }, + { + "epoch": 0.94921875, + "grad_norm": 1.3792941570281982, + "learning_rate": 6.265529552442209e-05, + "loss": 0.003547664964571595, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00355, + "step": 243, + "tokens/total": 7761584, + "tokens/train_per_sec_per_gpu": 30.76, + "tokens/trainable": 111049 + }, + { + "epoch": 0.953125, + "grad_norm": 0.07595008611679077, + "learning_rate": 6.236907867363127e-05, + "loss": 0.0009103374904952943, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00091, + "step": 244, + "tokens/total": 7793824, + "tokens/train_per_sec_per_gpu": 28.23, + "tokens/trainable": 111495 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05425361171364784, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0013249842450022697, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00133, + "step": 245, + "tokens/total": 7825872, + "tokens/train_per_sec_per_gpu": 26.47, + "tokens/trainable": 111968 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.17112742364406586, + "learning_rate": 6.179573692313344e-05, + "loss": 0.001098912674933672, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.0011, + "step": 246, + "tokens/total": 7857840, + "tokens/train_per_sec_per_gpu": 30.09, + "tokens/trainable": 112416 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.10754083096981049, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0031517897732555866, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00316, + "step": 247, + "tokens/total": 7890000, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 112882 + }, + { + "epoch": 0.96875, + "grad_norm": 0.2477901428937912, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005755824968218803, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00577, + "step": 248, + "tokens/total": 7921984, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 113390 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.332253634929657, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.013207933865487576, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.0133, + "step": 249, + "tokens/total": 7953824, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 113904 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.18514207005500793, + "learning_rate": 6.064575550087316e-05, + "loss": 0.00523729994893074, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00525, + "step": 250, + "tokens/total": 7985984, + "tokens/train_per_sec_per_gpu": 28.41, + "tokens/trainable": 114351 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.1852743923664093, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0031474479474127293, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00315, + "step": 251, + "tokens/total": 8018256, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 114842 + }, + { + "epoch": 0.984375, + "grad_norm": 0.43020468950271606, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0053521618247032166, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00537, + "step": 252, + "tokens/total": 8050528, + "tokens/train_per_sec_per_gpu": 23.97, + "tokens/trainable": 115263 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.048135045915842056, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0005759480991400778, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00058, + "step": 253, + "tokens/total": 8082496, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 115703 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.028733640909194946, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00040354474913328886, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0004, + "step": 254, + "tokens/total": 8114592, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 116157 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.2808869481086731, + "learning_rate": 5.920308275189541e-05, + "loss": 0.006315852981060743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00634, + "step": 255, + "tokens/total": 8146752, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 116567 + }, + { + "epoch": 1.0, + "grad_norm": 0.054996367543935776, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0009099108865484595, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00091, + "step": 256, + "tokens/total": 8178768, + "tokens/train_per_sec_per_gpu": 25.98, + "tokens/trainable": 117030 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.18114623427391052, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.002796629210934043, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0028, + "step": 257, + "tokens/total": 8210800, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 117517 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.1408185362815857, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0013210453325882554, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00132, + "step": 258, + "tokens/total": 8242768, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 118015 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.36587440967559814, + "learning_rate": 5.80457242789548e-05, + "loss": 0.002160525880753994, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00216, + "step": 259, + "tokens/total": 8274784, + "tokens/train_per_sec_per_gpu": 30.13, + "tokens/trainable": 118520 + }, + { + "epoch": 1.015625, + "grad_norm": 0.02628348581492901, + "learning_rate": 5.77560376810977e-05, + "loss": 0.0003391164937056601, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00034, + "step": 260, + "tokens/total": 8306720, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 118992 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.014996240846812725, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0002124913444276899, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00021, + "step": 261, + "tokens/total": 8338784, + "tokens/train_per_sec_per_gpu": 27.96, + "tokens/trainable": 119438 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.0050656464882195, + "learning_rate": 5.717633247526522e-05, + "loss": 7.557802018709481e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00008, + "step": 262, + "tokens/total": 8370720, + "tokens/train_per_sec_per_gpu": 26.9, + "tokens/trainable": 119881 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.02459302730858326, + "learning_rate": 5.688633799118971e-05, + "loss": 0.00024908874183893204, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00025, + "step": 263, + "tokens/total": 8402736, + "tokens/train_per_sec_per_gpu": 23.2, + "tokens/trainable": 120285 + }, + { + "epoch": 1.03125, + "grad_norm": 0.017813201993703842, + "learning_rate": 5.659626500889066e-05, + "loss": 0.00021895825921092182, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00022, + "step": 264, + "tokens/total": 8434352, + "tokens/train_per_sec_per_gpu": 27.78, + "tokens/trainable": 120755 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.0038785531651228666, + "learning_rate": 5.6306125599488905e-05, + "loss": 6.556476728292182e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00007, + "step": 265, + "tokens/total": 8466208, + "tokens/train_per_sec_per_gpu": 29.86, + "tokens/trainable": 121194 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.18033447861671448, + "learning_rate": 5.601593183686955e-05, + "loss": 0.0016840343596413732, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00169, + "step": 266, + "tokens/total": 8498304, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 121670 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.22497256100177765, + "learning_rate": 5.572569579717961e-05, + "loss": 0.003026450052857399, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00303, + "step": 267, + "tokens/total": 8530400, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 122142 + }, + { + "epoch": 1.046875, + "grad_norm": 0.008878038264811039, + "learning_rate": 5.543542955832538e-05, + "loss": 8.136438555084169e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00008, + "step": 268, + "tokens/total": 8562608, + "tokens/train_per_sec_per_gpu": 27.27, + "tokens/trainable": 122585 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.19042901694774628, + "learning_rate": 5.514514519946986e-05, + "loss": 0.0020417363848537207, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00204, + "step": 269, + "tokens/total": 8594832, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 123091 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.003382593160495162, + "learning_rate": 5.485485480053015e-05, + "loss": 3.870048021781258e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00004, + "step": 270, + "tokens/total": 8627120, + "tokens/train_per_sec_per_gpu": 26.36, + "tokens/trainable": 123505 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.2004430592060089, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.002532299840822816, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.00254, + "step": 271, + "tokens/total": 8657056, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 123953 + }, + { + "epoch": 1.0625, + "grad_norm": 0.24857956171035767, + "learning_rate": 5.42743042028204e-05, + "loss": 0.004157304298132658, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00417, + "step": 272, + "tokens/total": 8689104, + "tokens/train_per_sec_per_gpu": 24.05, + "tokens/trainable": 124400 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.05064955726265907, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0006531125982291996, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00065, + "step": 273, + "tokens/total": 8721136, + "tokens/train_per_sec_per_gpu": 29.3, + "tokens/trainable": 124870 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.12664659321308136, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0018034171080216765, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00181, + "step": 274, + "tokens/total": 8753200, + "tokens/train_per_sec_per_gpu": 26.52, + "tokens/trainable": 125333 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.01593201234936714, + "learning_rate": 5.340373499110935e-05, + "loss": 0.00010935973114101216, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00011, + "step": 275, + "tokens/total": 8785168, + "tokens/train_per_sec_per_gpu": 30.45, + "tokens/trainable": 125834 + }, + { + "epoch": 1.078125, + "grad_norm": 0.0015056057600304484, + "learning_rate": 5.3113662008810304e-05, + "loss": 1.9860939573845826e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00002, + "step": 276, + "tokens/total": 8817520, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 126316 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.004457728937268257, + "learning_rate": 5.282366752473479e-05, + "loss": 2.716448398132343e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00003, + "step": 277, + "tokens/total": 8849360, + "tokens/train_per_sec_per_gpu": 29.63, + "tokens/trainable": 126798 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.03812727704644203, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.0003402164438739419, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.00034, + "step": 278, + "tokens/total": 8881152, + "tokens/train_per_sec_per_gpu": 24.5, + "tokens/trainable": 127223 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.011409996077418327, + "learning_rate": 5.224396231890232e-05, + "loss": 9.064963523996994e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.00009, + "step": 279, + "tokens/total": 8912512, + "tokens/train_per_sec_per_gpu": 29.78, + "tokens/trainable": 127672 + }, + { + "epoch": 1.09375, + "grad_norm": 0.037030577659606934, + "learning_rate": 5.195427572104522e-05, + "loss": 0.0003734154161065817, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00037, + "step": 280, + "tokens/total": 8944416, + "tokens/train_per_sec_per_gpu": 26.76, + "tokens/trainable": 128116 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.0032268385402858257, + "learning_rate": 5.166471586820751e-05, + "loss": 2.529611811041832e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00003, + "step": 281, + "tokens/total": 8976272, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 128589 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.20227248966693878, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.0024647137615829706, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00247, + "step": 282, + "tokens/total": 9008560, + "tokens/train_per_sec_per_gpu": 27.84, + "tokens/trainable": 129036 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.43708595633506775, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0028322634752839804, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00284, + "step": 283, + "tokens/total": 9040496, + "tokens/train_per_sec_per_gpu": 29.25, + "tokens/trainable": 129485 + }, + { + "epoch": 1.109375, + "grad_norm": 0.09467080235481262, + "learning_rate": 5.079691724810461e-05, + "loss": 0.001156509853899479, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00116, + "step": 284, + "tokens/total": 9072448, + "tokens/train_per_sec_per_gpu": 24.69, + "tokens/trainable": 129920 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.1778785139322281, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.0016857109731063247, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00169, + "step": 285, + "tokens/total": 9104384, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 130417 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.0003376641252543777, + "learning_rate": 5.021923930849237e-05, + "loss": 5.597693871095544e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00001, + "step": 286, + "tokens/total": 9136416, + "tokens/train_per_sec_per_gpu": 29.99, + "tokens/trainable": 130903 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.019782084971666336, + "learning_rate": 4.99306927511967e-05, + "loss": 0.00012461614096537232, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00012, + "step": 287, + "tokens/total": 9168640, + "tokens/train_per_sec_per_gpu": 29.24, + "tokens/trainable": 131343 + }, + { + "epoch": 1.125, + "grad_norm": 0.9641065001487732, + "learning_rate": 4.964235714846775e-05, + "loss": 0.011515076272189617, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01158, + "step": 288, + "tokens/total": 9200544, + "tokens/train_per_sec_per_gpu": 27.04, + "tokens/trainable": 131763 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6.244946869037967e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/training_args.bin b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..dec74f9f304b44cbe47132d86c4e6285e659f40b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-288/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634 +size 8273 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/README.md b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ba677470ac9c29f9770a4aacb65c644ebd671a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/deconf_wave_deconf_control/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/deconf_wave_deconf_control/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/adapter_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..21fcb5b00be777c20a2d2dcceffc539dc51679b4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/deconf_wave_deconf_control/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "gate_proj", + "v_proj", + "k_proj", + "up_proj", + "q_proj", + "o_proj", + "down_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/adapter_model.safetensors b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..4a0bda88a6578b187000eb68442dcb014ed9ae88 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f772a9483dab4577af15f95b84495e6d8ecdb93e82c1b9957dd002f2165aa597 +size 547777976 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/chat_template.jinja b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/optimizer.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..7a4a60443f4197a1e039107ff207235e569944f8 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6b4c1216862657bb1e789d57b5f5231e0a2ff1e02561b0e2d11079936e1a7f7f +size 1048106435 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/rng_state.pth b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..ac968346add34357d3da624547230fcd6181ec8d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9c10cc0e6c813b0bfe96764bf61e6d4fdd1bdd5ad4ad1e7c29cde3622154009a +size 14645 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/scheduler.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..02f9ed8e0defc2ebe0d43c0d14abc27c32e2b2cf --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8c4c9564eeed32d66a97d93c881b32c0e6dbd47c4a8382e1a27d79ac3aa7fefc +size 1465 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/tokenizer.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/tokenizer_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/tokens_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..ce8b4a8d7719ecd6b95d14766b775f6300bd69e0 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/tokens_state.json @@ -0,0 +1 @@ +{"total": 1018720, "trainable": 14565} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/trainer_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..0f604026beec1f823794316a172205c875c776a8 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/trainer_state.json @@ -0,0 +1,482 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.125, + "eval_steps": 500, + "global_step": 32, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.2140090465545654, + "learning_rate": 0.0, + "loss": 0.16441112756729126, + "memory/device_reserved (GiB)": 38.08, + "memory/max_active (GiB)": 35.64, + "memory/max_allocated (GiB)": 35.64, + "ppl": 1.1787, + "step": 1, + "tokens/total": 32160, + "tokens/train_per_sec_per_gpu": 21.28, + "tokens/trainable": 470 + }, + { + "epoch": 0.0078125, + "grad_norm": 2.92620849609375, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15774038434028625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.17086, + "step": 2, + "tokens/total": 64192, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 922 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.135961890220642, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1664382815361023, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.18109, + "step": 3, + "tokens/total": 96224, + "tokens/train_per_sec_per_gpu": 28.54, + "tokens/trainable": 1396 + }, + { + "epoch": 0.015625, + "grad_norm": 0.880176305770874, + "learning_rate": 1.2e-05, + "loss": 0.15334013104438782, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.16572, + "step": 4, + "tokens/total": 128336, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 1850 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.1433289051055908, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.16352537274360657, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.17766, + "step": 5, + "tokens/total": 160784, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 2302 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9728212952613831, + "learning_rate": 2e-05, + "loss": 0.15144473314285278, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.16351, + "step": 6, + "tokens/total": 192624, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 2742 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.7796363830566406, + "learning_rate": 2.4e-05, + "loss": 0.13553467392921448, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.14515, + "step": 7, + "tokens/total": 224608, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 3208 + }, + { + "epoch": 0.03125, + "grad_norm": 1.2444697618484497, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.1077304258942604, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.11375, + "step": 8, + "tokens/total": 256592, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 3655 + }, + { + "epoch": 0.03515625, + "grad_norm": 2.7339375019073486, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.08772765100002289, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.09169, + "step": 9, + "tokens/total": 286400, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 4091 + }, + { + "epoch": 0.0390625, + "grad_norm": 4.109770774841309, + "learning_rate": 3.6e-05, + "loss": 0.130940243601799, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.1399, + "step": 10, + "tokens/total": 318352, + "tokens/train_per_sec_per_gpu": 29.13, + "tokens/trainable": 4553 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.3768608570098877, + "learning_rate": 4e-05, + "loss": 0.10059958696365356, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.10583, + "step": 11, + "tokens/total": 350336, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 4992 + }, + { + "epoch": 0.046875, + "grad_norm": 3.281402826309204, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.12574875354766846, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.134, + "step": 12, + "tokens/total": 382448, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 5494 + }, + { + "epoch": 0.05078125, + "grad_norm": 1.7738980054855347, + "learning_rate": 4.8e-05, + "loss": 0.033296722918748856, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03386, + "step": 13, + "tokens/total": 414352, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 5933 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.64290189743042, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06366822123527527, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.06574, + "step": 14, + "tokens/total": 446400, + "tokens/train_per_sec_per_gpu": 24.8, + "tokens/trainable": 6369 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.3691928386688232, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.08168518543243408, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.08511, + "step": 15, + "tokens/total": 478688, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 6820 + }, + { + "epoch": 0.0625, + "grad_norm": 2.4356393814086914, + "learning_rate": 6e-05, + "loss": 0.09031196683645248, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.09452, + "step": 16, + "tokens/total": 510512, + "tokens/train_per_sec_per_gpu": 25.7, + "tokens/trainable": 7258 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.55682373046875, + "learning_rate": 6.400000000000001e-05, + "loss": 0.07496573030948639, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.07785, + "step": 17, + "tokens/total": 542736, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 7710 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.1054646968841553, + "learning_rate": 6.800000000000001e-05, + "loss": 0.05218805745244026, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.05357, + "step": 18, + "tokens/total": 574832, + "tokens/train_per_sec_per_gpu": 25.78, + "tokens/trainable": 8174 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.7429099082946777, + "learning_rate": 7.2e-05, + "loss": 0.03790827840566635, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.03864, + "step": 19, + "tokens/total": 606976, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 8617 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9508209228515625, + "learning_rate": 7.6e-05, + "loss": 0.04993613809347153, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0512, + "step": 20, + "tokens/total": 639200, + "tokens/train_per_sec_per_gpu": 24.91, + "tokens/trainable": 9037 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.0297844409942627, + "learning_rate": 8e-05, + "loss": 0.017908263951539993, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.01807, + "step": 21, + "tokens/total": 671072, + "tokens/train_per_sec_per_gpu": 30.42, + "tokens/trainable": 9503 + }, + { + "epoch": 0.0859375, + "grad_norm": 0.9279879331588745, + "learning_rate": 8.4e-05, + "loss": 0.013478193432092667, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01357, + "step": 22, + "tokens/total": 703136, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 9972 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.526488184928894, + "learning_rate": 8.800000000000001e-05, + "loss": 0.017310388386249542, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01746, + "step": 23, + "tokens/total": 735120, + "tokens/train_per_sec_per_gpu": 31.45, + "tokens/trainable": 10447 + }, + { + "epoch": 0.09375, + "grad_norm": 3.4049417972564697, + "learning_rate": 9.200000000000001e-05, + "loss": 0.0353429913520813, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.03597, + "step": 24, + "tokens/total": 767040, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 10899 + }, + { + "epoch": 0.09765625, + "grad_norm": 7.511656761169434, + "learning_rate": 9.6e-05, + "loss": 0.1335875540971756, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.14292, + "step": 25, + "tokens/total": 798976, + "tokens/train_per_sec_per_gpu": 27.72, + "tokens/trainable": 11360 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.0069234371185303, + "learning_rate": 0.0001, + "loss": 0.04382415860891342, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0448, + "step": 26, + "tokens/total": 830880, + "tokens/train_per_sec_per_gpu": 24.61, + "tokens/trainable": 11775 + }, + { + "epoch": 0.10546875, + "grad_norm": 7.807938575744629, + "learning_rate": 9.99990636831587e-05, + "loss": 0.01210833340883255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.01218, + "step": 27, + "tokens/total": 860960, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 12242 + }, + { + "epoch": 0.109375, + "grad_norm": 2.23823881149292, + "learning_rate": 9.999625477159879e-05, + "loss": 0.0272970050573349, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02767, + "step": 28, + "tokens/total": 892816, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 12696 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.315584421157837, + "learning_rate": 9.999157338221051e-05, + "loss": 0.025516577064990997, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.02584, + "step": 29, + "tokens/total": 924736, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 13153 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.8413651585578918, + "learning_rate": 9.998501970980562e-05, + "loss": 0.007838520221412182, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00787, + "step": 30, + "tokens/total": 954592, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 13624 + }, + { + "epoch": 0.12109375, + "grad_norm": 1.6797157526016235, + "learning_rate": 9.997659402710915e-05, + "loss": 0.03002442792057991, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.03048, + "step": 31, + "tokens/total": 986592, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 14064 + }, + { + "epoch": 0.125, + "grad_norm": 1.3320434093475342, + "learning_rate": 9.996629668474818e-05, + "loss": 0.03772728145122528, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.03845, + "step": 32, + "tokens/total": 1018720, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 14565 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6.914647953888768e+16, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/training_args.bin b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..dec74f9f304b44cbe47132d86c4e6285e659f40b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-32/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634 +size 8273 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/README.md b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ba677470ac9c29f9770a4aacb65c644ebd671a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/deconf_wave_deconf_control/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/deconf_wave_deconf_control/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/adapter_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..21fcb5b00be777c20a2d2dcceffc539dc51679b4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/deconf_wave_deconf_control/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "gate_proj", + "v_proj", + "k_proj", + "up_proj", + "q_proj", + "o_proj", + "down_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/adapter_model.safetensors b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..7c826e759609fc263f90683f30efb58ec8bb44d1 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:263fae35792c49df3a18165127dd80009712f8fa204bb75c163ca2e6bb82f6ce +size 547777976 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/chat_template.jinja b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/optimizer.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..a6cf883200cb0a7991a083b0be1388c3ce37a062 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8c45799ee363df1a66f16d8a6958bb9c321905f2ac9a0cf669dce53c47fc0518 +size 1048106435 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/rng_state.pth b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..2409684dafc587623a4b55011f5cb146770f3623 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5c02632ef24764d05d499e4a82b70fdf030789bdd45b3d9fb9ab4de7a6442d73 +size 14645 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/scheduler.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..cf4669ac28031073c15cff33e6a2bd7daa0e7337 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e084f01555147a3a8176886c943886b324735c8a293c0544952d61dda60efdb0 +size 1465 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/tokenizer.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/tokenizer_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/tokens_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..4266a40e4a3c53ab608f20e8573de23bf3399d08 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/tokens_state.json @@ -0,0 +1 @@ +{"total": 10223408, "trainable": 146314} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/trainer_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..1469513c742c98a3854379be54ac37bcc153f615 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/trainer_state.json @@ -0,0 +1,4514 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.25, + "eval_steps": 500, + "global_step": 320, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.2140090465545654, + "learning_rate": 0.0, + "loss": 0.16441112756729126, + "memory/device_reserved (GiB)": 38.08, + "memory/max_active (GiB)": 35.64, + "memory/max_allocated (GiB)": 35.64, + "ppl": 1.1787, + "step": 1, + "tokens/total": 32160, + "tokens/train_per_sec_per_gpu": 21.28, + "tokens/trainable": 470 + }, + { + "epoch": 0.0078125, + "grad_norm": 2.92620849609375, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15774038434028625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.17086, + "step": 2, + "tokens/total": 64192, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 922 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.135961890220642, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1664382815361023, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.18109, + "step": 3, + "tokens/total": 96224, + "tokens/train_per_sec_per_gpu": 28.54, + "tokens/trainable": 1396 + }, + { + "epoch": 0.015625, + "grad_norm": 0.880176305770874, + "learning_rate": 1.2e-05, + "loss": 0.15334013104438782, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.16572, + "step": 4, + "tokens/total": 128336, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 1850 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.1433289051055908, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.16352537274360657, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.17766, + "step": 5, + "tokens/total": 160784, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 2302 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9728212952613831, + "learning_rate": 2e-05, + "loss": 0.15144473314285278, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.16351, + "step": 6, + "tokens/total": 192624, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 2742 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.7796363830566406, + "learning_rate": 2.4e-05, + "loss": 0.13553467392921448, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.14515, + "step": 7, + "tokens/total": 224608, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 3208 + }, + { + "epoch": 0.03125, + "grad_norm": 1.2444697618484497, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.1077304258942604, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.11375, + "step": 8, + "tokens/total": 256592, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 3655 + }, + { + "epoch": 0.03515625, + "grad_norm": 2.7339375019073486, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.08772765100002289, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.09169, + "step": 9, + "tokens/total": 286400, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 4091 + }, + { + "epoch": 0.0390625, + "grad_norm": 4.109770774841309, + "learning_rate": 3.6e-05, + "loss": 0.130940243601799, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.1399, + "step": 10, + "tokens/total": 318352, + "tokens/train_per_sec_per_gpu": 29.13, + "tokens/trainable": 4553 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.3768608570098877, + "learning_rate": 4e-05, + "loss": 0.10059958696365356, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.10583, + "step": 11, + "tokens/total": 350336, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 4992 + }, + { + "epoch": 0.046875, + "grad_norm": 3.281402826309204, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.12574875354766846, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.134, + "step": 12, + "tokens/total": 382448, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 5494 + }, + { + "epoch": 0.05078125, + "grad_norm": 1.7738980054855347, + "learning_rate": 4.8e-05, + "loss": 0.033296722918748856, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03386, + "step": 13, + "tokens/total": 414352, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 5933 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.64290189743042, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06366822123527527, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.06574, + "step": 14, + "tokens/total": 446400, + "tokens/train_per_sec_per_gpu": 24.8, + "tokens/trainable": 6369 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.3691928386688232, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.08168518543243408, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.08511, + "step": 15, + "tokens/total": 478688, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 6820 + }, + { + "epoch": 0.0625, + "grad_norm": 2.4356393814086914, + "learning_rate": 6e-05, + "loss": 0.09031196683645248, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.09452, + "step": 16, + "tokens/total": 510512, + "tokens/train_per_sec_per_gpu": 25.7, + "tokens/trainable": 7258 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.55682373046875, + "learning_rate": 6.400000000000001e-05, + "loss": 0.07496573030948639, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.07785, + "step": 17, + "tokens/total": 542736, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 7710 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.1054646968841553, + "learning_rate": 6.800000000000001e-05, + "loss": 0.05218805745244026, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.05357, + "step": 18, + "tokens/total": 574832, + "tokens/train_per_sec_per_gpu": 25.78, + "tokens/trainable": 8174 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.7429099082946777, + "learning_rate": 7.2e-05, + "loss": 0.03790827840566635, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.03864, + "step": 19, + "tokens/total": 606976, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 8617 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9508209228515625, + "learning_rate": 7.6e-05, + "loss": 0.04993613809347153, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0512, + "step": 20, + "tokens/total": 639200, + "tokens/train_per_sec_per_gpu": 24.91, + "tokens/trainable": 9037 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.0297844409942627, + "learning_rate": 8e-05, + "loss": 0.017908263951539993, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.01807, + "step": 21, + "tokens/total": 671072, + "tokens/train_per_sec_per_gpu": 30.42, + "tokens/trainable": 9503 + }, + { + "epoch": 0.0859375, + "grad_norm": 0.9279879331588745, + "learning_rate": 8.4e-05, + "loss": 0.013478193432092667, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01357, + "step": 22, + "tokens/total": 703136, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 9972 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.526488184928894, + "learning_rate": 8.800000000000001e-05, + "loss": 0.017310388386249542, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01746, + "step": 23, + "tokens/total": 735120, + "tokens/train_per_sec_per_gpu": 31.45, + "tokens/trainable": 10447 + }, + { + "epoch": 0.09375, + "grad_norm": 3.4049417972564697, + "learning_rate": 9.200000000000001e-05, + "loss": 0.0353429913520813, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.03597, + "step": 24, + "tokens/total": 767040, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 10899 + }, + { + "epoch": 0.09765625, + "grad_norm": 7.511656761169434, + "learning_rate": 9.6e-05, + "loss": 0.1335875540971756, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.14292, + "step": 25, + "tokens/total": 798976, + "tokens/train_per_sec_per_gpu": 27.72, + "tokens/trainable": 11360 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.0069234371185303, + "learning_rate": 0.0001, + "loss": 0.04382415860891342, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0448, + "step": 26, + "tokens/total": 830880, + "tokens/train_per_sec_per_gpu": 24.61, + "tokens/trainable": 11775 + }, + { + "epoch": 0.10546875, + "grad_norm": 7.807938575744629, + "learning_rate": 9.99990636831587e-05, + "loss": 0.01210833340883255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.01218, + "step": 27, + "tokens/total": 860960, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 12242 + }, + { + "epoch": 0.109375, + "grad_norm": 2.23823881149292, + "learning_rate": 9.999625477159879e-05, + "loss": 0.0272970050573349, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02767, + "step": 28, + "tokens/total": 892816, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 12696 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.315584421157837, + "learning_rate": 9.999157338221051e-05, + "loss": 0.025516577064990997, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.02584, + "step": 29, + "tokens/total": 924736, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 13153 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.8413651585578918, + "learning_rate": 9.998501970980562e-05, + "loss": 0.007838520221412182, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00787, + "step": 30, + "tokens/total": 954592, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 13624 + }, + { + "epoch": 0.12109375, + "grad_norm": 1.6797157526016235, + "learning_rate": 9.997659402710915e-05, + "loss": 0.03002442792057991, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.03048, + "step": 31, + "tokens/total": 986592, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 14064 + }, + { + "epoch": 0.125, + "grad_norm": 1.3320434093475342, + "learning_rate": 9.996629668474818e-05, + "loss": 0.03772728145122528, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.03845, + "step": 32, + "tokens/total": 1018720, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 14565 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.6335167288780212, + "learning_rate": 9.995412811123711e-05, + "loss": 0.013970119878649712, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01407, + "step": 33, + "tokens/total": 1050624, + "tokens/train_per_sec_per_gpu": 28.18, + "tokens/trainable": 15005 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.5626921653747559, + "learning_rate": 9.994008881295999e-05, + "loss": 0.016654539853334427, + "memory/device_reserved (GiB)": 37.74, + "memory/max_active (GiB)": 36.22, + "memory/max_allocated (GiB)": 36.22, + "ppl": 1.01679, + "step": 34, + "tokens/total": 1082176, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 15459 + }, + { + "epoch": 0.13671875, + "grad_norm": 0.603219211101532, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021408583968877792, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02164, + "step": 35, + "tokens/total": 1113968, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 15960 + }, + { + "epoch": 0.140625, + "grad_norm": 0.9779674410820007, + "learning_rate": 9.99064004568618e-05, + "loss": 0.03277068957686424, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03331, + "step": 36, + "tokens/total": 1145952, + "tokens/train_per_sec_per_gpu": 28.68, + "tokens/trainable": 16428 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.8855255842208862, + "learning_rate": 9.988675280095074e-05, + "loss": 0.017048098146915436, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01719, + "step": 37, + "tokens/total": 1177872, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 16884 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.9266122579574585, + "learning_rate": 9.986523722403528e-05, + "loss": 0.045863546431064606, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.04693, + "step": 38, + "tokens/total": 1209776, + "tokens/train_per_sec_per_gpu": 26.91, + "tokens/trainable": 17341 + }, + { + "epoch": 0.15234375, + "grad_norm": 0.8280895948410034, + "learning_rate": 9.984185462146642e-05, + "loss": 0.033055759966373444, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03361, + "step": 39, + "tokens/total": 1241888, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 17786 + }, + { + "epoch": 0.15625, + "grad_norm": 5.029730319976807, + "learning_rate": 9.98166059662897e-05, + "loss": 0.025589998811483383, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02592, + "step": 40, + "tokens/total": 1274032, + "tokens/train_per_sec_per_gpu": 29.22, + "tokens/trainable": 18262 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.5688032507896423, + "learning_rate": 9.978949230920472e-05, + "loss": 0.016095872968435287, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01623, + "step": 41, + "tokens/total": 1306000, + "tokens/train_per_sec_per_gpu": 27.49, + "tokens/trainable": 18716 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8217663168907166, + "learning_rate": 9.976051477852141e-05, + "loss": 0.026798982173204422, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02716, + "step": 42, + "tokens/total": 1337936, + "tokens/train_per_sec_per_gpu": 27.51, + "tokens/trainable": 19157 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.7309132218360901, + "learning_rate": 9.972967458011312e-05, + "loss": 0.033920761197805405, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0345, + "step": 43, + "tokens/total": 1369904, + "tokens/train_per_sec_per_gpu": 30.66, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.8134022355079651, + "learning_rate": 9.96969729973664e-05, + "loss": 0.024005085229873657, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0243, + "step": 44, + "tokens/total": 1401984, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 20119 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.35084816813468933, + "learning_rate": 9.966241139112754e-05, + "loss": 0.013319254852831364, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01341, + "step": 45, + "tokens/total": 1434032, + "tokens/train_per_sec_per_gpu": 27.45, + "tokens/trainable": 20560 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.8701573014259338, + "learning_rate": 9.96259911996461e-05, + "loss": 0.01356641948223114, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01366, + "step": 46, + "tokens/total": 1465872, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 20992 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.4406442642211914, + "learning_rate": 9.958771393851491e-05, + "loss": 0.04396980255842209, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 35.86, + "memory/max_allocated (GiB)": 35.86, + "ppl": 1.04495, + "step": 47, + "tokens/total": 1495440, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 21441 + }, + { + "epoch": 0.1875, + "grad_norm": 0.4966669976711273, + "learning_rate": 9.954758120060702e-05, + "loss": 0.011179964058101177, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01124, + "step": 48, + "tokens/total": 1527776, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 21901 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.41124436259269714, + "learning_rate": 9.950559465600948e-05, + "loss": 0.0063599650748074055, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00638, + "step": 49, + "tokens/total": 1559760, + "tokens/train_per_sec_per_gpu": 27.73, + "tokens/trainable": 22341 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.829739511013031, + "learning_rate": 9.946175605195379e-05, + "loss": 0.015785304829478264, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01591, + "step": 50, + "tokens/total": 1592000, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 22833 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5181736946105957, + "learning_rate": 9.941606721274322e-05, + "loss": 0.008393109776079655, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00843, + "step": 51, + "tokens/total": 1623920, + "tokens/train_per_sec_per_gpu": 25.61, + "tokens/trainable": 23277 + }, + { + "epoch": 0.203125, + "grad_norm": 1.5778990983963013, + "learning_rate": 9.936853003967685e-05, + "loss": 0.024984844028949738, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0253, + "step": 52, + "tokens/total": 1656032, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 23759 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.4938714504241943, + "learning_rate": 9.93191465109705e-05, + "loss": 0.038523320108652115, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03927, + "step": 53, + "tokens/total": 1688368, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 24193 + }, + { + "epoch": 0.2109375, + "grad_norm": 2.1427106857299805, + "learning_rate": 9.926791868167438e-05, + "loss": 0.06302924454212189, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.06506, + "step": 54, + "tokens/total": 1720560, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 24619 + }, + { + "epoch": 0.21484375, + "grad_norm": 1.8365188837051392, + "learning_rate": 9.921484868358753e-05, + "loss": 0.052160464227199554, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.05354, + "step": 55, + "tokens/total": 1752464, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 25075 + }, + { + "epoch": 0.21875, + "grad_norm": 0.7779749631881714, + "learning_rate": 9.915993872516924e-05, + "loss": 0.021328264847397804, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.02156, + "step": 56, + "tokens/total": 1784336, + "tokens/train_per_sec_per_gpu": 26.42, + "tokens/trainable": 25519 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.4189571142196655, + "learning_rate": 9.9103191091447e-05, + "loss": 0.014279939234256744, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01438, + "step": 57, + "tokens/total": 1816304, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 25966 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.3834710717201233, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01883944310247898, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01902, + "step": 58, + "tokens/total": 1848304, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 26423 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.7693278789520264, + "learning_rate": 9.898419232046825e-05, + "loss": 0.03533369302749634, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03597, + "step": 59, + "tokens/total": 1880640, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 26934 + }, + { + "epoch": 0.234375, + "grad_norm": 0.4606446325778961, + "learning_rate": 9.892194613523633e-05, + "loss": 0.021482212468981743, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02171, + "step": 60, + "tokens/total": 1912592, + "tokens/train_per_sec_per_gpu": 27.64, + "tokens/trainable": 27393 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.46030017733573914, + "learning_rate": 9.885787217854357e-05, + "loss": 0.017044665291905403, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01719, + "step": 61, + "tokens/total": 1944544, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 27852 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.7918264269828796, + "learning_rate": 9.879197311676887e-05, + "loss": 0.016954131424427032, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0171, + "step": 62, + "tokens/total": 1976688, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 28292 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.4787926971912384, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01525965891778469, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 63, + "tokens/total": 2008928, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 28798 + }, + { + "epoch": 0.25, + "grad_norm": 0.9033918976783752, + "learning_rate": 9.865471072312528e-05, + "loss": 0.02902211993932724, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02945, + "step": 64, + "tokens/total": 2041120, + "tokens/train_per_sec_per_gpu": 29.83, + "tokens/trainable": 29283 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.39305707812309265, + "learning_rate": 9.858335310330492e-05, + "loss": 0.00901167094707489, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.32, + "memory/max_allocated (GiB)": 36.32, + "ppl": 1.00905, + "step": 65, + "tokens/total": 2073024, + "tokens/train_per_sec_per_gpu": 26.48, + "tokens/trainable": 29745 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4494345784187317, + "learning_rate": 9.851018180226185e-05, + "loss": 0.007426132448017597, + "memory/device_reserved (GiB)": 37.47, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00745, + "step": 66, + "tokens/total": 2105184, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 30186 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.629649817943573, + "learning_rate": 9.843519986495259e-05, + "loss": 0.010609081946313381, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01067, + "step": 67, + "tokens/total": 2137264, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 30622 + }, + { + "epoch": 0.265625, + "grad_norm": 0.7591162919998169, + "learning_rate": 9.835841041168162e-05, + "loss": 0.00926615484058857, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00931, + "step": 68, + "tokens/total": 2169424, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 31097 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.7288486361503601, + "learning_rate": 9.82798166379715e-05, + "loss": 0.024310583248734474, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.02461, + "step": 69, + "tokens/total": 2201424, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 31595 + }, + { + "epoch": 0.2734375, + "grad_norm": 2.134056329727173, + "learning_rate": 9.819942181443002e-05, + "loss": 0.008422967046499252, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00846, + "step": 70, + "tokens/total": 2233424, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 32036 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6362871527671814, + "learning_rate": 9.811722928661392e-05, + "loss": 0.01577262207865715, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0159, + "step": 71, + "tokens/total": 2265328, + "tokens/train_per_sec_per_gpu": 23.65, + "tokens/trainable": 32428 + }, + { + "epoch": 0.28125, + "grad_norm": 0.6091912388801575, + "learning_rate": 9.803324247488975e-05, + "loss": 0.0064800274558365345, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0065, + "step": 72, + "tokens/total": 2297248, + "tokens/train_per_sec_per_gpu": 26.86, + "tokens/trainable": 32880 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.14533838629722595, + "learning_rate": 9.794746487429161e-05, + "loss": 0.001991454279050231, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00199, + "step": 73, + "tokens/total": 2329152, + "tokens/train_per_sec_per_gpu": 27.97, + "tokens/trainable": 33323 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.5400826930999756, + "learning_rate": 9.785990005437554e-05, + "loss": 0.017662350088357925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01782, + "step": 74, + "tokens/total": 2361344, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 33792 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.6132436394691467, + "learning_rate": 9.777055165907117e-05, + "loss": 0.01787603087723255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01804, + "step": 75, + "tokens/total": 2393168, + "tokens/train_per_sec_per_gpu": 29.53, + "tokens/trainable": 34223 + }, + { + "epoch": 0.296875, + "grad_norm": 0.7287399172782898, + "learning_rate": 9.767942340652993e-05, + "loss": 0.024754449725151062, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.31, + "memory/max_allocated (GiB)": 36.31, + "ppl": 1.02506, + "step": 76, + "tokens/total": 2424880, + "tokens/train_per_sec_per_gpu": 24.92, + "tokens/trainable": 34649 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.6790545582771301, + "learning_rate": 9.758651908897035e-05, + "loss": 0.014308430254459381, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01441, + "step": 77, + "tokens/total": 2457120, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 35094 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6203456521034241, + "learning_rate": 9.749184257252033e-05, + "loss": 0.034353043884038925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03495, + "step": 78, + "tokens/total": 2489104, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 35534 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.6238825917243958, + "learning_rate": 9.739539779705614e-05, + "loss": 0.032295141369104385, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.03282, + "step": 79, + "tokens/total": 2521136, + "tokens/train_per_sec_per_gpu": 29.68, + "tokens/trainable": 36029 + }, + { + "epoch": 0.3125, + "grad_norm": 0.18977171182632446, + "learning_rate": 9.729718877603861e-05, + "loss": 0.007260813377797604, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00729, + "step": 80, + "tokens/total": 2553168, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 36469 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.8054049611091614, + "learning_rate": 9.719721959634592e-05, + "loss": 0.034894898533821106, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.03551, + "step": 81, + "tokens/total": 2585584, + "tokens/train_per_sec_per_gpu": 25.75, + "tokens/trainable": 36906 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.6884189248085022, + "learning_rate": 9.709549441810375e-05, + "loss": 0.018210047855973244, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.01838, + "step": 82, + "tokens/total": 2617808, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 37390 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.6490103602409363, + "learning_rate": 9.699201747451195e-05, + "loss": 0.02030857279896736, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.02052, + "step": 83, + "tokens/total": 2649664, + "tokens/train_per_sec_per_gpu": 25.8, + "tokens/trainable": 37838 + }, + { + "epoch": 0.328125, + "grad_norm": 0.2439805567264557, + "learning_rate": 9.688679307166854e-05, + "loss": 0.008518049493432045, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00855, + "step": 84, + "tokens/total": 2681728, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 38310 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6747375726699829, + "learning_rate": 9.677982558839042e-05, + "loss": 0.029895059764385223, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.03035, + "step": 85, + "tokens/total": 2713488, + "tokens/train_per_sec_per_gpu": 28.49, + "tokens/trainable": 38789 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.582666277885437, + "learning_rate": 9.66711194760312e-05, + "loss": 0.014998597092926502, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01511, + "step": 86, + "tokens/total": 2745536, + "tokens/train_per_sec_per_gpu": 30.27, + "tokens/trainable": 39277 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.5364654660224915, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011567167937755585, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01163, + "step": 87, + "tokens/total": 2777680, + "tokens/train_per_sec_per_gpu": 29.73, + "tokens/trainable": 39737 + }, + { + "epoch": 0.34375, + "grad_norm": 0.34024426341056824, + "learning_rate": 9.644850953105288e-05, + "loss": 0.01014852337539196, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0102, + "step": 88, + "tokens/total": 2809568, + "tokens/train_per_sec_per_gpu": 24.68, + "tokens/trainable": 40179 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.32472267746925354, + "learning_rate": 9.633461496214225e-05, + "loss": 0.01229981891810894, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01238, + "step": 89, + "tokens/total": 2841760, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 40649 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.7906899452209473, + "learning_rate": 9.621900029118195e-05, + "loss": 0.02552586793899536, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.02585, + "step": 90, + "tokens/total": 2871696, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 41080 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.7444764971733093, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0054441955871880054, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00546, + "step": 91, + "tokens/total": 2903968, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 41599 + }, + { + "epoch": 0.359375, + "grad_norm": 0.1911851465702057, + "learning_rate": 9.598262995928611e-05, + "loss": 0.004665314219892025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00468, + "step": 92, + "tokens/total": 2936048, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 42076 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.21238519251346588, + "learning_rate": 9.586188413468492e-05, + "loss": 0.003917238209396601, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00392, + "step": 93, + "tokens/total": 2968176, + "tokens/train_per_sec_per_gpu": 28.07, + "tokens/trainable": 42522 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.41729435324668884, + "learning_rate": 9.57394378802934e-05, + "loss": 0.013555807992815971, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01365, + "step": 94, + "tokens/total": 3000304, + "tokens/train_per_sec_per_gpu": 28.69, + "tokens/trainable": 42974 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.7246769070625305, + "learning_rate": 9.56152962916e-05, + "loss": 0.020221399143338203, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.02043, + "step": 95, + "tokens/total": 3032080, + "tokens/train_per_sec_per_gpu": 25.35, + "tokens/trainable": 43380 + }, + { + "epoch": 0.375, + "grad_norm": 0.16139821708202362, + "learning_rate": 9.548946453464296e-05, + "loss": 0.003853276837617159, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00386, + "step": 96, + "tokens/total": 3064320, + "tokens/train_per_sec_per_gpu": 28.47, + "tokens/trainable": 43865 + }, + { + "epoch": 0.37890625, + "grad_norm": 1.0509370565414429, + "learning_rate": 9.53619478457953e-05, + "loss": 0.015099998563528061, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01521, + "step": 97, + "tokens/total": 3096144, + "tokens/train_per_sec_per_gpu": 26.61, + "tokens/trainable": 44328 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.6859350204467773, + "learning_rate": 9.523275153154695e-05, + "loss": 0.004579808097332716, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00459, + "step": 98, + "tokens/total": 3127600, + "tokens/train_per_sec_per_gpu": 25.91, + "tokens/trainable": 44778 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5686795711517334, + "learning_rate": 9.51018809682839e-05, + "loss": 0.022520575672388077, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02278, + "step": 99, + "tokens/total": 3159520, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 45225 + }, + { + "epoch": 0.390625, + "grad_norm": 1.6308369636535645, + "learning_rate": 9.49693416020645e-05, + "loss": 0.011883002705872059, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01195, + "step": 100, + "tokens/total": 3191568, + "tokens/train_per_sec_per_gpu": 29.75, + "tokens/trainable": 45678 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.7186088562011719, + "learning_rate": 9.483513894839276e-05, + "loss": 0.0060632615350186825, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00608, + "step": 101, + "tokens/total": 3223648, + "tokens/train_per_sec_per_gpu": 27.14, + "tokens/trainable": 46125 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.12036337703466415, + "learning_rate": 9.469927859198888e-05, + "loss": 0.0019118499476462603, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00191, + "step": 102, + "tokens/total": 3255744, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 46612 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.29928353428840637, + "learning_rate": 9.456176618655689e-05, + "loss": 0.00944911316037178, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00949, + "step": 103, + "tokens/total": 3287792, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 47059 + }, + { + "epoch": 0.40625, + "grad_norm": 0.3766293227672577, + "learning_rate": 9.442260745454927e-05, + "loss": 0.008984047919511795, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.68, + "memory/max_allocated (GiB)": 36.68, + "ppl": 1.00902, + "step": 104, + "tokens/total": 3320016, + "tokens/train_per_sec_per_gpu": 28.92, + "tokens/trainable": 47536 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3481820523738861, + "learning_rate": 9.428180818692884e-05, + "loss": 0.005444999784231186, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00546, + "step": 105, + "tokens/total": 3351952, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 48037 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.8414323329925537, + "learning_rate": 9.413937424292791e-05, + "loss": 0.032438624650239944, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03297, + "step": 106, + "tokens/total": 3384032, + "tokens/train_per_sec_per_gpu": 27.11, + "tokens/trainable": 48491 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.5067655444145203, + "learning_rate": 9.399531154980424e-05, + "loss": 0.008442888967692852, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00848, + "step": 107, + "tokens/total": 3416192, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 48940 + }, + { + "epoch": 0.421875, + "grad_norm": 0.41617852449417114, + "learning_rate": 9.384962610259455e-05, + "loss": 0.008386321365833282, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00842, + "step": 108, + "tokens/total": 3448080, + "tokens/train_per_sec_per_gpu": 26.56, + "tokens/trainable": 49389 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1587515026330948, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0031663556583225727, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.03, + "memory/max_allocated (GiB)": 36.03, + "ppl": 1.00317, + "step": 109, + "tokens/total": 3477952, + "tokens/train_per_sec_per_gpu": 29.8, + "tokens/trainable": 49838 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2903148829936981, + "learning_rate": 9.355341126345868e-05, + "loss": 0.007958847098052502, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00799, + "step": 110, + "tokens/total": 3509776, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 50310 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.642983078956604, + "learning_rate": 9.340289419824107e-05, + "loss": 0.006748045329004526, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00677, + "step": 111, + "tokens/total": 3541808, + "tokens/train_per_sec_per_gpu": 28.04, + "tokens/trainable": 50755 + }, + { + "epoch": 0.4375, + "grad_norm": 0.14928282797336578, + "learning_rate": 9.325077903184159e-05, + "loss": 0.0027839094400405884, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00279, + "step": 112, + "tokens/total": 3574064, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 51213 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.05922335386276245, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0009622900979593396, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00096, + "step": 113, + "tokens/total": 3605984, + "tokens/train_per_sec_per_gpu": 27.99, + "tokens/trainable": 51660 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.44227731227874756, + "learning_rate": 9.2941779782269e-05, + "loss": 0.002763954224064946, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00277, + "step": 114, + "tokens/total": 3638080, + "tokens/train_per_sec_per_gpu": 27.03, + "tokens/trainable": 52133 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5176818370819092, + "learning_rate": 9.278490855781596e-05, + "loss": 0.015623578801751137, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01575, + "step": 115, + "tokens/total": 3670016, + "tokens/train_per_sec_per_gpu": 31.44, + "tokens/trainable": 52580 + }, + { + "epoch": 0.453125, + "grad_norm": 0.4265497922897339, + "learning_rate": 9.262646494908604e-05, + "loss": 0.00794076919555664, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00797, + "step": 116, + "tokens/total": 3701984, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 53033 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.40457987785339355, + "learning_rate": 9.246645554956457e-05, + "loss": 0.022708112373948097, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02297, + "step": 117, + "tokens/total": 3733920, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 53517 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.01414525043219328, + "learning_rate": 9.230488701789578e-05, + "loss": 0.00020209309877827764, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.0002, + "step": 118, + "tokens/total": 3765840, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 53967 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.11359831690788269, + "learning_rate": 9.214176607760577e-05, + "loss": 0.0012168455868959427, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00122, + "step": 119, + "tokens/total": 3797680, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 54430 + }, + { + "epoch": 0.46875, + "grad_norm": 0.35982534289360046, + "learning_rate": 9.197709951682268e-05, + "loss": 0.004110721405595541, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00412, + "step": 120, + "tokens/total": 3829920, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 54896 + }, + { + "epoch": 0.47265625, + "grad_norm": 1.3690382242202759, + "learning_rate": 9.181089418799428e-05, + "loss": 0.02705790475010872, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02743, + "step": 121, + "tokens/total": 3862080, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 55379 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.351891428232193, + "learning_rate": 9.164315700760271e-05, + "loss": 0.0020163573790341616, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00202, + "step": 122, + "tokens/total": 3893968, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 55803 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4747961461544037, + "learning_rate": 9.147389495587671e-05, + "loss": 0.012588979676365852, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01267, + "step": 123, + "tokens/total": 3926160, + "tokens/train_per_sec_per_gpu": 30.14, + "tokens/trainable": 56249 + }, + { + "epoch": 0.484375, + "grad_norm": 0.4064250588417053, + "learning_rate": 9.130311507650116e-05, + "loss": 0.0022941348142921925, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0023, + "step": 124, + "tokens/total": 3958208, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 56713 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.22056929767131805, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0022175831254571676, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00222, + "step": 125, + "tokens/total": 3990208, + "tokens/train_per_sec_per_gpu": 32.57, + "tokens/trainable": 57196 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5698862671852112, + "learning_rate": 9.09570303250602e-05, + "loss": 0.010205100290477276, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01026, + "step": 126, + "tokens/total": 4022224, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 57680 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.2593887448310852, + "learning_rate": 9.078173985499394e-05, + "loss": 0.0032277877908200026, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00323, + "step": 127, + "tokens/total": 4053952, + "tokens/train_per_sec_per_gpu": 26.51, + "tokens/trainable": 58110 + }, + { + "epoch": 0.5, + "grad_norm": 0.3735230565071106, + "learning_rate": 9.060496036067713e-05, + "loss": 0.01031105499714613, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01036, + "step": 128, + "tokens/total": 4086144, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 58534 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.1332484930753708, + "learning_rate": 9.042669919862615e-05, + "loss": 0.0012390739284455776, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00124, + "step": 129, + "tokens/total": 4118096, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 58998 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.20182718336582184, + "learning_rate": 9.024696378701557e-05, + "loss": 0.004501686431467533, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00451, + "step": 130, + "tokens/total": 4150272, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 59448 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.529080331325531, + "learning_rate": 9.006576160536948e-05, + "loss": 0.00763288140296936, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00766, + "step": 131, + "tokens/total": 4181904, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 59906 + }, + { + "epoch": 0.515625, + "grad_norm": 0.21625548601150513, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0024378676898777485, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00244, + "step": 132, + "tokens/total": 4214032, + "tokens/train_per_sec_per_gpu": 28.36, + "tokens/trainable": 60350 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.37116539478302, + "learning_rate": 8.969898715494506e-05, + "loss": 0.0037849934305995703, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00379, + "step": 133, + "tokens/total": 4246064, + "tokens/train_per_sec_per_gpu": 30.18, + "tokens/trainable": 60831 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5646623373031616, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006517104804515839, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00654, + "step": 134, + "tokens/total": 4278224, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 61305 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.6371963620185852, + "learning_rate": 8.932643689864568e-05, + "loss": 0.008687382563948631, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00873, + "step": 135, + "tokens/total": 4310112, + "tokens/train_per_sec_per_gpu": 31.48, + "tokens/trainable": 61792 + }, + { + "epoch": 0.53125, + "grad_norm": 0.8210524916648865, + "learning_rate": 8.913801518498845e-05, + "loss": 0.008436474949121475, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00847, + "step": 136, + "tokens/total": 4342192, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 62253 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.5692176222801208, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017321188002824783, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.01747, + "step": 137, + "tokens/total": 4374096, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.12828375399112701, + "learning_rate": 8.875691779131569e-05, + "loss": 0.0007754197577014565, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00078, + "step": 138, + "tokens/total": 4406272, + "tokens/train_per_sec_per_gpu": 24.22, + "tokens/trainable": 63145 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.3188190162181854, + "learning_rate": 8.856425797031829e-05, + "loss": 0.006707335356622934, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00673, + "step": 139, + "tokens/total": 4438112, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 63587 + }, + { + "epoch": 0.546875, + "grad_norm": 0.6216707229614258, + "learning_rate": 8.837020140354295e-05, + "loss": 0.007024695165455341, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00705, + "step": 140, + "tokens/total": 4470128, + "tokens/train_per_sec_per_gpu": 30.64, + "tokens/trainable": 64052 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.1480305939912796, + "learning_rate": 8.817475616647554e-05, + "loss": 0.00229115248657763, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00229, + "step": 141, + "tokens/total": 4502096, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 64526 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22254763543605804, + "learning_rate": 8.797793039239017e-05, + "loss": 0.0029198499396443367, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00292, + "step": 142, + "tokens/total": 4534496, + "tokens/train_per_sec_per_gpu": 29.08, + "tokens/trainable": 64983 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.6901941895484924, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013051149435341358, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01314, + "step": 143, + "tokens/total": 4566720, + "tokens/train_per_sec_per_gpu": 31.56, + "tokens/trainable": 65492 + }, + { + "epoch": 0.5625, + "grad_norm": 0.9545458555221558, + "learning_rate": 8.758017005316988e-05, + "loss": 0.02235381305217743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.02261, + "step": 144, + "tokens/total": 4598880, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 65925 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.5515892505645752, + "learning_rate": 8.737925204046629e-05, + "loss": 0.02038375660777092, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.02059, + "step": 145, + "tokens/total": 4630864, + "tokens/train_per_sec_per_gpu": 26.3, + "tokens/trainable": 66383 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3512430191040039, + "learning_rate": 8.717698659491851e-05, + "loss": 0.008233271539211273, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00827, + "step": 146, + "tokens/total": 4662864, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 66840 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.6830361485481262, + "learning_rate": 8.697338213361735e-05, + "loss": 0.013419180177152157, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01351, + "step": 147, + "tokens/total": 4694912, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 67297 + }, + { + "epoch": 0.578125, + "grad_norm": 0.2117031067609787, + "learning_rate": 8.676844712937552e-05, + "loss": 0.004134457092732191, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00414, + "step": 148, + "tokens/total": 4727120, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 67769 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.378053218126297, + "learning_rate": 8.656219011037509e-05, + "loss": 0.009747720323503017, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0098, + "step": 149, + "tokens/total": 4759168, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 68257 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.21076741814613342, + "learning_rate": 8.63546196598125e-05, + "loss": 0.005035653710365295, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00505, + "step": 150, + "tokens/total": 4791008, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 68706 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.25075697898864746, + "learning_rate": 8.614574441554145e-05, + "loss": 0.006027051247656345, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00605, + "step": 151, + "tokens/total": 4822912, + "tokens/train_per_sec_per_gpu": 27.95, + "tokens/trainable": 69131 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5774204730987549, + "learning_rate": 8.593557306971349e-05, + "loss": 0.018617186695337296, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01879, + "step": 152, + "tokens/total": 4854816, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 69586 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.1364293098449707, + "learning_rate": 8.572411436841618e-05, + "loss": 0.002049289643764496, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00205, + "step": 153, + "tokens/total": 4886656, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 70061 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.25387004017829895, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0044304742477834225, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00444, + "step": 154, + "tokens/total": 4918768, + "tokens/train_per_sec_per_gpu": 23.26, + "tokens/trainable": 70467 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3047125041484833, + "learning_rate": 8.529737015125824e-05, + "loss": 0.0043389927595853806, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00435, + "step": 155, + "tokens/total": 4951008, + "tokens/train_per_sec_per_gpu": 27.53, + "tokens/trainable": 70933 + }, + { + "epoch": 0.609375, + "grad_norm": 0.1442601978778839, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0031973645091056824, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.0032, + "step": 156, + "tokens/total": 4982976, + "tokens/train_per_sec_per_gpu": 28.86, + "tokens/trainable": 71382 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.2608455717563629, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0032951058819890022, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0033, + "step": 157, + "tokens/total": 5014640, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 71808 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.637295663356781, + "learning_rate": 8.464782037243449e-05, + "loss": 0.01436957623809576, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.01447, + "step": 158, + "tokens/total": 5046672, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 72249 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.9834056496620178, + "learning_rate": 8.442882418044202e-05, + "loss": 0.02375311404466629, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.02404, + "step": 159, + "tokens/total": 5078528, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 72726 + }, + { + "epoch": 0.625, + "grad_norm": 0.35432615876197815, + "learning_rate": 8.420860333495179e-05, + "loss": 0.006119017023593187, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00614, + "step": 160, + "tokens/total": 5110704, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 73148 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.44270065426826477, + "learning_rate": 8.398716700025208e-05, + "loss": 0.01352179329842329, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.72, + "memory/max_allocated (GiB)": 36.72, + "ppl": 1.01361, + "step": 161, + "tokens/total": 5142944, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 73600 + }, + { + "epoch": 0.6328125, + "grad_norm": 1.2181634902954102, + "learning_rate": 8.376452439121266e-05, + "loss": 0.01820964366197586, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.01838, + "step": 162, + "tokens/total": 5174592, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 74068 + }, + { + "epoch": 0.63671875, + "grad_norm": 1.9321303367614746, + "learning_rate": 8.354068477290124e-05, + "loss": 0.009742327965795994, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00979, + "step": 163, + "tokens/total": 5206432, + "tokens/train_per_sec_per_gpu": 29.89, + "tokens/trainable": 74527 + }, + { + "epoch": 0.640625, + "grad_norm": 0.3545796275138855, + "learning_rate": 8.331565746019807e-05, + "loss": 0.00975855067372322, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00981, + "step": 164, + "tokens/total": 5238384, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 74992 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.35368478298187256, + "learning_rate": 8.308945181740812e-05, + "loss": 0.010195466689765453, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01025, + "step": 165, + "tokens/total": 5270272, + "tokens/train_per_sec_per_gpu": 29.41, + "tokens/trainable": 75442 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.34957799315452576, + "learning_rate": 8.286207725787153e-05, + "loss": 0.009373681619763374, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00942, + "step": 166, + "tokens/total": 5302368, + "tokens/train_per_sec_per_gpu": 26.2, + "tokens/trainable": 75887 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.3684624135494232, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008211556822061539, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00825, + "step": 167, + "tokens/total": 5334496, + "tokens/train_per_sec_per_gpu": 28.21, + "tokens/trainable": 76331 + }, + { + "epoch": 0.65625, + "grad_norm": 0.12826858460903168, + "learning_rate": 8.240385928474219e-05, + "loss": 0.004492453299462795, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0045, + "step": 168, + "tokens/total": 5364288, + "tokens/train_per_sec_per_gpu": 30.16, + "tokens/trainable": 76745 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.31828173995018005, + "learning_rate": 8.217303493946967e-05, + "loss": 0.005733514670282602, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00575, + "step": 169, + "tokens/total": 5396512, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 77201 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6174353361129761, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01915649324655533, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01934, + "step": 170, + "tokens/total": 5428624, + "tokens/train_per_sec_per_gpu": 27.75, + "tokens/trainable": 77655 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.2902314066886902, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0036087254993617535, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00362, + "step": 171, + "tokens/total": 5460880, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 78122 + }, + { + "epoch": 0.671875, + "grad_norm": 0.3267754912376404, + "learning_rate": 8.147381587530713e-05, + "loss": 0.008090222254395485, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00812, + "step": 172, + "tokens/total": 5493088, + "tokens/train_per_sec_per_gpu": 27.86, + "tokens/trainable": 78576 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.4086189866065979, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006351984106004238, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00637, + "step": 173, + "tokens/total": 5525312, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 79059 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5261490345001221, + "learning_rate": 8.100214524900103e-05, + "loss": 0.008662862703204155, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.0087, + "step": 174, + "tokens/total": 5556880, + "tokens/train_per_sec_per_gpu": 24.62, + "tokens/trainable": 79498 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11292761564254761, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0027105510234832764, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00271, + "step": 175, + "tokens/total": 5588976, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 79922 + }, + { + "epoch": 0.6875, + "grad_norm": 0.33315548300743103, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0030926030594855547, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.0031, + "step": 176, + "tokens/total": 5618992, + "tokens/train_per_sec_per_gpu": 29.51, + "tokens/trainable": 80383 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.3136921226978302, + "learning_rate": 8.028654871074489e-05, + "loss": 0.005057850852608681, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00507, + "step": 177, + "tokens/total": 5651072, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 80824 + }, + { + "epoch": 0.6953125, + "grad_norm": 1.0526509284973145, + "learning_rate": 8.004589869885986e-05, + "loss": 0.010707498528063297, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01077, + "step": 178, + "tokens/total": 5683296, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 81243 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.1329648196697235, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0017975447699427605, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.0018, + "step": 179, + "tokens/total": 5715520, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 81716 + }, + { + "epoch": 0.703125, + "grad_norm": 1.0749201774597168, + "learning_rate": 7.95614819466576e-05, + "loss": 0.025617048144340515, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02595, + "step": 180, + "tokens/total": 5747488, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 82181 + }, + { + "epoch": 0.70703125, + "grad_norm": 1.0446465015411377, + "learning_rate": 7.931773536489872e-05, + "loss": 0.030236585065722466, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 35.9, + "memory/max_allocated (GiB)": 35.9, + "ppl": 1.0307, + "step": 181, + "tokens/total": 5777264, + "tokens/train_per_sec_per_gpu": 28.89, + "tokens/trainable": 82626 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22853964567184448, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0038214433006942272, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00383, + "step": 182, + "tokens/total": 5809264, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 83089 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.057821910828351974, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0004995397757738829, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0005, + "step": 183, + "tokens/total": 5841488, + "tokens/train_per_sec_per_gpu": 29.34, + "tokens/trainable": 83590 + }, + { + "epoch": 0.71875, + "grad_norm": 1.3264167308807373, + "learning_rate": 7.85804646415409e-05, + "loss": 0.013232000172138214, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01332, + "step": 184, + "tokens/total": 5873696, + "tokens/train_per_sec_per_gpu": 29.33, + "tokens/trainable": 84046 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.6027824878692627, + "learning_rate": 7.833273149760207e-05, + "loss": 0.012215660884976387, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01229, + "step": 185, + "tokens/total": 5905408, + "tokens/train_per_sec_per_gpu": 28.25, + "tokens/trainable": 84517 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.5848041772842407, + "learning_rate": 7.808402738346527e-05, + "loss": 0.024627480655908585, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.02493, + "step": 186, + "tokens/total": 5937392, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 84974 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.6823275089263916, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0028813849203288555, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00289, + "step": 187, + "tokens/total": 5969344, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 85460 + }, + { + "epoch": 0.734375, + "grad_norm": 0.41855958104133606, + "learning_rate": 7.758374768294647e-05, + "loss": 0.009796380065381527, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00984, + "step": 188, + "tokens/total": 6001296, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 85939 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6849080324172974, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004100777208805084, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00411, + "step": 189, + "tokens/total": 6031216, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 86377 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.15838374197483063, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003986812196671963, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00399, + "step": 190, + "tokens/total": 6063296, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 86834 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.3117486536502838, + "learning_rate": 7.682630588562518e-05, + "loss": 0.009132719598710537, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00917, + "step": 191, + "tokens/total": 6095216, + "tokens/train_per_sec_per_gpu": 26.65, + "tokens/trainable": 87265 + }, + { + "epoch": 0.75, + "grad_norm": 0.7837114334106445, + "learning_rate": 7.657199467573129e-05, + "loss": 0.007239446043968201, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00727, + "step": 192, + "tokens/total": 6127488, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 87747 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21777178347110748, + "learning_rate": 7.631678576708561e-05, + "loss": 0.002932134782895446, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00294, + "step": 193, + "tokens/total": 6159488, + "tokens/train_per_sec_per_gpu": 30.37, + "tokens/trainable": 88239 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.19159242510795593, + "learning_rate": 7.606068977997255e-05, + "loss": 0.003356864908710122, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00336, + "step": 194, + "tokens/total": 6191296, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 88718 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.3977915346622467, + "learning_rate": 7.580371737159148e-05, + "loss": 0.008549144491553307, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00859, + "step": 195, + "tokens/total": 6223120, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 89129 + }, + { + "epoch": 0.765625, + "grad_norm": 0.3875303566455841, + "learning_rate": 7.554587923561324e-05, + "loss": 0.006131038535386324, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00615, + "step": 196, + "tokens/total": 6254880, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 89567 + }, + { + "epoch": 0.76953125, + "grad_norm": 3.4133481979370117, + "learning_rate": 7.528718610173511e-05, + "loss": 0.01836150512099266, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01853, + "step": 197, + "tokens/total": 6286960, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 89988 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.2509997487068176, + "learning_rate": 7.502764873523431e-05, + "loss": 0.003821226069703698, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00383, + "step": 198, + "tokens/total": 6318880, + "tokens/train_per_sec_per_gpu": 30.29, + "tokens/trainable": 90434 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.9550195932388306, + "learning_rate": 7.476727793652011e-05, + "loss": 0.006872436031699181, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0069, + "step": 199, + "tokens/total": 6350928, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 90913 + }, + { + "epoch": 0.78125, + "grad_norm": 0.411724716424942, + "learning_rate": 7.450608454068415e-05, + "loss": 0.014243390411138535, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01435, + "step": 200, + "tokens/total": 6382976, + "tokens/train_per_sec_per_gpu": 28.29, + "tokens/trainable": 91355 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.34264692664146423, + "learning_rate": 7.424407941704987e-05, + "loss": 0.005309835076332092, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00532, + "step": 201, + "tokens/total": 6415120, + "tokens/train_per_sec_per_gpu": 31.41, + "tokens/trainable": 91856 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.41955795884132385, + "learning_rate": 7.398127346871986e-05, + "loss": 0.009196512401103973, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00924, + "step": 202, + "tokens/total": 6447216, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 92311 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.03107454814016819, + "learning_rate": 7.371767763212238e-05, + "loss": 0.0004908143309876323, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00049, + "step": 203, + "tokens/total": 6479280, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 92764 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39578482508659363, + "learning_rate": 7.345330287655617e-05, + "loss": 0.006089594680815935, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00611, + "step": 204, + "tokens/total": 6511040, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 93227 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.6444572806358337, + "learning_rate": 7.31881602037339e-05, + "loss": 0.010252210311591625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0103, + "step": 205, + "tokens/total": 6543120, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 93675 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.3283638060092926, + "learning_rate": 7.29222606473245e-05, + "loss": 0.00761406822130084, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00764, + "step": 206, + "tokens/total": 6575520, + "tokens/train_per_sec_per_gpu": 27.02, + "tokens/trainable": 94120 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.09952464699745178, + "learning_rate": 7.265561527249383e-05, + "loss": 0.001628706930205226, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00163, + "step": 207, + "tokens/total": 6607408, + "tokens/train_per_sec_per_gpu": 29.43, + "tokens/trainable": 94557 + }, + { + "epoch": 0.8125, + "grad_norm": 0.31704938411712646, + "learning_rate": 7.238823517544436e-05, + "loss": 0.005896817892789841, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00591, + "step": 208, + "tokens/total": 6639488, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 95035 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.0999862551689148, + "learning_rate": 7.212013148295333e-05, + "loss": 0.001754446537233889, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00176, + "step": 209, + "tokens/total": 6671520, + "tokens/train_per_sec_per_gpu": 31.08, + "tokens/trainable": 95517 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.1405109018087387, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0017940793186426163, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0018, + "step": 210, + "tokens/total": 6703536, + "tokens/train_per_sec_per_gpu": 30.3, + "tokens/trainable": 96026 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.09185238927602768, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0015102234901860356, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00151, + "step": 211, + "tokens/total": 6735616, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 96477 + }, + { + "epoch": 0.828125, + "grad_norm": 1.0464473962783813, + "learning_rate": 7.131159054949273e-05, + "loss": 0.016731251031160355, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01687, + "step": 212, + "tokens/total": 6767584, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 96951 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.3517152965068817, + "learning_rate": 7.104070433827139e-05, + "loss": 0.007984626106917858, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00802, + "step": 213, + "tokens/total": 6799488, + "tokens/train_per_sec_per_gpu": 25.83, + "tokens/trainable": 97350 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.28192469477653503, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00557567086070776, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00559, + "step": 214, + "tokens/total": 6831552, + "tokens/train_per_sec_per_gpu": 24.66, + "tokens/trainable": 97792 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.46135279536247253, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0170141588896513, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.01716, + "step": 215, + "tokens/total": 6863488, + "tokens/train_per_sec_per_gpu": 30.38, + "tokens/trainable": 98257 + }, + { + "epoch": 0.84375, + "grad_norm": 0.4277898073196411, + "learning_rate": 7.022408581865382e-05, + "loss": 0.009399959817528725, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.71, + "memory/max_allocated (GiB)": 36.71, + "ppl": 1.00944, + "step": 216, + "tokens/total": 6895888, + "tokens/train_per_sec_per_gpu": 27.28, + "tokens/trainable": 98731 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.34988704323768616, + "learning_rate": 6.99505974422157e-05, + "loss": 0.002909669652581215, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00291, + "step": 217, + "tokens/total": 6928000, + "tokens/train_per_sec_per_gpu": 30.25, + "tokens/trainable": 99212 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.4028005599975586, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0058855339884757996, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0059, + "step": 218, + "tokens/total": 6960096, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 99654 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.3000575304031372, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0025230362080037594, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00253, + "step": 219, + "tokens/total": 6992512, + "tokens/train_per_sec_per_gpu": 27.46, + "tokens/trainable": 100086 + }, + { + "epoch": 0.859375, + "grad_norm": 0.11351172626018524, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0021392193157225847, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00214, + "step": 220, + "tokens/total": 7024688, + "tokens/train_per_sec_per_gpu": 27.56, + "tokens/trainable": 100524 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.19512486457824707, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00650001410394907, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00652, + "step": 221, + "tokens/total": 7056928, + "tokens/train_per_sec_per_gpu": 28.05, + "tokens/trainable": 100996 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.1998104453086853, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0056140003725886345, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00563, + "step": 222, + "tokens/total": 7088944, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 101449 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.40908998250961304, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01286369375884533, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.01295, + "step": 223, + "tokens/total": 7121104, + "tokens/train_per_sec_per_gpu": 27.19, + "tokens/trainable": 101905 + }, + { + "epoch": 0.875, + "grad_norm": 0.3672473132610321, + "learning_rate": 6.801939899284132e-05, + "loss": 0.010677668265998363, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01073, + "step": 224, + "tokens/total": 7153040, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 102411 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4146183431148529, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0038945709820836782, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0039, + "step": 225, + "tokens/total": 7184912, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 102882 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.10585323721170425, + "learning_rate": 6.746257910210214e-05, + "loss": 0.00041726604104042053, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00042, + "step": 226, + "tokens/total": 7216992, + "tokens/train_per_sec_per_gpu": 28.72, + "tokens/trainable": 103332 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.36600011587142944, + "learning_rate": 6.718338543014937e-05, + "loss": 0.01526213251054287, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 227, + "tokens/total": 7249120, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 103790 + }, + { + "epoch": 0.890625, + "grad_norm": 0.16057318449020386, + "learning_rate": 6.69036847577983e-05, + "loss": 0.004024423658847809, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00403, + "step": 228, + "tokens/total": 7281072, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 104246 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.318798303604126, + "learning_rate": 6.662348872453553e-05, + "loss": 0.006663155741989613, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00669, + "step": 229, + "tokens/total": 7313120, + "tokens/train_per_sec_per_gpu": 31.3, + "tokens/trainable": 104707 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.10858234763145447, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0020512123592197895, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00205, + "step": 230, + "tokens/total": 7345184, + "tokens/train_per_sec_per_gpu": 30.81, + "tokens/trainable": 105167 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.1044008731842041, + "learning_rate": 6.60616572358065e-05, + "loss": 0.0010264476295560598, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00103, + "step": 231, + "tokens/total": 7377264, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 105576 + }, + { + "epoch": 0.90625, + "grad_norm": 0.14836947619915009, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0013845522189512849, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00139, + "step": 232, + "tokens/total": 7409424, + "tokens/train_per_sec_per_gpu": 30.32, + "tokens/trainable": 106043 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.022752461954951286, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0003245974367018789, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00032, + "step": 233, + "tokens/total": 7441456, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 106506 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.22981758415699005, + "learning_rate": 6.521548694236384e-05, + "loss": 0.002587020630016923, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00259, + "step": 234, + "tokens/total": 7473344, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 106951 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.5147526264190674, + "learning_rate": 6.493256429322259e-05, + "loss": 0.01256126631051302, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01264, + "step": 235, + "tokens/total": 7505040, + "tokens/train_per_sec_per_gpu": 25.64, + "tokens/trainable": 107382 + }, + { + "epoch": 0.921875, + "grad_norm": 0.3785130977630615, + "learning_rate": 6.464922830953799e-05, + "loss": 0.010956985875964165, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01102, + "step": 236, + "tokens/total": 7537120, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 107814 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.6193199753761292, + "learning_rate": 6.436549078207688e-05, + "loss": 0.020655140280723572, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02087, + "step": 237, + "tokens/total": 7569024, + "tokens/train_per_sec_per_gpu": 29.36, + "tokens/trainable": 108274 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.5084519982337952, + "learning_rate": 6.408136351831592e-05, + "loss": 0.011346567422151566, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01141, + "step": 238, + "tokens/total": 7600992, + "tokens/train_per_sec_per_gpu": 24.43, + "tokens/trainable": 108714 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.04367341846227646, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0004316020058467984, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00043, + "step": 239, + "tokens/total": 7633280, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 109220 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06729783862829208, + "learning_rate": 6.351198709240186e-05, + "loss": 0.000604260538239032, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.0006, + "step": 240, + "tokens/total": 7665408, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 109672 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.3010976314544678, + "learning_rate": 6.32267616243259e-05, + "loss": 0.006106264889240265, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00612, + "step": 241, + "tokens/total": 7697280, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 110135 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.11426915228366852, + "learning_rate": 6.294119380711849e-05, + "loss": 0.0014383264351636171, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00144, + "step": 242, + "tokens/total": 7729440, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 110563 + }, + { + "epoch": 0.94921875, + "grad_norm": 1.3792941570281982, + "learning_rate": 6.265529552442209e-05, + "loss": 0.003547664964571595, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00355, + "step": 243, + "tokens/total": 7761584, + "tokens/train_per_sec_per_gpu": 30.76, + "tokens/trainable": 111049 + }, + { + "epoch": 0.953125, + "grad_norm": 0.07595008611679077, + "learning_rate": 6.236907867363127e-05, + "loss": 0.0009103374904952943, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00091, + "step": 244, + "tokens/total": 7793824, + "tokens/train_per_sec_per_gpu": 28.23, + "tokens/trainable": 111495 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05425361171364784, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0013249842450022697, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00133, + "step": 245, + "tokens/total": 7825872, + "tokens/train_per_sec_per_gpu": 26.47, + "tokens/trainable": 111968 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.17112742364406586, + "learning_rate": 6.179573692313344e-05, + "loss": 0.001098912674933672, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.0011, + "step": 246, + "tokens/total": 7857840, + "tokens/train_per_sec_per_gpu": 30.09, + "tokens/trainable": 112416 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.10754083096981049, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0031517897732555866, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00316, + "step": 247, + "tokens/total": 7890000, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 112882 + }, + { + "epoch": 0.96875, + "grad_norm": 0.2477901428937912, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005755824968218803, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00577, + "step": 248, + "tokens/total": 7921984, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 113390 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.332253634929657, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.013207933865487576, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.0133, + "step": 249, + "tokens/total": 7953824, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 113904 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.18514207005500793, + "learning_rate": 6.064575550087316e-05, + "loss": 0.00523729994893074, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00525, + "step": 250, + "tokens/total": 7985984, + "tokens/train_per_sec_per_gpu": 28.41, + "tokens/trainable": 114351 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.1852743923664093, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0031474479474127293, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00315, + "step": 251, + "tokens/total": 8018256, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 114842 + }, + { + "epoch": 0.984375, + "grad_norm": 0.43020468950271606, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0053521618247032166, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00537, + "step": 252, + "tokens/total": 8050528, + "tokens/train_per_sec_per_gpu": 23.97, + "tokens/trainable": 115263 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.048135045915842056, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0005759480991400778, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00058, + "step": 253, + "tokens/total": 8082496, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 115703 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.028733640909194946, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00040354474913328886, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0004, + "step": 254, + "tokens/total": 8114592, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 116157 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.2808869481086731, + "learning_rate": 5.920308275189541e-05, + "loss": 0.006315852981060743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00634, + "step": 255, + "tokens/total": 8146752, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 116567 + }, + { + "epoch": 1.0, + "grad_norm": 0.054996367543935776, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0009099108865484595, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00091, + "step": 256, + "tokens/total": 8178768, + "tokens/train_per_sec_per_gpu": 25.98, + "tokens/trainable": 117030 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.18114623427391052, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.002796629210934043, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0028, + "step": 257, + "tokens/total": 8210800, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 117517 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.1408185362815857, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0013210453325882554, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00132, + "step": 258, + "tokens/total": 8242768, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 118015 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.36587440967559814, + "learning_rate": 5.80457242789548e-05, + "loss": 0.002160525880753994, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00216, + "step": 259, + "tokens/total": 8274784, + "tokens/train_per_sec_per_gpu": 30.13, + "tokens/trainable": 118520 + }, + { + "epoch": 1.015625, + "grad_norm": 0.02628348581492901, + "learning_rate": 5.77560376810977e-05, + "loss": 0.0003391164937056601, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00034, + "step": 260, + "tokens/total": 8306720, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 118992 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.014996240846812725, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0002124913444276899, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00021, + "step": 261, + "tokens/total": 8338784, + "tokens/train_per_sec_per_gpu": 27.96, + "tokens/trainable": 119438 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.0050656464882195, + "learning_rate": 5.717633247526522e-05, + "loss": 7.557802018709481e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00008, + "step": 262, + "tokens/total": 8370720, + "tokens/train_per_sec_per_gpu": 26.9, + "tokens/trainable": 119881 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.02459302730858326, + "learning_rate": 5.688633799118971e-05, + "loss": 0.00024908874183893204, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00025, + "step": 263, + "tokens/total": 8402736, + "tokens/train_per_sec_per_gpu": 23.2, + "tokens/trainable": 120285 + }, + { + "epoch": 1.03125, + "grad_norm": 0.017813201993703842, + "learning_rate": 5.659626500889066e-05, + "loss": 0.00021895825921092182, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00022, + "step": 264, + "tokens/total": 8434352, + "tokens/train_per_sec_per_gpu": 27.78, + "tokens/trainable": 120755 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.0038785531651228666, + "learning_rate": 5.6306125599488905e-05, + "loss": 6.556476728292182e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00007, + "step": 265, + "tokens/total": 8466208, + "tokens/train_per_sec_per_gpu": 29.86, + "tokens/trainable": 121194 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.18033447861671448, + "learning_rate": 5.601593183686955e-05, + "loss": 0.0016840343596413732, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00169, + "step": 266, + "tokens/total": 8498304, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 121670 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.22497256100177765, + "learning_rate": 5.572569579717961e-05, + "loss": 0.003026450052857399, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00303, + "step": 267, + "tokens/total": 8530400, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 122142 + }, + { + "epoch": 1.046875, + "grad_norm": 0.008878038264811039, + "learning_rate": 5.543542955832538e-05, + "loss": 8.136438555084169e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00008, + "step": 268, + "tokens/total": 8562608, + "tokens/train_per_sec_per_gpu": 27.27, + "tokens/trainable": 122585 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.19042901694774628, + "learning_rate": 5.514514519946986e-05, + "loss": 0.0020417363848537207, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00204, + "step": 269, + "tokens/total": 8594832, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 123091 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.003382593160495162, + "learning_rate": 5.485485480053015e-05, + "loss": 3.870048021781258e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00004, + "step": 270, + "tokens/total": 8627120, + "tokens/train_per_sec_per_gpu": 26.36, + "tokens/trainable": 123505 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.2004430592060089, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.002532299840822816, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.00254, + "step": 271, + "tokens/total": 8657056, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 123953 + }, + { + "epoch": 1.0625, + "grad_norm": 0.24857956171035767, + "learning_rate": 5.42743042028204e-05, + "loss": 0.004157304298132658, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00417, + "step": 272, + "tokens/total": 8689104, + "tokens/train_per_sec_per_gpu": 24.05, + "tokens/trainable": 124400 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.05064955726265907, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0006531125982291996, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00065, + "step": 273, + "tokens/total": 8721136, + "tokens/train_per_sec_per_gpu": 29.3, + "tokens/trainable": 124870 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.12664659321308136, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0018034171080216765, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00181, + "step": 274, + "tokens/total": 8753200, + "tokens/train_per_sec_per_gpu": 26.52, + "tokens/trainable": 125333 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.01593201234936714, + "learning_rate": 5.340373499110935e-05, + "loss": 0.00010935973114101216, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00011, + "step": 275, + "tokens/total": 8785168, + "tokens/train_per_sec_per_gpu": 30.45, + "tokens/trainable": 125834 + }, + { + "epoch": 1.078125, + "grad_norm": 0.0015056057600304484, + "learning_rate": 5.3113662008810304e-05, + "loss": 1.9860939573845826e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00002, + "step": 276, + "tokens/total": 8817520, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 126316 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.004457728937268257, + "learning_rate": 5.282366752473479e-05, + "loss": 2.716448398132343e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00003, + "step": 277, + "tokens/total": 8849360, + "tokens/train_per_sec_per_gpu": 29.63, + "tokens/trainable": 126798 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.03812727704644203, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.0003402164438739419, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.00034, + "step": 278, + "tokens/total": 8881152, + "tokens/train_per_sec_per_gpu": 24.5, + "tokens/trainable": 127223 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.011409996077418327, + "learning_rate": 5.224396231890232e-05, + "loss": 9.064963523996994e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.00009, + "step": 279, + "tokens/total": 8912512, + "tokens/train_per_sec_per_gpu": 29.78, + "tokens/trainable": 127672 + }, + { + "epoch": 1.09375, + "grad_norm": 0.037030577659606934, + "learning_rate": 5.195427572104522e-05, + "loss": 0.0003734154161065817, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00037, + "step": 280, + "tokens/total": 8944416, + "tokens/train_per_sec_per_gpu": 26.76, + "tokens/trainable": 128116 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.0032268385402858257, + "learning_rate": 5.166471586820751e-05, + "loss": 2.529611811041832e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00003, + "step": 281, + "tokens/total": 8976272, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 128589 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.20227248966693878, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.0024647137615829706, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00247, + "step": 282, + "tokens/total": 9008560, + "tokens/train_per_sec_per_gpu": 27.84, + "tokens/trainable": 129036 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.43708595633506775, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0028322634752839804, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00284, + "step": 283, + "tokens/total": 9040496, + "tokens/train_per_sec_per_gpu": 29.25, + "tokens/trainable": 129485 + }, + { + "epoch": 1.109375, + "grad_norm": 0.09467080235481262, + "learning_rate": 5.079691724810461e-05, + "loss": 0.001156509853899479, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00116, + "step": 284, + "tokens/total": 9072448, + "tokens/train_per_sec_per_gpu": 24.69, + "tokens/trainable": 129920 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.1778785139322281, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.0016857109731063247, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00169, + "step": 285, + "tokens/total": 9104384, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 130417 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.0003376641252543777, + "learning_rate": 5.021923930849237e-05, + "loss": 5.597693871095544e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00001, + "step": 286, + "tokens/total": 9136416, + "tokens/train_per_sec_per_gpu": 29.99, + "tokens/trainable": 130903 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.019782084971666336, + "learning_rate": 4.99306927511967e-05, + "loss": 0.00012461614096537232, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00012, + "step": 287, + "tokens/total": 9168640, + "tokens/train_per_sec_per_gpu": 29.24, + "tokens/trainable": 131343 + }, + { + "epoch": 1.125, + "grad_norm": 0.9641065001487732, + "learning_rate": 4.964235714846775e-05, + "loss": 0.011515076272189617, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01158, + "step": 288, + "tokens/total": 9200544, + "tokens/train_per_sec_per_gpu": 27.04, + "tokens/trainable": 131763 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0010062052169814706, + "learning_rate": 4.9354244499126866e-05, + "loss": 9.489545846008696e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00001, + "step": 289, + "tokens/total": 9232624, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 132229 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.002373398980125785, + "learning_rate": 4.90663667927174e-05, + "loss": 1.573777262819931e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00002, + "step": 290, + "tokens/total": 9264768, + "tokens/train_per_sec_per_gpu": 29.5, + "tokens/trainable": 132678 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.0014718567254021764, + "learning_rate": 4.877873600900581e-05, + "loss": 9.888429303828161e-06, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00001, + "step": 291, + "tokens/total": 9296944, + "tokens/train_per_sec_per_gpu": 24.47, + "tokens/trainable": 133136 + }, + { + "epoch": 1.140625, + "grad_norm": 0.0018440725980326533, + "learning_rate": 4.849136411748306e-05, + "loss": 8.275845175376162e-06, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00001, + "step": 292, + "tokens/total": 9328848, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 133608 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.12918472290039062, + "learning_rate": 4.8204263076866574e-05, + "loss": 0.0009128568926826119, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00091, + "step": 293, + "tokens/total": 9360928, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 134108 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.019195031374692917, + "learning_rate": 4.791744483460251e-05, + "loss": 9.439548011869192e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00009, + "step": 294, + "tokens/total": 9393040, + "tokens/train_per_sec_per_gpu": 25.71, + "tokens/trainable": 134541 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.1602204144001007, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0004640200058929622, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00046, + "step": 295, + "tokens/total": 9425248, + "tokens/train_per_sec_per_gpu": 24.5, + "tokens/trainable": 134966 + }, + { + "epoch": 1.15625, + "grad_norm": 0.5406288504600525, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.0034592358861118555, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00347, + "step": 296, + "tokens/total": 9457264, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 135402 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.004333238583058119, + "learning_rate": 4.705880619288153e-05, + "loss": 3.075868880841881e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00003, + "step": 297, + "tokens/total": 9489056, + "tokens/train_per_sec_per_gpu": 24.87, + "tokens/trainable": 135804 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.006247827783226967, + "learning_rate": 4.677323837567412e-05, + "loss": 2.383375249337405e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00002, + "step": 298, + "tokens/total": 9521024, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 136231 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.0779612734913826, + "learning_rate": 4.6488012907598146e-05, + "loss": 0.0004679379053413868, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00047, + "step": 299, + "tokens/total": 9553120, + "tokens/train_per_sec_per_gpu": 28.37, + "tokens/trainable": 136705 + }, + { + "epoch": 1.171875, + "grad_norm": 0.01816786825656891, + "learning_rate": 4.620314165804964e-05, + "loss": 6.801338167861104e-05, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00007, + "step": 300, + "tokens/total": 9585040, + "tokens/train_per_sec_per_gpu": 28.78, + "tokens/trainable": 137178 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.9530329704284668, + "learning_rate": 4.591863648168407e-05, + "loss": 0.008843549527227879, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00888, + "step": 301, + "tokens/total": 9617120, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 137643 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0371168851852417, + "learning_rate": 4.5634509217923135e-05, + "loss": 0.00014635953994002193, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00015, + "step": 302, + "tokens/total": 9649024, + "tokens/train_per_sec_per_gpu": 26.74, + "tokens/trainable": 138078 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.08589725196361542, + "learning_rate": 4.535077169046201e-05, + "loss": 0.00043714733328670263, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00044, + "step": 303, + "tokens/total": 9680832, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 138515 + }, + { + "epoch": 1.1875, + "grad_norm": 0.358940988779068, + "learning_rate": 4.506743570677743e-05, + "loss": 0.006240838672965765, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00626, + "step": 304, + "tokens/total": 9713040, + "tokens/train_per_sec_per_gpu": 29.29, + "tokens/trainable": 138948 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.12659867107868195, + "learning_rate": 4.478451305763618e-05, + "loss": 0.0008667556685395539, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00087, + "step": 305, + "tokens/total": 9744976, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 139408 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.08289908617734909, + "learning_rate": 4.450201551660454e-05, + "loss": 0.0006384386215358973, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.02, + "memory/max_allocated (GiB)": 36.02, + "ppl": 1.00064, + "step": 306, + "tokens/total": 9774784, + "tokens/train_per_sec_per_gpu": 27.76, + "tokens/trainable": 139840 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.19195082783699036, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.002342985011637211, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00235, + "step": 307, + "tokens/total": 9806656, + "tokens/train_per_sec_per_gpu": 27.6, + "tokens/trainable": 140281 + }, + { + "epoch": 1.203125, + "grad_norm": 0.07421132177114487, + "learning_rate": 4.393834276419352e-05, + "loss": 0.0007085398538038135, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00071, + "step": 308, + "tokens/total": 9838720, + "tokens/train_per_sec_per_gpu": 30.88, + "tokens/trainable": 140776 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.004596009384840727, + "learning_rate": 4.36571910095382e-05, + "loss": 2.8969257982680574e-05, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00003, + "step": 309, + "tokens/total": 9870784, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 141228 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.04099060595035553, + "learning_rate": 4.337651127546448e-05, + "loss": 0.00030985785997472703, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00031, + "step": 310, + "tokens/total": 9902784, + "tokens/train_per_sec_per_gpu": 30.8, + "tokens/trainable": 141679 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.18840794265270233, + "learning_rate": 4.3096315242201736e-05, + "loss": 0.0020186181645840406, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00202, + "step": 311, + "tokens/total": 9934768, + "tokens/train_per_sec_per_gpu": 26.72, + "tokens/trainable": 142122 + }, + { + "epoch": 1.21875, + "grad_norm": 0.5018841028213501, + "learning_rate": 4.2816614569850635e-05, + "loss": 0.006148474756628275, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00617, + "step": 312, + "tokens/total": 9966960, + "tokens/train_per_sec_per_gpu": 27.27, + "tokens/trainable": 142561 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.1730002760887146, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.003579454030841589, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00359, + "step": 313, + "tokens/total": 9999104, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 143046 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.6586350202560425, + "learning_rate": 4.225874584473174e-05, + "loss": 0.0015231478027999401, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00152, + "step": 314, + "tokens/total": 10031136, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 143493 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.011105087585747242, + "learning_rate": 4.19806010071587e-05, + "loss": 0.00011838486534543335, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00012, + "step": 315, + "tokens/total": 10063184, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 143956 + }, + { + "epoch": 1.234375, + "grad_norm": 0.2628834843635559, + "learning_rate": 4.170299795992081e-05, + "loss": 0.003402995876967907, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00341, + "step": 316, + "tokens/total": 10095312, + "tokens/train_per_sec_per_gpu": 28.29, + "tokens/trainable": 144411 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.13821958005428314, + "learning_rate": 4.142594825521398e-05, + "loss": 0.0016696923412382603, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00167, + "step": 317, + "tokens/total": 10127440, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 144892 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.0023418583441525698, + "learning_rate": 4.114946342220728e-05, + "loss": 2.6753859856398776e-05, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00003, + "step": 318, + "tokens/total": 10159184, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 145339 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.0008433983894065022, + "learning_rate": 4.087355496656321e-05, + "loss": 1.602486736373976e-05, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00002, + "step": 319, + "tokens/total": 10191216, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 145811 + }, + { + "epoch": 1.25, + "grad_norm": 0.12588869035243988, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0011584729654714465, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00116, + "step": 320, + "tokens/total": 10223408, + "tokens/train_per_sec_per_gpu": 32.27, + "tokens/trainable": 146314 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6.939224439391596e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/training_args.bin b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..dec74f9f304b44cbe47132d86c4e6285e659f40b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-320/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634 +size 8273 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/README.md b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ba677470ac9c29f9770a4aacb65c644ebd671a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/deconf_wave_deconf_control/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/deconf_wave_deconf_control/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/adapter_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..21fcb5b00be777c20a2d2dcceffc539dc51679b4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/deconf_wave_deconf_control/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "gate_proj", + "v_proj", + "k_proj", + "up_proj", + "q_proj", + "o_proj", + "down_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/adapter_model.safetensors b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..279e6fec358ad731272a11fbc58e4d663d0fc59f --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f9890c99e57c0fceff6825077b047dba561c930ed2dfb35d1252cdeeb3e1af93 +size 547777976 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/chat_template.jinja b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/optimizer.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..9c939dde478224d0f640ae4095967584806f1704 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:58e55c29ad23f3d7b8a0408bcdc8565910857f28490ff49d0353b8188da1e01c +size 1048106435 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/rng_state.pth b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..de5eb959342c20f9a2e2b71626f609172bf3cc1a --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a233590e8ae7b63a7844fe1bd1dec95c80b2124e32e247705c3cc4ffe157fee7 +size 14645 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/scheduler.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..1d2b8d34c008174b230d6dbbee4469d934686b9f --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:153174d2675ff0dc957d8edec3db026478f6ffc8ae455dadcc1c6ec96b4b4ce7 +size 1465 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/tokenizer.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/tokenizer_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/tokens_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..f8e635fc3668005973746175cedb24fd6f9bdf64 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/tokens_state.json @@ -0,0 +1 @@ +{"total": 11249088, "trainable": 160952} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/trainer_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..30e54de373a13fb4eda5fa80d3b417e499dd7738 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/trainer_state.json @@ -0,0 +1,4962 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.375, + "eval_steps": 500, + "global_step": 352, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.2140090465545654, + "learning_rate": 0.0, + "loss": 0.16441112756729126, + "memory/device_reserved (GiB)": 38.08, + "memory/max_active (GiB)": 35.64, + "memory/max_allocated (GiB)": 35.64, + "ppl": 1.1787, + "step": 1, + "tokens/total": 32160, + "tokens/train_per_sec_per_gpu": 21.28, + "tokens/trainable": 470 + }, + { + "epoch": 0.0078125, + "grad_norm": 2.92620849609375, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15774038434028625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.17086, + "step": 2, + "tokens/total": 64192, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 922 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.135961890220642, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1664382815361023, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.18109, + "step": 3, + "tokens/total": 96224, + "tokens/train_per_sec_per_gpu": 28.54, + "tokens/trainable": 1396 + }, + { + "epoch": 0.015625, + "grad_norm": 0.880176305770874, + "learning_rate": 1.2e-05, + "loss": 0.15334013104438782, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.16572, + "step": 4, + "tokens/total": 128336, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 1850 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.1433289051055908, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.16352537274360657, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.17766, + "step": 5, + "tokens/total": 160784, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 2302 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9728212952613831, + "learning_rate": 2e-05, + "loss": 0.15144473314285278, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.16351, + "step": 6, + "tokens/total": 192624, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 2742 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.7796363830566406, + "learning_rate": 2.4e-05, + "loss": 0.13553467392921448, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.14515, + "step": 7, + "tokens/total": 224608, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 3208 + }, + { + "epoch": 0.03125, + "grad_norm": 1.2444697618484497, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.1077304258942604, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.11375, + "step": 8, + "tokens/total": 256592, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 3655 + }, + { + "epoch": 0.03515625, + "grad_norm": 2.7339375019073486, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.08772765100002289, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.09169, + "step": 9, + "tokens/total": 286400, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 4091 + }, + { + "epoch": 0.0390625, + "grad_norm": 4.109770774841309, + "learning_rate": 3.6e-05, + "loss": 0.130940243601799, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.1399, + "step": 10, + "tokens/total": 318352, + "tokens/train_per_sec_per_gpu": 29.13, + "tokens/trainable": 4553 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.3768608570098877, + "learning_rate": 4e-05, + "loss": 0.10059958696365356, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.10583, + "step": 11, + "tokens/total": 350336, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 4992 + }, + { + "epoch": 0.046875, + "grad_norm": 3.281402826309204, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.12574875354766846, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.134, + "step": 12, + "tokens/total": 382448, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 5494 + }, + { + "epoch": 0.05078125, + "grad_norm": 1.7738980054855347, + "learning_rate": 4.8e-05, + "loss": 0.033296722918748856, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03386, + "step": 13, + "tokens/total": 414352, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 5933 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.64290189743042, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06366822123527527, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.06574, + "step": 14, + "tokens/total": 446400, + "tokens/train_per_sec_per_gpu": 24.8, + "tokens/trainable": 6369 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.3691928386688232, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.08168518543243408, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.08511, + "step": 15, + "tokens/total": 478688, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 6820 + }, + { + "epoch": 0.0625, + "grad_norm": 2.4356393814086914, + "learning_rate": 6e-05, + "loss": 0.09031196683645248, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.09452, + "step": 16, + "tokens/total": 510512, + "tokens/train_per_sec_per_gpu": 25.7, + "tokens/trainable": 7258 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.55682373046875, + "learning_rate": 6.400000000000001e-05, + "loss": 0.07496573030948639, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.07785, + "step": 17, + "tokens/total": 542736, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 7710 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.1054646968841553, + "learning_rate": 6.800000000000001e-05, + "loss": 0.05218805745244026, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.05357, + "step": 18, + "tokens/total": 574832, + "tokens/train_per_sec_per_gpu": 25.78, + "tokens/trainable": 8174 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.7429099082946777, + "learning_rate": 7.2e-05, + "loss": 0.03790827840566635, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.03864, + "step": 19, + "tokens/total": 606976, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 8617 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9508209228515625, + "learning_rate": 7.6e-05, + "loss": 0.04993613809347153, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0512, + "step": 20, + "tokens/total": 639200, + "tokens/train_per_sec_per_gpu": 24.91, + "tokens/trainable": 9037 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.0297844409942627, + "learning_rate": 8e-05, + "loss": 0.017908263951539993, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.01807, + "step": 21, + "tokens/total": 671072, + "tokens/train_per_sec_per_gpu": 30.42, + "tokens/trainable": 9503 + }, + { + "epoch": 0.0859375, + "grad_norm": 0.9279879331588745, + "learning_rate": 8.4e-05, + "loss": 0.013478193432092667, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01357, + "step": 22, + "tokens/total": 703136, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 9972 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.526488184928894, + "learning_rate": 8.800000000000001e-05, + "loss": 0.017310388386249542, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01746, + "step": 23, + "tokens/total": 735120, + "tokens/train_per_sec_per_gpu": 31.45, + "tokens/trainable": 10447 + }, + { + "epoch": 0.09375, + "grad_norm": 3.4049417972564697, + "learning_rate": 9.200000000000001e-05, + "loss": 0.0353429913520813, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.03597, + "step": 24, + "tokens/total": 767040, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 10899 + }, + { + "epoch": 0.09765625, + "grad_norm": 7.511656761169434, + "learning_rate": 9.6e-05, + "loss": 0.1335875540971756, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.14292, + "step": 25, + "tokens/total": 798976, + "tokens/train_per_sec_per_gpu": 27.72, + "tokens/trainable": 11360 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.0069234371185303, + "learning_rate": 0.0001, + "loss": 0.04382415860891342, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0448, + "step": 26, + "tokens/total": 830880, + "tokens/train_per_sec_per_gpu": 24.61, + "tokens/trainable": 11775 + }, + { + "epoch": 0.10546875, + "grad_norm": 7.807938575744629, + "learning_rate": 9.99990636831587e-05, + "loss": 0.01210833340883255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.01218, + "step": 27, + "tokens/total": 860960, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 12242 + }, + { + "epoch": 0.109375, + "grad_norm": 2.23823881149292, + "learning_rate": 9.999625477159879e-05, + "loss": 0.0272970050573349, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02767, + "step": 28, + "tokens/total": 892816, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 12696 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.315584421157837, + "learning_rate": 9.999157338221051e-05, + "loss": 0.025516577064990997, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.02584, + "step": 29, + "tokens/total": 924736, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 13153 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.8413651585578918, + "learning_rate": 9.998501970980562e-05, + "loss": 0.007838520221412182, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00787, + "step": 30, + "tokens/total": 954592, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 13624 + }, + { + "epoch": 0.12109375, + "grad_norm": 1.6797157526016235, + "learning_rate": 9.997659402710915e-05, + "loss": 0.03002442792057991, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.03048, + "step": 31, + "tokens/total": 986592, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 14064 + }, + { + "epoch": 0.125, + "grad_norm": 1.3320434093475342, + "learning_rate": 9.996629668474818e-05, + "loss": 0.03772728145122528, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.03845, + "step": 32, + "tokens/total": 1018720, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 14565 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.6335167288780212, + "learning_rate": 9.995412811123711e-05, + "loss": 0.013970119878649712, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01407, + "step": 33, + "tokens/total": 1050624, + "tokens/train_per_sec_per_gpu": 28.18, + "tokens/trainable": 15005 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.5626921653747559, + "learning_rate": 9.994008881295999e-05, + "loss": 0.016654539853334427, + "memory/device_reserved (GiB)": 37.74, + "memory/max_active (GiB)": 36.22, + "memory/max_allocated (GiB)": 36.22, + "ppl": 1.01679, + "step": 34, + "tokens/total": 1082176, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 15459 + }, + { + "epoch": 0.13671875, + "grad_norm": 0.603219211101532, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021408583968877792, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02164, + "step": 35, + "tokens/total": 1113968, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 15960 + }, + { + "epoch": 0.140625, + "grad_norm": 0.9779674410820007, + "learning_rate": 9.99064004568618e-05, + "loss": 0.03277068957686424, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03331, + "step": 36, + "tokens/total": 1145952, + "tokens/train_per_sec_per_gpu": 28.68, + "tokens/trainable": 16428 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.8855255842208862, + "learning_rate": 9.988675280095074e-05, + "loss": 0.017048098146915436, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01719, + "step": 37, + "tokens/total": 1177872, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 16884 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.9266122579574585, + "learning_rate": 9.986523722403528e-05, + "loss": 0.045863546431064606, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.04693, + "step": 38, + "tokens/total": 1209776, + "tokens/train_per_sec_per_gpu": 26.91, + "tokens/trainable": 17341 + }, + { + "epoch": 0.15234375, + "grad_norm": 0.8280895948410034, + "learning_rate": 9.984185462146642e-05, + "loss": 0.033055759966373444, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03361, + "step": 39, + "tokens/total": 1241888, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 17786 + }, + { + "epoch": 0.15625, + "grad_norm": 5.029730319976807, + "learning_rate": 9.98166059662897e-05, + "loss": 0.025589998811483383, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02592, + "step": 40, + "tokens/total": 1274032, + "tokens/train_per_sec_per_gpu": 29.22, + "tokens/trainable": 18262 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.5688032507896423, + "learning_rate": 9.978949230920472e-05, + "loss": 0.016095872968435287, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01623, + "step": 41, + "tokens/total": 1306000, + "tokens/train_per_sec_per_gpu": 27.49, + "tokens/trainable": 18716 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8217663168907166, + "learning_rate": 9.976051477852141e-05, + "loss": 0.026798982173204422, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02716, + "step": 42, + "tokens/total": 1337936, + "tokens/train_per_sec_per_gpu": 27.51, + "tokens/trainable": 19157 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.7309132218360901, + "learning_rate": 9.972967458011312e-05, + "loss": 0.033920761197805405, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0345, + "step": 43, + "tokens/total": 1369904, + "tokens/train_per_sec_per_gpu": 30.66, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.8134022355079651, + "learning_rate": 9.96969729973664e-05, + "loss": 0.024005085229873657, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0243, + "step": 44, + "tokens/total": 1401984, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 20119 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.35084816813468933, + "learning_rate": 9.966241139112754e-05, + "loss": 0.013319254852831364, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01341, + "step": 45, + "tokens/total": 1434032, + "tokens/train_per_sec_per_gpu": 27.45, + "tokens/trainable": 20560 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.8701573014259338, + "learning_rate": 9.96259911996461e-05, + "loss": 0.01356641948223114, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01366, + "step": 46, + "tokens/total": 1465872, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 20992 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.4406442642211914, + "learning_rate": 9.958771393851491e-05, + "loss": 0.04396980255842209, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 35.86, + "memory/max_allocated (GiB)": 35.86, + "ppl": 1.04495, + "step": 47, + "tokens/total": 1495440, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 21441 + }, + { + "epoch": 0.1875, + "grad_norm": 0.4966669976711273, + "learning_rate": 9.954758120060702e-05, + "loss": 0.011179964058101177, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01124, + "step": 48, + "tokens/total": 1527776, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 21901 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.41124436259269714, + "learning_rate": 9.950559465600948e-05, + "loss": 0.0063599650748074055, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00638, + "step": 49, + "tokens/total": 1559760, + "tokens/train_per_sec_per_gpu": 27.73, + "tokens/trainable": 22341 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.829739511013031, + "learning_rate": 9.946175605195379e-05, + "loss": 0.015785304829478264, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01591, + "step": 50, + "tokens/total": 1592000, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 22833 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5181736946105957, + "learning_rate": 9.941606721274322e-05, + "loss": 0.008393109776079655, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00843, + "step": 51, + "tokens/total": 1623920, + "tokens/train_per_sec_per_gpu": 25.61, + "tokens/trainable": 23277 + }, + { + "epoch": 0.203125, + "grad_norm": 1.5778990983963013, + "learning_rate": 9.936853003967685e-05, + "loss": 0.024984844028949738, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0253, + "step": 52, + "tokens/total": 1656032, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 23759 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.4938714504241943, + "learning_rate": 9.93191465109705e-05, + "loss": 0.038523320108652115, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03927, + "step": 53, + "tokens/total": 1688368, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 24193 + }, + { + "epoch": 0.2109375, + "grad_norm": 2.1427106857299805, + "learning_rate": 9.926791868167438e-05, + "loss": 0.06302924454212189, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.06506, + "step": 54, + "tokens/total": 1720560, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 24619 + }, + { + "epoch": 0.21484375, + "grad_norm": 1.8365188837051392, + "learning_rate": 9.921484868358753e-05, + "loss": 0.052160464227199554, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.05354, + "step": 55, + "tokens/total": 1752464, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 25075 + }, + { + "epoch": 0.21875, + "grad_norm": 0.7779749631881714, + "learning_rate": 9.915993872516924e-05, + "loss": 0.021328264847397804, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.02156, + "step": 56, + "tokens/total": 1784336, + "tokens/train_per_sec_per_gpu": 26.42, + "tokens/trainable": 25519 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.4189571142196655, + "learning_rate": 9.9103191091447e-05, + "loss": 0.014279939234256744, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01438, + "step": 57, + "tokens/total": 1816304, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 25966 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.3834710717201233, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01883944310247898, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01902, + "step": 58, + "tokens/total": 1848304, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 26423 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.7693278789520264, + "learning_rate": 9.898419232046825e-05, + "loss": 0.03533369302749634, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03597, + "step": 59, + "tokens/total": 1880640, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 26934 + }, + { + "epoch": 0.234375, + "grad_norm": 0.4606446325778961, + "learning_rate": 9.892194613523633e-05, + "loss": 0.021482212468981743, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02171, + "step": 60, + "tokens/total": 1912592, + "tokens/train_per_sec_per_gpu": 27.64, + "tokens/trainable": 27393 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.46030017733573914, + "learning_rate": 9.885787217854357e-05, + "loss": 0.017044665291905403, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01719, + "step": 61, + "tokens/total": 1944544, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 27852 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.7918264269828796, + "learning_rate": 9.879197311676887e-05, + "loss": 0.016954131424427032, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0171, + "step": 62, + "tokens/total": 1976688, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 28292 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.4787926971912384, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01525965891778469, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 63, + "tokens/total": 2008928, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 28798 + }, + { + "epoch": 0.25, + "grad_norm": 0.9033918976783752, + "learning_rate": 9.865471072312528e-05, + "loss": 0.02902211993932724, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02945, + "step": 64, + "tokens/total": 2041120, + "tokens/train_per_sec_per_gpu": 29.83, + "tokens/trainable": 29283 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.39305707812309265, + "learning_rate": 9.858335310330492e-05, + "loss": 0.00901167094707489, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.32, + "memory/max_allocated (GiB)": 36.32, + "ppl": 1.00905, + "step": 65, + "tokens/total": 2073024, + "tokens/train_per_sec_per_gpu": 26.48, + "tokens/trainable": 29745 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4494345784187317, + "learning_rate": 9.851018180226185e-05, + "loss": 0.007426132448017597, + "memory/device_reserved (GiB)": 37.47, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00745, + "step": 66, + "tokens/total": 2105184, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 30186 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.629649817943573, + "learning_rate": 9.843519986495259e-05, + "loss": 0.010609081946313381, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01067, + "step": 67, + "tokens/total": 2137264, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 30622 + }, + { + "epoch": 0.265625, + "grad_norm": 0.7591162919998169, + "learning_rate": 9.835841041168162e-05, + "loss": 0.00926615484058857, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00931, + "step": 68, + "tokens/total": 2169424, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 31097 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.7288486361503601, + "learning_rate": 9.82798166379715e-05, + "loss": 0.024310583248734474, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.02461, + "step": 69, + "tokens/total": 2201424, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 31595 + }, + { + "epoch": 0.2734375, + "grad_norm": 2.134056329727173, + "learning_rate": 9.819942181443002e-05, + "loss": 0.008422967046499252, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00846, + "step": 70, + "tokens/total": 2233424, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 32036 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6362871527671814, + "learning_rate": 9.811722928661392e-05, + "loss": 0.01577262207865715, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0159, + "step": 71, + "tokens/total": 2265328, + "tokens/train_per_sec_per_gpu": 23.65, + "tokens/trainable": 32428 + }, + { + "epoch": 0.28125, + "grad_norm": 0.6091912388801575, + "learning_rate": 9.803324247488975e-05, + "loss": 0.0064800274558365345, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0065, + "step": 72, + "tokens/total": 2297248, + "tokens/train_per_sec_per_gpu": 26.86, + "tokens/trainable": 32880 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.14533838629722595, + "learning_rate": 9.794746487429161e-05, + "loss": 0.001991454279050231, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00199, + "step": 73, + "tokens/total": 2329152, + "tokens/train_per_sec_per_gpu": 27.97, + "tokens/trainable": 33323 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.5400826930999756, + "learning_rate": 9.785990005437554e-05, + "loss": 0.017662350088357925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01782, + "step": 74, + "tokens/total": 2361344, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 33792 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.6132436394691467, + "learning_rate": 9.777055165907117e-05, + "loss": 0.01787603087723255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01804, + "step": 75, + "tokens/total": 2393168, + "tokens/train_per_sec_per_gpu": 29.53, + "tokens/trainable": 34223 + }, + { + "epoch": 0.296875, + "grad_norm": 0.7287399172782898, + "learning_rate": 9.767942340652993e-05, + "loss": 0.024754449725151062, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.31, + "memory/max_allocated (GiB)": 36.31, + "ppl": 1.02506, + "step": 76, + "tokens/total": 2424880, + "tokens/train_per_sec_per_gpu": 24.92, + "tokens/trainable": 34649 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.6790545582771301, + "learning_rate": 9.758651908897035e-05, + "loss": 0.014308430254459381, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01441, + "step": 77, + "tokens/total": 2457120, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 35094 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6203456521034241, + "learning_rate": 9.749184257252033e-05, + "loss": 0.034353043884038925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03495, + "step": 78, + "tokens/total": 2489104, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 35534 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.6238825917243958, + "learning_rate": 9.739539779705614e-05, + "loss": 0.032295141369104385, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.03282, + "step": 79, + "tokens/total": 2521136, + "tokens/train_per_sec_per_gpu": 29.68, + "tokens/trainable": 36029 + }, + { + "epoch": 0.3125, + "grad_norm": 0.18977171182632446, + "learning_rate": 9.729718877603861e-05, + "loss": 0.007260813377797604, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00729, + "step": 80, + "tokens/total": 2553168, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 36469 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.8054049611091614, + "learning_rate": 9.719721959634592e-05, + "loss": 0.034894898533821106, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.03551, + "step": 81, + "tokens/total": 2585584, + "tokens/train_per_sec_per_gpu": 25.75, + "tokens/trainable": 36906 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.6884189248085022, + "learning_rate": 9.709549441810375e-05, + "loss": 0.018210047855973244, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.01838, + "step": 82, + "tokens/total": 2617808, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 37390 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.6490103602409363, + "learning_rate": 9.699201747451195e-05, + "loss": 0.02030857279896736, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.02052, + "step": 83, + "tokens/total": 2649664, + "tokens/train_per_sec_per_gpu": 25.8, + "tokens/trainable": 37838 + }, + { + "epoch": 0.328125, + "grad_norm": 0.2439805567264557, + "learning_rate": 9.688679307166854e-05, + "loss": 0.008518049493432045, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00855, + "step": 84, + "tokens/total": 2681728, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 38310 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6747375726699829, + "learning_rate": 9.677982558839042e-05, + "loss": 0.029895059764385223, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.03035, + "step": 85, + "tokens/total": 2713488, + "tokens/train_per_sec_per_gpu": 28.49, + "tokens/trainable": 38789 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.582666277885437, + "learning_rate": 9.66711194760312e-05, + "loss": 0.014998597092926502, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01511, + "step": 86, + "tokens/total": 2745536, + "tokens/train_per_sec_per_gpu": 30.27, + "tokens/trainable": 39277 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.5364654660224915, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011567167937755585, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01163, + "step": 87, + "tokens/total": 2777680, + "tokens/train_per_sec_per_gpu": 29.73, + "tokens/trainable": 39737 + }, + { + "epoch": 0.34375, + "grad_norm": 0.34024426341056824, + "learning_rate": 9.644850953105288e-05, + "loss": 0.01014852337539196, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0102, + "step": 88, + "tokens/total": 2809568, + "tokens/train_per_sec_per_gpu": 24.68, + "tokens/trainable": 40179 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.32472267746925354, + "learning_rate": 9.633461496214225e-05, + "loss": 0.01229981891810894, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01238, + "step": 89, + "tokens/total": 2841760, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 40649 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.7906899452209473, + "learning_rate": 9.621900029118195e-05, + "loss": 0.02552586793899536, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.02585, + "step": 90, + "tokens/total": 2871696, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 41080 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.7444764971733093, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0054441955871880054, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00546, + "step": 91, + "tokens/total": 2903968, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 41599 + }, + { + "epoch": 0.359375, + "grad_norm": 0.1911851465702057, + "learning_rate": 9.598262995928611e-05, + "loss": 0.004665314219892025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00468, + "step": 92, + "tokens/total": 2936048, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 42076 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.21238519251346588, + "learning_rate": 9.586188413468492e-05, + "loss": 0.003917238209396601, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00392, + "step": 93, + "tokens/total": 2968176, + "tokens/train_per_sec_per_gpu": 28.07, + "tokens/trainable": 42522 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.41729435324668884, + "learning_rate": 9.57394378802934e-05, + "loss": 0.013555807992815971, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01365, + "step": 94, + "tokens/total": 3000304, + "tokens/train_per_sec_per_gpu": 28.69, + "tokens/trainable": 42974 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.7246769070625305, + "learning_rate": 9.56152962916e-05, + "loss": 0.020221399143338203, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.02043, + "step": 95, + "tokens/total": 3032080, + "tokens/train_per_sec_per_gpu": 25.35, + "tokens/trainable": 43380 + }, + { + "epoch": 0.375, + "grad_norm": 0.16139821708202362, + "learning_rate": 9.548946453464296e-05, + "loss": 0.003853276837617159, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00386, + "step": 96, + "tokens/total": 3064320, + "tokens/train_per_sec_per_gpu": 28.47, + "tokens/trainable": 43865 + }, + { + "epoch": 0.37890625, + "grad_norm": 1.0509370565414429, + "learning_rate": 9.53619478457953e-05, + "loss": 0.015099998563528061, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01521, + "step": 97, + "tokens/total": 3096144, + "tokens/train_per_sec_per_gpu": 26.61, + "tokens/trainable": 44328 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.6859350204467773, + "learning_rate": 9.523275153154695e-05, + "loss": 0.004579808097332716, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00459, + "step": 98, + "tokens/total": 3127600, + "tokens/train_per_sec_per_gpu": 25.91, + "tokens/trainable": 44778 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5686795711517334, + "learning_rate": 9.51018809682839e-05, + "loss": 0.022520575672388077, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02278, + "step": 99, + "tokens/total": 3159520, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 45225 + }, + { + "epoch": 0.390625, + "grad_norm": 1.6308369636535645, + "learning_rate": 9.49693416020645e-05, + "loss": 0.011883002705872059, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01195, + "step": 100, + "tokens/total": 3191568, + "tokens/train_per_sec_per_gpu": 29.75, + "tokens/trainable": 45678 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.7186088562011719, + "learning_rate": 9.483513894839276e-05, + "loss": 0.0060632615350186825, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00608, + "step": 101, + "tokens/total": 3223648, + "tokens/train_per_sec_per_gpu": 27.14, + "tokens/trainable": 46125 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.12036337703466415, + "learning_rate": 9.469927859198888e-05, + "loss": 0.0019118499476462603, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00191, + "step": 102, + "tokens/total": 3255744, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 46612 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.29928353428840637, + "learning_rate": 9.456176618655689e-05, + "loss": 0.00944911316037178, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00949, + "step": 103, + "tokens/total": 3287792, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 47059 + }, + { + "epoch": 0.40625, + "grad_norm": 0.3766293227672577, + "learning_rate": 9.442260745454927e-05, + "loss": 0.008984047919511795, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.68, + "memory/max_allocated (GiB)": 36.68, + "ppl": 1.00902, + "step": 104, + "tokens/total": 3320016, + "tokens/train_per_sec_per_gpu": 28.92, + "tokens/trainable": 47536 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3481820523738861, + "learning_rate": 9.428180818692884e-05, + "loss": 0.005444999784231186, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00546, + "step": 105, + "tokens/total": 3351952, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 48037 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.8414323329925537, + "learning_rate": 9.413937424292791e-05, + "loss": 0.032438624650239944, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03297, + "step": 106, + "tokens/total": 3384032, + "tokens/train_per_sec_per_gpu": 27.11, + "tokens/trainable": 48491 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.5067655444145203, + "learning_rate": 9.399531154980424e-05, + "loss": 0.008442888967692852, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00848, + "step": 107, + "tokens/total": 3416192, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 48940 + }, + { + "epoch": 0.421875, + "grad_norm": 0.41617852449417114, + "learning_rate": 9.384962610259455e-05, + "loss": 0.008386321365833282, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00842, + "step": 108, + "tokens/total": 3448080, + "tokens/train_per_sec_per_gpu": 26.56, + "tokens/trainable": 49389 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1587515026330948, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0031663556583225727, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.03, + "memory/max_allocated (GiB)": 36.03, + "ppl": 1.00317, + "step": 109, + "tokens/total": 3477952, + "tokens/train_per_sec_per_gpu": 29.8, + "tokens/trainable": 49838 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2903148829936981, + "learning_rate": 9.355341126345868e-05, + "loss": 0.007958847098052502, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00799, + "step": 110, + "tokens/total": 3509776, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 50310 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.642983078956604, + "learning_rate": 9.340289419824107e-05, + "loss": 0.006748045329004526, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00677, + "step": 111, + "tokens/total": 3541808, + "tokens/train_per_sec_per_gpu": 28.04, + "tokens/trainable": 50755 + }, + { + "epoch": 0.4375, + "grad_norm": 0.14928282797336578, + "learning_rate": 9.325077903184159e-05, + "loss": 0.0027839094400405884, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00279, + "step": 112, + "tokens/total": 3574064, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 51213 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.05922335386276245, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0009622900979593396, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00096, + "step": 113, + "tokens/total": 3605984, + "tokens/train_per_sec_per_gpu": 27.99, + "tokens/trainable": 51660 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.44227731227874756, + "learning_rate": 9.2941779782269e-05, + "loss": 0.002763954224064946, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00277, + "step": 114, + "tokens/total": 3638080, + "tokens/train_per_sec_per_gpu": 27.03, + "tokens/trainable": 52133 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5176818370819092, + "learning_rate": 9.278490855781596e-05, + "loss": 0.015623578801751137, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01575, + "step": 115, + "tokens/total": 3670016, + "tokens/train_per_sec_per_gpu": 31.44, + "tokens/trainable": 52580 + }, + { + "epoch": 0.453125, + "grad_norm": 0.4265497922897339, + "learning_rate": 9.262646494908604e-05, + "loss": 0.00794076919555664, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00797, + "step": 116, + "tokens/total": 3701984, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 53033 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.40457987785339355, + "learning_rate": 9.246645554956457e-05, + "loss": 0.022708112373948097, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02297, + "step": 117, + "tokens/total": 3733920, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 53517 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.01414525043219328, + "learning_rate": 9.230488701789578e-05, + "loss": 0.00020209309877827764, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.0002, + "step": 118, + "tokens/total": 3765840, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 53967 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.11359831690788269, + "learning_rate": 9.214176607760577e-05, + "loss": 0.0012168455868959427, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00122, + "step": 119, + "tokens/total": 3797680, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 54430 + }, + { + "epoch": 0.46875, + "grad_norm": 0.35982534289360046, + "learning_rate": 9.197709951682268e-05, + "loss": 0.004110721405595541, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00412, + "step": 120, + "tokens/total": 3829920, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 54896 + }, + { + "epoch": 0.47265625, + "grad_norm": 1.3690382242202759, + "learning_rate": 9.181089418799428e-05, + "loss": 0.02705790475010872, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02743, + "step": 121, + "tokens/total": 3862080, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 55379 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.351891428232193, + "learning_rate": 9.164315700760271e-05, + "loss": 0.0020163573790341616, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00202, + "step": 122, + "tokens/total": 3893968, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 55803 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4747961461544037, + "learning_rate": 9.147389495587671e-05, + "loss": 0.012588979676365852, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01267, + "step": 123, + "tokens/total": 3926160, + "tokens/train_per_sec_per_gpu": 30.14, + "tokens/trainable": 56249 + }, + { + "epoch": 0.484375, + "grad_norm": 0.4064250588417053, + "learning_rate": 9.130311507650116e-05, + "loss": 0.0022941348142921925, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0023, + "step": 124, + "tokens/total": 3958208, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 56713 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.22056929767131805, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0022175831254571676, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00222, + "step": 125, + "tokens/total": 3990208, + "tokens/train_per_sec_per_gpu": 32.57, + "tokens/trainable": 57196 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5698862671852112, + "learning_rate": 9.09570303250602e-05, + "loss": 0.010205100290477276, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01026, + "step": 126, + "tokens/total": 4022224, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 57680 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.2593887448310852, + "learning_rate": 9.078173985499394e-05, + "loss": 0.0032277877908200026, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00323, + "step": 127, + "tokens/total": 4053952, + "tokens/train_per_sec_per_gpu": 26.51, + "tokens/trainable": 58110 + }, + { + "epoch": 0.5, + "grad_norm": 0.3735230565071106, + "learning_rate": 9.060496036067713e-05, + "loss": 0.01031105499714613, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01036, + "step": 128, + "tokens/total": 4086144, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 58534 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.1332484930753708, + "learning_rate": 9.042669919862615e-05, + "loss": 0.0012390739284455776, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00124, + "step": 129, + "tokens/total": 4118096, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 58998 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.20182718336582184, + "learning_rate": 9.024696378701557e-05, + "loss": 0.004501686431467533, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00451, + "step": 130, + "tokens/total": 4150272, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 59448 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.529080331325531, + "learning_rate": 9.006576160536948e-05, + "loss": 0.00763288140296936, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00766, + "step": 131, + "tokens/total": 4181904, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 59906 + }, + { + "epoch": 0.515625, + "grad_norm": 0.21625548601150513, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0024378676898777485, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00244, + "step": 132, + "tokens/total": 4214032, + "tokens/train_per_sec_per_gpu": 28.36, + "tokens/trainable": 60350 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.37116539478302, + "learning_rate": 8.969898715494506e-05, + "loss": 0.0037849934305995703, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00379, + "step": 133, + "tokens/total": 4246064, + "tokens/train_per_sec_per_gpu": 30.18, + "tokens/trainable": 60831 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5646623373031616, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006517104804515839, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00654, + "step": 134, + "tokens/total": 4278224, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 61305 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.6371963620185852, + "learning_rate": 8.932643689864568e-05, + "loss": 0.008687382563948631, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00873, + "step": 135, + "tokens/total": 4310112, + "tokens/train_per_sec_per_gpu": 31.48, + "tokens/trainable": 61792 + }, + { + "epoch": 0.53125, + "grad_norm": 0.8210524916648865, + "learning_rate": 8.913801518498845e-05, + "loss": 0.008436474949121475, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00847, + "step": 136, + "tokens/total": 4342192, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 62253 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.5692176222801208, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017321188002824783, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.01747, + "step": 137, + "tokens/total": 4374096, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.12828375399112701, + "learning_rate": 8.875691779131569e-05, + "loss": 0.0007754197577014565, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00078, + "step": 138, + "tokens/total": 4406272, + "tokens/train_per_sec_per_gpu": 24.22, + "tokens/trainable": 63145 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.3188190162181854, + "learning_rate": 8.856425797031829e-05, + "loss": 0.006707335356622934, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00673, + "step": 139, + "tokens/total": 4438112, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 63587 + }, + { + "epoch": 0.546875, + "grad_norm": 0.6216707229614258, + "learning_rate": 8.837020140354295e-05, + "loss": 0.007024695165455341, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00705, + "step": 140, + "tokens/total": 4470128, + "tokens/train_per_sec_per_gpu": 30.64, + "tokens/trainable": 64052 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.1480305939912796, + "learning_rate": 8.817475616647554e-05, + "loss": 0.00229115248657763, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00229, + "step": 141, + "tokens/total": 4502096, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 64526 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22254763543605804, + "learning_rate": 8.797793039239017e-05, + "loss": 0.0029198499396443367, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00292, + "step": 142, + "tokens/total": 4534496, + "tokens/train_per_sec_per_gpu": 29.08, + "tokens/trainable": 64983 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.6901941895484924, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013051149435341358, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01314, + "step": 143, + "tokens/total": 4566720, + "tokens/train_per_sec_per_gpu": 31.56, + "tokens/trainable": 65492 + }, + { + "epoch": 0.5625, + "grad_norm": 0.9545458555221558, + "learning_rate": 8.758017005316988e-05, + "loss": 0.02235381305217743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.02261, + "step": 144, + "tokens/total": 4598880, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 65925 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.5515892505645752, + "learning_rate": 8.737925204046629e-05, + "loss": 0.02038375660777092, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.02059, + "step": 145, + "tokens/total": 4630864, + "tokens/train_per_sec_per_gpu": 26.3, + "tokens/trainable": 66383 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3512430191040039, + "learning_rate": 8.717698659491851e-05, + "loss": 0.008233271539211273, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00827, + "step": 146, + "tokens/total": 4662864, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 66840 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.6830361485481262, + "learning_rate": 8.697338213361735e-05, + "loss": 0.013419180177152157, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01351, + "step": 147, + "tokens/total": 4694912, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 67297 + }, + { + "epoch": 0.578125, + "grad_norm": 0.2117031067609787, + "learning_rate": 8.676844712937552e-05, + "loss": 0.004134457092732191, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00414, + "step": 148, + "tokens/total": 4727120, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 67769 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.378053218126297, + "learning_rate": 8.656219011037509e-05, + "loss": 0.009747720323503017, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0098, + "step": 149, + "tokens/total": 4759168, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 68257 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.21076741814613342, + "learning_rate": 8.63546196598125e-05, + "loss": 0.005035653710365295, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00505, + "step": 150, + "tokens/total": 4791008, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 68706 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.25075697898864746, + "learning_rate": 8.614574441554145e-05, + "loss": 0.006027051247656345, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00605, + "step": 151, + "tokens/total": 4822912, + "tokens/train_per_sec_per_gpu": 27.95, + "tokens/trainable": 69131 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5774204730987549, + "learning_rate": 8.593557306971349e-05, + "loss": 0.018617186695337296, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01879, + "step": 152, + "tokens/total": 4854816, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 69586 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.1364293098449707, + "learning_rate": 8.572411436841618e-05, + "loss": 0.002049289643764496, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00205, + "step": 153, + "tokens/total": 4886656, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 70061 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.25387004017829895, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0044304742477834225, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00444, + "step": 154, + "tokens/total": 4918768, + "tokens/train_per_sec_per_gpu": 23.26, + "tokens/trainable": 70467 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3047125041484833, + "learning_rate": 8.529737015125824e-05, + "loss": 0.0043389927595853806, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00435, + "step": 155, + "tokens/total": 4951008, + "tokens/train_per_sec_per_gpu": 27.53, + "tokens/trainable": 70933 + }, + { + "epoch": 0.609375, + "grad_norm": 0.1442601978778839, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0031973645091056824, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.0032, + "step": 156, + "tokens/total": 4982976, + "tokens/train_per_sec_per_gpu": 28.86, + "tokens/trainable": 71382 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.2608455717563629, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0032951058819890022, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0033, + "step": 157, + "tokens/total": 5014640, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 71808 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.637295663356781, + "learning_rate": 8.464782037243449e-05, + "loss": 0.01436957623809576, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.01447, + "step": 158, + "tokens/total": 5046672, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 72249 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.9834056496620178, + "learning_rate": 8.442882418044202e-05, + "loss": 0.02375311404466629, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.02404, + "step": 159, + "tokens/total": 5078528, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 72726 + }, + { + "epoch": 0.625, + "grad_norm": 0.35432615876197815, + "learning_rate": 8.420860333495179e-05, + "loss": 0.006119017023593187, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00614, + "step": 160, + "tokens/total": 5110704, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 73148 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.44270065426826477, + "learning_rate": 8.398716700025208e-05, + "loss": 0.01352179329842329, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.72, + "memory/max_allocated (GiB)": 36.72, + "ppl": 1.01361, + "step": 161, + "tokens/total": 5142944, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 73600 + }, + { + "epoch": 0.6328125, + "grad_norm": 1.2181634902954102, + "learning_rate": 8.376452439121266e-05, + "loss": 0.01820964366197586, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.01838, + "step": 162, + "tokens/total": 5174592, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 74068 + }, + { + "epoch": 0.63671875, + "grad_norm": 1.9321303367614746, + "learning_rate": 8.354068477290124e-05, + "loss": 0.009742327965795994, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00979, + "step": 163, + "tokens/total": 5206432, + "tokens/train_per_sec_per_gpu": 29.89, + "tokens/trainable": 74527 + }, + { + "epoch": 0.640625, + "grad_norm": 0.3545796275138855, + "learning_rate": 8.331565746019807e-05, + "loss": 0.00975855067372322, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00981, + "step": 164, + "tokens/total": 5238384, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 74992 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.35368478298187256, + "learning_rate": 8.308945181740812e-05, + "loss": 0.010195466689765453, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01025, + "step": 165, + "tokens/total": 5270272, + "tokens/train_per_sec_per_gpu": 29.41, + "tokens/trainable": 75442 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.34957799315452576, + "learning_rate": 8.286207725787153e-05, + "loss": 0.009373681619763374, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00942, + "step": 166, + "tokens/total": 5302368, + "tokens/train_per_sec_per_gpu": 26.2, + "tokens/trainable": 75887 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.3684624135494232, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008211556822061539, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00825, + "step": 167, + "tokens/total": 5334496, + "tokens/train_per_sec_per_gpu": 28.21, + "tokens/trainable": 76331 + }, + { + "epoch": 0.65625, + "grad_norm": 0.12826858460903168, + "learning_rate": 8.240385928474219e-05, + "loss": 0.004492453299462795, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0045, + "step": 168, + "tokens/total": 5364288, + "tokens/train_per_sec_per_gpu": 30.16, + "tokens/trainable": 76745 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.31828173995018005, + "learning_rate": 8.217303493946967e-05, + "loss": 0.005733514670282602, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00575, + "step": 169, + "tokens/total": 5396512, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 77201 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6174353361129761, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01915649324655533, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01934, + "step": 170, + "tokens/total": 5428624, + "tokens/train_per_sec_per_gpu": 27.75, + "tokens/trainable": 77655 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.2902314066886902, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0036087254993617535, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00362, + "step": 171, + "tokens/total": 5460880, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 78122 + }, + { + "epoch": 0.671875, + "grad_norm": 0.3267754912376404, + "learning_rate": 8.147381587530713e-05, + "loss": 0.008090222254395485, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00812, + "step": 172, + "tokens/total": 5493088, + "tokens/train_per_sec_per_gpu": 27.86, + "tokens/trainable": 78576 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.4086189866065979, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006351984106004238, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00637, + "step": 173, + "tokens/total": 5525312, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 79059 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5261490345001221, + "learning_rate": 8.100214524900103e-05, + "loss": 0.008662862703204155, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.0087, + "step": 174, + "tokens/total": 5556880, + "tokens/train_per_sec_per_gpu": 24.62, + "tokens/trainable": 79498 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11292761564254761, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0027105510234832764, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00271, + "step": 175, + "tokens/total": 5588976, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 79922 + }, + { + "epoch": 0.6875, + "grad_norm": 0.33315548300743103, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0030926030594855547, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.0031, + "step": 176, + "tokens/total": 5618992, + "tokens/train_per_sec_per_gpu": 29.51, + "tokens/trainable": 80383 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.3136921226978302, + "learning_rate": 8.028654871074489e-05, + "loss": 0.005057850852608681, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00507, + "step": 177, + "tokens/total": 5651072, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 80824 + }, + { + "epoch": 0.6953125, + "grad_norm": 1.0526509284973145, + "learning_rate": 8.004589869885986e-05, + "loss": 0.010707498528063297, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01077, + "step": 178, + "tokens/total": 5683296, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 81243 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.1329648196697235, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0017975447699427605, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.0018, + "step": 179, + "tokens/total": 5715520, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 81716 + }, + { + "epoch": 0.703125, + "grad_norm": 1.0749201774597168, + "learning_rate": 7.95614819466576e-05, + "loss": 0.025617048144340515, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02595, + "step": 180, + "tokens/total": 5747488, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 82181 + }, + { + "epoch": 0.70703125, + "grad_norm": 1.0446465015411377, + "learning_rate": 7.931773536489872e-05, + "loss": 0.030236585065722466, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 35.9, + "memory/max_allocated (GiB)": 35.9, + "ppl": 1.0307, + "step": 181, + "tokens/total": 5777264, + "tokens/train_per_sec_per_gpu": 28.89, + "tokens/trainable": 82626 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22853964567184448, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0038214433006942272, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00383, + "step": 182, + "tokens/total": 5809264, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 83089 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.057821910828351974, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0004995397757738829, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0005, + "step": 183, + "tokens/total": 5841488, + "tokens/train_per_sec_per_gpu": 29.34, + "tokens/trainable": 83590 + }, + { + "epoch": 0.71875, + "grad_norm": 1.3264167308807373, + "learning_rate": 7.85804646415409e-05, + "loss": 0.013232000172138214, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01332, + "step": 184, + "tokens/total": 5873696, + "tokens/train_per_sec_per_gpu": 29.33, + "tokens/trainable": 84046 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.6027824878692627, + "learning_rate": 7.833273149760207e-05, + "loss": 0.012215660884976387, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01229, + "step": 185, + "tokens/total": 5905408, + "tokens/train_per_sec_per_gpu": 28.25, + "tokens/trainable": 84517 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.5848041772842407, + "learning_rate": 7.808402738346527e-05, + "loss": 0.024627480655908585, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.02493, + "step": 186, + "tokens/total": 5937392, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 84974 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.6823275089263916, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0028813849203288555, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00289, + "step": 187, + "tokens/total": 5969344, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 85460 + }, + { + "epoch": 0.734375, + "grad_norm": 0.41855958104133606, + "learning_rate": 7.758374768294647e-05, + "loss": 0.009796380065381527, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00984, + "step": 188, + "tokens/total": 6001296, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 85939 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6849080324172974, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004100777208805084, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00411, + "step": 189, + "tokens/total": 6031216, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 86377 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.15838374197483063, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003986812196671963, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00399, + "step": 190, + "tokens/total": 6063296, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 86834 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.3117486536502838, + "learning_rate": 7.682630588562518e-05, + "loss": 0.009132719598710537, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00917, + "step": 191, + "tokens/total": 6095216, + "tokens/train_per_sec_per_gpu": 26.65, + "tokens/trainable": 87265 + }, + { + "epoch": 0.75, + "grad_norm": 0.7837114334106445, + "learning_rate": 7.657199467573129e-05, + "loss": 0.007239446043968201, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00727, + "step": 192, + "tokens/total": 6127488, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 87747 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21777178347110748, + "learning_rate": 7.631678576708561e-05, + "loss": 0.002932134782895446, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00294, + "step": 193, + "tokens/total": 6159488, + "tokens/train_per_sec_per_gpu": 30.37, + "tokens/trainable": 88239 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.19159242510795593, + "learning_rate": 7.606068977997255e-05, + "loss": 0.003356864908710122, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00336, + "step": 194, + "tokens/total": 6191296, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 88718 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.3977915346622467, + "learning_rate": 7.580371737159148e-05, + "loss": 0.008549144491553307, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00859, + "step": 195, + "tokens/total": 6223120, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 89129 + }, + { + "epoch": 0.765625, + "grad_norm": 0.3875303566455841, + "learning_rate": 7.554587923561324e-05, + "loss": 0.006131038535386324, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00615, + "step": 196, + "tokens/total": 6254880, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 89567 + }, + { + "epoch": 0.76953125, + "grad_norm": 3.4133481979370117, + "learning_rate": 7.528718610173511e-05, + "loss": 0.01836150512099266, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01853, + "step": 197, + "tokens/total": 6286960, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 89988 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.2509997487068176, + "learning_rate": 7.502764873523431e-05, + "loss": 0.003821226069703698, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00383, + "step": 198, + "tokens/total": 6318880, + "tokens/train_per_sec_per_gpu": 30.29, + "tokens/trainable": 90434 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.9550195932388306, + "learning_rate": 7.476727793652011e-05, + "loss": 0.006872436031699181, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0069, + "step": 199, + "tokens/total": 6350928, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 90913 + }, + { + "epoch": 0.78125, + "grad_norm": 0.411724716424942, + "learning_rate": 7.450608454068415e-05, + "loss": 0.014243390411138535, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01435, + "step": 200, + "tokens/total": 6382976, + "tokens/train_per_sec_per_gpu": 28.29, + "tokens/trainable": 91355 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.34264692664146423, + "learning_rate": 7.424407941704987e-05, + "loss": 0.005309835076332092, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00532, + "step": 201, + "tokens/total": 6415120, + "tokens/train_per_sec_per_gpu": 31.41, + "tokens/trainable": 91856 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.41955795884132385, + "learning_rate": 7.398127346871986e-05, + "loss": 0.009196512401103973, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00924, + "step": 202, + "tokens/total": 6447216, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 92311 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.03107454814016819, + "learning_rate": 7.371767763212238e-05, + "loss": 0.0004908143309876323, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00049, + "step": 203, + "tokens/total": 6479280, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 92764 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39578482508659363, + "learning_rate": 7.345330287655617e-05, + "loss": 0.006089594680815935, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00611, + "step": 204, + "tokens/total": 6511040, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 93227 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.6444572806358337, + "learning_rate": 7.31881602037339e-05, + "loss": 0.010252210311591625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0103, + "step": 205, + "tokens/total": 6543120, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 93675 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.3283638060092926, + "learning_rate": 7.29222606473245e-05, + "loss": 0.00761406822130084, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00764, + "step": 206, + "tokens/total": 6575520, + "tokens/train_per_sec_per_gpu": 27.02, + "tokens/trainable": 94120 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.09952464699745178, + "learning_rate": 7.265561527249383e-05, + "loss": 0.001628706930205226, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00163, + "step": 207, + "tokens/total": 6607408, + "tokens/train_per_sec_per_gpu": 29.43, + "tokens/trainable": 94557 + }, + { + "epoch": 0.8125, + "grad_norm": 0.31704938411712646, + "learning_rate": 7.238823517544436e-05, + "loss": 0.005896817892789841, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00591, + "step": 208, + "tokens/total": 6639488, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 95035 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.0999862551689148, + "learning_rate": 7.212013148295333e-05, + "loss": 0.001754446537233889, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00176, + "step": 209, + "tokens/total": 6671520, + "tokens/train_per_sec_per_gpu": 31.08, + "tokens/trainable": 95517 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.1405109018087387, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0017940793186426163, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0018, + "step": 210, + "tokens/total": 6703536, + "tokens/train_per_sec_per_gpu": 30.3, + "tokens/trainable": 96026 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.09185238927602768, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0015102234901860356, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00151, + "step": 211, + "tokens/total": 6735616, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 96477 + }, + { + "epoch": 0.828125, + "grad_norm": 1.0464473962783813, + "learning_rate": 7.131159054949273e-05, + "loss": 0.016731251031160355, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01687, + "step": 212, + "tokens/total": 6767584, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 96951 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.3517152965068817, + "learning_rate": 7.104070433827139e-05, + "loss": 0.007984626106917858, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00802, + "step": 213, + "tokens/total": 6799488, + "tokens/train_per_sec_per_gpu": 25.83, + "tokens/trainable": 97350 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.28192469477653503, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00557567086070776, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00559, + "step": 214, + "tokens/total": 6831552, + "tokens/train_per_sec_per_gpu": 24.66, + "tokens/trainable": 97792 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.46135279536247253, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0170141588896513, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.01716, + "step": 215, + "tokens/total": 6863488, + "tokens/train_per_sec_per_gpu": 30.38, + "tokens/trainable": 98257 + }, + { + "epoch": 0.84375, + "grad_norm": 0.4277898073196411, + "learning_rate": 7.022408581865382e-05, + "loss": 0.009399959817528725, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.71, + "memory/max_allocated (GiB)": 36.71, + "ppl": 1.00944, + "step": 216, + "tokens/total": 6895888, + "tokens/train_per_sec_per_gpu": 27.28, + "tokens/trainable": 98731 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.34988704323768616, + "learning_rate": 6.99505974422157e-05, + "loss": 0.002909669652581215, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00291, + "step": 217, + "tokens/total": 6928000, + "tokens/train_per_sec_per_gpu": 30.25, + "tokens/trainable": 99212 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.4028005599975586, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0058855339884757996, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0059, + "step": 218, + "tokens/total": 6960096, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 99654 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.3000575304031372, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0025230362080037594, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00253, + "step": 219, + "tokens/total": 6992512, + "tokens/train_per_sec_per_gpu": 27.46, + "tokens/trainable": 100086 + }, + { + "epoch": 0.859375, + "grad_norm": 0.11351172626018524, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0021392193157225847, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00214, + "step": 220, + "tokens/total": 7024688, + "tokens/train_per_sec_per_gpu": 27.56, + "tokens/trainable": 100524 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.19512486457824707, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00650001410394907, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00652, + "step": 221, + "tokens/total": 7056928, + "tokens/train_per_sec_per_gpu": 28.05, + "tokens/trainable": 100996 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.1998104453086853, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0056140003725886345, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00563, + "step": 222, + "tokens/total": 7088944, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 101449 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.40908998250961304, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01286369375884533, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.01295, + "step": 223, + "tokens/total": 7121104, + "tokens/train_per_sec_per_gpu": 27.19, + "tokens/trainable": 101905 + }, + { + "epoch": 0.875, + "grad_norm": 0.3672473132610321, + "learning_rate": 6.801939899284132e-05, + "loss": 0.010677668265998363, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01073, + "step": 224, + "tokens/total": 7153040, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 102411 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4146183431148529, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0038945709820836782, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0039, + "step": 225, + "tokens/total": 7184912, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 102882 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.10585323721170425, + "learning_rate": 6.746257910210214e-05, + "loss": 0.00041726604104042053, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00042, + "step": 226, + "tokens/total": 7216992, + "tokens/train_per_sec_per_gpu": 28.72, + "tokens/trainable": 103332 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.36600011587142944, + "learning_rate": 6.718338543014937e-05, + "loss": 0.01526213251054287, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 227, + "tokens/total": 7249120, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 103790 + }, + { + "epoch": 0.890625, + "grad_norm": 0.16057318449020386, + "learning_rate": 6.69036847577983e-05, + "loss": 0.004024423658847809, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00403, + "step": 228, + "tokens/total": 7281072, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 104246 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.318798303604126, + "learning_rate": 6.662348872453553e-05, + "loss": 0.006663155741989613, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00669, + "step": 229, + "tokens/total": 7313120, + "tokens/train_per_sec_per_gpu": 31.3, + "tokens/trainable": 104707 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.10858234763145447, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0020512123592197895, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00205, + "step": 230, + "tokens/total": 7345184, + "tokens/train_per_sec_per_gpu": 30.81, + "tokens/trainable": 105167 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.1044008731842041, + "learning_rate": 6.60616572358065e-05, + "loss": 0.0010264476295560598, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00103, + "step": 231, + "tokens/total": 7377264, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 105576 + }, + { + "epoch": 0.90625, + "grad_norm": 0.14836947619915009, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0013845522189512849, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00139, + "step": 232, + "tokens/total": 7409424, + "tokens/train_per_sec_per_gpu": 30.32, + "tokens/trainable": 106043 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.022752461954951286, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0003245974367018789, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00032, + "step": 233, + "tokens/total": 7441456, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 106506 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.22981758415699005, + "learning_rate": 6.521548694236384e-05, + "loss": 0.002587020630016923, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00259, + "step": 234, + "tokens/total": 7473344, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 106951 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.5147526264190674, + "learning_rate": 6.493256429322259e-05, + "loss": 0.01256126631051302, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01264, + "step": 235, + "tokens/total": 7505040, + "tokens/train_per_sec_per_gpu": 25.64, + "tokens/trainable": 107382 + }, + { + "epoch": 0.921875, + "grad_norm": 0.3785130977630615, + "learning_rate": 6.464922830953799e-05, + "loss": 0.010956985875964165, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01102, + "step": 236, + "tokens/total": 7537120, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 107814 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.6193199753761292, + "learning_rate": 6.436549078207688e-05, + "loss": 0.020655140280723572, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02087, + "step": 237, + "tokens/total": 7569024, + "tokens/train_per_sec_per_gpu": 29.36, + "tokens/trainable": 108274 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.5084519982337952, + "learning_rate": 6.408136351831592e-05, + "loss": 0.011346567422151566, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01141, + "step": 238, + "tokens/total": 7600992, + "tokens/train_per_sec_per_gpu": 24.43, + "tokens/trainable": 108714 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.04367341846227646, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0004316020058467984, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00043, + "step": 239, + "tokens/total": 7633280, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 109220 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06729783862829208, + "learning_rate": 6.351198709240186e-05, + "loss": 0.000604260538239032, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.0006, + "step": 240, + "tokens/total": 7665408, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 109672 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.3010976314544678, + "learning_rate": 6.32267616243259e-05, + "loss": 0.006106264889240265, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00612, + "step": 241, + "tokens/total": 7697280, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 110135 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.11426915228366852, + "learning_rate": 6.294119380711849e-05, + "loss": 0.0014383264351636171, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00144, + "step": 242, + "tokens/total": 7729440, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 110563 + }, + { + "epoch": 0.94921875, + "grad_norm": 1.3792941570281982, + "learning_rate": 6.265529552442209e-05, + "loss": 0.003547664964571595, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00355, + "step": 243, + "tokens/total": 7761584, + "tokens/train_per_sec_per_gpu": 30.76, + "tokens/trainable": 111049 + }, + { + "epoch": 0.953125, + "grad_norm": 0.07595008611679077, + "learning_rate": 6.236907867363127e-05, + "loss": 0.0009103374904952943, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00091, + "step": 244, + "tokens/total": 7793824, + "tokens/train_per_sec_per_gpu": 28.23, + "tokens/trainable": 111495 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05425361171364784, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0013249842450022697, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00133, + "step": 245, + "tokens/total": 7825872, + "tokens/train_per_sec_per_gpu": 26.47, + "tokens/trainable": 111968 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.17112742364406586, + "learning_rate": 6.179573692313344e-05, + "loss": 0.001098912674933672, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.0011, + "step": 246, + "tokens/total": 7857840, + "tokens/train_per_sec_per_gpu": 30.09, + "tokens/trainable": 112416 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.10754083096981049, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0031517897732555866, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00316, + "step": 247, + "tokens/total": 7890000, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 112882 + }, + { + "epoch": 0.96875, + "grad_norm": 0.2477901428937912, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005755824968218803, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00577, + "step": 248, + "tokens/total": 7921984, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 113390 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.332253634929657, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.013207933865487576, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.0133, + "step": 249, + "tokens/total": 7953824, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 113904 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.18514207005500793, + "learning_rate": 6.064575550087316e-05, + "loss": 0.00523729994893074, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00525, + "step": 250, + "tokens/total": 7985984, + "tokens/train_per_sec_per_gpu": 28.41, + "tokens/trainable": 114351 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.1852743923664093, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0031474479474127293, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00315, + "step": 251, + "tokens/total": 8018256, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 114842 + }, + { + "epoch": 0.984375, + "grad_norm": 0.43020468950271606, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0053521618247032166, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00537, + "step": 252, + "tokens/total": 8050528, + "tokens/train_per_sec_per_gpu": 23.97, + "tokens/trainable": 115263 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.048135045915842056, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0005759480991400778, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00058, + "step": 253, + "tokens/total": 8082496, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 115703 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.028733640909194946, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00040354474913328886, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0004, + "step": 254, + "tokens/total": 8114592, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 116157 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.2808869481086731, + "learning_rate": 5.920308275189541e-05, + "loss": 0.006315852981060743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00634, + "step": 255, + "tokens/total": 8146752, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 116567 + }, + { + "epoch": 1.0, + "grad_norm": 0.054996367543935776, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0009099108865484595, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00091, + "step": 256, + "tokens/total": 8178768, + "tokens/train_per_sec_per_gpu": 25.98, + "tokens/trainable": 117030 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.18114623427391052, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.002796629210934043, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0028, + "step": 257, + "tokens/total": 8210800, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 117517 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.1408185362815857, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0013210453325882554, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00132, + "step": 258, + "tokens/total": 8242768, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 118015 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.36587440967559814, + "learning_rate": 5.80457242789548e-05, + "loss": 0.002160525880753994, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00216, + "step": 259, + "tokens/total": 8274784, + "tokens/train_per_sec_per_gpu": 30.13, + "tokens/trainable": 118520 + }, + { + "epoch": 1.015625, + "grad_norm": 0.02628348581492901, + "learning_rate": 5.77560376810977e-05, + "loss": 0.0003391164937056601, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00034, + "step": 260, + "tokens/total": 8306720, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 118992 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.014996240846812725, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0002124913444276899, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00021, + "step": 261, + "tokens/total": 8338784, + "tokens/train_per_sec_per_gpu": 27.96, + "tokens/trainable": 119438 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.0050656464882195, + "learning_rate": 5.717633247526522e-05, + "loss": 7.557802018709481e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00008, + "step": 262, + "tokens/total": 8370720, + "tokens/train_per_sec_per_gpu": 26.9, + "tokens/trainable": 119881 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.02459302730858326, + "learning_rate": 5.688633799118971e-05, + "loss": 0.00024908874183893204, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00025, + "step": 263, + "tokens/total": 8402736, + "tokens/train_per_sec_per_gpu": 23.2, + "tokens/trainable": 120285 + }, + { + "epoch": 1.03125, + "grad_norm": 0.017813201993703842, + "learning_rate": 5.659626500889066e-05, + "loss": 0.00021895825921092182, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00022, + "step": 264, + "tokens/total": 8434352, + "tokens/train_per_sec_per_gpu": 27.78, + "tokens/trainable": 120755 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.0038785531651228666, + "learning_rate": 5.6306125599488905e-05, + "loss": 6.556476728292182e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00007, + "step": 265, + "tokens/total": 8466208, + "tokens/train_per_sec_per_gpu": 29.86, + "tokens/trainable": 121194 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.18033447861671448, + "learning_rate": 5.601593183686955e-05, + "loss": 0.0016840343596413732, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00169, + "step": 266, + "tokens/total": 8498304, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 121670 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.22497256100177765, + "learning_rate": 5.572569579717961e-05, + "loss": 0.003026450052857399, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00303, + "step": 267, + "tokens/total": 8530400, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 122142 + }, + { + "epoch": 1.046875, + "grad_norm": 0.008878038264811039, + "learning_rate": 5.543542955832538e-05, + "loss": 8.136438555084169e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00008, + "step": 268, + "tokens/total": 8562608, + "tokens/train_per_sec_per_gpu": 27.27, + "tokens/trainable": 122585 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.19042901694774628, + "learning_rate": 5.514514519946986e-05, + "loss": 0.0020417363848537207, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00204, + "step": 269, + "tokens/total": 8594832, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 123091 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.003382593160495162, + "learning_rate": 5.485485480053015e-05, + "loss": 3.870048021781258e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00004, + "step": 270, + "tokens/total": 8627120, + "tokens/train_per_sec_per_gpu": 26.36, + "tokens/trainable": 123505 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.2004430592060089, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.002532299840822816, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.00254, + "step": 271, + "tokens/total": 8657056, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 123953 + }, + { + "epoch": 1.0625, + "grad_norm": 0.24857956171035767, + "learning_rate": 5.42743042028204e-05, + "loss": 0.004157304298132658, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00417, + "step": 272, + "tokens/total": 8689104, + "tokens/train_per_sec_per_gpu": 24.05, + "tokens/trainable": 124400 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.05064955726265907, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0006531125982291996, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00065, + "step": 273, + "tokens/total": 8721136, + "tokens/train_per_sec_per_gpu": 29.3, + "tokens/trainable": 124870 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.12664659321308136, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0018034171080216765, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00181, + "step": 274, + "tokens/total": 8753200, + "tokens/train_per_sec_per_gpu": 26.52, + "tokens/trainable": 125333 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.01593201234936714, + "learning_rate": 5.340373499110935e-05, + "loss": 0.00010935973114101216, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00011, + "step": 275, + "tokens/total": 8785168, + "tokens/train_per_sec_per_gpu": 30.45, + "tokens/trainable": 125834 + }, + { + "epoch": 1.078125, + "grad_norm": 0.0015056057600304484, + "learning_rate": 5.3113662008810304e-05, + "loss": 1.9860939573845826e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00002, + "step": 276, + "tokens/total": 8817520, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 126316 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.004457728937268257, + "learning_rate": 5.282366752473479e-05, + "loss": 2.716448398132343e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00003, + "step": 277, + "tokens/total": 8849360, + "tokens/train_per_sec_per_gpu": 29.63, + "tokens/trainable": 126798 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.03812727704644203, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.0003402164438739419, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.00034, + "step": 278, + "tokens/total": 8881152, + "tokens/train_per_sec_per_gpu": 24.5, + "tokens/trainable": 127223 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.011409996077418327, + "learning_rate": 5.224396231890232e-05, + "loss": 9.064963523996994e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.00009, + "step": 279, + "tokens/total": 8912512, + "tokens/train_per_sec_per_gpu": 29.78, + "tokens/trainable": 127672 + }, + { + "epoch": 1.09375, + "grad_norm": 0.037030577659606934, + "learning_rate": 5.195427572104522e-05, + "loss": 0.0003734154161065817, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00037, + "step": 280, + "tokens/total": 8944416, + "tokens/train_per_sec_per_gpu": 26.76, + "tokens/trainable": 128116 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.0032268385402858257, + "learning_rate": 5.166471586820751e-05, + "loss": 2.529611811041832e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00003, + "step": 281, + "tokens/total": 8976272, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 128589 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.20227248966693878, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.0024647137615829706, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00247, + "step": 282, + "tokens/total": 9008560, + "tokens/train_per_sec_per_gpu": 27.84, + "tokens/trainable": 129036 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.43708595633506775, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0028322634752839804, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00284, + "step": 283, + "tokens/total": 9040496, + "tokens/train_per_sec_per_gpu": 29.25, + "tokens/trainable": 129485 + }, + { + "epoch": 1.109375, + "grad_norm": 0.09467080235481262, + "learning_rate": 5.079691724810461e-05, + "loss": 0.001156509853899479, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00116, + "step": 284, + "tokens/total": 9072448, + "tokens/train_per_sec_per_gpu": 24.69, + "tokens/trainable": 129920 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.1778785139322281, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.0016857109731063247, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00169, + "step": 285, + "tokens/total": 9104384, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 130417 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.0003376641252543777, + "learning_rate": 5.021923930849237e-05, + "loss": 5.597693871095544e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00001, + "step": 286, + "tokens/total": 9136416, + "tokens/train_per_sec_per_gpu": 29.99, + "tokens/trainable": 130903 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.019782084971666336, + "learning_rate": 4.99306927511967e-05, + "loss": 0.00012461614096537232, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00012, + "step": 287, + "tokens/total": 9168640, + "tokens/train_per_sec_per_gpu": 29.24, + "tokens/trainable": 131343 + }, + { + "epoch": 1.125, + "grad_norm": 0.9641065001487732, + "learning_rate": 4.964235714846775e-05, + "loss": 0.011515076272189617, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01158, + "step": 288, + "tokens/total": 9200544, + "tokens/train_per_sec_per_gpu": 27.04, + "tokens/trainable": 131763 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0010062052169814706, + "learning_rate": 4.9354244499126866e-05, + "loss": 9.489545846008696e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00001, + "step": 289, + "tokens/total": 9232624, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 132229 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.002373398980125785, + "learning_rate": 4.90663667927174e-05, + "loss": 1.573777262819931e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00002, + "step": 290, + "tokens/total": 9264768, + "tokens/train_per_sec_per_gpu": 29.5, + "tokens/trainable": 132678 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.0014718567254021764, + "learning_rate": 4.877873600900581e-05, + "loss": 9.888429303828161e-06, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00001, + "step": 291, + "tokens/total": 9296944, + "tokens/train_per_sec_per_gpu": 24.47, + "tokens/trainable": 133136 + }, + { + "epoch": 1.140625, + "grad_norm": 0.0018440725980326533, + "learning_rate": 4.849136411748306e-05, + "loss": 8.275845175376162e-06, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00001, + "step": 292, + "tokens/total": 9328848, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 133608 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.12918472290039062, + "learning_rate": 4.8204263076866574e-05, + "loss": 0.0009128568926826119, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00091, + "step": 293, + "tokens/total": 9360928, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 134108 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.019195031374692917, + "learning_rate": 4.791744483460251e-05, + "loss": 9.439548011869192e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00009, + "step": 294, + "tokens/total": 9393040, + "tokens/train_per_sec_per_gpu": 25.71, + "tokens/trainable": 134541 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.1602204144001007, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0004640200058929622, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00046, + "step": 295, + "tokens/total": 9425248, + "tokens/train_per_sec_per_gpu": 24.5, + "tokens/trainable": 134966 + }, + { + "epoch": 1.15625, + "grad_norm": 0.5406288504600525, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.0034592358861118555, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00347, + "step": 296, + "tokens/total": 9457264, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 135402 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.004333238583058119, + "learning_rate": 4.705880619288153e-05, + "loss": 3.075868880841881e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00003, + "step": 297, + "tokens/total": 9489056, + "tokens/train_per_sec_per_gpu": 24.87, + "tokens/trainable": 135804 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.006247827783226967, + "learning_rate": 4.677323837567412e-05, + "loss": 2.383375249337405e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00002, + "step": 298, + "tokens/total": 9521024, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 136231 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.0779612734913826, + "learning_rate": 4.6488012907598146e-05, + "loss": 0.0004679379053413868, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00047, + "step": 299, + "tokens/total": 9553120, + "tokens/train_per_sec_per_gpu": 28.37, + "tokens/trainable": 136705 + }, + { + "epoch": 1.171875, + "grad_norm": 0.01816786825656891, + "learning_rate": 4.620314165804964e-05, + "loss": 6.801338167861104e-05, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00007, + "step": 300, + "tokens/total": 9585040, + "tokens/train_per_sec_per_gpu": 28.78, + "tokens/trainable": 137178 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.9530329704284668, + "learning_rate": 4.591863648168407e-05, + "loss": 0.008843549527227879, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00888, + "step": 301, + "tokens/total": 9617120, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 137643 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0371168851852417, + "learning_rate": 4.5634509217923135e-05, + "loss": 0.00014635953994002193, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00015, + "step": 302, + "tokens/total": 9649024, + "tokens/train_per_sec_per_gpu": 26.74, + "tokens/trainable": 138078 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.08589725196361542, + "learning_rate": 4.535077169046201e-05, + "loss": 0.00043714733328670263, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00044, + "step": 303, + "tokens/total": 9680832, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 138515 + }, + { + "epoch": 1.1875, + "grad_norm": 0.358940988779068, + "learning_rate": 4.506743570677743e-05, + "loss": 0.006240838672965765, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00626, + "step": 304, + "tokens/total": 9713040, + "tokens/train_per_sec_per_gpu": 29.29, + "tokens/trainable": 138948 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.12659867107868195, + "learning_rate": 4.478451305763618e-05, + "loss": 0.0008667556685395539, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00087, + "step": 305, + "tokens/total": 9744976, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 139408 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.08289908617734909, + "learning_rate": 4.450201551660454e-05, + "loss": 0.0006384386215358973, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.02, + "memory/max_allocated (GiB)": 36.02, + "ppl": 1.00064, + "step": 306, + "tokens/total": 9774784, + "tokens/train_per_sec_per_gpu": 27.76, + "tokens/trainable": 139840 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.19195082783699036, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.002342985011637211, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00235, + "step": 307, + "tokens/total": 9806656, + "tokens/train_per_sec_per_gpu": 27.6, + "tokens/trainable": 140281 + }, + { + "epoch": 1.203125, + "grad_norm": 0.07421132177114487, + "learning_rate": 4.393834276419352e-05, + "loss": 0.0007085398538038135, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00071, + "step": 308, + "tokens/total": 9838720, + "tokens/train_per_sec_per_gpu": 30.88, + "tokens/trainable": 140776 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.004596009384840727, + "learning_rate": 4.36571910095382e-05, + "loss": 2.8969257982680574e-05, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00003, + "step": 309, + "tokens/total": 9870784, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 141228 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.04099060595035553, + "learning_rate": 4.337651127546448e-05, + "loss": 0.00030985785997472703, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00031, + "step": 310, + "tokens/total": 9902784, + "tokens/train_per_sec_per_gpu": 30.8, + "tokens/trainable": 141679 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.18840794265270233, + "learning_rate": 4.3096315242201736e-05, + "loss": 0.0020186181645840406, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00202, + "step": 311, + "tokens/total": 9934768, + "tokens/train_per_sec_per_gpu": 26.72, + "tokens/trainable": 142122 + }, + { + "epoch": 1.21875, + "grad_norm": 0.5018841028213501, + "learning_rate": 4.2816614569850635e-05, + "loss": 0.006148474756628275, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00617, + "step": 312, + "tokens/total": 9966960, + "tokens/train_per_sec_per_gpu": 27.27, + "tokens/trainable": 142561 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.1730002760887146, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.003579454030841589, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00359, + "step": 313, + "tokens/total": 9999104, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 143046 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.6586350202560425, + "learning_rate": 4.225874584473174e-05, + "loss": 0.0015231478027999401, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00152, + "step": 314, + "tokens/total": 10031136, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 143493 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.011105087585747242, + "learning_rate": 4.19806010071587e-05, + "loss": 0.00011838486534543335, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00012, + "step": 315, + "tokens/total": 10063184, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 143956 + }, + { + "epoch": 1.234375, + "grad_norm": 0.2628834843635559, + "learning_rate": 4.170299795992081e-05, + "loss": 0.003402995876967907, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00341, + "step": 316, + "tokens/total": 10095312, + "tokens/train_per_sec_per_gpu": 28.29, + "tokens/trainable": 144411 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.13821958005428314, + "learning_rate": 4.142594825521398e-05, + "loss": 0.0016696923412382603, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00167, + "step": 317, + "tokens/total": 10127440, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 144892 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.0023418583441525698, + "learning_rate": 4.114946342220728e-05, + "loss": 2.6753859856398776e-05, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00003, + "step": 318, + "tokens/total": 10159184, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 145339 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.0008433983894065022, + "learning_rate": 4.087355496656321e-05, + "loss": 1.602486736373976e-05, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00002, + "step": 319, + "tokens/total": 10191216, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 145811 + }, + { + "epoch": 1.25, + "grad_norm": 0.12588869035243988, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0011584729654714465, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00116, + "step": 320, + "tokens/total": 10223408, + "tokens/train_per_sec_per_gpu": 32.27, + "tokens/trainable": 146314 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.028395529836416245, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.00015757072833366692, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00016, + "step": 321, + "tokens/total": 10255536, + "tokens/train_per_sec_per_gpu": 26.68, + "tokens/trainable": 146781 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.039599668234586716, + "learning_rate": 4.004940255778431e-05, + "loss": 0.000321136845741421, + "memory/device_reserved (GiB)": 37.39, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00032, + "step": 322, + "tokens/total": 10287664, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 147225 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.11097095906734467, + "learning_rate": 3.977591418134619e-05, + "loss": 0.0004549310833681375, + "memory/device_reserved (GiB)": 37.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00046, + "step": 323, + "tokens/total": 10320080, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 147673 + }, + { + "epoch": 1.265625, + "grad_norm": 0.19018623232841492, + "learning_rate": 3.95030593412612e-05, + "loss": 0.0028336881659924984, + "memory/device_reserved (GiB)": 38.53, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00284, + "step": 324, + "tokens/total": 10351984, + "tokens/train_per_sec_per_gpu": 27.33, + "tokens/trainable": 148103 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.4650349020957947, + "learning_rate": 3.923084939213296e-05, + "loss": 0.006978687364608049, + "memory/device_reserved (GiB)": 38.53, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.007, + "step": 325, + "tokens/total": 10384256, + "tokens/train_per_sec_per_gpu": 26.21, + "tokens/trainable": 148535 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.14811205863952637, + "learning_rate": 3.895929566172861e-05, + "loss": 0.0014966176822781563, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0015, + "step": 326, + "tokens/total": 10416208, + "tokens/train_per_sec_per_gpu": 28.37, + "tokens/trainable": 148999 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.11302439123392105, + "learning_rate": 3.868840945050728e-05, + "loss": 0.0011318891774863005, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00113, + "step": 327, + "tokens/total": 10448224, + "tokens/train_per_sec_per_gpu": 26.37, + "tokens/trainable": 149431 + }, + { + "epoch": 1.28125, + "grad_norm": 0.10916705429553986, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0009605163359083235, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00096, + "step": 328, + "tokens/total": 10480448, + "tokens/train_per_sec_per_gpu": 28.27, + "tokens/trainable": 149886 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.0004018457839265466, + "learning_rate": 3.814868464809027e-05, + "loss": 7.040077434794512e-06, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00001, + "step": 329, + "tokens/total": 10512096, + "tokens/train_per_sec_per_gpu": 25.92, + "tokens/trainable": 150288 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.03717898577451706, + "learning_rate": 3.787986851704667e-05, + "loss": 0.0001680658315308392, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00017, + "step": 330, + "tokens/total": 10544000, + "tokens/train_per_sec_per_gpu": 28.52, + "tokens/trainable": 150733 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.21975846588611603, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00015534088015556335, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00016, + "step": 331, + "tokens/total": 10576048, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 151207 + }, + { + "epoch": 1.296875, + "grad_norm": 0.10715529322624207, + "learning_rate": 3.734438472750619e-05, + "loss": 0.0012046336196362972, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00121, + "step": 332, + "tokens/total": 10608048, + "tokens/train_per_sec_per_gpu": 31.08, + "tokens/trainable": 151694 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.13602837920188904, + "learning_rate": 3.707773935267552e-05, + "loss": 0.00039335948531515896, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00039, + "step": 333, + "tokens/total": 10640176, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 152188 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.002963080769404769, + "learning_rate": 3.68118397962661e-05, + "loss": 6.6289721871726215e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00001, + "step": 334, + "tokens/total": 10672224, + "tokens/train_per_sec_per_gpu": 25.55, + "tokens/trainable": 152596 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.007236327510327101, + "learning_rate": 3.654669712344384e-05, + "loss": 5.341085125110112e-05, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00005, + "step": 335, + "tokens/total": 10704192, + "tokens/train_per_sec_per_gpu": 30.44, + "tokens/trainable": 153052 + }, + { + "epoch": 1.3125, + "grad_norm": 0.0897810310125351, + "learning_rate": 3.628232236787763e-05, + "loss": 0.0008687702938914299, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00087, + "step": 336, + "tokens/total": 10736272, + "tokens/train_per_sec_per_gpu": 25.33, + "tokens/trainable": 153491 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.011829929426312447, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.00011201453162357211, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00011, + "step": 337, + "tokens/total": 10768528, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 153983 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.25218063592910767, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0022679665125906467, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00227, + "step": 338, + "tokens/total": 10800464, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 154464 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.00034703267738223076, + "learning_rate": 3.549391545931585e-05, + "loss": 2.884747800635523e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.0, + "step": 339, + "tokens/total": 10832512, + "tokens/train_per_sec_per_gpu": 29.7, + "tokens/trainable": 154924 + }, + { + "epoch": 1.328125, + "grad_norm": 0.004890776239335537, + "learning_rate": 3.5232722063479914e-05, + "loss": 2.2697908207192086e-05, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00002, + "step": 340, + "tokens/total": 10864624, + "tokens/train_per_sec_per_gpu": 25.46, + "tokens/trainable": 155373 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.0013245025184005499, + "learning_rate": 3.49723512647657e-05, + "loss": 8.980642633105163e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00001, + "step": 341, + "tokens/total": 10896816, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 155873 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.03804129734635353, + "learning_rate": 3.471281389826491e-05, + "loss": 0.00021880728309042752, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00022, + "step": 342, + "tokens/total": 10928864, + "tokens/train_per_sec_per_gpu": 25.31, + "tokens/trainable": 156278 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.4880214333534241, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.005397590808570385, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00541, + "step": 343, + "tokens/total": 10960816, + "tokens/train_per_sec_per_gpu": 29.58, + "tokens/trainable": 156733 + }, + { + "epoch": 1.34375, + "grad_norm": 0.37780871987342834, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0029298821464180946, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00293, + "step": 344, + "tokens/total": 10992912, + "tokens/train_per_sec_per_gpu": 28.4, + "tokens/trainable": 157203 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.0034261250402778387, + "learning_rate": 3.3939310220027456e-05, + "loss": 2.1256186300888658e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00002, + "step": 345, + "tokens/total": 11024832, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 157707 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.0005507867899723351, + "learning_rate": 3.3683214232914404e-05, + "loss": 6.991718692006543e-06, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00001, + "step": 346, + "tokens/total": 11056992, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 158180 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.0016383324982598424, + "learning_rate": 3.342800532426873e-05, + "loss": 1.2630136552616023e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00001, + "step": 347, + "tokens/total": 11089264, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 158646 + }, + { + "epoch": 1.359375, + "grad_norm": 0.0020912184845656157, + "learning_rate": 3.317369411437484e-05, + "loss": 1.536883064545691e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00002, + "step": 348, + "tokens/total": 11120976, + "tokens/train_per_sec_per_gpu": 29.34, + "tokens/trainable": 159115 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.0016828920925036073, + "learning_rate": 3.292029118616024e-05, + "loss": 1.334734679403482e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00001, + "step": 349, + "tokens/total": 11152976, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 159538 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.02520265430212021, + "learning_rate": 3.266780708475511e-05, + "loss": 9.595962183084339e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0001, + "step": 350, + "tokens/total": 11185056, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 160016 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.009836713783442974, + "learning_rate": 3.241625231705354e-05, + "loss": 7.521865336457267e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00008, + "step": 351, + "tokens/total": 11217088, + "tokens/train_per_sec_per_gpu": 28.51, + "tokens/trainable": 160475 + }, + { + "epoch": 1.375, + "grad_norm": 0.0698060467839241, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0003666624252218753, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00037, + "step": 352, + "tokens/total": 11249088, + "tokens/train_per_sec_per_gpu": 28.31, + "tokens/trainable": 160952 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 7.635413393505055e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/training_args.bin b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..dec74f9f304b44cbe47132d86c4e6285e659f40b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-352/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634 +size 8273 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/README.md b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ba677470ac9c29f9770a4aacb65c644ebd671a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/deconf_wave_deconf_control/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/deconf_wave_deconf_control/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/adapter_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..21fcb5b00be777c20a2d2dcceffc539dc51679b4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/deconf_wave_deconf_control/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "gate_proj", + "v_proj", + "k_proj", + "up_proj", + "q_proj", + "o_proj", + "down_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/adapter_model.safetensors b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..49630b7b9aa852b48313cee1872837ad951c8d46 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:834bd4e38c4009c9984ea7f7a0279929d85e4e7899319c059884d20f0350e555 +size 547777976 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/chat_template.jinja b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/optimizer.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..d34306d895d623f04b8879d2a5d6174846ce391b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ffe030ff67dde34c1b5c21d0aff24f4f9e864cd5ef85a774b40a5522c7ea0e9a +size 1048106435 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/rng_state.pth b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..b3292f047c4e6d2821bd568a4f217393be32016c --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6dda0d951907a9f38dc914f0c69c278d37e12d59f369e206aeccea593ef56f8e +size 14645 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/scheduler.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..6c676299e10f0344839a2b08f1cc1e004410cac0 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8908527ed67c1624f630ff35c9d5ed61abd3040f0eae468424d1042bb8809a79 +size 1465 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/tokenizer.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/tokenizer_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/tokens_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..39784035936074f0dcb32a7d706e82ae6f224d3d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/tokens_state.json @@ -0,0 +1 @@ +{"total": 12271600, "trainable": 175586} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/trainer_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..e7327a944810016b853e01aff7d37cd046846f62 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/trainer_state.json @@ -0,0 +1,5410 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.5, + "eval_steps": 500, + "global_step": 384, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.2140090465545654, + "learning_rate": 0.0, + "loss": 0.16441112756729126, + "memory/device_reserved (GiB)": 38.08, + "memory/max_active (GiB)": 35.64, + "memory/max_allocated (GiB)": 35.64, + "ppl": 1.1787, + "step": 1, + "tokens/total": 32160, + "tokens/train_per_sec_per_gpu": 21.28, + "tokens/trainable": 470 + }, + { + "epoch": 0.0078125, + "grad_norm": 2.92620849609375, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15774038434028625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.17086, + "step": 2, + "tokens/total": 64192, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 922 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.135961890220642, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1664382815361023, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.18109, + "step": 3, + "tokens/total": 96224, + "tokens/train_per_sec_per_gpu": 28.54, + "tokens/trainable": 1396 + }, + { + "epoch": 0.015625, + "grad_norm": 0.880176305770874, + "learning_rate": 1.2e-05, + "loss": 0.15334013104438782, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.16572, + "step": 4, + "tokens/total": 128336, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 1850 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.1433289051055908, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.16352537274360657, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.17766, + "step": 5, + "tokens/total": 160784, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 2302 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9728212952613831, + "learning_rate": 2e-05, + "loss": 0.15144473314285278, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.16351, + "step": 6, + "tokens/total": 192624, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 2742 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.7796363830566406, + "learning_rate": 2.4e-05, + "loss": 0.13553467392921448, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.14515, + "step": 7, + "tokens/total": 224608, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 3208 + }, + { + "epoch": 0.03125, + "grad_norm": 1.2444697618484497, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.1077304258942604, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.11375, + "step": 8, + "tokens/total": 256592, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 3655 + }, + { + "epoch": 0.03515625, + "grad_norm": 2.7339375019073486, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.08772765100002289, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.09169, + "step": 9, + "tokens/total": 286400, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 4091 + }, + { + "epoch": 0.0390625, + "grad_norm": 4.109770774841309, + "learning_rate": 3.6e-05, + "loss": 0.130940243601799, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.1399, + "step": 10, + "tokens/total": 318352, + "tokens/train_per_sec_per_gpu": 29.13, + "tokens/trainable": 4553 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.3768608570098877, + "learning_rate": 4e-05, + "loss": 0.10059958696365356, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.10583, + "step": 11, + "tokens/total": 350336, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 4992 + }, + { + "epoch": 0.046875, + "grad_norm": 3.281402826309204, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.12574875354766846, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.134, + "step": 12, + "tokens/total": 382448, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 5494 + }, + { + "epoch": 0.05078125, + "grad_norm": 1.7738980054855347, + "learning_rate": 4.8e-05, + "loss": 0.033296722918748856, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03386, + "step": 13, + "tokens/total": 414352, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 5933 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.64290189743042, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06366822123527527, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.06574, + "step": 14, + "tokens/total": 446400, + "tokens/train_per_sec_per_gpu": 24.8, + "tokens/trainable": 6369 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.3691928386688232, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.08168518543243408, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.08511, + "step": 15, + "tokens/total": 478688, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 6820 + }, + { + "epoch": 0.0625, + "grad_norm": 2.4356393814086914, + "learning_rate": 6e-05, + "loss": 0.09031196683645248, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.09452, + "step": 16, + "tokens/total": 510512, + "tokens/train_per_sec_per_gpu": 25.7, + "tokens/trainable": 7258 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.55682373046875, + "learning_rate": 6.400000000000001e-05, + "loss": 0.07496573030948639, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.07785, + "step": 17, + "tokens/total": 542736, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 7710 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.1054646968841553, + "learning_rate": 6.800000000000001e-05, + "loss": 0.05218805745244026, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.05357, + "step": 18, + "tokens/total": 574832, + "tokens/train_per_sec_per_gpu": 25.78, + "tokens/trainable": 8174 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.7429099082946777, + "learning_rate": 7.2e-05, + "loss": 0.03790827840566635, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.03864, + "step": 19, + "tokens/total": 606976, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 8617 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9508209228515625, + "learning_rate": 7.6e-05, + "loss": 0.04993613809347153, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0512, + "step": 20, + "tokens/total": 639200, + "tokens/train_per_sec_per_gpu": 24.91, + "tokens/trainable": 9037 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.0297844409942627, + "learning_rate": 8e-05, + "loss": 0.017908263951539993, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.01807, + "step": 21, + "tokens/total": 671072, + "tokens/train_per_sec_per_gpu": 30.42, + "tokens/trainable": 9503 + }, + { + "epoch": 0.0859375, + "grad_norm": 0.9279879331588745, + "learning_rate": 8.4e-05, + "loss": 0.013478193432092667, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01357, + "step": 22, + "tokens/total": 703136, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 9972 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.526488184928894, + "learning_rate": 8.800000000000001e-05, + "loss": 0.017310388386249542, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01746, + "step": 23, + "tokens/total": 735120, + "tokens/train_per_sec_per_gpu": 31.45, + "tokens/trainable": 10447 + }, + { + "epoch": 0.09375, + "grad_norm": 3.4049417972564697, + "learning_rate": 9.200000000000001e-05, + "loss": 0.0353429913520813, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.03597, + "step": 24, + "tokens/total": 767040, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 10899 + }, + { + "epoch": 0.09765625, + "grad_norm": 7.511656761169434, + "learning_rate": 9.6e-05, + "loss": 0.1335875540971756, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.14292, + "step": 25, + "tokens/total": 798976, + "tokens/train_per_sec_per_gpu": 27.72, + "tokens/trainable": 11360 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.0069234371185303, + "learning_rate": 0.0001, + "loss": 0.04382415860891342, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0448, + "step": 26, + "tokens/total": 830880, + "tokens/train_per_sec_per_gpu": 24.61, + "tokens/trainable": 11775 + }, + { + "epoch": 0.10546875, + "grad_norm": 7.807938575744629, + "learning_rate": 9.99990636831587e-05, + "loss": 0.01210833340883255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.01218, + "step": 27, + "tokens/total": 860960, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 12242 + }, + { + "epoch": 0.109375, + "grad_norm": 2.23823881149292, + "learning_rate": 9.999625477159879e-05, + "loss": 0.0272970050573349, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02767, + "step": 28, + "tokens/total": 892816, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 12696 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.315584421157837, + "learning_rate": 9.999157338221051e-05, + "loss": 0.025516577064990997, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.02584, + "step": 29, + "tokens/total": 924736, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 13153 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.8413651585578918, + "learning_rate": 9.998501970980562e-05, + "loss": 0.007838520221412182, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00787, + "step": 30, + "tokens/total": 954592, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 13624 + }, + { + "epoch": 0.12109375, + "grad_norm": 1.6797157526016235, + "learning_rate": 9.997659402710915e-05, + "loss": 0.03002442792057991, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.03048, + "step": 31, + "tokens/total": 986592, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 14064 + }, + { + "epoch": 0.125, + "grad_norm": 1.3320434093475342, + "learning_rate": 9.996629668474818e-05, + "loss": 0.03772728145122528, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.03845, + "step": 32, + "tokens/total": 1018720, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 14565 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.6335167288780212, + "learning_rate": 9.995412811123711e-05, + "loss": 0.013970119878649712, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01407, + "step": 33, + "tokens/total": 1050624, + "tokens/train_per_sec_per_gpu": 28.18, + "tokens/trainable": 15005 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.5626921653747559, + "learning_rate": 9.994008881295999e-05, + "loss": 0.016654539853334427, + "memory/device_reserved (GiB)": 37.74, + "memory/max_active (GiB)": 36.22, + "memory/max_allocated (GiB)": 36.22, + "ppl": 1.01679, + "step": 34, + "tokens/total": 1082176, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 15459 + }, + { + "epoch": 0.13671875, + "grad_norm": 0.603219211101532, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021408583968877792, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02164, + "step": 35, + "tokens/total": 1113968, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 15960 + }, + { + "epoch": 0.140625, + "grad_norm": 0.9779674410820007, + "learning_rate": 9.99064004568618e-05, + "loss": 0.03277068957686424, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03331, + "step": 36, + "tokens/total": 1145952, + "tokens/train_per_sec_per_gpu": 28.68, + "tokens/trainable": 16428 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.8855255842208862, + "learning_rate": 9.988675280095074e-05, + "loss": 0.017048098146915436, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01719, + "step": 37, + "tokens/total": 1177872, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 16884 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.9266122579574585, + "learning_rate": 9.986523722403528e-05, + "loss": 0.045863546431064606, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.04693, + "step": 38, + "tokens/total": 1209776, + "tokens/train_per_sec_per_gpu": 26.91, + "tokens/trainable": 17341 + }, + { + "epoch": 0.15234375, + "grad_norm": 0.8280895948410034, + "learning_rate": 9.984185462146642e-05, + "loss": 0.033055759966373444, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03361, + "step": 39, + "tokens/total": 1241888, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 17786 + }, + { + "epoch": 0.15625, + "grad_norm": 5.029730319976807, + "learning_rate": 9.98166059662897e-05, + "loss": 0.025589998811483383, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02592, + "step": 40, + "tokens/total": 1274032, + "tokens/train_per_sec_per_gpu": 29.22, + "tokens/trainable": 18262 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.5688032507896423, + "learning_rate": 9.978949230920472e-05, + "loss": 0.016095872968435287, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01623, + "step": 41, + "tokens/total": 1306000, + "tokens/train_per_sec_per_gpu": 27.49, + "tokens/trainable": 18716 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8217663168907166, + "learning_rate": 9.976051477852141e-05, + "loss": 0.026798982173204422, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02716, + "step": 42, + "tokens/total": 1337936, + "tokens/train_per_sec_per_gpu": 27.51, + "tokens/trainable": 19157 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.7309132218360901, + "learning_rate": 9.972967458011312e-05, + "loss": 0.033920761197805405, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0345, + "step": 43, + "tokens/total": 1369904, + "tokens/train_per_sec_per_gpu": 30.66, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.8134022355079651, + "learning_rate": 9.96969729973664e-05, + "loss": 0.024005085229873657, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0243, + "step": 44, + "tokens/total": 1401984, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 20119 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.35084816813468933, + "learning_rate": 9.966241139112754e-05, + "loss": 0.013319254852831364, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01341, + "step": 45, + "tokens/total": 1434032, + "tokens/train_per_sec_per_gpu": 27.45, + "tokens/trainable": 20560 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.8701573014259338, + "learning_rate": 9.96259911996461e-05, + "loss": 0.01356641948223114, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01366, + "step": 46, + "tokens/total": 1465872, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 20992 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.4406442642211914, + "learning_rate": 9.958771393851491e-05, + "loss": 0.04396980255842209, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 35.86, + "memory/max_allocated (GiB)": 35.86, + "ppl": 1.04495, + "step": 47, + "tokens/total": 1495440, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 21441 + }, + { + "epoch": 0.1875, + "grad_norm": 0.4966669976711273, + "learning_rate": 9.954758120060702e-05, + "loss": 0.011179964058101177, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01124, + "step": 48, + "tokens/total": 1527776, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 21901 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.41124436259269714, + "learning_rate": 9.950559465600948e-05, + "loss": 0.0063599650748074055, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00638, + "step": 49, + "tokens/total": 1559760, + "tokens/train_per_sec_per_gpu": 27.73, + "tokens/trainable": 22341 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.829739511013031, + "learning_rate": 9.946175605195379e-05, + "loss": 0.015785304829478264, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01591, + "step": 50, + "tokens/total": 1592000, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 22833 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5181736946105957, + "learning_rate": 9.941606721274322e-05, + "loss": 0.008393109776079655, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00843, + "step": 51, + "tokens/total": 1623920, + "tokens/train_per_sec_per_gpu": 25.61, + "tokens/trainable": 23277 + }, + { + "epoch": 0.203125, + "grad_norm": 1.5778990983963013, + "learning_rate": 9.936853003967685e-05, + "loss": 0.024984844028949738, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0253, + "step": 52, + "tokens/total": 1656032, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 23759 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.4938714504241943, + "learning_rate": 9.93191465109705e-05, + "loss": 0.038523320108652115, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03927, + "step": 53, + "tokens/total": 1688368, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 24193 + }, + { + "epoch": 0.2109375, + "grad_norm": 2.1427106857299805, + "learning_rate": 9.926791868167438e-05, + "loss": 0.06302924454212189, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.06506, + "step": 54, + "tokens/total": 1720560, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 24619 + }, + { + "epoch": 0.21484375, + "grad_norm": 1.8365188837051392, + "learning_rate": 9.921484868358753e-05, + "loss": 0.052160464227199554, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.05354, + "step": 55, + "tokens/total": 1752464, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 25075 + }, + { + "epoch": 0.21875, + "grad_norm": 0.7779749631881714, + "learning_rate": 9.915993872516924e-05, + "loss": 0.021328264847397804, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.02156, + "step": 56, + "tokens/total": 1784336, + "tokens/train_per_sec_per_gpu": 26.42, + "tokens/trainable": 25519 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.4189571142196655, + "learning_rate": 9.9103191091447e-05, + "loss": 0.014279939234256744, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01438, + "step": 57, + "tokens/total": 1816304, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 25966 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.3834710717201233, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01883944310247898, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01902, + "step": 58, + "tokens/total": 1848304, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 26423 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.7693278789520264, + "learning_rate": 9.898419232046825e-05, + "loss": 0.03533369302749634, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03597, + "step": 59, + "tokens/total": 1880640, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 26934 + }, + { + "epoch": 0.234375, + "grad_norm": 0.4606446325778961, + "learning_rate": 9.892194613523633e-05, + "loss": 0.021482212468981743, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02171, + "step": 60, + "tokens/total": 1912592, + "tokens/train_per_sec_per_gpu": 27.64, + "tokens/trainable": 27393 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.46030017733573914, + "learning_rate": 9.885787217854357e-05, + "loss": 0.017044665291905403, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01719, + "step": 61, + "tokens/total": 1944544, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 27852 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.7918264269828796, + "learning_rate": 9.879197311676887e-05, + "loss": 0.016954131424427032, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0171, + "step": 62, + "tokens/total": 1976688, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 28292 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.4787926971912384, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01525965891778469, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 63, + "tokens/total": 2008928, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 28798 + }, + { + "epoch": 0.25, + "grad_norm": 0.9033918976783752, + "learning_rate": 9.865471072312528e-05, + "loss": 0.02902211993932724, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02945, + "step": 64, + "tokens/total": 2041120, + "tokens/train_per_sec_per_gpu": 29.83, + "tokens/trainable": 29283 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.39305707812309265, + "learning_rate": 9.858335310330492e-05, + "loss": 0.00901167094707489, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.32, + "memory/max_allocated (GiB)": 36.32, + "ppl": 1.00905, + "step": 65, + "tokens/total": 2073024, + "tokens/train_per_sec_per_gpu": 26.48, + "tokens/trainable": 29745 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4494345784187317, + "learning_rate": 9.851018180226185e-05, + "loss": 0.007426132448017597, + "memory/device_reserved (GiB)": 37.47, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00745, + "step": 66, + "tokens/total": 2105184, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 30186 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.629649817943573, + "learning_rate": 9.843519986495259e-05, + "loss": 0.010609081946313381, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01067, + "step": 67, + "tokens/total": 2137264, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 30622 + }, + { + "epoch": 0.265625, + "grad_norm": 0.7591162919998169, + "learning_rate": 9.835841041168162e-05, + "loss": 0.00926615484058857, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00931, + "step": 68, + "tokens/total": 2169424, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 31097 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.7288486361503601, + "learning_rate": 9.82798166379715e-05, + "loss": 0.024310583248734474, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.02461, + "step": 69, + "tokens/total": 2201424, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 31595 + }, + { + "epoch": 0.2734375, + "grad_norm": 2.134056329727173, + "learning_rate": 9.819942181443002e-05, + "loss": 0.008422967046499252, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00846, + "step": 70, + "tokens/total": 2233424, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 32036 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6362871527671814, + "learning_rate": 9.811722928661392e-05, + "loss": 0.01577262207865715, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0159, + "step": 71, + "tokens/total": 2265328, + "tokens/train_per_sec_per_gpu": 23.65, + "tokens/trainable": 32428 + }, + { + "epoch": 0.28125, + "grad_norm": 0.6091912388801575, + "learning_rate": 9.803324247488975e-05, + "loss": 0.0064800274558365345, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0065, + "step": 72, + "tokens/total": 2297248, + "tokens/train_per_sec_per_gpu": 26.86, + "tokens/trainable": 32880 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.14533838629722595, + "learning_rate": 9.794746487429161e-05, + "loss": 0.001991454279050231, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00199, + "step": 73, + "tokens/total": 2329152, + "tokens/train_per_sec_per_gpu": 27.97, + "tokens/trainable": 33323 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.5400826930999756, + "learning_rate": 9.785990005437554e-05, + "loss": 0.017662350088357925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01782, + "step": 74, + "tokens/total": 2361344, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 33792 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.6132436394691467, + "learning_rate": 9.777055165907117e-05, + "loss": 0.01787603087723255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01804, + "step": 75, + "tokens/total": 2393168, + "tokens/train_per_sec_per_gpu": 29.53, + "tokens/trainable": 34223 + }, + { + "epoch": 0.296875, + "grad_norm": 0.7287399172782898, + "learning_rate": 9.767942340652993e-05, + "loss": 0.024754449725151062, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.31, + "memory/max_allocated (GiB)": 36.31, + "ppl": 1.02506, + "step": 76, + "tokens/total": 2424880, + "tokens/train_per_sec_per_gpu": 24.92, + "tokens/trainable": 34649 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.6790545582771301, + "learning_rate": 9.758651908897035e-05, + "loss": 0.014308430254459381, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01441, + "step": 77, + "tokens/total": 2457120, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 35094 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6203456521034241, + "learning_rate": 9.749184257252033e-05, + "loss": 0.034353043884038925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03495, + "step": 78, + "tokens/total": 2489104, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 35534 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.6238825917243958, + "learning_rate": 9.739539779705614e-05, + "loss": 0.032295141369104385, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.03282, + "step": 79, + "tokens/total": 2521136, + "tokens/train_per_sec_per_gpu": 29.68, + "tokens/trainable": 36029 + }, + { + "epoch": 0.3125, + "grad_norm": 0.18977171182632446, + "learning_rate": 9.729718877603861e-05, + "loss": 0.007260813377797604, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00729, + "step": 80, + "tokens/total": 2553168, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 36469 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.8054049611091614, + "learning_rate": 9.719721959634592e-05, + "loss": 0.034894898533821106, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.03551, + "step": 81, + "tokens/total": 2585584, + "tokens/train_per_sec_per_gpu": 25.75, + "tokens/trainable": 36906 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.6884189248085022, + "learning_rate": 9.709549441810375e-05, + "loss": 0.018210047855973244, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.01838, + "step": 82, + "tokens/total": 2617808, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 37390 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.6490103602409363, + "learning_rate": 9.699201747451195e-05, + "loss": 0.02030857279896736, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.02052, + "step": 83, + "tokens/total": 2649664, + "tokens/train_per_sec_per_gpu": 25.8, + "tokens/trainable": 37838 + }, + { + "epoch": 0.328125, + "grad_norm": 0.2439805567264557, + "learning_rate": 9.688679307166854e-05, + "loss": 0.008518049493432045, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00855, + "step": 84, + "tokens/total": 2681728, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 38310 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6747375726699829, + "learning_rate": 9.677982558839042e-05, + "loss": 0.029895059764385223, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.03035, + "step": 85, + "tokens/total": 2713488, + "tokens/train_per_sec_per_gpu": 28.49, + "tokens/trainable": 38789 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.582666277885437, + "learning_rate": 9.66711194760312e-05, + "loss": 0.014998597092926502, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01511, + "step": 86, + "tokens/total": 2745536, + "tokens/train_per_sec_per_gpu": 30.27, + "tokens/trainable": 39277 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.5364654660224915, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011567167937755585, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01163, + "step": 87, + "tokens/total": 2777680, + "tokens/train_per_sec_per_gpu": 29.73, + "tokens/trainable": 39737 + }, + { + "epoch": 0.34375, + "grad_norm": 0.34024426341056824, + "learning_rate": 9.644850953105288e-05, + "loss": 0.01014852337539196, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0102, + "step": 88, + "tokens/total": 2809568, + "tokens/train_per_sec_per_gpu": 24.68, + "tokens/trainable": 40179 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.32472267746925354, + "learning_rate": 9.633461496214225e-05, + "loss": 0.01229981891810894, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01238, + "step": 89, + "tokens/total": 2841760, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 40649 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.7906899452209473, + "learning_rate": 9.621900029118195e-05, + "loss": 0.02552586793899536, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.02585, + "step": 90, + "tokens/total": 2871696, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 41080 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.7444764971733093, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0054441955871880054, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00546, + "step": 91, + "tokens/total": 2903968, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 41599 + }, + { + "epoch": 0.359375, + "grad_norm": 0.1911851465702057, + "learning_rate": 9.598262995928611e-05, + "loss": 0.004665314219892025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00468, + "step": 92, + "tokens/total": 2936048, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 42076 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.21238519251346588, + "learning_rate": 9.586188413468492e-05, + "loss": 0.003917238209396601, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00392, + "step": 93, + "tokens/total": 2968176, + "tokens/train_per_sec_per_gpu": 28.07, + "tokens/trainable": 42522 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.41729435324668884, + "learning_rate": 9.57394378802934e-05, + "loss": 0.013555807992815971, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01365, + "step": 94, + "tokens/total": 3000304, + "tokens/train_per_sec_per_gpu": 28.69, + "tokens/trainable": 42974 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.7246769070625305, + "learning_rate": 9.56152962916e-05, + "loss": 0.020221399143338203, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.02043, + "step": 95, + "tokens/total": 3032080, + "tokens/train_per_sec_per_gpu": 25.35, + "tokens/trainable": 43380 + }, + { + "epoch": 0.375, + "grad_norm": 0.16139821708202362, + "learning_rate": 9.548946453464296e-05, + "loss": 0.003853276837617159, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00386, + "step": 96, + "tokens/total": 3064320, + "tokens/train_per_sec_per_gpu": 28.47, + "tokens/trainable": 43865 + }, + { + "epoch": 0.37890625, + "grad_norm": 1.0509370565414429, + "learning_rate": 9.53619478457953e-05, + "loss": 0.015099998563528061, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01521, + "step": 97, + "tokens/total": 3096144, + "tokens/train_per_sec_per_gpu": 26.61, + "tokens/trainable": 44328 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.6859350204467773, + "learning_rate": 9.523275153154695e-05, + "loss": 0.004579808097332716, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00459, + "step": 98, + "tokens/total": 3127600, + "tokens/train_per_sec_per_gpu": 25.91, + "tokens/trainable": 44778 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5686795711517334, + "learning_rate": 9.51018809682839e-05, + "loss": 0.022520575672388077, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02278, + "step": 99, + "tokens/total": 3159520, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 45225 + }, + { + "epoch": 0.390625, + "grad_norm": 1.6308369636535645, + "learning_rate": 9.49693416020645e-05, + "loss": 0.011883002705872059, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01195, + "step": 100, + "tokens/total": 3191568, + "tokens/train_per_sec_per_gpu": 29.75, + "tokens/trainable": 45678 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.7186088562011719, + "learning_rate": 9.483513894839276e-05, + "loss": 0.0060632615350186825, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00608, + "step": 101, + "tokens/total": 3223648, + "tokens/train_per_sec_per_gpu": 27.14, + "tokens/trainable": 46125 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.12036337703466415, + "learning_rate": 9.469927859198888e-05, + "loss": 0.0019118499476462603, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00191, + "step": 102, + "tokens/total": 3255744, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 46612 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.29928353428840637, + "learning_rate": 9.456176618655689e-05, + "loss": 0.00944911316037178, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00949, + "step": 103, + "tokens/total": 3287792, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 47059 + }, + { + "epoch": 0.40625, + "grad_norm": 0.3766293227672577, + "learning_rate": 9.442260745454927e-05, + "loss": 0.008984047919511795, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.68, + "memory/max_allocated (GiB)": 36.68, + "ppl": 1.00902, + "step": 104, + "tokens/total": 3320016, + "tokens/train_per_sec_per_gpu": 28.92, + "tokens/trainable": 47536 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3481820523738861, + "learning_rate": 9.428180818692884e-05, + "loss": 0.005444999784231186, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00546, + "step": 105, + "tokens/total": 3351952, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 48037 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.8414323329925537, + "learning_rate": 9.413937424292791e-05, + "loss": 0.032438624650239944, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03297, + "step": 106, + "tokens/total": 3384032, + "tokens/train_per_sec_per_gpu": 27.11, + "tokens/trainable": 48491 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.5067655444145203, + "learning_rate": 9.399531154980424e-05, + "loss": 0.008442888967692852, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00848, + "step": 107, + "tokens/total": 3416192, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 48940 + }, + { + "epoch": 0.421875, + "grad_norm": 0.41617852449417114, + "learning_rate": 9.384962610259455e-05, + "loss": 0.008386321365833282, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00842, + "step": 108, + "tokens/total": 3448080, + "tokens/train_per_sec_per_gpu": 26.56, + "tokens/trainable": 49389 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1587515026330948, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0031663556583225727, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.03, + "memory/max_allocated (GiB)": 36.03, + "ppl": 1.00317, + "step": 109, + "tokens/total": 3477952, + "tokens/train_per_sec_per_gpu": 29.8, + "tokens/trainable": 49838 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2903148829936981, + "learning_rate": 9.355341126345868e-05, + "loss": 0.007958847098052502, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00799, + "step": 110, + "tokens/total": 3509776, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 50310 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.642983078956604, + "learning_rate": 9.340289419824107e-05, + "loss": 0.006748045329004526, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00677, + "step": 111, + "tokens/total": 3541808, + "tokens/train_per_sec_per_gpu": 28.04, + "tokens/trainable": 50755 + }, + { + "epoch": 0.4375, + "grad_norm": 0.14928282797336578, + "learning_rate": 9.325077903184159e-05, + "loss": 0.0027839094400405884, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00279, + "step": 112, + "tokens/total": 3574064, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 51213 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.05922335386276245, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0009622900979593396, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00096, + "step": 113, + "tokens/total": 3605984, + "tokens/train_per_sec_per_gpu": 27.99, + "tokens/trainable": 51660 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.44227731227874756, + "learning_rate": 9.2941779782269e-05, + "loss": 0.002763954224064946, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00277, + "step": 114, + "tokens/total": 3638080, + "tokens/train_per_sec_per_gpu": 27.03, + "tokens/trainable": 52133 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5176818370819092, + "learning_rate": 9.278490855781596e-05, + "loss": 0.015623578801751137, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01575, + "step": 115, + "tokens/total": 3670016, + "tokens/train_per_sec_per_gpu": 31.44, + "tokens/trainable": 52580 + }, + { + "epoch": 0.453125, + "grad_norm": 0.4265497922897339, + "learning_rate": 9.262646494908604e-05, + "loss": 0.00794076919555664, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00797, + "step": 116, + "tokens/total": 3701984, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 53033 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.40457987785339355, + "learning_rate": 9.246645554956457e-05, + "loss": 0.022708112373948097, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02297, + "step": 117, + "tokens/total": 3733920, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 53517 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.01414525043219328, + "learning_rate": 9.230488701789578e-05, + "loss": 0.00020209309877827764, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.0002, + "step": 118, + "tokens/total": 3765840, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 53967 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.11359831690788269, + "learning_rate": 9.214176607760577e-05, + "loss": 0.0012168455868959427, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00122, + "step": 119, + "tokens/total": 3797680, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 54430 + }, + { + "epoch": 0.46875, + "grad_norm": 0.35982534289360046, + "learning_rate": 9.197709951682268e-05, + "loss": 0.004110721405595541, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00412, + "step": 120, + "tokens/total": 3829920, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 54896 + }, + { + "epoch": 0.47265625, + "grad_norm": 1.3690382242202759, + "learning_rate": 9.181089418799428e-05, + "loss": 0.02705790475010872, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02743, + "step": 121, + "tokens/total": 3862080, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 55379 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.351891428232193, + "learning_rate": 9.164315700760271e-05, + "loss": 0.0020163573790341616, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00202, + "step": 122, + "tokens/total": 3893968, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 55803 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4747961461544037, + "learning_rate": 9.147389495587671e-05, + "loss": 0.012588979676365852, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01267, + "step": 123, + "tokens/total": 3926160, + "tokens/train_per_sec_per_gpu": 30.14, + "tokens/trainable": 56249 + }, + { + "epoch": 0.484375, + "grad_norm": 0.4064250588417053, + "learning_rate": 9.130311507650116e-05, + "loss": 0.0022941348142921925, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0023, + "step": 124, + "tokens/total": 3958208, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 56713 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.22056929767131805, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0022175831254571676, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00222, + "step": 125, + "tokens/total": 3990208, + "tokens/train_per_sec_per_gpu": 32.57, + "tokens/trainable": 57196 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5698862671852112, + "learning_rate": 9.09570303250602e-05, + "loss": 0.010205100290477276, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01026, + "step": 126, + "tokens/total": 4022224, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 57680 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.2593887448310852, + "learning_rate": 9.078173985499394e-05, + "loss": 0.0032277877908200026, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00323, + "step": 127, + "tokens/total": 4053952, + "tokens/train_per_sec_per_gpu": 26.51, + "tokens/trainable": 58110 + }, + { + "epoch": 0.5, + "grad_norm": 0.3735230565071106, + "learning_rate": 9.060496036067713e-05, + "loss": 0.01031105499714613, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01036, + "step": 128, + "tokens/total": 4086144, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 58534 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.1332484930753708, + "learning_rate": 9.042669919862615e-05, + "loss": 0.0012390739284455776, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00124, + "step": 129, + "tokens/total": 4118096, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 58998 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.20182718336582184, + "learning_rate": 9.024696378701557e-05, + "loss": 0.004501686431467533, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00451, + "step": 130, + "tokens/total": 4150272, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 59448 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.529080331325531, + "learning_rate": 9.006576160536948e-05, + "loss": 0.00763288140296936, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00766, + "step": 131, + "tokens/total": 4181904, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 59906 + }, + { + "epoch": 0.515625, + "grad_norm": 0.21625548601150513, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0024378676898777485, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00244, + "step": 132, + "tokens/total": 4214032, + "tokens/train_per_sec_per_gpu": 28.36, + "tokens/trainable": 60350 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.37116539478302, + "learning_rate": 8.969898715494506e-05, + "loss": 0.0037849934305995703, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00379, + "step": 133, + "tokens/total": 4246064, + "tokens/train_per_sec_per_gpu": 30.18, + "tokens/trainable": 60831 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5646623373031616, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006517104804515839, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00654, + "step": 134, + "tokens/total": 4278224, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 61305 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.6371963620185852, + "learning_rate": 8.932643689864568e-05, + "loss": 0.008687382563948631, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00873, + "step": 135, + "tokens/total": 4310112, + "tokens/train_per_sec_per_gpu": 31.48, + "tokens/trainable": 61792 + }, + { + "epoch": 0.53125, + "grad_norm": 0.8210524916648865, + "learning_rate": 8.913801518498845e-05, + "loss": 0.008436474949121475, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00847, + "step": 136, + "tokens/total": 4342192, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 62253 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.5692176222801208, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017321188002824783, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.01747, + "step": 137, + "tokens/total": 4374096, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.12828375399112701, + "learning_rate": 8.875691779131569e-05, + "loss": 0.0007754197577014565, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00078, + "step": 138, + "tokens/total": 4406272, + "tokens/train_per_sec_per_gpu": 24.22, + "tokens/trainable": 63145 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.3188190162181854, + "learning_rate": 8.856425797031829e-05, + "loss": 0.006707335356622934, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00673, + "step": 139, + "tokens/total": 4438112, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 63587 + }, + { + "epoch": 0.546875, + "grad_norm": 0.6216707229614258, + "learning_rate": 8.837020140354295e-05, + "loss": 0.007024695165455341, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00705, + "step": 140, + "tokens/total": 4470128, + "tokens/train_per_sec_per_gpu": 30.64, + "tokens/trainable": 64052 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.1480305939912796, + "learning_rate": 8.817475616647554e-05, + "loss": 0.00229115248657763, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00229, + "step": 141, + "tokens/total": 4502096, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 64526 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22254763543605804, + "learning_rate": 8.797793039239017e-05, + "loss": 0.0029198499396443367, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00292, + "step": 142, + "tokens/total": 4534496, + "tokens/train_per_sec_per_gpu": 29.08, + "tokens/trainable": 64983 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.6901941895484924, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013051149435341358, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01314, + "step": 143, + "tokens/total": 4566720, + "tokens/train_per_sec_per_gpu": 31.56, + "tokens/trainable": 65492 + }, + { + "epoch": 0.5625, + "grad_norm": 0.9545458555221558, + "learning_rate": 8.758017005316988e-05, + "loss": 0.02235381305217743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.02261, + "step": 144, + "tokens/total": 4598880, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 65925 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.5515892505645752, + "learning_rate": 8.737925204046629e-05, + "loss": 0.02038375660777092, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.02059, + "step": 145, + "tokens/total": 4630864, + "tokens/train_per_sec_per_gpu": 26.3, + "tokens/trainable": 66383 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3512430191040039, + "learning_rate": 8.717698659491851e-05, + "loss": 0.008233271539211273, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00827, + "step": 146, + "tokens/total": 4662864, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 66840 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.6830361485481262, + "learning_rate": 8.697338213361735e-05, + "loss": 0.013419180177152157, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01351, + "step": 147, + "tokens/total": 4694912, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 67297 + }, + { + "epoch": 0.578125, + "grad_norm": 0.2117031067609787, + "learning_rate": 8.676844712937552e-05, + "loss": 0.004134457092732191, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00414, + "step": 148, + "tokens/total": 4727120, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 67769 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.378053218126297, + "learning_rate": 8.656219011037509e-05, + "loss": 0.009747720323503017, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0098, + "step": 149, + "tokens/total": 4759168, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 68257 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.21076741814613342, + "learning_rate": 8.63546196598125e-05, + "loss": 0.005035653710365295, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00505, + "step": 150, + "tokens/total": 4791008, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 68706 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.25075697898864746, + "learning_rate": 8.614574441554145e-05, + "loss": 0.006027051247656345, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00605, + "step": 151, + "tokens/total": 4822912, + "tokens/train_per_sec_per_gpu": 27.95, + "tokens/trainable": 69131 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5774204730987549, + "learning_rate": 8.593557306971349e-05, + "loss": 0.018617186695337296, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01879, + "step": 152, + "tokens/total": 4854816, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 69586 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.1364293098449707, + "learning_rate": 8.572411436841618e-05, + "loss": 0.002049289643764496, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00205, + "step": 153, + "tokens/total": 4886656, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 70061 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.25387004017829895, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0044304742477834225, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00444, + "step": 154, + "tokens/total": 4918768, + "tokens/train_per_sec_per_gpu": 23.26, + "tokens/trainable": 70467 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3047125041484833, + "learning_rate": 8.529737015125824e-05, + "loss": 0.0043389927595853806, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00435, + "step": 155, + "tokens/total": 4951008, + "tokens/train_per_sec_per_gpu": 27.53, + "tokens/trainable": 70933 + }, + { + "epoch": 0.609375, + "grad_norm": 0.1442601978778839, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0031973645091056824, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.0032, + "step": 156, + "tokens/total": 4982976, + "tokens/train_per_sec_per_gpu": 28.86, + "tokens/trainable": 71382 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.2608455717563629, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0032951058819890022, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0033, + "step": 157, + "tokens/total": 5014640, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 71808 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.637295663356781, + "learning_rate": 8.464782037243449e-05, + "loss": 0.01436957623809576, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.01447, + "step": 158, + "tokens/total": 5046672, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 72249 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.9834056496620178, + "learning_rate": 8.442882418044202e-05, + "loss": 0.02375311404466629, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.02404, + "step": 159, + "tokens/total": 5078528, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 72726 + }, + { + "epoch": 0.625, + "grad_norm": 0.35432615876197815, + "learning_rate": 8.420860333495179e-05, + "loss": 0.006119017023593187, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00614, + "step": 160, + "tokens/total": 5110704, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 73148 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.44270065426826477, + "learning_rate": 8.398716700025208e-05, + "loss": 0.01352179329842329, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.72, + "memory/max_allocated (GiB)": 36.72, + "ppl": 1.01361, + "step": 161, + "tokens/total": 5142944, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 73600 + }, + { + "epoch": 0.6328125, + "grad_norm": 1.2181634902954102, + "learning_rate": 8.376452439121266e-05, + "loss": 0.01820964366197586, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.01838, + "step": 162, + "tokens/total": 5174592, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 74068 + }, + { + "epoch": 0.63671875, + "grad_norm": 1.9321303367614746, + "learning_rate": 8.354068477290124e-05, + "loss": 0.009742327965795994, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00979, + "step": 163, + "tokens/total": 5206432, + "tokens/train_per_sec_per_gpu": 29.89, + "tokens/trainable": 74527 + }, + { + "epoch": 0.640625, + "grad_norm": 0.3545796275138855, + "learning_rate": 8.331565746019807e-05, + "loss": 0.00975855067372322, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00981, + "step": 164, + "tokens/total": 5238384, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 74992 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.35368478298187256, + "learning_rate": 8.308945181740812e-05, + "loss": 0.010195466689765453, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01025, + "step": 165, + "tokens/total": 5270272, + "tokens/train_per_sec_per_gpu": 29.41, + "tokens/trainable": 75442 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.34957799315452576, + "learning_rate": 8.286207725787153e-05, + "loss": 0.009373681619763374, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00942, + "step": 166, + "tokens/total": 5302368, + "tokens/train_per_sec_per_gpu": 26.2, + "tokens/trainable": 75887 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.3684624135494232, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008211556822061539, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00825, + "step": 167, + "tokens/total": 5334496, + "tokens/train_per_sec_per_gpu": 28.21, + "tokens/trainable": 76331 + }, + { + "epoch": 0.65625, + "grad_norm": 0.12826858460903168, + "learning_rate": 8.240385928474219e-05, + "loss": 0.004492453299462795, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0045, + "step": 168, + "tokens/total": 5364288, + "tokens/train_per_sec_per_gpu": 30.16, + "tokens/trainable": 76745 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.31828173995018005, + "learning_rate": 8.217303493946967e-05, + "loss": 0.005733514670282602, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00575, + "step": 169, + "tokens/total": 5396512, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 77201 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6174353361129761, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01915649324655533, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01934, + "step": 170, + "tokens/total": 5428624, + "tokens/train_per_sec_per_gpu": 27.75, + "tokens/trainable": 77655 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.2902314066886902, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0036087254993617535, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00362, + "step": 171, + "tokens/total": 5460880, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 78122 + }, + { + "epoch": 0.671875, + "grad_norm": 0.3267754912376404, + "learning_rate": 8.147381587530713e-05, + "loss": 0.008090222254395485, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00812, + "step": 172, + "tokens/total": 5493088, + "tokens/train_per_sec_per_gpu": 27.86, + "tokens/trainable": 78576 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.4086189866065979, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006351984106004238, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00637, + "step": 173, + "tokens/total": 5525312, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 79059 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5261490345001221, + "learning_rate": 8.100214524900103e-05, + "loss": 0.008662862703204155, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.0087, + "step": 174, + "tokens/total": 5556880, + "tokens/train_per_sec_per_gpu": 24.62, + "tokens/trainable": 79498 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11292761564254761, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0027105510234832764, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00271, + "step": 175, + "tokens/total": 5588976, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 79922 + }, + { + "epoch": 0.6875, + "grad_norm": 0.33315548300743103, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0030926030594855547, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.0031, + "step": 176, + "tokens/total": 5618992, + "tokens/train_per_sec_per_gpu": 29.51, + "tokens/trainable": 80383 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.3136921226978302, + "learning_rate": 8.028654871074489e-05, + "loss": 0.005057850852608681, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00507, + "step": 177, + "tokens/total": 5651072, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 80824 + }, + { + "epoch": 0.6953125, + "grad_norm": 1.0526509284973145, + "learning_rate": 8.004589869885986e-05, + "loss": 0.010707498528063297, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01077, + "step": 178, + "tokens/total": 5683296, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 81243 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.1329648196697235, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0017975447699427605, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.0018, + "step": 179, + "tokens/total": 5715520, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 81716 + }, + { + "epoch": 0.703125, + "grad_norm": 1.0749201774597168, + "learning_rate": 7.95614819466576e-05, + "loss": 0.025617048144340515, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02595, + "step": 180, + "tokens/total": 5747488, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 82181 + }, + { + "epoch": 0.70703125, + "grad_norm": 1.0446465015411377, + "learning_rate": 7.931773536489872e-05, + "loss": 0.030236585065722466, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 35.9, + "memory/max_allocated (GiB)": 35.9, + "ppl": 1.0307, + "step": 181, + "tokens/total": 5777264, + "tokens/train_per_sec_per_gpu": 28.89, + "tokens/trainable": 82626 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22853964567184448, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0038214433006942272, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00383, + "step": 182, + "tokens/total": 5809264, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 83089 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.057821910828351974, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0004995397757738829, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0005, + "step": 183, + "tokens/total": 5841488, + "tokens/train_per_sec_per_gpu": 29.34, + "tokens/trainable": 83590 + }, + { + "epoch": 0.71875, + "grad_norm": 1.3264167308807373, + "learning_rate": 7.85804646415409e-05, + "loss": 0.013232000172138214, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01332, + "step": 184, + "tokens/total": 5873696, + "tokens/train_per_sec_per_gpu": 29.33, + "tokens/trainable": 84046 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.6027824878692627, + "learning_rate": 7.833273149760207e-05, + "loss": 0.012215660884976387, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01229, + "step": 185, + "tokens/total": 5905408, + "tokens/train_per_sec_per_gpu": 28.25, + "tokens/trainable": 84517 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.5848041772842407, + "learning_rate": 7.808402738346527e-05, + "loss": 0.024627480655908585, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.02493, + "step": 186, + "tokens/total": 5937392, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 84974 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.6823275089263916, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0028813849203288555, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00289, + "step": 187, + "tokens/total": 5969344, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 85460 + }, + { + "epoch": 0.734375, + "grad_norm": 0.41855958104133606, + "learning_rate": 7.758374768294647e-05, + "loss": 0.009796380065381527, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00984, + "step": 188, + "tokens/total": 6001296, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 85939 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6849080324172974, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004100777208805084, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00411, + "step": 189, + "tokens/total": 6031216, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 86377 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.15838374197483063, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003986812196671963, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00399, + "step": 190, + "tokens/total": 6063296, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 86834 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.3117486536502838, + "learning_rate": 7.682630588562518e-05, + "loss": 0.009132719598710537, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00917, + "step": 191, + "tokens/total": 6095216, + "tokens/train_per_sec_per_gpu": 26.65, + "tokens/trainable": 87265 + }, + { + "epoch": 0.75, + "grad_norm": 0.7837114334106445, + "learning_rate": 7.657199467573129e-05, + "loss": 0.007239446043968201, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00727, + "step": 192, + "tokens/total": 6127488, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 87747 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21777178347110748, + "learning_rate": 7.631678576708561e-05, + "loss": 0.002932134782895446, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00294, + "step": 193, + "tokens/total": 6159488, + "tokens/train_per_sec_per_gpu": 30.37, + "tokens/trainable": 88239 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.19159242510795593, + "learning_rate": 7.606068977997255e-05, + "loss": 0.003356864908710122, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00336, + "step": 194, + "tokens/total": 6191296, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 88718 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.3977915346622467, + "learning_rate": 7.580371737159148e-05, + "loss": 0.008549144491553307, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00859, + "step": 195, + "tokens/total": 6223120, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 89129 + }, + { + "epoch": 0.765625, + "grad_norm": 0.3875303566455841, + "learning_rate": 7.554587923561324e-05, + "loss": 0.006131038535386324, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00615, + "step": 196, + "tokens/total": 6254880, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 89567 + }, + { + "epoch": 0.76953125, + "grad_norm": 3.4133481979370117, + "learning_rate": 7.528718610173511e-05, + "loss": 0.01836150512099266, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01853, + "step": 197, + "tokens/total": 6286960, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 89988 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.2509997487068176, + "learning_rate": 7.502764873523431e-05, + "loss": 0.003821226069703698, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00383, + "step": 198, + "tokens/total": 6318880, + "tokens/train_per_sec_per_gpu": 30.29, + "tokens/trainable": 90434 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.9550195932388306, + "learning_rate": 7.476727793652011e-05, + "loss": 0.006872436031699181, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0069, + "step": 199, + "tokens/total": 6350928, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 90913 + }, + { + "epoch": 0.78125, + "grad_norm": 0.411724716424942, + "learning_rate": 7.450608454068415e-05, + "loss": 0.014243390411138535, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01435, + "step": 200, + "tokens/total": 6382976, + "tokens/train_per_sec_per_gpu": 28.29, + "tokens/trainable": 91355 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.34264692664146423, + "learning_rate": 7.424407941704987e-05, + "loss": 0.005309835076332092, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00532, + "step": 201, + "tokens/total": 6415120, + "tokens/train_per_sec_per_gpu": 31.41, + "tokens/trainable": 91856 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.41955795884132385, + "learning_rate": 7.398127346871986e-05, + "loss": 0.009196512401103973, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00924, + "step": 202, + "tokens/total": 6447216, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 92311 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.03107454814016819, + "learning_rate": 7.371767763212238e-05, + "loss": 0.0004908143309876323, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00049, + "step": 203, + "tokens/total": 6479280, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 92764 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39578482508659363, + "learning_rate": 7.345330287655617e-05, + "loss": 0.006089594680815935, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00611, + "step": 204, + "tokens/total": 6511040, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 93227 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.6444572806358337, + "learning_rate": 7.31881602037339e-05, + "loss": 0.010252210311591625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0103, + "step": 205, + "tokens/total": 6543120, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 93675 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.3283638060092926, + "learning_rate": 7.29222606473245e-05, + "loss": 0.00761406822130084, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00764, + "step": 206, + "tokens/total": 6575520, + "tokens/train_per_sec_per_gpu": 27.02, + "tokens/trainable": 94120 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.09952464699745178, + "learning_rate": 7.265561527249383e-05, + "loss": 0.001628706930205226, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00163, + "step": 207, + "tokens/total": 6607408, + "tokens/train_per_sec_per_gpu": 29.43, + "tokens/trainable": 94557 + }, + { + "epoch": 0.8125, + "grad_norm": 0.31704938411712646, + "learning_rate": 7.238823517544436e-05, + "loss": 0.005896817892789841, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00591, + "step": 208, + "tokens/total": 6639488, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 95035 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.0999862551689148, + "learning_rate": 7.212013148295333e-05, + "loss": 0.001754446537233889, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00176, + "step": 209, + "tokens/total": 6671520, + "tokens/train_per_sec_per_gpu": 31.08, + "tokens/trainable": 95517 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.1405109018087387, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0017940793186426163, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0018, + "step": 210, + "tokens/total": 6703536, + "tokens/train_per_sec_per_gpu": 30.3, + "tokens/trainable": 96026 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.09185238927602768, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0015102234901860356, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00151, + "step": 211, + "tokens/total": 6735616, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 96477 + }, + { + "epoch": 0.828125, + "grad_norm": 1.0464473962783813, + "learning_rate": 7.131159054949273e-05, + "loss": 0.016731251031160355, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01687, + "step": 212, + "tokens/total": 6767584, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 96951 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.3517152965068817, + "learning_rate": 7.104070433827139e-05, + "loss": 0.007984626106917858, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00802, + "step": 213, + "tokens/total": 6799488, + "tokens/train_per_sec_per_gpu": 25.83, + "tokens/trainable": 97350 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.28192469477653503, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00557567086070776, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00559, + "step": 214, + "tokens/total": 6831552, + "tokens/train_per_sec_per_gpu": 24.66, + "tokens/trainable": 97792 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.46135279536247253, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0170141588896513, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.01716, + "step": 215, + "tokens/total": 6863488, + "tokens/train_per_sec_per_gpu": 30.38, + "tokens/trainable": 98257 + }, + { + "epoch": 0.84375, + "grad_norm": 0.4277898073196411, + "learning_rate": 7.022408581865382e-05, + "loss": 0.009399959817528725, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.71, + "memory/max_allocated (GiB)": 36.71, + "ppl": 1.00944, + "step": 216, + "tokens/total": 6895888, + "tokens/train_per_sec_per_gpu": 27.28, + "tokens/trainable": 98731 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.34988704323768616, + "learning_rate": 6.99505974422157e-05, + "loss": 0.002909669652581215, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00291, + "step": 217, + "tokens/total": 6928000, + "tokens/train_per_sec_per_gpu": 30.25, + "tokens/trainable": 99212 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.4028005599975586, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0058855339884757996, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0059, + "step": 218, + "tokens/total": 6960096, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 99654 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.3000575304031372, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0025230362080037594, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00253, + "step": 219, + "tokens/total": 6992512, + "tokens/train_per_sec_per_gpu": 27.46, + "tokens/trainable": 100086 + }, + { + "epoch": 0.859375, + "grad_norm": 0.11351172626018524, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0021392193157225847, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00214, + "step": 220, + "tokens/total": 7024688, + "tokens/train_per_sec_per_gpu": 27.56, + "tokens/trainable": 100524 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.19512486457824707, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00650001410394907, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00652, + "step": 221, + "tokens/total": 7056928, + "tokens/train_per_sec_per_gpu": 28.05, + "tokens/trainable": 100996 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.1998104453086853, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0056140003725886345, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00563, + "step": 222, + "tokens/total": 7088944, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 101449 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.40908998250961304, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01286369375884533, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.01295, + "step": 223, + "tokens/total": 7121104, + "tokens/train_per_sec_per_gpu": 27.19, + "tokens/trainable": 101905 + }, + { + "epoch": 0.875, + "grad_norm": 0.3672473132610321, + "learning_rate": 6.801939899284132e-05, + "loss": 0.010677668265998363, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01073, + "step": 224, + "tokens/total": 7153040, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 102411 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4146183431148529, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0038945709820836782, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0039, + "step": 225, + "tokens/total": 7184912, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 102882 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.10585323721170425, + "learning_rate": 6.746257910210214e-05, + "loss": 0.00041726604104042053, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00042, + "step": 226, + "tokens/total": 7216992, + "tokens/train_per_sec_per_gpu": 28.72, + "tokens/trainable": 103332 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.36600011587142944, + "learning_rate": 6.718338543014937e-05, + "loss": 0.01526213251054287, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 227, + "tokens/total": 7249120, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 103790 + }, + { + "epoch": 0.890625, + "grad_norm": 0.16057318449020386, + "learning_rate": 6.69036847577983e-05, + "loss": 0.004024423658847809, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00403, + "step": 228, + "tokens/total": 7281072, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 104246 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.318798303604126, + "learning_rate": 6.662348872453553e-05, + "loss": 0.006663155741989613, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00669, + "step": 229, + "tokens/total": 7313120, + "tokens/train_per_sec_per_gpu": 31.3, + "tokens/trainable": 104707 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.10858234763145447, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0020512123592197895, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00205, + "step": 230, + "tokens/total": 7345184, + "tokens/train_per_sec_per_gpu": 30.81, + "tokens/trainable": 105167 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.1044008731842041, + "learning_rate": 6.60616572358065e-05, + "loss": 0.0010264476295560598, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00103, + "step": 231, + "tokens/total": 7377264, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 105576 + }, + { + "epoch": 0.90625, + "grad_norm": 0.14836947619915009, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0013845522189512849, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00139, + "step": 232, + "tokens/total": 7409424, + "tokens/train_per_sec_per_gpu": 30.32, + "tokens/trainable": 106043 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.022752461954951286, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0003245974367018789, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00032, + "step": 233, + "tokens/total": 7441456, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 106506 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.22981758415699005, + "learning_rate": 6.521548694236384e-05, + "loss": 0.002587020630016923, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00259, + "step": 234, + "tokens/total": 7473344, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 106951 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.5147526264190674, + "learning_rate": 6.493256429322259e-05, + "loss": 0.01256126631051302, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01264, + "step": 235, + "tokens/total": 7505040, + "tokens/train_per_sec_per_gpu": 25.64, + "tokens/trainable": 107382 + }, + { + "epoch": 0.921875, + "grad_norm": 0.3785130977630615, + "learning_rate": 6.464922830953799e-05, + "loss": 0.010956985875964165, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01102, + "step": 236, + "tokens/total": 7537120, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 107814 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.6193199753761292, + "learning_rate": 6.436549078207688e-05, + "loss": 0.020655140280723572, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02087, + "step": 237, + "tokens/total": 7569024, + "tokens/train_per_sec_per_gpu": 29.36, + "tokens/trainable": 108274 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.5084519982337952, + "learning_rate": 6.408136351831592e-05, + "loss": 0.011346567422151566, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01141, + "step": 238, + "tokens/total": 7600992, + "tokens/train_per_sec_per_gpu": 24.43, + "tokens/trainable": 108714 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.04367341846227646, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0004316020058467984, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00043, + "step": 239, + "tokens/total": 7633280, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 109220 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06729783862829208, + "learning_rate": 6.351198709240186e-05, + "loss": 0.000604260538239032, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.0006, + "step": 240, + "tokens/total": 7665408, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 109672 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.3010976314544678, + "learning_rate": 6.32267616243259e-05, + "loss": 0.006106264889240265, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00612, + "step": 241, + "tokens/total": 7697280, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 110135 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.11426915228366852, + "learning_rate": 6.294119380711849e-05, + "loss": 0.0014383264351636171, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00144, + "step": 242, + "tokens/total": 7729440, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 110563 + }, + { + "epoch": 0.94921875, + "grad_norm": 1.3792941570281982, + "learning_rate": 6.265529552442209e-05, + "loss": 0.003547664964571595, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00355, + "step": 243, + "tokens/total": 7761584, + "tokens/train_per_sec_per_gpu": 30.76, + "tokens/trainable": 111049 + }, + { + "epoch": 0.953125, + "grad_norm": 0.07595008611679077, + "learning_rate": 6.236907867363127e-05, + "loss": 0.0009103374904952943, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00091, + "step": 244, + "tokens/total": 7793824, + "tokens/train_per_sec_per_gpu": 28.23, + "tokens/trainable": 111495 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05425361171364784, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0013249842450022697, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00133, + "step": 245, + "tokens/total": 7825872, + "tokens/train_per_sec_per_gpu": 26.47, + "tokens/trainable": 111968 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.17112742364406586, + "learning_rate": 6.179573692313344e-05, + "loss": 0.001098912674933672, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.0011, + "step": 246, + "tokens/total": 7857840, + "tokens/train_per_sec_per_gpu": 30.09, + "tokens/trainable": 112416 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.10754083096981049, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0031517897732555866, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00316, + "step": 247, + "tokens/total": 7890000, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 112882 + }, + { + "epoch": 0.96875, + "grad_norm": 0.2477901428937912, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005755824968218803, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00577, + "step": 248, + "tokens/total": 7921984, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 113390 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.332253634929657, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.013207933865487576, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.0133, + "step": 249, + "tokens/total": 7953824, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 113904 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.18514207005500793, + "learning_rate": 6.064575550087316e-05, + "loss": 0.00523729994893074, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00525, + "step": 250, + "tokens/total": 7985984, + "tokens/train_per_sec_per_gpu": 28.41, + "tokens/trainable": 114351 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.1852743923664093, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0031474479474127293, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00315, + "step": 251, + "tokens/total": 8018256, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 114842 + }, + { + "epoch": 0.984375, + "grad_norm": 0.43020468950271606, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0053521618247032166, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00537, + "step": 252, + "tokens/total": 8050528, + "tokens/train_per_sec_per_gpu": 23.97, + "tokens/trainable": 115263 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.048135045915842056, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0005759480991400778, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00058, + "step": 253, + "tokens/total": 8082496, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 115703 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.028733640909194946, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00040354474913328886, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0004, + "step": 254, + "tokens/total": 8114592, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 116157 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.2808869481086731, + "learning_rate": 5.920308275189541e-05, + "loss": 0.006315852981060743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00634, + "step": 255, + "tokens/total": 8146752, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 116567 + }, + { + "epoch": 1.0, + "grad_norm": 0.054996367543935776, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0009099108865484595, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00091, + "step": 256, + "tokens/total": 8178768, + "tokens/train_per_sec_per_gpu": 25.98, + "tokens/trainable": 117030 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.18114623427391052, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.002796629210934043, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0028, + "step": 257, + "tokens/total": 8210800, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 117517 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.1408185362815857, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0013210453325882554, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00132, + "step": 258, + "tokens/total": 8242768, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 118015 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.36587440967559814, + "learning_rate": 5.80457242789548e-05, + "loss": 0.002160525880753994, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00216, + "step": 259, + "tokens/total": 8274784, + "tokens/train_per_sec_per_gpu": 30.13, + "tokens/trainable": 118520 + }, + { + "epoch": 1.015625, + "grad_norm": 0.02628348581492901, + "learning_rate": 5.77560376810977e-05, + "loss": 0.0003391164937056601, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00034, + "step": 260, + "tokens/total": 8306720, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 118992 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.014996240846812725, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0002124913444276899, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00021, + "step": 261, + "tokens/total": 8338784, + "tokens/train_per_sec_per_gpu": 27.96, + "tokens/trainable": 119438 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.0050656464882195, + "learning_rate": 5.717633247526522e-05, + "loss": 7.557802018709481e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00008, + "step": 262, + "tokens/total": 8370720, + "tokens/train_per_sec_per_gpu": 26.9, + "tokens/trainable": 119881 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.02459302730858326, + "learning_rate": 5.688633799118971e-05, + "loss": 0.00024908874183893204, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00025, + "step": 263, + "tokens/total": 8402736, + "tokens/train_per_sec_per_gpu": 23.2, + "tokens/trainable": 120285 + }, + { + "epoch": 1.03125, + "grad_norm": 0.017813201993703842, + "learning_rate": 5.659626500889066e-05, + "loss": 0.00021895825921092182, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00022, + "step": 264, + "tokens/total": 8434352, + "tokens/train_per_sec_per_gpu": 27.78, + "tokens/trainable": 120755 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.0038785531651228666, + "learning_rate": 5.6306125599488905e-05, + "loss": 6.556476728292182e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00007, + "step": 265, + "tokens/total": 8466208, + "tokens/train_per_sec_per_gpu": 29.86, + "tokens/trainable": 121194 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.18033447861671448, + "learning_rate": 5.601593183686955e-05, + "loss": 0.0016840343596413732, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00169, + "step": 266, + "tokens/total": 8498304, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 121670 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.22497256100177765, + "learning_rate": 5.572569579717961e-05, + "loss": 0.003026450052857399, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00303, + "step": 267, + "tokens/total": 8530400, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 122142 + }, + { + "epoch": 1.046875, + "grad_norm": 0.008878038264811039, + "learning_rate": 5.543542955832538e-05, + "loss": 8.136438555084169e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00008, + "step": 268, + "tokens/total": 8562608, + "tokens/train_per_sec_per_gpu": 27.27, + "tokens/trainable": 122585 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.19042901694774628, + "learning_rate": 5.514514519946986e-05, + "loss": 0.0020417363848537207, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00204, + "step": 269, + "tokens/total": 8594832, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 123091 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.003382593160495162, + "learning_rate": 5.485485480053015e-05, + "loss": 3.870048021781258e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00004, + "step": 270, + "tokens/total": 8627120, + "tokens/train_per_sec_per_gpu": 26.36, + "tokens/trainable": 123505 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.2004430592060089, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.002532299840822816, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.00254, + "step": 271, + "tokens/total": 8657056, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 123953 + }, + { + "epoch": 1.0625, + "grad_norm": 0.24857956171035767, + "learning_rate": 5.42743042028204e-05, + "loss": 0.004157304298132658, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00417, + "step": 272, + "tokens/total": 8689104, + "tokens/train_per_sec_per_gpu": 24.05, + "tokens/trainable": 124400 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.05064955726265907, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0006531125982291996, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00065, + "step": 273, + "tokens/total": 8721136, + "tokens/train_per_sec_per_gpu": 29.3, + "tokens/trainable": 124870 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.12664659321308136, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0018034171080216765, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00181, + "step": 274, + "tokens/total": 8753200, + "tokens/train_per_sec_per_gpu": 26.52, + "tokens/trainable": 125333 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.01593201234936714, + "learning_rate": 5.340373499110935e-05, + "loss": 0.00010935973114101216, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00011, + "step": 275, + "tokens/total": 8785168, + "tokens/train_per_sec_per_gpu": 30.45, + "tokens/trainable": 125834 + }, + { + "epoch": 1.078125, + "grad_norm": 0.0015056057600304484, + "learning_rate": 5.3113662008810304e-05, + "loss": 1.9860939573845826e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00002, + "step": 276, + "tokens/total": 8817520, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 126316 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.004457728937268257, + "learning_rate": 5.282366752473479e-05, + "loss": 2.716448398132343e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00003, + "step": 277, + "tokens/total": 8849360, + "tokens/train_per_sec_per_gpu": 29.63, + "tokens/trainable": 126798 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.03812727704644203, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.0003402164438739419, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.00034, + "step": 278, + "tokens/total": 8881152, + "tokens/train_per_sec_per_gpu": 24.5, + "tokens/trainable": 127223 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.011409996077418327, + "learning_rate": 5.224396231890232e-05, + "loss": 9.064963523996994e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.00009, + "step": 279, + "tokens/total": 8912512, + "tokens/train_per_sec_per_gpu": 29.78, + "tokens/trainable": 127672 + }, + { + "epoch": 1.09375, + "grad_norm": 0.037030577659606934, + "learning_rate": 5.195427572104522e-05, + "loss": 0.0003734154161065817, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00037, + "step": 280, + "tokens/total": 8944416, + "tokens/train_per_sec_per_gpu": 26.76, + "tokens/trainable": 128116 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.0032268385402858257, + "learning_rate": 5.166471586820751e-05, + "loss": 2.529611811041832e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00003, + "step": 281, + "tokens/total": 8976272, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 128589 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.20227248966693878, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.0024647137615829706, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00247, + "step": 282, + "tokens/total": 9008560, + "tokens/train_per_sec_per_gpu": 27.84, + "tokens/trainable": 129036 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.43708595633506775, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0028322634752839804, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00284, + "step": 283, + "tokens/total": 9040496, + "tokens/train_per_sec_per_gpu": 29.25, + "tokens/trainable": 129485 + }, + { + "epoch": 1.109375, + "grad_norm": 0.09467080235481262, + "learning_rate": 5.079691724810461e-05, + "loss": 0.001156509853899479, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00116, + "step": 284, + "tokens/total": 9072448, + "tokens/train_per_sec_per_gpu": 24.69, + "tokens/trainable": 129920 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.1778785139322281, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.0016857109731063247, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00169, + "step": 285, + "tokens/total": 9104384, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 130417 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.0003376641252543777, + "learning_rate": 5.021923930849237e-05, + "loss": 5.597693871095544e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00001, + "step": 286, + "tokens/total": 9136416, + "tokens/train_per_sec_per_gpu": 29.99, + "tokens/trainable": 130903 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.019782084971666336, + "learning_rate": 4.99306927511967e-05, + "loss": 0.00012461614096537232, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00012, + "step": 287, + "tokens/total": 9168640, + "tokens/train_per_sec_per_gpu": 29.24, + "tokens/trainable": 131343 + }, + { + "epoch": 1.125, + "grad_norm": 0.9641065001487732, + "learning_rate": 4.964235714846775e-05, + "loss": 0.011515076272189617, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01158, + "step": 288, + "tokens/total": 9200544, + "tokens/train_per_sec_per_gpu": 27.04, + "tokens/trainable": 131763 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0010062052169814706, + "learning_rate": 4.9354244499126866e-05, + "loss": 9.489545846008696e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00001, + "step": 289, + "tokens/total": 9232624, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 132229 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.002373398980125785, + "learning_rate": 4.90663667927174e-05, + "loss": 1.573777262819931e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00002, + "step": 290, + "tokens/total": 9264768, + "tokens/train_per_sec_per_gpu": 29.5, + "tokens/trainable": 132678 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.0014718567254021764, + "learning_rate": 4.877873600900581e-05, + "loss": 9.888429303828161e-06, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00001, + "step": 291, + "tokens/total": 9296944, + "tokens/train_per_sec_per_gpu": 24.47, + "tokens/trainable": 133136 + }, + { + "epoch": 1.140625, + "grad_norm": 0.0018440725980326533, + "learning_rate": 4.849136411748306e-05, + "loss": 8.275845175376162e-06, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00001, + "step": 292, + "tokens/total": 9328848, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 133608 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.12918472290039062, + "learning_rate": 4.8204263076866574e-05, + "loss": 0.0009128568926826119, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00091, + "step": 293, + "tokens/total": 9360928, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 134108 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.019195031374692917, + "learning_rate": 4.791744483460251e-05, + "loss": 9.439548011869192e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00009, + "step": 294, + "tokens/total": 9393040, + "tokens/train_per_sec_per_gpu": 25.71, + "tokens/trainable": 134541 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.1602204144001007, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0004640200058929622, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00046, + "step": 295, + "tokens/total": 9425248, + "tokens/train_per_sec_per_gpu": 24.5, + "tokens/trainable": 134966 + }, + { + "epoch": 1.15625, + "grad_norm": 0.5406288504600525, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.0034592358861118555, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00347, + "step": 296, + "tokens/total": 9457264, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 135402 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.004333238583058119, + "learning_rate": 4.705880619288153e-05, + "loss": 3.075868880841881e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00003, + "step": 297, + "tokens/total": 9489056, + "tokens/train_per_sec_per_gpu": 24.87, + "tokens/trainable": 135804 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.006247827783226967, + "learning_rate": 4.677323837567412e-05, + "loss": 2.383375249337405e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00002, + "step": 298, + "tokens/total": 9521024, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 136231 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.0779612734913826, + "learning_rate": 4.6488012907598146e-05, + "loss": 0.0004679379053413868, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00047, + "step": 299, + "tokens/total": 9553120, + "tokens/train_per_sec_per_gpu": 28.37, + "tokens/trainable": 136705 + }, + { + "epoch": 1.171875, + "grad_norm": 0.01816786825656891, + "learning_rate": 4.620314165804964e-05, + "loss": 6.801338167861104e-05, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00007, + "step": 300, + "tokens/total": 9585040, + "tokens/train_per_sec_per_gpu": 28.78, + "tokens/trainable": 137178 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.9530329704284668, + "learning_rate": 4.591863648168407e-05, + "loss": 0.008843549527227879, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00888, + "step": 301, + "tokens/total": 9617120, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 137643 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0371168851852417, + "learning_rate": 4.5634509217923135e-05, + "loss": 0.00014635953994002193, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00015, + "step": 302, + "tokens/total": 9649024, + "tokens/train_per_sec_per_gpu": 26.74, + "tokens/trainable": 138078 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.08589725196361542, + "learning_rate": 4.535077169046201e-05, + "loss": 0.00043714733328670263, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00044, + "step": 303, + "tokens/total": 9680832, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 138515 + }, + { + "epoch": 1.1875, + "grad_norm": 0.358940988779068, + "learning_rate": 4.506743570677743e-05, + "loss": 0.006240838672965765, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00626, + "step": 304, + "tokens/total": 9713040, + "tokens/train_per_sec_per_gpu": 29.29, + "tokens/trainable": 138948 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.12659867107868195, + "learning_rate": 4.478451305763618e-05, + "loss": 0.0008667556685395539, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00087, + "step": 305, + "tokens/total": 9744976, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 139408 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.08289908617734909, + "learning_rate": 4.450201551660454e-05, + "loss": 0.0006384386215358973, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.02, + "memory/max_allocated (GiB)": 36.02, + "ppl": 1.00064, + "step": 306, + "tokens/total": 9774784, + "tokens/train_per_sec_per_gpu": 27.76, + "tokens/trainable": 139840 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.19195082783699036, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.002342985011637211, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00235, + "step": 307, + "tokens/total": 9806656, + "tokens/train_per_sec_per_gpu": 27.6, + "tokens/trainable": 140281 + }, + { + "epoch": 1.203125, + "grad_norm": 0.07421132177114487, + "learning_rate": 4.393834276419352e-05, + "loss": 0.0007085398538038135, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00071, + "step": 308, + "tokens/total": 9838720, + "tokens/train_per_sec_per_gpu": 30.88, + "tokens/trainable": 140776 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.004596009384840727, + "learning_rate": 4.36571910095382e-05, + "loss": 2.8969257982680574e-05, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00003, + "step": 309, + "tokens/total": 9870784, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 141228 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.04099060595035553, + "learning_rate": 4.337651127546448e-05, + "loss": 0.00030985785997472703, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00031, + "step": 310, + "tokens/total": 9902784, + "tokens/train_per_sec_per_gpu": 30.8, + "tokens/trainable": 141679 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.18840794265270233, + "learning_rate": 4.3096315242201736e-05, + "loss": 0.0020186181645840406, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00202, + "step": 311, + "tokens/total": 9934768, + "tokens/train_per_sec_per_gpu": 26.72, + "tokens/trainable": 142122 + }, + { + "epoch": 1.21875, + "grad_norm": 0.5018841028213501, + "learning_rate": 4.2816614569850635e-05, + "loss": 0.006148474756628275, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00617, + "step": 312, + "tokens/total": 9966960, + "tokens/train_per_sec_per_gpu": 27.27, + "tokens/trainable": 142561 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.1730002760887146, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.003579454030841589, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00359, + "step": 313, + "tokens/total": 9999104, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 143046 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.6586350202560425, + "learning_rate": 4.225874584473174e-05, + "loss": 0.0015231478027999401, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00152, + "step": 314, + "tokens/total": 10031136, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 143493 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.011105087585747242, + "learning_rate": 4.19806010071587e-05, + "loss": 0.00011838486534543335, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00012, + "step": 315, + "tokens/total": 10063184, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 143956 + }, + { + "epoch": 1.234375, + "grad_norm": 0.2628834843635559, + "learning_rate": 4.170299795992081e-05, + "loss": 0.003402995876967907, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00341, + "step": 316, + "tokens/total": 10095312, + "tokens/train_per_sec_per_gpu": 28.29, + "tokens/trainable": 144411 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.13821958005428314, + "learning_rate": 4.142594825521398e-05, + "loss": 0.0016696923412382603, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00167, + "step": 317, + "tokens/total": 10127440, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 144892 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.0023418583441525698, + "learning_rate": 4.114946342220728e-05, + "loss": 2.6753859856398776e-05, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00003, + "step": 318, + "tokens/total": 10159184, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 145339 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.0008433983894065022, + "learning_rate": 4.087355496656321e-05, + "loss": 1.602486736373976e-05, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00002, + "step": 319, + "tokens/total": 10191216, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 145811 + }, + { + "epoch": 1.25, + "grad_norm": 0.12588869035243988, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0011584729654714465, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00116, + "step": 320, + "tokens/total": 10223408, + "tokens/train_per_sec_per_gpu": 32.27, + "tokens/trainable": 146314 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.028395529836416245, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.00015757072833366692, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00016, + "step": 321, + "tokens/total": 10255536, + "tokens/train_per_sec_per_gpu": 26.68, + "tokens/trainable": 146781 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.039599668234586716, + "learning_rate": 4.004940255778431e-05, + "loss": 0.000321136845741421, + "memory/device_reserved (GiB)": 37.39, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00032, + "step": 322, + "tokens/total": 10287664, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 147225 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.11097095906734467, + "learning_rate": 3.977591418134619e-05, + "loss": 0.0004549310833681375, + "memory/device_reserved (GiB)": 37.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00046, + "step": 323, + "tokens/total": 10320080, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 147673 + }, + { + "epoch": 1.265625, + "grad_norm": 0.19018623232841492, + "learning_rate": 3.95030593412612e-05, + "loss": 0.0028336881659924984, + "memory/device_reserved (GiB)": 38.53, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00284, + "step": 324, + "tokens/total": 10351984, + "tokens/train_per_sec_per_gpu": 27.33, + "tokens/trainable": 148103 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.4650349020957947, + "learning_rate": 3.923084939213296e-05, + "loss": 0.006978687364608049, + "memory/device_reserved (GiB)": 38.53, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.007, + "step": 325, + "tokens/total": 10384256, + "tokens/train_per_sec_per_gpu": 26.21, + "tokens/trainable": 148535 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.14811205863952637, + "learning_rate": 3.895929566172861e-05, + "loss": 0.0014966176822781563, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0015, + "step": 326, + "tokens/total": 10416208, + "tokens/train_per_sec_per_gpu": 28.37, + "tokens/trainable": 148999 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.11302439123392105, + "learning_rate": 3.868840945050728e-05, + "loss": 0.0011318891774863005, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00113, + "step": 327, + "tokens/total": 10448224, + "tokens/train_per_sec_per_gpu": 26.37, + "tokens/trainable": 149431 + }, + { + "epoch": 1.28125, + "grad_norm": 0.10916705429553986, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0009605163359083235, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00096, + "step": 328, + "tokens/total": 10480448, + "tokens/train_per_sec_per_gpu": 28.27, + "tokens/trainable": 149886 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.0004018457839265466, + "learning_rate": 3.814868464809027e-05, + "loss": 7.040077434794512e-06, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00001, + "step": 329, + "tokens/total": 10512096, + "tokens/train_per_sec_per_gpu": 25.92, + "tokens/trainable": 150288 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.03717898577451706, + "learning_rate": 3.787986851704667e-05, + "loss": 0.0001680658315308392, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00017, + "step": 330, + "tokens/total": 10544000, + "tokens/train_per_sec_per_gpu": 28.52, + "tokens/trainable": 150733 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.21975846588611603, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00015534088015556335, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00016, + "step": 331, + "tokens/total": 10576048, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 151207 + }, + { + "epoch": 1.296875, + "grad_norm": 0.10715529322624207, + "learning_rate": 3.734438472750619e-05, + "loss": 0.0012046336196362972, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00121, + "step": 332, + "tokens/total": 10608048, + "tokens/train_per_sec_per_gpu": 31.08, + "tokens/trainable": 151694 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.13602837920188904, + "learning_rate": 3.707773935267552e-05, + "loss": 0.00039335948531515896, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00039, + "step": 333, + "tokens/total": 10640176, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 152188 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.002963080769404769, + "learning_rate": 3.68118397962661e-05, + "loss": 6.6289721871726215e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00001, + "step": 334, + "tokens/total": 10672224, + "tokens/train_per_sec_per_gpu": 25.55, + "tokens/trainable": 152596 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.007236327510327101, + "learning_rate": 3.654669712344384e-05, + "loss": 5.341085125110112e-05, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00005, + "step": 335, + "tokens/total": 10704192, + "tokens/train_per_sec_per_gpu": 30.44, + "tokens/trainable": 153052 + }, + { + "epoch": 1.3125, + "grad_norm": 0.0897810310125351, + "learning_rate": 3.628232236787763e-05, + "loss": 0.0008687702938914299, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00087, + "step": 336, + "tokens/total": 10736272, + "tokens/train_per_sec_per_gpu": 25.33, + "tokens/trainable": 153491 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.011829929426312447, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.00011201453162357211, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00011, + "step": 337, + "tokens/total": 10768528, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 153983 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.25218063592910767, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0022679665125906467, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00227, + "step": 338, + "tokens/total": 10800464, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 154464 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.00034703267738223076, + "learning_rate": 3.549391545931585e-05, + "loss": 2.884747800635523e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.0, + "step": 339, + "tokens/total": 10832512, + "tokens/train_per_sec_per_gpu": 29.7, + "tokens/trainable": 154924 + }, + { + "epoch": 1.328125, + "grad_norm": 0.004890776239335537, + "learning_rate": 3.5232722063479914e-05, + "loss": 2.2697908207192086e-05, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00002, + "step": 340, + "tokens/total": 10864624, + "tokens/train_per_sec_per_gpu": 25.46, + "tokens/trainable": 155373 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.0013245025184005499, + "learning_rate": 3.49723512647657e-05, + "loss": 8.980642633105163e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00001, + "step": 341, + "tokens/total": 10896816, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 155873 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.03804129734635353, + "learning_rate": 3.471281389826491e-05, + "loss": 0.00021880728309042752, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00022, + "step": 342, + "tokens/total": 10928864, + "tokens/train_per_sec_per_gpu": 25.31, + "tokens/trainable": 156278 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.4880214333534241, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.005397590808570385, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00541, + "step": 343, + "tokens/total": 10960816, + "tokens/train_per_sec_per_gpu": 29.58, + "tokens/trainable": 156733 + }, + { + "epoch": 1.34375, + "grad_norm": 0.37780871987342834, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0029298821464180946, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00293, + "step": 344, + "tokens/total": 10992912, + "tokens/train_per_sec_per_gpu": 28.4, + "tokens/trainable": 157203 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.0034261250402778387, + "learning_rate": 3.3939310220027456e-05, + "loss": 2.1256186300888658e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00002, + "step": 345, + "tokens/total": 11024832, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 157707 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.0005507867899723351, + "learning_rate": 3.3683214232914404e-05, + "loss": 6.991718692006543e-06, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00001, + "step": 346, + "tokens/total": 11056992, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 158180 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.0016383324982598424, + "learning_rate": 3.342800532426873e-05, + "loss": 1.2630136552616023e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00001, + "step": 347, + "tokens/total": 11089264, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 158646 + }, + { + "epoch": 1.359375, + "grad_norm": 0.0020912184845656157, + "learning_rate": 3.317369411437484e-05, + "loss": 1.536883064545691e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00002, + "step": 348, + "tokens/total": 11120976, + "tokens/train_per_sec_per_gpu": 29.34, + "tokens/trainable": 159115 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.0016828920925036073, + "learning_rate": 3.292029118616024e-05, + "loss": 1.334734679403482e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00001, + "step": 349, + "tokens/total": 11152976, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 159538 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.02520265430212021, + "learning_rate": 3.266780708475511e-05, + "loss": 9.595962183084339e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0001, + "step": 350, + "tokens/total": 11185056, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 160016 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.009836713783442974, + "learning_rate": 3.241625231705354e-05, + "loss": 7.521865336457267e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00008, + "step": 351, + "tokens/total": 11217088, + "tokens/train_per_sec_per_gpu": 28.51, + "tokens/trainable": 160475 + }, + { + "epoch": 1.375, + "grad_norm": 0.0698060467839241, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0003666624252218753, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00037, + "step": 352, + "tokens/total": 11249088, + "tokens/train_per_sec_per_gpu": 28.31, + "tokens/trainable": 160952 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.0018428800394758582, + "learning_rate": 3.191597261653475e-05, + "loss": 1.3194729945098516e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00001, + "step": 353, + "tokens/total": 11281328, + "tokens/train_per_sec_per_gpu": 27.4, + "tokens/trainable": 161420 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.020005319267511368, + "learning_rate": 3.166726850239794e-05, + "loss": 0.00017299644241575152, + "memory/device_reserved (GiB)": 38.47, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00017, + "step": 354, + "tokens/total": 11313200, + "tokens/train_per_sec_per_gpu": 28.31, + "tokens/trainable": 161868 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.00018166302470490336, + "learning_rate": 3.141953535845912e-05, + "loss": 4.138089934713207e-06, + "memory/device_reserved (GiB)": 38.47, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0, + "step": 355, + "tokens/total": 11345008, + "tokens/train_per_sec_per_gpu": 29.74, + "tokens/trainable": 162318 + }, + { + "epoch": 1.390625, + "grad_norm": 0.009165152907371521, + "learning_rate": 3.11727834939056e-05, + "loss": 6.729022425133735e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00007, + "step": 356, + "tokens/total": 11377216, + "tokens/train_per_sec_per_gpu": 26.26, + "tokens/trainable": 162789 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.2891274690628052, + "learning_rate": 3.092702317708967e-05, + "loss": 0.005585570354014635, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.0056, + "step": 357, + "tokens/total": 11409536, + "tokens/train_per_sec_per_gpu": 30.34, + "tokens/trainable": 163254 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.0007154783816076815, + "learning_rate": 3.0682264635101276e-05, + "loss": 5.591478839050978e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00001, + "step": 358, + "tokens/total": 11441552, + "tokens/train_per_sec_per_gpu": 28.74, + "tokens/trainable": 163715 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.01731656678020954, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.0001050045175361447, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00011, + "step": 359, + "tokens/total": 11473376, + "tokens/train_per_sec_per_gpu": 28.79, + "tokens/trainable": 164197 + }, + { + "epoch": 1.40625, + "grad_norm": 0.11494546383619308, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.00048459687968716025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00048, + "step": 360, + "tokens/total": 11505312, + "tokens/train_per_sec_per_gpu": 26.54, + "tokens/trainable": 164661 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.08215854316949844, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.0002725277154240757, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00027, + "step": 361, + "tokens/total": 11537536, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 165111 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.024266909807920456, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.0001677784021012485, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 35.94, + "memory/max_allocated (GiB)": 35.94, + "ppl": 1.00017, + "step": 362, + "tokens/total": 11567216, + "tokens/train_per_sec_per_gpu": 25.3, + "tokens/trainable": 165552 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.008205167017877102, + "learning_rate": 2.9473853553877484e-05, + "loss": 3.423610905883834e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00003, + "step": 363, + "tokens/total": 11599312, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 166026 + }, + { + "epoch": 1.421875, + "grad_norm": 0.4230109453201294, + "learning_rate": 2.9235318065647e-05, + "loss": 0.005893740337342024, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00591, + "step": 364, + "tokens/total": 11631344, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 166486 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.1095249131321907, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.0006186347454786301, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00062, + "step": 365, + "tokens/total": 11663632, + "tokens/train_per_sec_per_gpu": 30.17, + "tokens/trainable": 166959 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.0014034683117642999, + "learning_rate": 2.8761473491751258e-05, + "loss": 1.2728614819934592e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.27, + "memory/max_allocated (GiB)": 36.27, + "ppl": 1.00001, + "step": 366, + "tokens/total": 11695360, + "tokens/train_per_sec_per_gpu": 28.19, + "tokens/trainable": 167394 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.044087354093790054, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.0001466910180170089, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00015, + "step": 367, + "tokens/total": 11727184, + "tokens/train_per_sec_per_gpu": 23.64, + "tokens/trainable": 167832 + }, + { + "epoch": 1.4375, + "grad_norm": 0.0005008528823964298, + "learning_rate": 2.829199644117484e-05, + "loss": 1.1319095392536838e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00001, + "step": 368, + "tokens/total": 11759312, + "tokens/train_per_sec_per_gpu": 25.79, + "tokens/trainable": 168242 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.02015153504908085, + "learning_rate": 2.8058920186702553e-05, + "loss": 0.00011158352572238073, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00011, + "step": 369, + "tokens/total": 11791536, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 168715 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.00031807494815438986, + "learning_rate": 2.782696506053033e-05, + "loss": 6.533119631058071e-06, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00001, + "step": 370, + "tokens/total": 11823632, + "tokens/train_per_sec_per_gpu": 26.68, + "tokens/trainable": 169147 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.4510185122489929, + "learning_rate": 2.7596140715257824e-05, + "loss": 0.004431337118148804, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00444, + "step": 371, + "tokens/total": 11855392, + "tokens/train_per_sec_per_gpu": 28.61, + "tokens/trainable": 169628 + }, + { + "epoch": 1.453125, + "grad_norm": 0.0034806979820132256, + "learning_rate": 2.7366456756428184e-05, + "loss": 2.722550561884418e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00003, + "step": 372, + "tokens/total": 11887424, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 170085 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.018526704981923103, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.00013849925016984344, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00014, + "step": 373, + "tokens/total": 11919280, + "tokens/train_per_sec_per_gpu": 29.39, + "tokens/trainable": 170584 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.010734346695244312, + "learning_rate": 2.691054818259188e-05, + "loss": 1.963317481568083e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00002, + "step": 374, + "tokens/total": 11951488, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 171058 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.004906357266008854, + "learning_rate": 2.6684342539801933e-05, + "loss": 4.44888137280941e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00004, + "step": 375, + "tokens/total": 11983504, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 171518 + }, + { + "epoch": 1.46875, + "grad_norm": 0.0035627244506031275, + "learning_rate": 2.645931522709877e-05, + "loss": 1.879796946013812e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00002, + "step": 376, + "tokens/total": 12015360, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 172007 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.014949376694858074, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.0001050513528753072, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00011, + "step": 377, + "tokens/total": 12047296, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 172468 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.024432353675365448, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.00010428359382785857, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0001, + "step": 378, + "tokens/total": 12079392, + "tokens/train_per_sec_per_gpu": 27.91, + "tokens/trainable": 172923 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.5905564427375793, + "learning_rate": 2.579139666504821e-05, + "loss": 0.019045401364564896, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01923, + "step": 379, + "tokens/total": 12111616, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 173370 + }, + { + "epoch": 1.484375, + "grad_norm": 0.0027490397915244102, + "learning_rate": 2.557117581955798e-05, + "loss": 3.106705116806552e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00003, + "step": 380, + "tokens/total": 12143536, + "tokens/train_per_sec_per_gpu": 28.66, + "tokens/trainable": 173837 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.10220319032669067, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.0007754361140541732, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00078, + "step": 381, + "tokens/total": 12175792, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 174294 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.06388501822948456, + "learning_rate": 2.5134417202396277e-05, + "loss": 2.506363671272993e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00003, + "step": 382, + "tokens/total": 12207472, + "tokens/train_per_sec_per_gpu": 25.85, + "tokens/trainable": 174701 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.006029566749930382, + "learning_rate": 2.491789760603361e-05, + "loss": 5.214785414864309e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00005, + "step": 383, + "tokens/total": 12239520, + "tokens/train_per_sec_per_gpu": 26.67, + "tokens/trainable": 175131 + }, + { + "epoch": 1.5, + "grad_norm": 0.015079713426530361, + "learning_rate": 2.4702629848741764e-05, + "loss": 0.00011988497863058001, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00012, + "step": 384, + "tokens/total": 12271600, + "tokens/train_per_sec_per_gpu": 28.23, + "tokens/trainable": 175586 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 8.329452040888704e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/training_args.bin b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..dec74f9f304b44cbe47132d86c4e6285e659f40b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-384/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634 +size 8273 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/README.md b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ba677470ac9c29f9770a4aacb65c644ebd671a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/deconf_wave_deconf_control/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/deconf_wave_deconf_control/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/adapter_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..21fcb5b00be777c20a2d2dcceffc539dc51679b4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/deconf_wave_deconf_control/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "gate_proj", + "v_proj", + "k_proj", + "up_proj", + "q_proj", + "o_proj", + "down_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/adapter_model.safetensors b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..c09e4c1035d7222192a3211b45027b684b60bcd3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:de1a8f00f7cce35d617852d93a55d1fcaecd8cc3d2262ffdb8e4a7c45014046e +size 547777976 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/chat_template.jinja b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/optimizer.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..b223045eed122c5be3af91b5e9e213fcca0b7cc2 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:41196749ab2b88f648fb2c57dcd3b5cda0e88ee74d3effc38687d4becd62c1bc +size 1048106435 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/rng_state.pth b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..ae7227a8ca2fe7407dbf69f96397a8f1a0ec78a2 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1a83de7169fc700b7228ad5136408c67bee753dfa65a1332ab9a3124f2461909 +size 14645 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/scheduler.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..20a981d8c8f57b8d2e18429ceb299414229ad6ac --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2ed29f28c9b7651cfc581e35c7a694a1573d310ed4e8afccc5b01f2e33bedecc +size 1465 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/tokenizer.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/tokenizer_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/tokens_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..ddf3e0a37da6c1584f71fec2d86ac69104d300c7 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/tokens_state.json @@ -0,0 +1 @@ +{"total": 13294480, "trainable": 190306} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/trainer_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..74a42d0d3ddba67ebd1d3c49167f0ef90af5fdeb --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/trainer_state.json @@ -0,0 +1,5858 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.625, + "eval_steps": 500, + "global_step": 416, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.2140090465545654, + "learning_rate": 0.0, + "loss": 0.16441112756729126, + "memory/device_reserved (GiB)": 38.08, + "memory/max_active (GiB)": 35.64, + "memory/max_allocated (GiB)": 35.64, + "ppl": 1.1787, + "step": 1, + "tokens/total": 32160, + "tokens/train_per_sec_per_gpu": 21.28, + "tokens/trainable": 470 + }, + { + "epoch": 0.0078125, + "grad_norm": 2.92620849609375, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15774038434028625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.17086, + "step": 2, + "tokens/total": 64192, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 922 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.135961890220642, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1664382815361023, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.18109, + "step": 3, + "tokens/total": 96224, + "tokens/train_per_sec_per_gpu": 28.54, + "tokens/trainable": 1396 + }, + { + "epoch": 0.015625, + "grad_norm": 0.880176305770874, + "learning_rate": 1.2e-05, + "loss": 0.15334013104438782, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.16572, + "step": 4, + "tokens/total": 128336, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 1850 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.1433289051055908, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.16352537274360657, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.17766, + "step": 5, + "tokens/total": 160784, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 2302 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9728212952613831, + "learning_rate": 2e-05, + "loss": 0.15144473314285278, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.16351, + "step": 6, + "tokens/total": 192624, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 2742 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.7796363830566406, + "learning_rate": 2.4e-05, + "loss": 0.13553467392921448, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.14515, + "step": 7, + "tokens/total": 224608, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 3208 + }, + { + "epoch": 0.03125, + "grad_norm": 1.2444697618484497, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.1077304258942604, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.11375, + "step": 8, + "tokens/total": 256592, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 3655 + }, + { + "epoch": 0.03515625, + "grad_norm": 2.7339375019073486, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.08772765100002289, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.09169, + "step": 9, + "tokens/total": 286400, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 4091 + }, + { + "epoch": 0.0390625, + "grad_norm": 4.109770774841309, + "learning_rate": 3.6e-05, + "loss": 0.130940243601799, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.1399, + "step": 10, + "tokens/total": 318352, + "tokens/train_per_sec_per_gpu": 29.13, + "tokens/trainable": 4553 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.3768608570098877, + "learning_rate": 4e-05, + "loss": 0.10059958696365356, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.10583, + "step": 11, + "tokens/total": 350336, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 4992 + }, + { + "epoch": 0.046875, + "grad_norm": 3.281402826309204, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.12574875354766846, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.134, + "step": 12, + "tokens/total": 382448, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 5494 + }, + { + "epoch": 0.05078125, + "grad_norm": 1.7738980054855347, + "learning_rate": 4.8e-05, + "loss": 0.033296722918748856, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03386, + "step": 13, + "tokens/total": 414352, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 5933 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.64290189743042, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06366822123527527, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.06574, + "step": 14, + "tokens/total": 446400, + "tokens/train_per_sec_per_gpu": 24.8, + "tokens/trainable": 6369 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.3691928386688232, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.08168518543243408, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.08511, + "step": 15, + "tokens/total": 478688, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 6820 + }, + { + "epoch": 0.0625, + "grad_norm": 2.4356393814086914, + "learning_rate": 6e-05, + "loss": 0.09031196683645248, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.09452, + "step": 16, + "tokens/total": 510512, + "tokens/train_per_sec_per_gpu": 25.7, + "tokens/trainable": 7258 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.55682373046875, + "learning_rate": 6.400000000000001e-05, + "loss": 0.07496573030948639, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.07785, + "step": 17, + "tokens/total": 542736, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 7710 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.1054646968841553, + "learning_rate": 6.800000000000001e-05, + "loss": 0.05218805745244026, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.05357, + "step": 18, + "tokens/total": 574832, + "tokens/train_per_sec_per_gpu": 25.78, + "tokens/trainable": 8174 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.7429099082946777, + "learning_rate": 7.2e-05, + "loss": 0.03790827840566635, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.03864, + "step": 19, + "tokens/total": 606976, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 8617 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9508209228515625, + "learning_rate": 7.6e-05, + "loss": 0.04993613809347153, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0512, + "step": 20, + "tokens/total": 639200, + "tokens/train_per_sec_per_gpu": 24.91, + "tokens/trainable": 9037 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.0297844409942627, + "learning_rate": 8e-05, + "loss": 0.017908263951539993, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.01807, + "step": 21, + "tokens/total": 671072, + "tokens/train_per_sec_per_gpu": 30.42, + "tokens/trainable": 9503 + }, + { + "epoch": 0.0859375, + "grad_norm": 0.9279879331588745, + "learning_rate": 8.4e-05, + "loss": 0.013478193432092667, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01357, + "step": 22, + "tokens/total": 703136, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 9972 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.526488184928894, + "learning_rate": 8.800000000000001e-05, + "loss": 0.017310388386249542, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01746, + "step": 23, + "tokens/total": 735120, + "tokens/train_per_sec_per_gpu": 31.45, + "tokens/trainable": 10447 + }, + { + "epoch": 0.09375, + "grad_norm": 3.4049417972564697, + "learning_rate": 9.200000000000001e-05, + "loss": 0.0353429913520813, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.03597, + "step": 24, + "tokens/total": 767040, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 10899 + }, + { + "epoch": 0.09765625, + "grad_norm": 7.511656761169434, + "learning_rate": 9.6e-05, + "loss": 0.1335875540971756, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.14292, + "step": 25, + "tokens/total": 798976, + "tokens/train_per_sec_per_gpu": 27.72, + "tokens/trainable": 11360 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.0069234371185303, + "learning_rate": 0.0001, + "loss": 0.04382415860891342, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0448, + "step": 26, + "tokens/total": 830880, + "tokens/train_per_sec_per_gpu": 24.61, + "tokens/trainable": 11775 + }, + { + "epoch": 0.10546875, + "grad_norm": 7.807938575744629, + "learning_rate": 9.99990636831587e-05, + "loss": 0.01210833340883255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.01218, + "step": 27, + "tokens/total": 860960, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 12242 + }, + { + "epoch": 0.109375, + "grad_norm": 2.23823881149292, + "learning_rate": 9.999625477159879e-05, + "loss": 0.0272970050573349, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02767, + "step": 28, + "tokens/total": 892816, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 12696 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.315584421157837, + "learning_rate": 9.999157338221051e-05, + "loss": 0.025516577064990997, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.02584, + "step": 29, + "tokens/total": 924736, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 13153 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.8413651585578918, + "learning_rate": 9.998501970980562e-05, + "loss": 0.007838520221412182, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00787, + "step": 30, + "tokens/total": 954592, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 13624 + }, + { + "epoch": 0.12109375, + "grad_norm": 1.6797157526016235, + "learning_rate": 9.997659402710915e-05, + "loss": 0.03002442792057991, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.03048, + "step": 31, + "tokens/total": 986592, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 14064 + }, + { + "epoch": 0.125, + "grad_norm": 1.3320434093475342, + "learning_rate": 9.996629668474818e-05, + "loss": 0.03772728145122528, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.03845, + "step": 32, + "tokens/total": 1018720, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 14565 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.6335167288780212, + "learning_rate": 9.995412811123711e-05, + "loss": 0.013970119878649712, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01407, + "step": 33, + "tokens/total": 1050624, + "tokens/train_per_sec_per_gpu": 28.18, + "tokens/trainable": 15005 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.5626921653747559, + "learning_rate": 9.994008881295999e-05, + "loss": 0.016654539853334427, + "memory/device_reserved (GiB)": 37.74, + "memory/max_active (GiB)": 36.22, + "memory/max_allocated (GiB)": 36.22, + "ppl": 1.01679, + "step": 34, + "tokens/total": 1082176, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 15459 + }, + { + "epoch": 0.13671875, + "grad_norm": 0.603219211101532, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021408583968877792, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02164, + "step": 35, + "tokens/total": 1113968, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 15960 + }, + { + "epoch": 0.140625, + "grad_norm": 0.9779674410820007, + "learning_rate": 9.99064004568618e-05, + "loss": 0.03277068957686424, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03331, + "step": 36, + "tokens/total": 1145952, + "tokens/train_per_sec_per_gpu": 28.68, + "tokens/trainable": 16428 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.8855255842208862, + "learning_rate": 9.988675280095074e-05, + "loss": 0.017048098146915436, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01719, + "step": 37, + "tokens/total": 1177872, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 16884 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.9266122579574585, + "learning_rate": 9.986523722403528e-05, + "loss": 0.045863546431064606, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.04693, + "step": 38, + "tokens/total": 1209776, + "tokens/train_per_sec_per_gpu": 26.91, + "tokens/trainable": 17341 + }, + { + "epoch": 0.15234375, + "grad_norm": 0.8280895948410034, + "learning_rate": 9.984185462146642e-05, + "loss": 0.033055759966373444, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03361, + "step": 39, + "tokens/total": 1241888, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 17786 + }, + { + "epoch": 0.15625, + "grad_norm": 5.029730319976807, + "learning_rate": 9.98166059662897e-05, + "loss": 0.025589998811483383, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02592, + "step": 40, + "tokens/total": 1274032, + "tokens/train_per_sec_per_gpu": 29.22, + "tokens/trainable": 18262 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.5688032507896423, + "learning_rate": 9.978949230920472e-05, + "loss": 0.016095872968435287, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01623, + "step": 41, + "tokens/total": 1306000, + "tokens/train_per_sec_per_gpu": 27.49, + "tokens/trainable": 18716 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8217663168907166, + "learning_rate": 9.976051477852141e-05, + "loss": 0.026798982173204422, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02716, + "step": 42, + "tokens/total": 1337936, + "tokens/train_per_sec_per_gpu": 27.51, + "tokens/trainable": 19157 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.7309132218360901, + "learning_rate": 9.972967458011312e-05, + "loss": 0.033920761197805405, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0345, + "step": 43, + "tokens/total": 1369904, + "tokens/train_per_sec_per_gpu": 30.66, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.8134022355079651, + "learning_rate": 9.96969729973664e-05, + "loss": 0.024005085229873657, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0243, + "step": 44, + "tokens/total": 1401984, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 20119 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.35084816813468933, + "learning_rate": 9.966241139112754e-05, + "loss": 0.013319254852831364, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01341, + "step": 45, + "tokens/total": 1434032, + "tokens/train_per_sec_per_gpu": 27.45, + "tokens/trainable": 20560 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.8701573014259338, + "learning_rate": 9.96259911996461e-05, + "loss": 0.01356641948223114, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01366, + "step": 46, + "tokens/total": 1465872, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 20992 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.4406442642211914, + "learning_rate": 9.958771393851491e-05, + "loss": 0.04396980255842209, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 35.86, + "memory/max_allocated (GiB)": 35.86, + "ppl": 1.04495, + "step": 47, + "tokens/total": 1495440, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 21441 + }, + { + "epoch": 0.1875, + "grad_norm": 0.4966669976711273, + "learning_rate": 9.954758120060702e-05, + "loss": 0.011179964058101177, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01124, + "step": 48, + "tokens/total": 1527776, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 21901 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.41124436259269714, + "learning_rate": 9.950559465600948e-05, + "loss": 0.0063599650748074055, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00638, + "step": 49, + "tokens/total": 1559760, + "tokens/train_per_sec_per_gpu": 27.73, + "tokens/trainable": 22341 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.829739511013031, + "learning_rate": 9.946175605195379e-05, + "loss": 0.015785304829478264, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01591, + "step": 50, + "tokens/total": 1592000, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 22833 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5181736946105957, + "learning_rate": 9.941606721274322e-05, + "loss": 0.008393109776079655, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00843, + "step": 51, + "tokens/total": 1623920, + "tokens/train_per_sec_per_gpu": 25.61, + "tokens/trainable": 23277 + }, + { + "epoch": 0.203125, + "grad_norm": 1.5778990983963013, + "learning_rate": 9.936853003967685e-05, + "loss": 0.024984844028949738, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0253, + "step": 52, + "tokens/total": 1656032, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 23759 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.4938714504241943, + "learning_rate": 9.93191465109705e-05, + "loss": 0.038523320108652115, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03927, + "step": 53, + "tokens/total": 1688368, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 24193 + }, + { + "epoch": 0.2109375, + "grad_norm": 2.1427106857299805, + "learning_rate": 9.926791868167438e-05, + "loss": 0.06302924454212189, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.06506, + "step": 54, + "tokens/total": 1720560, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 24619 + }, + { + "epoch": 0.21484375, + "grad_norm": 1.8365188837051392, + "learning_rate": 9.921484868358753e-05, + "loss": 0.052160464227199554, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.05354, + "step": 55, + "tokens/total": 1752464, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 25075 + }, + { + "epoch": 0.21875, + "grad_norm": 0.7779749631881714, + "learning_rate": 9.915993872516924e-05, + "loss": 0.021328264847397804, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.02156, + "step": 56, + "tokens/total": 1784336, + "tokens/train_per_sec_per_gpu": 26.42, + "tokens/trainable": 25519 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.4189571142196655, + "learning_rate": 9.9103191091447e-05, + "loss": 0.014279939234256744, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01438, + "step": 57, + "tokens/total": 1816304, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 25966 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.3834710717201233, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01883944310247898, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01902, + "step": 58, + "tokens/total": 1848304, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 26423 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.7693278789520264, + "learning_rate": 9.898419232046825e-05, + "loss": 0.03533369302749634, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03597, + "step": 59, + "tokens/total": 1880640, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 26934 + }, + { + "epoch": 0.234375, + "grad_norm": 0.4606446325778961, + "learning_rate": 9.892194613523633e-05, + "loss": 0.021482212468981743, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02171, + "step": 60, + "tokens/total": 1912592, + "tokens/train_per_sec_per_gpu": 27.64, + "tokens/trainable": 27393 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.46030017733573914, + "learning_rate": 9.885787217854357e-05, + "loss": 0.017044665291905403, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01719, + "step": 61, + "tokens/total": 1944544, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 27852 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.7918264269828796, + "learning_rate": 9.879197311676887e-05, + "loss": 0.016954131424427032, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0171, + "step": 62, + "tokens/total": 1976688, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 28292 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.4787926971912384, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01525965891778469, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 63, + "tokens/total": 2008928, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 28798 + }, + { + "epoch": 0.25, + "grad_norm": 0.9033918976783752, + "learning_rate": 9.865471072312528e-05, + "loss": 0.02902211993932724, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02945, + "step": 64, + "tokens/total": 2041120, + "tokens/train_per_sec_per_gpu": 29.83, + "tokens/trainable": 29283 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.39305707812309265, + "learning_rate": 9.858335310330492e-05, + "loss": 0.00901167094707489, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.32, + "memory/max_allocated (GiB)": 36.32, + "ppl": 1.00905, + "step": 65, + "tokens/total": 2073024, + "tokens/train_per_sec_per_gpu": 26.48, + "tokens/trainable": 29745 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4494345784187317, + "learning_rate": 9.851018180226185e-05, + "loss": 0.007426132448017597, + "memory/device_reserved (GiB)": 37.47, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00745, + "step": 66, + "tokens/total": 2105184, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 30186 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.629649817943573, + "learning_rate": 9.843519986495259e-05, + "loss": 0.010609081946313381, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01067, + "step": 67, + "tokens/total": 2137264, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 30622 + }, + { + "epoch": 0.265625, + "grad_norm": 0.7591162919998169, + "learning_rate": 9.835841041168162e-05, + "loss": 0.00926615484058857, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00931, + "step": 68, + "tokens/total": 2169424, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 31097 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.7288486361503601, + "learning_rate": 9.82798166379715e-05, + "loss": 0.024310583248734474, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.02461, + "step": 69, + "tokens/total": 2201424, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 31595 + }, + { + "epoch": 0.2734375, + "grad_norm": 2.134056329727173, + "learning_rate": 9.819942181443002e-05, + "loss": 0.008422967046499252, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00846, + "step": 70, + "tokens/total": 2233424, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 32036 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6362871527671814, + "learning_rate": 9.811722928661392e-05, + "loss": 0.01577262207865715, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0159, + "step": 71, + "tokens/total": 2265328, + "tokens/train_per_sec_per_gpu": 23.65, + "tokens/trainable": 32428 + }, + { + "epoch": 0.28125, + "grad_norm": 0.6091912388801575, + "learning_rate": 9.803324247488975e-05, + "loss": 0.0064800274558365345, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0065, + "step": 72, + "tokens/total": 2297248, + "tokens/train_per_sec_per_gpu": 26.86, + "tokens/trainable": 32880 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.14533838629722595, + "learning_rate": 9.794746487429161e-05, + "loss": 0.001991454279050231, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00199, + "step": 73, + "tokens/total": 2329152, + "tokens/train_per_sec_per_gpu": 27.97, + "tokens/trainable": 33323 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.5400826930999756, + "learning_rate": 9.785990005437554e-05, + "loss": 0.017662350088357925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01782, + "step": 74, + "tokens/total": 2361344, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 33792 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.6132436394691467, + "learning_rate": 9.777055165907117e-05, + "loss": 0.01787603087723255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01804, + "step": 75, + "tokens/total": 2393168, + "tokens/train_per_sec_per_gpu": 29.53, + "tokens/trainable": 34223 + }, + { + "epoch": 0.296875, + "grad_norm": 0.7287399172782898, + "learning_rate": 9.767942340652993e-05, + "loss": 0.024754449725151062, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.31, + "memory/max_allocated (GiB)": 36.31, + "ppl": 1.02506, + "step": 76, + "tokens/total": 2424880, + "tokens/train_per_sec_per_gpu": 24.92, + "tokens/trainable": 34649 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.6790545582771301, + "learning_rate": 9.758651908897035e-05, + "loss": 0.014308430254459381, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01441, + "step": 77, + "tokens/total": 2457120, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 35094 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6203456521034241, + "learning_rate": 9.749184257252033e-05, + "loss": 0.034353043884038925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03495, + "step": 78, + "tokens/total": 2489104, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 35534 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.6238825917243958, + "learning_rate": 9.739539779705614e-05, + "loss": 0.032295141369104385, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.03282, + "step": 79, + "tokens/total": 2521136, + "tokens/train_per_sec_per_gpu": 29.68, + "tokens/trainable": 36029 + }, + { + "epoch": 0.3125, + "grad_norm": 0.18977171182632446, + "learning_rate": 9.729718877603861e-05, + "loss": 0.007260813377797604, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00729, + "step": 80, + "tokens/total": 2553168, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 36469 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.8054049611091614, + "learning_rate": 9.719721959634592e-05, + "loss": 0.034894898533821106, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.03551, + "step": 81, + "tokens/total": 2585584, + "tokens/train_per_sec_per_gpu": 25.75, + "tokens/trainable": 36906 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.6884189248085022, + "learning_rate": 9.709549441810375e-05, + "loss": 0.018210047855973244, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.01838, + "step": 82, + "tokens/total": 2617808, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 37390 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.6490103602409363, + "learning_rate": 9.699201747451195e-05, + "loss": 0.02030857279896736, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.02052, + "step": 83, + "tokens/total": 2649664, + "tokens/train_per_sec_per_gpu": 25.8, + "tokens/trainable": 37838 + }, + { + "epoch": 0.328125, + "grad_norm": 0.2439805567264557, + "learning_rate": 9.688679307166854e-05, + "loss": 0.008518049493432045, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00855, + "step": 84, + "tokens/total": 2681728, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 38310 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6747375726699829, + "learning_rate": 9.677982558839042e-05, + "loss": 0.029895059764385223, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.03035, + "step": 85, + "tokens/total": 2713488, + "tokens/train_per_sec_per_gpu": 28.49, + "tokens/trainable": 38789 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.582666277885437, + "learning_rate": 9.66711194760312e-05, + "loss": 0.014998597092926502, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01511, + "step": 86, + "tokens/total": 2745536, + "tokens/train_per_sec_per_gpu": 30.27, + "tokens/trainable": 39277 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.5364654660224915, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011567167937755585, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01163, + "step": 87, + "tokens/total": 2777680, + "tokens/train_per_sec_per_gpu": 29.73, + "tokens/trainable": 39737 + }, + { + "epoch": 0.34375, + "grad_norm": 0.34024426341056824, + "learning_rate": 9.644850953105288e-05, + "loss": 0.01014852337539196, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0102, + "step": 88, + "tokens/total": 2809568, + "tokens/train_per_sec_per_gpu": 24.68, + "tokens/trainable": 40179 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.32472267746925354, + "learning_rate": 9.633461496214225e-05, + "loss": 0.01229981891810894, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01238, + "step": 89, + "tokens/total": 2841760, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 40649 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.7906899452209473, + "learning_rate": 9.621900029118195e-05, + "loss": 0.02552586793899536, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.02585, + "step": 90, + "tokens/total": 2871696, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 41080 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.7444764971733093, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0054441955871880054, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00546, + "step": 91, + "tokens/total": 2903968, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 41599 + }, + { + "epoch": 0.359375, + "grad_norm": 0.1911851465702057, + "learning_rate": 9.598262995928611e-05, + "loss": 0.004665314219892025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00468, + "step": 92, + "tokens/total": 2936048, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 42076 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.21238519251346588, + "learning_rate": 9.586188413468492e-05, + "loss": 0.003917238209396601, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00392, + "step": 93, + "tokens/total": 2968176, + "tokens/train_per_sec_per_gpu": 28.07, + "tokens/trainable": 42522 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.41729435324668884, + "learning_rate": 9.57394378802934e-05, + "loss": 0.013555807992815971, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01365, + "step": 94, + "tokens/total": 3000304, + "tokens/train_per_sec_per_gpu": 28.69, + "tokens/trainable": 42974 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.7246769070625305, + "learning_rate": 9.56152962916e-05, + "loss": 0.020221399143338203, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.02043, + "step": 95, + "tokens/total": 3032080, + "tokens/train_per_sec_per_gpu": 25.35, + "tokens/trainable": 43380 + }, + { + "epoch": 0.375, + "grad_norm": 0.16139821708202362, + "learning_rate": 9.548946453464296e-05, + "loss": 0.003853276837617159, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00386, + "step": 96, + "tokens/total": 3064320, + "tokens/train_per_sec_per_gpu": 28.47, + "tokens/trainable": 43865 + }, + { + "epoch": 0.37890625, + "grad_norm": 1.0509370565414429, + "learning_rate": 9.53619478457953e-05, + "loss": 0.015099998563528061, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01521, + "step": 97, + "tokens/total": 3096144, + "tokens/train_per_sec_per_gpu": 26.61, + "tokens/trainable": 44328 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.6859350204467773, + "learning_rate": 9.523275153154695e-05, + "loss": 0.004579808097332716, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00459, + "step": 98, + "tokens/total": 3127600, + "tokens/train_per_sec_per_gpu": 25.91, + "tokens/trainable": 44778 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5686795711517334, + "learning_rate": 9.51018809682839e-05, + "loss": 0.022520575672388077, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02278, + "step": 99, + "tokens/total": 3159520, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 45225 + }, + { + "epoch": 0.390625, + "grad_norm": 1.6308369636535645, + "learning_rate": 9.49693416020645e-05, + "loss": 0.011883002705872059, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01195, + "step": 100, + "tokens/total": 3191568, + "tokens/train_per_sec_per_gpu": 29.75, + "tokens/trainable": 45678 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.7186088562011719, + "learning_rate": 9.483513894839276e-05, + "loss": 0.0060632615350186825, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00608, + "step": 101, + "tokens/total": 3223648, + "tokens/train_per_sec_per_gpu": 27.14, + "tokens/trainable": 46125 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.12036337703466415, + "learning_rate": 9.469927859198888e-05, + "loss": 0.0019118499476462603, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00191, + "step": 102, + "tokens/total": 3255744, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 46612 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.29928353428840637, + "learning_rate": 9.456176618655689e-05, + "loss": 0.00944911316037178, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00949, + "step": 103, + "tokens/total": 3287792, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 47059 + }, + { + "epoch": 0.40625, + "grad_norm": 0.3766293227672577, + "learning_rate": 9.442260745454927e-05, + "loss": 0.008984047919511795, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.68, + "memory/max_allocated (GiB)": 36.68, + "ppl": 1.00902, + "step": 104, + "tokens/total": 3320016, + "tokens/train_per_sec_per_gpu": 28.92, + "tokens/trainable": 47536 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3481820523738861, + "learning_rate": 9.428180818692884e-05, + "loss": 0.005444999784231186, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00546, + "step": 105, + "tokens/total": 3351952, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 48037 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.8414323329925537, + "learning_rate": 9.413937424292791e-05, + "loss": 0.032438624650239944, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03297, + "step": 106, + "tokens/total": 3384032, + "tokens/train_per_sec_per_gpu": 27.11, + "tokens/trainable": 48491 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.5067655444145203, + "learning_rate": 9.399531154980424e-05, + "loss": 0.008442888967692852, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00848, + "step": 107, + "tokens/total": 3416192, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 48940 + }, + { + "epoch": 0.421875, + "grad_norm": 0.41617852449417114, + "learning_rate": 9.384962610259455e-05, + "loss": 0.008386321365833282, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00842, + "step": 108, + "tokens/total": 3448080, + "tokens/train_per_sec_per_gpu": 26.56, + "tokens/trainable": 49389 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1587515026330948, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0031663556583225727, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.03, + "memory/max_allocated (GiB)": 36.03, + "ppl": 1.00317, + "step": 109, + "tokens/total": 3477952, + "tokens/train_per_sec_per_gpu": 29.8, + "tokens/trainable": 49838 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2903148829936981, + "learning_rate": 9.355341126345868e-05, + "loss": 0.007958847098052502, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00799, + "step": 110, + "tokens/total": 3509776, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 50310 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.642983078956604, + "learning_rate": 9.340289419824107e-05, + "loss": 0.006748045329004526, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00677, + "step": 111, + "tokens/total": 3541808, + "tokens/train_per_sec_per_gpu": 28.04, + "tokens/trainable": 50755 + }, + { + "epoch": 0.4375, + "grad_norm": 0.14928282797336578, + "learning_rate": 9.325077903184159e-05, + "loss": 0.0027839094400405884, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00279, + "step": 112, + "tokens/total": 3574064, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 51213 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.05922335386276245, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0009622900979593396, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00096, + "step": 113, + "tokens/total": 3605984, + "tokens/train_per_sec_per_gpu": 27.99, + "tokens/trainable": 51660 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.44227731227874756, + "learning_rate": 9.2941779782269e-05, + "loss": 0.002763954224064946, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00277, + "step": 114, + "tokens/total": 3638080, + "tokens/train_per_sec_per_gpu": 27.03, + "tokens/trainable": 52133 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5176818370819092, + "learning_rate": 9.278490855781596e-05, + "loss": 0.015623578801751137, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01575, + "step": 115, + "tokens/total": 3670016, + "tokens/train_per_sec_per_gpu": 31.44, + "tokens/trainable": 52580 + }, + { + "epoch": 0.453125, + "grad_norm": 0.4265497922897339, + "learning_rate": 9.262646494908604e-05, + "loss": 0.00794076919555664, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00797, + "step": 116, + "tokens/total": 3701984, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 53033 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.40457987785339355, + "learning_rate": 9.246645554956457e-05, + "loss": 0.022708112373948097, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02297, + "step": 117, + "tokens/total": 3733920, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 53517 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.01414525043219328, + "learning_rate": 9.230488701789578e-05, + "loss": 0.00020209309877827764, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.0002, + "step": 118, + "tokens/total": 3765840, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 53967 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.11359831690788269, + "learning_rate": 9.214176607760577e-05, + "loss": 0.0012168455868959427, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00122, + "step": 119, + "tokens/total": 3797680, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 54430 + }, + { + "epoch": 0.46875, + "grad_norm": 0.35982534289360046, + "learning_rate": 9.197709951682268e-05, + "loss": 0.004110721405595541, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00412, + "step": 120, + "tokens/total": 3829920, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 54896 + }, + { + "epoch": 0.47265625, + "grad_norm": 1.3690382242202759, + "learning_rate": 9.181089418799428e-05, + "loss": 0.02705790475010872, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02743, + "step": 121, + "tokens/total": 3862080, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 55379 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.351891428232193, + "learning_rate": 9.164315700760271e-05, + "loss": 0.0020163573790341616, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00202, + "step": 122, + "tokens/total": 3893968, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 55803 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4747961461544037, + "learning_rate": 9.147389495587671e-05, + "loss": 0.012588979676365852, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01267, + "step": 123, + "tokens/total": 3926160, + "tokens/train_per_sec_per_gpu": 30.14, + "tokens/trainable": 56249 + }, + { + "epoch": 0.484375, + "grad_norm": 0.4064250588417053, + "learning_rate": 9.130311507650116e-05, + "loss": 0.0022941348142921925, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0023, + "step": 124, + "tokens/total": 3958208, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 56713 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.22056929767131805, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0022175831254571676, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00222, + "step": 125, + "tokens/total": 3990208, + "tokens/train_per_sec_per_gpu": 32.57, + "tokens/trainable": 57196 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5698862671852112, + "learning_rate": 9.09570303250602e-05, + "loss": 0.010205100290477276, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01026, + "step": 126, + "tokens/total": 4022224, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 57680 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.2593887448310852, + "learning_rate": 9.078173985499394e-05, + "loss": 0.0032277877908200026, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00323, + "step": 127, + "tokens/total": 4053952, + "tokens/train_per_sec_per_gpu": 26.51, + "tokens/trainable": 58110 + }, + { + "epoch": 0.5, + "grad_norm": 0.3735230565071106, + "learning_rate": 9.060496036067713e-05, + "loss": 0.01031105499714613, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01036, + "step": 128, + "tokens/total": 4086144, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 58534 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.1332484930753708, + "learning_rate": 9.042669919862615e-05, + "loss": 0.0012390739284455776, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00124, + "step": 129, + "tokens/total": 4118096, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 58998 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.20182718336582184, + "learning_rate": 9.024696378701557e-05, + "loss": 0.004501686431467533, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00451, + "step": 130, + "tokens/total": 4150272, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 59448 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.529080331325531, + "learning_rate": 9.006576160536948e-05, + "loss": 0.00763288140296936, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00766, + "step": 131, + "tokens/total": 4181904, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 59906 + }, + { + "epoch": 0.515625, + "grad_norm": 0.21625548601150513, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0024378676898777485, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00244, + "step": 132, + "tokens/total": 4214032, + "tokens/train_per_sec_per_gpu": 28.36, + "tokens/trainable": 60350 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.37116539478302, + "learning_rate": 8.969898715494506e-05, + "loss": 0.0037849934305995703, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00379, + "step": 133, + "tokens/total": 4246064, + "tokens/train_per_sec_per_gpu": 30.18, + "tokens/trainable": 60831 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5646623373031616, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006517104804515839, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00654, + "step": 134, + "tokens/total": 4278224, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 61305 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.6371963620185852, + "learning_rate": 8.932643689864568e-05, + "loss": 0.008687382563948631, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00873, + "step": 135, + "tokens/total": 4310112, + "tokens/train_per_sec_per_gpu": 31.48, + "tokens/trainable": 61792 + }, + { + "epoch": 0.53125, + "grad_norm": 0.8210524916648865, + "learning_rate": 8.913801518498845e-05, + "loss": 0.008436474949121475, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00847, + "step": 136, + "tokens/total": 4342192, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 62253 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.5692176222801208, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017321188002824783, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.01747, + "step": 137, + "tokens/total": 4374096, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.12828375399112701, + "learning_rate": 8.875691779131569e-05, + "loss": 0.0007754197577014565, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00078, + "step": 138, + "tokens/total": 4406272, + "tokens/train_per_sec_per_gpu": 24.22, + "tokens/trainable": 63145 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.3188190162181854, + "learning_rate": 8.856425797031829e-05, + "loss": 0.006707335356622934, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00673, + "step": 139, + "tokens/total": 4438112, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 63587 + }, + { + "epoch": 0.546875, + "grad_norm": 0.6216707229614258, + "learning_rate": 8.837020140354295e-05, + "loss": 0.007024695165455341, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00705, + "step": 140, + "tokens/total": 4470128, + "tokens/train_per_sec_per_gpu": 30.64, + "tokens/trainable": 64052 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.1480305939912796, + "learning_rate": 8.817475616647554e-05, + "loss": 0.00229115248657763, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00229, + "step": 141, + "tokens/total": 4502096, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 64526 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22254763543605804, + "learning_rate": 8.797793039239017e-05, + "loss": 0.0029198499396443367, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00292, + "step": 142, + "tokens/total": 4534496, + "tokens/train_per_sec_per_gpu": 29.08, + "tokens/trainable": 64983 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.6901941895484924, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013051149435341358, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01314, + "step": 143, + "tokens/total": 4566720, + "tokens/train_per_sec_per_gpu": 31.56, + "tokens/trainable": 65492 + }, + { + "epoch": 0.5625, + "grad_norm": 0.9545458555221558, + "learning_rate": 8.758017005316988e-05, + "loss": 0.02235381305217743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.02261, + "step": 144, + "tokens/total": 4598880, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 65925 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.5515892505645752, + "learning_rate": 8.737925204046629e-05, + "loss": 0.02038375660777092, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.02059, + "step": 145, + "tokens/total": 4630864, + "tokens/train_per_sec_per_gpu": 26.3, + "tokens/trainable": 66383 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3512430191040039, + "learning_rate": 8.717698659491851e-05, + "loss": 0.008233271539211273, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00827, + "step": 146, + "tokens/total": 4662864, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 66840 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.6830361485481262, + "learning_rate": 8.697338213361735e-05, + "loss": 0.013419180177152157, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01351, + "step": 147, + "tokens/total": 4694912, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 67297 + }, + { + "epoch": 0.578125, + "grad_norm": 0.2117031067609787, + "learning_rate": 8.676844712937552e-05, + "loss": 0.004134457092732191, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00414, + "step": 148, + "tokens/total": 4727120, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 67769 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.378053218126297, + "learning_rate": 8.656219011037509e-05, + "loss": 0.009747720323503017, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0098, + "step": 149, + "tokens/total": 4759168, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 68257 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.21076741814613342, + "learning_rate": 8.63546196598125e-05, + "loss": 0.005035653710365295, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00505, + "step": 150, + "tokens/total": 4791008, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 68706 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.25075697898864746, + "learning_rate": 8.614574441554145e-05, + "loss": 0.006027051247656345, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00605, + "step": 151, + "tokens/total": 4822912, + "tokens/train_per_sec_per_gpu": 27.95, + "tokens/trainable": 69131 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5774204730987549, + "learning_rate": 8.593557306971349e-05, + "loss": 0.018617186695337296, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01879, + "step": 152, + "tokens/total": 4854816, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 69586 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.1364293098449707, + "learning_rate": 8.572411436841618e-05, + "loss": 0.002049289643764496, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00205, + "step": 153, + "tokens/total": 4886656, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 70061 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.25387004017829895, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0044304742477834225, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00444, + "step": 154, + "tokens/total": 4918768, + "tokens/train_per_sec_per_gpu": 23.26, + "tokens/trainable": 70467 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3047125041484833, + "learning_rate": 8.529737015125824e-05, + "loss": 0.0043389927595853806, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00435, + "step": 155, + "tokens/total": 4951008, + "tokens/train_per_sec_per_gpu": 27.53, + "tokens/trainable": 70933 + }, + { + "epoch": 0.609375, + "grad_norm": 0.1442601978778839, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0031973645091056824, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.0032, + "step": 156, + "tokens/total": 4982976, + "tokens/train_per_sec_per_gpu": 28.86, + "tokens/trainable": 71382 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.2608455717563629, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0032951058819890022, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0033, + "step": 157, + "tokens/total": 5014640, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 71808 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.637295663356781, + "learning_rate": 8.464782037243449e-05, + "loss": 0.01436957623809576, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.01447, + "step": 158, + "tokens/total": 5046672, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 72249 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.9834056496620178, + "learning_rate": 8.442882418044202e-05, + "loss": 0.02375311404466629, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.02404, + "step": 159, + "tokens/total": 5078528, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 72726 + }, + { + "epoch": 0.625, + "grad_norm": 0.35432615876197815, + "learning_rate": 8.420860333495179e-05, + "loss": 0.006119017023593187, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00614, + "step": 160, + "tokens/total": 5110704, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 73148 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.44270065426826477, + "learning_rate": 8.398716700025208e-05, + "loss": 0.01352179329842329, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.72, + "memory/max_allocated (GiB)": 36.72, + "ppl": 1.01361, + "step": 161, + "tokens/total": 5142944, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 73600 + }, + { + "epoch": 0.6328125, + "grad_norm": 1.2181634902954102, + "learning_rate": 8.376452439121266e-05, + "loss": 0.01820964366197586, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.01838, + "step": 162, + "tokens/total": 5174592, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 74068 + }, + { + "epoch": 0.63671875, + "grad_norm": 1.9321303367614746, + "learning_rate": 8.354068477290124e-05, + "loss": 0.009742327965795994, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00979, + "step": 163, + "tokens/total": 5206432, + "tokens/train_per_sec_per_gpu": 29.89, + "tokens/trainable": 74527 + }, + { + "epoch": 0.640625, + "grad_norm": 0.3545796275138855, + "learning_rate": 8.331565746019807e-05, + "loss": 0.00975855067372322, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00981, + "step": 164, + "tokens/total": 5238384, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 74992 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.35368478298187256, + "learning_rate": 8.308945181740812e-05, + "loss": 0.010195466689765453, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01025, + "step": 165, + "tokens/total": 5270272, + "tokens/train_per_sec_per_gpu": 29.41, + "tokens/trainable": 75442 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.34957799315452576, + "learning_rate": 8.286207725787153e-05, + "loss": 0.009373681619763374, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00942, + "step": 166, + "tokens/total": 5302368, + "tokens/train_per_sec_per_gpu": 26.2, + "tokens/trainable": 75887 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.3684624135494232, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008211556822061539, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00825, + "step": 167, + "tokens/total": 5334496, + "tokens/train_per_sec_per_gpu": 28.21, + "tokens/trainable": 76331 + }, + { + "epoch": 0.65625, + "grad_norm": 0.12826858460903168, + "learning_rate": 8.240385928474219e-05, + "loss": 0.004492453299462795, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0045, + "step": 168, + "tokens/total": 5364288, + "tokens/train_per_sec_per_gpu": 30.16, + "tokens/trainable": 76745 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.31828173995018005, + "learning_rate": 8.217303493946967e-05, + "loss": 0.005733514670282602, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00575, + "step": 169, + "tokens/total": 5396512, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 77201 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6174353361129761, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01915649324655533, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01934, + "step": 170, + "tokens/total": 5428624, + "tokens/train_per_sec_per_gpu": 27.75, + "tokens/trainable": 77655 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.2902314066886902, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0036087254993617535, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00362, + "step": 171, + "tokens/total": 5460880, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 78122 + }, + { + "epoch": 0.671875, + "grad_norm": 0.3267754912376404, + "learning_rate": 8.147381587530713e-05, + "loss": 0.008090222254395485, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00812, + "step": 172, + "tokens/total": 5493088, + "tokens/train_per_sec_per_gpu": 27.86, + "tokens/trainable": 78576 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.4086189866065979, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006351984106004238, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00637, + "step": 173, + "tokens/total": 5525312, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 79059 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5261490345001221, + "learning_rate": 8.100214524900103e-05, + "loss": 0.008662862703204155, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.0087, + "step": 174, + "tokens/total": 5556880, + "tokens/train_per_sec_per_gpu": 24.62, + "tokens/trainable": 79498 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11292761564254761, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0027105510234832764, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00271, + "step": 175, + "tokens/total": 5588976, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 79922 + }, + { + "epoch": 0.6875, + "grad_norm": 0.33315548300743103, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0030926030594855547, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.0031, + "step": 176, + "tokens/total": 5618992, + "tokens/train_per_sec_per_gpu": 29.51, + "tokens/trainable": 80383 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.3136921226978302, + "learning_rate": 8.028654871074489e-05, + "loss": 0.005057850852608681, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00507, + "step": 177, + "tokens/total": 5651072, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 80824 + }, + { + "epoch": 0.6953125, + "grad_norm": 1.0526509284973145, + "learning_rate": 8.004589869885986e-05, + "loss": 0.010707498528063297, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01077, + "step": 178, + "tokens/total": 5683296, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 81243 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.1329648196697235, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0017975447699427605, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.0018, + "step": 179, + "tokens/total": 5715520, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 81716 + }, + { + "epoch": 0.703125, + "grad_norm": 1.0749201774597168, + "learning_rate": 7.95614819466576e-05, + "loss": 0.025617048144340515, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02595, + "step": 180, + "tokens/total": 5747488, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 82181 + }, + { + "epoch": 0.70703125, + "grad_norm": 1.0446465015411377, + "learning_rate": 7.931773536489872e-05, + "loss": 0.030236585065722466, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 35.9, + "memory/max_allocated (GiB)": 35.9, + "ppl": 1.0307, + "step": 181, + "tokens/total": 5777264, + "tokens/train_per_sec_per_gpu": 28.89, + "tokens/trainable": 82626 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22853964567184448, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0038214433006942272, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00383, + "step": 182, + "tokens/total": 5809264, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 83089 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.057821910828351974, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0004995397757738829, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0005, + "step": 183, + "tokens/total": 5841488, + "tokens/train_per_sec_per_gpu": 29.34, + "tokens/trainable": 83590 + }, + { + "epoch": 0.71875, + "grad_norm": 1.3264167308807373, + "learning_rate": 7.85804646415409e-05, + "loss": 0.013232000172138214, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01332, + "step": 184, + "tokens/total": 5873696, + "tokens/train_per_sec_per_gpu": 29.33, + "tokens/trainable": 84046 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.6027824878692627, + "learning_rate": 7.833273149760207e-05, + "loss": 0.012215660884976387, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01229, + "step": 185, + "tokens/total": 5905408, + "tokens/train_per_sec_per_gpu": 28.25, + "tokens/trainable": 84517 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.5848041772842407, + "learning_rate": 7.808402738346527e-05, + "loss": 0.024627480655908585, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.02493, + "step": 186, + "tokens/total": 5937392, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 84974 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.6823275089263916, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0028813849203288555, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00289, + "step": 187, + "tokens/total": 5969344, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 85460 + }, + { + "epoch": 0.734375, + "grad_norm": 0.41855958104133606, + "learning_rate": 7.758374768294647e-05, + "loss": 0.009796380065381527, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00984, + "step": 188, + "tokens/total": 6001296, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 85939 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6849080324172974, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004100777208805084, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00411, + "step": 189, + "tokens/total": 6031216, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 86377 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.15838374197483063, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003986812196671963, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00399, + "step": 190, + "tokens/total": 6063296, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 86834 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.3117486536502838, + "learning_rate": 7.682630588562518e-05, + "loss": 0.009132719598710537, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00917, + "step": 191, + "tokens/total": 6095216, + "tokens/train_per_sec_per_gpu": 26.65, + "tokens/trainable": 87265 + }, + { + "epoch": 0.75, + "grad_norm": 0.7837114334106445, + "learning_rate": 7.657199467573129e-05, + "loss": 0.007239446043968201, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00727, + "step": 192, + "tokens/total": 6127488, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 87747 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21777178347110748, + "learning_rate": 7.631678576708561e-05, + "loss": 0.002932134782895446, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00294, + "step": 193, + "tokens/total": 6159488, + "tokens/train_per_sec_per_gpu": 30.37, + "tokens/trainable": 88239 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.19159242510795593, + "learning_rate": 7.606068977997255e-05, + "loss": 0.003356864908710122, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00336, + "step": 194, + "tokens/total": 6191296, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 88718 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.3977915346622467, + "learning_rate": 7.580371737159148e-05, + "loss": 0.008549144491553307, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00859, + "step": 195, + "tokens/total": 6223120, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 89129 + }, + { + "epoch": 0.765625, + "grad_norm": 0.3875303566455841, + "learning_rate": 7.554587923561324e-05, + "loss": 0.006131038535386324, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00615, + "step": 196, + "tokens/total": 6254880, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 89567 + }, + { + "epoch": 0.76953125, + "grad_norm": 3.4133481979370117, + "learning_rate": 7.528718610173511e-05, + "loss": 0.01836150512099266, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01853, + "step": 197, + "tokens/total": 6286960, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 89988 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.2509997487068176, + "learning_rate": 7.502764873523431e-05, + "loss": 0.003821226069703698, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00383, + "step": 198, + "tokens/total": 6318880, + "tokens/train_per_sec_per_gpu": 30.29, + "tokens/trainable": 90434 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.9550195932388306, + "learning_rate": 7.476727793652011e-05, + "loss": 0.006872436031699181, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0069, + "step": 199, + "tokens/total": 6350928, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 90913 + }, + { + "epoch": 0.78125, + "grad_norm": 0.411724716424942, + "learning_rate": 7.450608454068415e-05, + "loss": 0.014243390411138535, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01435, + "step": 200, + "tokens/total": 6382976, + "tokens/train_per_sec_per_gpu": 28.29, + "tokens/trainable": 91355 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.34264692664146423, + "learning_rate": 7.424407941704987e-05, + "loss": 0.005309835076332092, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00532, + "step": 201, + "tokens/total": 6415120, + "tokens/train_per_sec_per_gpu": 31.41, + "tokens/trainable": 91856 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.41955795884132385, + "learning_rate": 7.398127346871986e-05, + "loss": 0.009196512401103973, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00924, + "step": 202, + "tokens/total": 6447216, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 92311 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.03107454814016819, + "learning_rate": 7.371767763212238e-05, + "loss": 0.0004908143309876323, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00049, + "step": 203, + "tokens/total": 6479280, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 92764 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39578482508659363, + "learning_rate": 7.345330287655617e-05, + "loss": 0.006089594680815935, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00611, + "step": 204, + "tokens/total": 6511040, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 93227 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.6444572806358337, + "learning_rate": 7.31881602037339e-05, + "loss": 0.010252210311591625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0103, + "step": 205, + "tokens/total": 6543120, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 93675 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.3283638060092926, + "learning_rate": 7.29222606473245e-05, + "loss": 0.00761406822130084, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00764, + "step": 206, + "tokens/total": 6575520, + "tokens/train_per_sec_per_gpu": 27.02, + "tokens/trainable": 94120 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.09952464699745178, + "learning_rate": 7.265561527249383e-05, + "loss": 0.001628706930205226, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00163, + "step": 207, + "tokens/total": 6607408, + "tokens/train_per_sec_per_gpu": 29.43, + "tokens/trainable": 94557 + }, + { + "epoch": 0.8125, + "grad_norm": 0.31704938411712646, + "learning_rate": 7.238823517544436e-05, + "loss": 0.005896817892789841, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00591, + "step": 208, + "tokens/total": 6639488, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 95035 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.0999862551689148, + "learning_rate": 7.212013148295333e-05, + "loss": 0.001754446537233889, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00176, + "step": 209, + "tokens/total": 6671520, + "tokens/train_per_sec_per_gpu": 31.08, + "tokens/trainable": 95517 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.1405109018087387, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0017940793186426163, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0018, + "step": 210, + "tokens/total": 6703536, + "tokens/train_per_sec_per_gpu": 30.3, + "tokens/trainable": 96026 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.09185238927602768, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0015102234901860356, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00151, + "step": 211, + "tokens/total": 6735616, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 96477 + }, + { + "epoch": 0.828125, + "grad_norm": 1.0464473962783813, + "learning_rate": 7.131159054949273e-05, + "loss": 0.016731251031160355, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01687, + "step": 212, + "tokens/total": 6767584, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 96951 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.3517152965068817, + "learning_rate": 7.104070433827139e-05, + "loss": 0.007984626106917858, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00802, + "step": 213, + "tokens/total": 6799488, + "tokens/train_per_sec_per_gpu": 25.83, + "tokens/trainable": 97350 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.28192469477653503, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00557567086070776, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00559, + "step": 214, + "tokens/total": 6831552, + "tokens/train_per_sec_per_gpu": 24.66, + "tokens/trainable": 97792 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.46135279536247253, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0170141588896513, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.01716, + "step": 215, + "tokens/total": 6863488, + "tokens/train_per_sec_per_gpu": 30.38, + "tokens/trainable": 98257 + }, + { + "epoch": 0.84375, + "grad_norm": 0.4277898073196411, + "learning_rate": 7.022408581865382e-05, + "loss": 0.009399959817528725, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.71, + "memory/max_allocated (GiB)": 36.71, + "ppl": 1.00944, + "step": 216, + "tokens/total": 6895888, + "tokens/train_per_sec_per_gpu": 27.28, + "tokens/trainable": 98731 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.34988704323768616, + "learning_rate": 6.99505974422157e-05, + "loss": 0.002909669652581215, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00291, + "step": 217, + "tokens/total": 6928000, + "tokens/train_per_sec_per_gpu": 30.25, + "tokens/trainable": 99212 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.4028005599975586, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0058855339884757996, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0059, + "step": 218, + "tokens/total": 6960096, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 99654 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.3000575304031372, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0025230362080037594, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00253, + "step": 219, + "tokens/total": 6992512, + "tokens/train_per_sec_per_gpu": 27.46, + "tokens/trainable": 100086 + }, + { + "epoch": 0.859375, + "grad_norm": 0.11351172626018524, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0021392193157225847, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00214, + "step": 220, + "tokens/total": 7024688, + "tokens/train_per_sec_per_gpu": 27.56, + "tokens/trainable": 100524 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.19512486457824707, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00650001410394907, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00652, + "step": 221, + "tokens/total": 7056928, + "tokens/train_per_sec_per_gpu": 28.05, + "tokens/trainable": 100996 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.1998104453086853, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0056140003725886345, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00563, + "step": 222, + "tokens/total": 7088944, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 101449 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.40908998250961304, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01286369375884533, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.01295, + "step": 223, + "tokens/total": 7121104, + "tokens/train_per_sec_per_gpu": 27.19, + "tokens/trainable": 101905 + }, + { + "epoch": 0.875, + "grad_norm": 0.3672473132610321, + "learning_rate": 6.801939899284132e-05, + "loss": 0.010677668265998363, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01073, + "step": 224, + "tokens/total": 7153040, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 102411 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4146183431148529, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0038945709820836782, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0039, + "step": 225, + "tokens/total": 7184912, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 102882 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.10585323721170425, + "learning_rate": 6.746257910210214e-05, + "loss": 0.00041726604104042053, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00042, + "step": 226, + "tokens/total": 7216992, + "tokens/train_per_sec_per_gpu": 28.72, + "tokens/trainable": 103332 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.36600011587142944, + "learning_rate": 6.718338543014937e-05, + "loss": 0.01526213251054287, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 227, + "tokens/total": 7249120, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 103790 + }, + { + "epoch": 0.890625, + "grad_norm": 0.16057318449020386, + "learning_rate": 6.69036847577983e-05, + "loss": 0.004024423658847809, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00403, + "step": 228, + "tokens/total": 7281072, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 104246 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.318798303604126, + "learning_rate": 6.662348872453553e-05, + "loss": 0.006663155741989613, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00669, + "step": 229, + "tokens/total": 7313120, + "tokens/train_per_sec_per_gpu": 31.3, + "tokens/trainable": 104707 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.10858234763145447, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0020512123592197895, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00205, + "step": 230, + "tokens/total": 7345184, + "tokens/train_per_sec_per_gpu": 30.81, + "tokens/trainable": 105167 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.1044008731842041, + "learning_rate": 6.60616572358065e-05, + "loss": 0.0010264476295560598, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00103, + "step": 231, + "tokens/total": 7377264, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 105576 + }, + { + "epoch": 0.90625, + "grad_norm": 0.14836947619915009, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0013845522189512849, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00139, + "step": 232, + "tokens/total": 7409424, + "tokens/train_per_sec_per_gpu": 30.32, + "tokens/trainable": 106043 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.022752461954951286, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0003245974367018789, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00032, + "step": 233, + "tokens/total": 7441456, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 106506 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.22981758415699005, + "learning_rate": 6.521548694236384e-05, + "loss": 0.002587020630016923, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00259, + "step": 234, + "tokens/total": 7473344, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 106951 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.5147526264190674, + "learning_rate": 6.493256429322259e-05, + "loss": 0.01256126631051302, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01264, + "step": 235, + "tokens/total": 7505040, + "tokens/train_per_sec_per_gpu": 25.64, + "tokens/trainable": 107382 + }, + { + "epoch": 0.921875, + "grad_norm": 0.3785130977630615, + "learning_rate": 6.464922830953799e-05, + "loss": 0.010956985875964165, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01102, + "step": 236, + "tokens/total": 7537120, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 107814 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.6193199753761292, + "learning_rate": 6.436549078207688e-05, + "loss": 0.020655140280723572, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02087, + "step": 237, + "tokens/total": 7569024, + "tokens/train_per_sec_per_gpu": 29.36, + "tokens/trainable": 108274 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.5084519982337952, + "learning_rate": 6.408136351831592e-05, + "loss": 0.011346567422151566, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01141, + "step": 238, + "tokens/total": 7600992, + "tokens/train_per_sec_per_gpu": 24.43, + "tokens/trainable": 108714 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.04367341846227646, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0004316020058467984, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00043, + "step": 239, + "tokens/total": 7633280, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 109220 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06729783862829208, + "learning_rate": 6.351198709240186e-05, + "loss": 0.000604260538239032, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.0006, + "step": 240, + "tokens/total": 7665408, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 109672 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.3010976314544678, + "learning_rate": 6.32267616243259e-05, + "loss": 0.006106264889240265, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00612, + "step": 241, + "tokens/total": 7697280, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 110135 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.11426915228366852, + "learning_rate": 6.294119380711849e-05, + "loss": 0.0014383264351636171, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00144, + "step": 242, + "tokens/total": 7729440, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 110563 + }, + { + "epoch": 0.94921875, + "grad_norm": 1.3792941570281982, + "learning_rate": 6.265529552442209e-05, + "loss": 0.003547664964571595, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00355, + "step": 243, + "tokens/total": 7761584, + "tokens/train_per_sec_per_gpu": 30.76, + "tokens/trainable": 111049 + }, + { + "epoch": 0.953125, + "grad_norm": 0.07595008611679077, + "learning_rate": 6.236907867363127e-05, + "loss": 0.0009103374904952943, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00091, + "step": 244, + "tokens/total": 7793824, + "tokens/train_per_sec_per_gpu": 28.23, + "tokens/trainable": 111495 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05425361171364784, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0013249842450022697, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00133, + "step": 245, + "tokens/total": 7825872, + "tokens/train_per_sec_per_gpu": 26.47, + "tokens/trainable": 111968 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.17112742364406586, + "learning_rate": 6.179573692313344e-05, + "loss": 0.001098912674933672, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.0011, + "step": 246, + "tokens/total": 7857840, + "tokens/train_per_sec_per_gpu": 30.09, + "tokens/trainable": 112416 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.10754083096981049, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0031517897732555866, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00316, + "step": 247, + "tokens/total": 7890000, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 112882 + }, + { + "epoch": 0.96875, + "grad_norm": 0.2477901428937912, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005755824968218803, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00577, + "step": 248, + "tokens/total": 7921984, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 113390 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.332253634929657, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.013207933865487576, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.0133, + "step": 249, + "tokens/total": 7953824, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 113904 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.18514207005500793, + "learning_rate": 6.064575550087316e-05, + "loss": 0.00523729994893074, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00525, + "step": 250, + "tokens/total": 7985984, + "tokens/train_per_sec_per_gpu": 28.41, + "tokens/trainable": 114351 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.1852743923664093, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0031474479474127293, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00315, + "step": 251, + "tokens/total": 8018256, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 114842 + }, + { + "epoch": 0.984375, + "grad_norm": 0.43020468950271606, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0053521618247032166, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00537, + "step": 252, + "tokens/total": 8050528, + "tokens/train_per_sec_per_gpu": 23.97, + "tokens/trainable": 115263 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.048135045915842056, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0005759480991400778, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00058, + "step": 253, + "tokens/total": 8082496, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 115703 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.028733640909194946, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00040354474913328886, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0004, + "step": 254, + "tokens/total": 8114592, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 116157 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.2808869481086731, + "learning_rate": 5.920308275189541e-05, + "loss": 0.006315852981060743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00634, + "step": 255, + "tokens/total": 8146752, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 116567 + }, + { + "epoch": 1.0, + "grad_norm": 0.054996367543935776, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0009099108865484595, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00091, + "step": 256, + "tokens/total": 8178768, + "tokens/train_per_sec_per_gpu": 25.98, + "tokens/trainable": 117030 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.18114623427391052, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.002796629210934043, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0028, + "step": 257, + "tokens/total": 8210800, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 117517 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.1408185362815857, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0013210453325882554, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00132, + "step": 258, + "tokens/total": 8242768, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 118015 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.36587440967559814, + "learning_rate": 5.80457242789548e-05, + "loss": 0.002160525880753994, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00216, + "step": 259, + "tokens/total": 8274784, + "tokens/train_per_sec_per_gpu": 30.13, + "tokens/trainable": 118520 + }, + { + "epoch": 1.015625, + "grad_norm": 0.02628348581492901, + "learning_rate": 5.77560376810977e-05, + "loss": 0.0003391164937056601, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00034, + "step": 260, + "tokens/total": 8306720, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 118992 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.014996240846812725, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0002124913444276899, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00021, + "step": 261, + "tokens/total": 8338784, + "tokens/train_per_sec_per_gpu": 27.96, + "tokens/trainable": 119438 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.0050656464882195, + "learning_rate": 5.717633247526522e-05, + "loss": 7.557802018709481e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00008, + "step": 262, + "tokens/total": 8370720, + "tokens/train_per_sec_per_gpu": 26.9, + "tokens/trainable": 119881 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.02459302730858326, + "learning_rate": 5.688633799118971e-05, + "loss": 0.00024908874183893204, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00025, + "step": 263, + "tokens/total": 8402736, + "tokens/train_per_sec_per_gpu": 23.2, + "tokens/trainable": 120285 + }, + { + "epoch": 1.03125, + "grad_norm": 0.017813201993703842, + "learning_rate": 5.659626500889066e-05, + "loss": 0.00021895825921092182, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00022, + "step": 264, + "tokens/total": 8434352, + "tokens/train_per_sec_per_gpu": 27.78, + "tokens/trainable": 120755 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.0038785531651228666, + "learning_rate": 5.6306125599488905e-05, + "loss": 6.556476728292182e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00007, + "step": 265, + "tokens/total": 8466208, + "tokens/train_per_sec_per_gpu": 29.86, + "tokens/trainable": 121194 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.18033447861671448, + "learning_rate": 5.601593183686955e-05, + "loss": 0.0016840343596413732, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00169, + "step": 266, + "tokens/total": 8498304, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 121670 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.22497256100177765, + "learning_rate": 5.572569579717961e-05, + "loss": 0.003026450052857399, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00303, + "step": 267, + "tokens/total": 8530400, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 122142 + }, + { + "epoch": 1.046875, + "grad_norm": 0.008878038264811039, + "learning_rate": 5.543542955832538e-05, + "loss": 8.136438555084169e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00008, + "step": 268, + "tokens/total": 8562608, + "tokens/train_per_sec_per_gpu": 27.27, + "tokens/trainable": 122585 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.19042901694774628, + "learning_rate": 5.514514519946986e-05, + "loss": 0.0020417363848537207, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00204, + "step": 269, + "tokens/total": 8594832, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 123091 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.003382593160495162, + "learning_rate": 5.485485480053015e-05, + "loss": 3.870048021781258e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00004, + "step": 270, + "tokens/total": 8627120, + "tokens/train_per_sec_per_gpu": 26.36, + "tokens/trainable": 123505 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.2004430592060089, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.002532299840822816, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.00254, + "step": 271, + "tokens/total": 8657056, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 123953 + }, + { + "epoch": 1.0625, + "grad_norm": 0.24857956171035767, + "learning_rate": 5.42743042028204e-05, + "loss": 0.004157304298132658, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00417, + "step": 272, + "tokens/total": 8689104, + "tokens/train_per_sec_per_gpu": 24.05, + "tokens/trainable": 124400 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.05064955726265907, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0006531125982291996, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00065, + "step": 273, + "tokens/total": 8721136, + "tokens/train_per_sec_per_gpu": 29.3, + "tokens/trainable": 124870 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.12664659321308136, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0018034171080216765, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00181, + "step": 274, + "tokens/total": 8753200, + "tokens/train_per_sec_per_gpu": 26.52, + "tokens/trainable": 125333 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.01593201234936714, + "learning_rate": 5.340373499110935e-05, + "loss": 0.00010935973114101216, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00011, + "step": 275, + "tokens/total": 8785168, + "tokens/train_per_sec_per_gpu": 30.45, + "tokens/trainable": 125834 + }, + { + "epoch": 1.078125, + "grad_norm": 0.0015056057600304484, + "learning_rate": 5.3113662008810304e-05, + "loss": 1.9860939573845826e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00002, + "step": 276, + "tokens/total": 8817520, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 126316 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.004457728937268257, + "learning_rate": 5.282366752473479e-05, + "loss": 2.716448398132343e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00003, + "step": 277, + "tokens/total": 8849360, + "tokens/train_per_sec_per_gpu": 29.63, + "tokens/trainable": 126798 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.03812727704644203, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.0003402164438739419, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.00034, + "step": 278, + "tokens/total": 8881152, + "tokens/train_per_sec_per_gpu": 24.5, + "tokens/trainable": 127223 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.011409996077418327, + "learning_rate": 5.224396231890232e-05, + "loss": 9.064963523996994e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.00009, + "step": 279, + "tokens/total": 8912512, + "tokens/train_per_sec_per_gpu": 29.78, + "tokens/trainable": 127672 + }, + { + "epoch": 1.09375, + "grad_norm": 0.037030577659606934, + "learning_rate": 5.195427572104522e-05, + "loss": 0.0003734154161065817, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00037, + "step": 280, + "tokens/total": 8944416, + "tokens/train_per_sec_per_gpu": 26.76, + "tokens/trainable": 128116 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.0032268385402858257, + "learning_rate": 5.166471586820751e-05, + "loss": 2.529611811041832e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00003, + "step": 281, + "tokens/total": 8976272, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 128589 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.20227248966693878, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.0024647137615829706, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00247, + "step": 282, + "tokens/total": 9008560, + "tokens/train_per_sec_per_gpu": 27.84, + "tokens/trainable": 129036 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.43708595633506775, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0028322634752839804, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00284, + "step": 283, + "tokens/total": 9040496, + "tokens/train_per_sec_per_gpu": 29.25, + "tokens/trainable": 129485 + }, + { + "epoch": 1.109375, + "grad_norm": 0.09467080235481262, + "learning_rate": 5.079691724810461e-05, + "loss": 0.001156509853899479, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00116, + "step": 284, + "tokens/total": 9072448, + "tokens/train_per_sec_per_gpu": 24.69, + "tokens/trainable": 129920 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.1778785139322281, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.0016857109731063247, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00169, + "step": 285, + "tokens/total": 9104384, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 130417 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.0003376641252543777, + "learning_rate": 5.021923930849237e-05, + "loss": 5.597693871095544e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00001, + "step": 286, + "tokens/total": 9136416, + "tokens/train_per_sec_per_gpu": 29.99, + "tokens/trainable": 130903 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.019782084971666336, + "learning_rate": 4.99306927511967e-05, + "loss": 0.00012461614096537232, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00012, + "step": 287, + "tokens/total": 9168640, + "tokens/train_per_sec_per_gpu": 29.24, + "tokens/trainable": 131343 + }, + { + "epoch": 1.125, + "grad_norm": 0.9641065001487732, + "learning_rate": 4.964235714846775e-05, + "loss": 0.011515076272189617, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01158, + "step": 288, + "tokens/total": 9200544, + "tokens/train_per_sec_per_gpu": 27.04, + "tokens/trainable": 131763 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0010062052169814706, + "learning_rate": 4.9354244499126866e-05, + "loss": 9.489545846008696e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00001, + "step": 289, + "tokens/total": 9232624, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 132229 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.002373398980125785, + "learning_rate": 4.90663667927174e-05, + "loss": 1.573777262819931e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00002, + "step": 290, + "tokens/total": 9264768, + "tokens/train_per_sec_per_gpu": 29.5, + "tokens/trainable": 132678 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.0014718567254021764, + "learning_rate": 4.877873600900581e-05, + "loss": 9.888429303828161e-06, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00001, + "step": 291, + "tokens/total": 9296944, + "tokens/train_per_sec_per_gpu": 24.47, + "tokens/trainable": 133136 + }, + { + "epoch": 1.140625, + "grad_norm": 0.0018440725980326533, + "learning_rate": 4.849136411748306e-05, + "loss": 8.275845175376162e-06, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00001, + "step": 292, + "tokens/total": 9328848, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 133608 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.12918472290039062, + "learning_rate": 4.8204263076866574e-05, + "loss": 0.0009128568926826119, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00091, + "step": 293, + "tokens/total": 9360928, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 134108 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.019195031374692917, + "learning_rate": 4.791744483460251e-05, + "loss": 9.439548011869192e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00009, + "step": 294, + "tokens/total": 9393040, + "tokens/train_per_sec_per_gpu": 25.71, + "tokens/trainable": 134541 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.1602204144001007, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0004640200058929622, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00046, + "step": 295, + "tokens/total": 9425248, + "tokens/train_per_sec_per_gpu": 24.5, + "tokens/trainable": 134966 + }, + { + "epoch": 1.15625, + "grad_norm": 0.5406288504600525, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.0034592358861118555, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00347, + "step": 296, + "tokens/total": 9457264, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 135402 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.004333238583058119, + "learning_rate": 4.705880619288153e-05, + "loss": 3.075868880841881e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00003, + "step": 297, + "tokens/total": 9489056, + "tokens/train_per_sec_per_gpu": 24.87, + "tokens/trainable": 135804 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.006247827783226967, + "learning_rate": 4.677323837567412e-05, + "loss": 2.383375249337405e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00002, + "step": 298, + "tokens/total": 9521024, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 136231 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.0779612734913826, + "learning_rate": 4.6488012907598146e-05, + "loss": 0.0004679379053413868, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00047, + "step": 299, + "tokens/total": 9553120, + "tokens/train_per_sec_per_gpu": 28.37, + "tokens/trainable": 136705 + }, + { + "epoch": 1.171875, + "grad_norm": 0.01816786825656891, + "learning_rate": 4.620314165804964e-05, + "loss": 6.801338167861104e-05, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00007, + "step": 300, + "tokens/total": 9585040, + "tokens/train_per_sec_per_gpu": 28.78, + "tokens/trainable": 137178 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.9530329704284668, + "learning_rate": 4.591863648168407e-05, + "loss": 0.008843549527227879, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00888, + "step": 301, + "tokens/total": 9617120, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 137643 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0371168851852417, + "learning_rate": 4.5634509217923135e-05, + "loss": 0.00014635953994002193, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00015, + "step": 302, + "tokens/total": 9649024, + "tokens/train_per_sec_per_gpu": 26.74, + "tokens/trainable": 138078 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.08589725196361542, + "learning_rate": 4.535077169046201e-05, + "loss": 0.00043714733328670263, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00044, + "step": 303, + "tokens/total": 9680832, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 138515 + }, + { + "epoch": 1.1875, + "grad_norm": 0.358940988779068, + "learning_rate": 4.506743570677743e-05, + "loss": 0.006240838672965765, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00626, + "step": 304, + "tokens/total": 9713040, + "tokens/train_per_sec_per_gpu": 29.29, + "tokens/trainable": 138948 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.12659867107868195, + "learning_rate": 4.478451305763618e-05, + "loss": 0.0008667556685395539, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00087, + "step": 305, + "tokens/total": 9744976, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 139408 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.08289908617734909, + "learning_rate": 4.450201551660454e-05, + "loss": 0.0006384386215358973, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.02, + "memory/max_allocated (GiB)": 36.02, + "ppl": 1.00064, + "step": 306, + "tokens/total": 9774784, + "tokens/train_per_sec_per_gpu": 27.76, + "tokens/trainable": 139840 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.19195082783699036, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.002342985011637211, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00235, + "step": 307, + "tokens/total": 9806656, + "tokens/train_per_sec_per_gpu": 27.6, + "tokens/trainable": 140281 + }, + { + "epoch": 1.203125, + "grad_norm": 0.07421132177114487, + "learning_rate": 4.393834276419352e-05, + "loss": 0.0007085398538038135, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00071, + "step": 308, + "tokens/total": 9838720, + "tokens/train_per_sec_per_gpu": 30.88, + "tokens/trainable": 140776 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.004596009384840727, + "learning_rate": 4.36571910095382e-05, + "loss": 2.8969257982680574e-05, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00003, + "step": 309, + "tokens/total": 9870784, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 141228 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.04099060595035553, + "learning_rate": 4.337651127546448e-05, + "loss": 0.00030985785997472703, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00031, + "step": 310, + "tokens/total": 9902784, + "tokens/train_per_sec_per_gpu": 30.8, + "tokens/trainable": 141679 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.18840794265270233, + "learning_rate": 4.3096315242201736e-05, + "loss": 0.0020186181645840406, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00202, + "step": 311, + "tokens/total": 9934768, + "tokens/train_per_sec_per_gpu": 26.72, + "tokens/trainable": 142122 + }, + { + "epoch": 1.21875, + "grad_norm": 0.5018841028213501, + "learning_rate": 4.2816614569850635e-05, + "loss": 0.006148474756628275, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00617, + "step": 312, + "tokens/total": 9966960, + "tokens/train_per_sec_per_gpu": 27.27, + "tokens/trainable": 142561 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.1730002760887146, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.003579454030841589, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00359, + "step": 313, + "tokens/total": 9999104, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 143046 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.6586350202560425, + "learning_rate": 4.225874584473174e-05, + "loss": 0.0015231478027999401, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00152, + "step": 314, + "tokens/total": 10031136, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 143493 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.011105087585747242, + "learning_rate": 4.19806010071587e-05, + "loss": 0.00011838486534543335, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00012, + "step": 315, + "tokens/total": 10063184, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 143956 + }, + { + "epoch": 1.234375, + "grad_norm": 0.2628834843635559, + "learning_rate": 4.170299795992081e-05, + "loss": 0.003402995876967907, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00341, + "step": 316, + "tokens/total": 10095312, + "tokens/train_per_sec_per_gpu": 28.29, + "tokens/trainable": 144411 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.13821958005428314, + "learning_rate": 4.142594825521398e-05, + "loss": 0.0016696923412382603, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00167, + "step": 317, + "tokens/total": 10127440, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 144892 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.0023418583441525698, + "learning_rate": 4.114946342220728e-05, + "loss": 2.6753859856398776e-05, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00003, + "step": 318, + "tokens/total": 10159184, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 145339 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.0008433983894065022, + "learning_rate": 4.087355496656321e-05, + "loss": 1.602486736373976e-05, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00002, + "step": 319, + "tokens/total": 10191216, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 145811 + }, + { + "epoch": 1.25, + "grad_norm": 0.12588869035243988, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0011584729654714465, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00116, + "step": 320, + "tokens/total": 10223408, + "tokens/train_per_sec_per_gpu": 32.27, + "tokens/trainable": 146314 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.028395529836416245, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.00015757072833366692, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00016, + "step": 321, + "tokens/total": 10255536, + "tokens/train_per_sec_per_gpu": 26.68, + "tokens/trainable": 146781 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.039599668234586716, + "learning_rate": 4.004940255778431e-05, + "loss": 0.000321136845741421, + "memory/device_reserved (GiB)": 37.39, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00032, + "step": 322, + "tokens/total": 10287664, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 147225 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.11097095906734467, + "learning_rate": 3.977591418134619e-05, + "loss": 0.0004549310833681375, + "memory/device_reserved (GiB)": 37.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00046, + "step": 323, + "tokens/total": 10320080, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 147673 + }, + { + "epoch": 1.265625, + "grad_norm": 0.19018623232841492, + "learning_rate": 3.95030593412612e-05, + "loss": 0.0028336881659924984, + "memory/device_reserved (GiB)": 38.53, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00284, + "step": 324, + "tokens/total": 10351984, + "tokens/train_per_sec_per_gpu": 27.33, + "tokens/trainable": 148103 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.4650349020957947, + "learning_rate": 3.923084939213296e-05, + "loss": 0.006978687364608049, + "memory/device_reserved (GiB)": 38.53, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.007, + "step": 325, + "tokens/total": 10384256, + "tokens/train_per_sec_per_gpu": 26.21, + "tokens/trainable": 148535 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.14811205863952637, + "learning_rate": 3.895929566172861e-05, + "loss": 0.0014966176822781563, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0015, + "step": 326, + "tokens/total": 10416208, + "tokens/train_per_sec_per_gpu": 28.37, + "tokens/trainable": 148999 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.11302439123392105, + "learning_rate": 3.868840945050728e-05, + "loss": 0.0011318891774863005, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00113, + "step": 327, + "tokens/total": 10448224, + "tokens/train_per_sec_per_gpu": 26.37, + "tokens/trainable": 149431 + }, + { + "epoch": 1.28125, + "grad_norm": 0.10916705429553986, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0009605163359083235, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00096, + "step": 328, + "tokens/total": 10480448, + "tokens/train_per_sec_per_gpu": 28.27, + "tokens/trainable": 149886 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.0004018457839265466, + "learning_rate": 3.814868464809027e-05, + "loss": 7.040077434794512e-06, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00001, + "step": 329, + "tokens/total": 10512096, + "tokens/train_per_sec_per_gpu": 25.92, + "tokens/trainable": 150288 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.03717898577451706, + "learning_rate": 3.787986851704667e-05, + "loss": 0.0001680658315308392, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00017, + "step": 330, + "tokens/total": 10544000, + "tokens/train_per_sec_per_gpu": 28.52, + "tokens/trainable": 150733 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.21975846588611603, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00015534088015556335, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00016, + "step": 331, + "tokens/total": 10576048, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 151207 + }, + { + "epoch": 1.296875, + "grad_norm": 0.10715529322624207, + "learning_rate": 3.734438472750619e-05, + "loss": 0.0012046336196362972, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00121, + "step": 332, + "tokens/total": 10608048, + "tokens/train_per_sec_per_gpu": 31.08, + "tokens/trainable": 151694 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.13602837920188904, + "learning_rate": 3.707773935267552e-05, + "loss": 0.00039335948531515896, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00039, + "step": 333, + "tokens/total": 10640176, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 152188 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.002963080769404769, + "learning_rate": 3.68118397962661e-05, + "loss": 6.6289721871726215e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00001, + "step": 334, + "tokens/total": 10672224, + "tokens/train_per_sec_per_gpu": 25.55, + "tokens/trainable": 152596 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.007236327510327101, + "learning_rate": 3.654669712344384e-05, + "loss": 5.341085125110112e-05, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00005, + "step": 335, + "tokens/total": 10704192, + "tokens/train_per_sec_per_gpu": 30.44, + "tokens/trainable": 153052 + }, + { + "epoch": 1.3125, + "grad_norm": 0.0897810310125351, + "learning_rate": 3.628232236787763e-05, + "loss": 0.0008687702938914299, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00087, + "step": 336, + "tokens/total": 10736272, + "tokens/train_per_sec_per_gpu": 25.33, + "tokens/trainable": 153491 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.011829929426312447, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.00011201453162357211, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00011, + "step": 337, + "tokens/total": 10768528, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 153983 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.25218063592910767, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0022679665125906467, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00227, + "step": 338, + "tokens/total": 10800464, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 154464 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.00034703267738223076, + "learning_rate": 3.549391545931585e-05, + "loss": 2.884747800635523e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.0, + "step": 339, + "tokens/total": 10832512, + "tokens/train_per_sec_per_gpu": 29.7, + "tokens/trainable": 154924 + }, + { + "epoch": 1.328125, + "grad_norm": 0.004890776239335537, + "learning_rate": 3.5232722063479914e-05, + "loss": 2.2697908207192086e-05, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00002, + "step": 340, + "tokens/total": 10864624, + "tokens/train_per_sec_per_gpu": 25.46, + "tokens/trainable": 155373 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.0013245025184005499, + "learning_rate": 3.49723512647657e-05, + "loss": 8.980642633105163e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00001, + "step": 341, + "tokens/total": 10896816, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 155873 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.03804129734635353, + "learning_rate": 3.471281389826491e-05, + "loss": 0.00021880728309042752, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00022, + "step": 342, + "tokens/total": 10928864, + "tokens/train_per_sec_per_gpu": 25.31, + "tokens/trainable": 156278 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.4880214333534241, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.005397590808570385, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00541, + "step": 343, + "tokens/total": 10960816, + "tokens/train_per_sec_per_gpu": 29.58, + "tokens/trainable": 156733 + }, + { + "epoch": 1.34375, + "grad_norm": 0.37780871987342834, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0029298821464180946, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00293, + "step": 344, + "tokens/total": 10992912, + "tokens/train_per_sec_per_gpu": 28.4, + "tokens/trainable": 157203 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.0034261250402778387, + "learning_rate": 3.3939310220027456e-05, + "loss": 2.1256186300888658e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00002, + "step": 345, + "tokens/total": 11024832, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 157707 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.0005507867899723351, + "learning_rate": 3.3683214232914404e-05, + "loss": 6.991718692006543e-06, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00001, + "step": 346, + "tokens/total": 11056992, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 158180 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.0016383324982598424, + "learning_rate": 3.342800532426873e-05, + "loss": 1.2630136552616023e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00001, + "step": 347, + "tokens/total": 11089264, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 158646 + }, + { + "epoch": 1.359375, + "grad_norm": 0.0020912184845656157, + "learning_rate": 3.317369411437484e-05, + "loss": 1.536883064545691e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00002, + "step": 348, + "tokens/total": 11120976, + "tokens/train_per_sec_per_gpu": 29.34, + "tokens/trainable": 159115 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.0016828920925036073, + "learning_rate": 3.292029118616024e-05, + "loss": 1.334734679403482e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00001, + "step": 349, + "tokens/total": 11152976, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 159538 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.02520265430212021, + "learning_rate": 3.266780708475511e-05, + "loss": 9.595962183084339e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0001, + "step": 350, + "tokens/total": 11185056, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 160016 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.009836713783442974, + "learning_rate": 3.241625231705354e-05, + "loss": 7.521865336457267e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00008, + "step": 351, + "tokens/total": 11217088, + "tokens/train_per_sec_per_gpu": 28.51, + "tokens/trainable": 160475 + }, + { + "epoch": 1.375, + "grad_norm": 0.0698060467839241, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0003666624252218753, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00037, + "step": 352, + "tokens/total": 11249088, + "tokens/train_per_sec_per_gpu": 28.31, + "tokens/trainable": 160952 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.0018428800394758582, + "learning_rate": 3.191597261653475e-05, + "loss": 1.3194729945098516e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00001, + "step": 353, + "tokens/total": 11281328, + "tokens/train_per_sec_per_gpu": 27.4, + "tokens/trainable": 161420 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.020005319267511368, + "learning_rate": 3.166726850239794e-05, + "loss": 0.00017299644241575152, + "memory/device_reserved (GiB)": 38.47, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00017, + "step": 354, + "tokens/total": 11313200, + "tokens/train_per_sec_per_gpu": 28.31, + "tokens/trainable": 161868 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.00018166302470490336, + "learning_rate": 3.141953535845912e-05, + "loss": 4.138089934713207e-06, + "memory/device_reserved (GiB)": 38.47, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0, + "step": 355, + "tokens/total": 11345008, + "tokens/train_per_sec_per_gpu": 29.74, + "tokens/trainable": 162318 + }, + { + "epoch": 1.390625, + "grad_norm": 0.009165152907371521, + "learning_rate": 3.11727834939056e-05, + "loss": 6.729022425133735e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00007, + "step": 356, + "tokens/total": 11377216, + "tokens/train_per_sec_per_gpu": 26.26, + "tokens/trainable": 162789 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.2891274690628052, + "learning_rate": 3.092702317708967e-05, + "loss": 0.005585570354014635, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.0056, + "step": 357, + "tokens/total": 11409536, + "tokens/train_per_sec_per_gpu": 30.34, + "tokens/trainable": 163254 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.0007154783816076815, + "learning_rate": 3.0682264635101276e-05, + "loss": 5.591478839050978e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00001, + "step": 358, + "tokens/total": 11441552, + "tokens/train_per_sec_per_gpu": 28.74, + "tokens/trainable": 163715 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.01731656678020954, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.0001050045175361447, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00011, + "step": 359, + "tokens/total": 11473376, + "tokens/train_per_sec_per_gpu": 28.79, + "tokens/trainable": 164197 + }, + { + "epoch": 1.40625, + "grad_norm": 0.11494546383619308, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.00048459687968716025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00048, + "step": 360, + "tokens/total": 11505312, + "tokens/train_per_sec_per_gpu": 26.54, + "tokens/trainable": 164661 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.08215854316949844, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.0002725277154240757, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00027, + "step": 361, + "tokens/total": 11537536, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 165111 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.024266909807920456, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.0001677784021012485, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 35.94, + "memory/max_allocated (GiB)": 35.94, + "ppl": 1.00017, + "step": 362, + "tokens/total": 11567216, + "tokens/train_per_sec_per_gpu": 25.3, + "tokens/trainable": 165552 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.008205167017877102, + "learning_rate": 2.9473853553877484e-05, + "loss": 3.423610905883834e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00003, + "step": 363, + "tokens/total": 11599312, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 166026 + }, + { + "epoch": 1.421875, + "grad_norm": 0.4230109453201294, + "learning_rate": 2.9235318065647e-05, + "loss": 0.005893740337342024, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00591, + "step": 364, + "tokens/total": 11631344, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 166486 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.1095249131321907, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.0006186347454786301, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00062, + "step": 365, + "tokens/total": 11663632, + "tokens/train_per_sec_per_gpu": 30.17, + "tokens/trainable": 166959 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.0014034683117642999, + "learning_rate": 2.8761473491751258e-05, + "loss": 1.2728614819934592e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.27, + "memory/max_allocated (GiB)": 36.27, + "ppl": 1.00001, + "step": 366, + "tokens/total": 11695360, + "tokens/train_per_sec_per_gpu": 28.19, + "tokens/trainable": 167394 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.044087354093790054, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.0001466910180170089, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00015, + "step": 367, + "tokens/total": 11727184, + "tokens/train_per_sec_per_gpu": 23.64, + "tokens/trainable": 167832 + }, + { + "epoch": 1.4375, + "grad_norm": 0.0005008528823964298, + "learning_rate": 2.829199644117484e-05, + "loss": 1.1319095392536838e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00001, + "step": 368, + "tokens/total": 11759312, + "tokens/train_per_sec_per_gpu": 25.79, + "tokens/trainable": 168242 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.02015153504908085, + "learning_rate": 2.8058920186702553e-05, + "loss": 0.00011158352572238073, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00011, + "step": 369, + "tokens/total": 11791536, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 168715 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.00031807494815438986, + "learning_rate": 2.782696506053033e-05, + "loss": 6.533119631058071e-06, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00001, + "step": 370, + "tokens/total": 11823632, + "tokens/train_per_sec_per_gpu": 26.68, + "tokens/trainable": 169147 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.4510185122489929, + "learning_rate": 2.7596140715257824e-05, + "loss": 0.004431337118148804, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00444, + "step": 371, + "tokens/total": 11855392, + "tokens/train_per_sec_per_gpu": 28.61, + "tokens/trainable": 169628 + }, + { + "epoch": 1.453125, + "grad_norm": 0.0034806979820132256, + "learning_rate": 2.7366456756428184e-05, + "loss": 2.722550561884418e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00003, + "step": 372, + "tokens/total": 11887424, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 170085 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.018526704981923103, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.00013849925016984344, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00014, + "step": 373, + "tokens/total": 11919280, + "tokens/train_per_sec_per_gpu": 29.39, + "tokens/trainable": 170584 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.010734346695244312, + "learning_rate": 2.691054818259188e-05, + "loss": 1.963317481568083e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00002, + "step": 374, + "tokens/total": 11951488, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 171058 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.004906357266008854, + "learning_rate": 2.6684342539801933e-05, + "loss": 4.44888137280941e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00004, + "step": 375, + "tokens/total": 11983504, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 171518 + }, + { + "epoch": 1.46875, + "grad_norm": 0.0035627244506031275, + "learning_rate": 2.645931522709877e-05, + "loss": 1.879796946013812e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00002, + "step": 376, + "tokens/total": 12015360, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 172007 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.014949376694858074, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.0001050513528753072, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00011, + "step": 377, + "tokens/total": 12047296, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 172468 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.024432353675365448, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.00010428359382785857, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0001, + "step": 378, + "tokens/total": 12079392, + "tokens/train_per_sec_per_gpu": 27.91, + "tokens/trainable": 172923 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.5905564427375793, + "learning_rate": 2.579139666504821e-05, + "loss": 0.019045401364564896, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01923, + "step": 379, + "tokens/total": 12111616, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 173370 + }, + { + "epoch": 1.484375, + "grad_norm": 0.0027490397915244102, + "learning_rate": 2.557117581955798e-05, + "loss": 3.106705116806552e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00003, + "step": 380, + "tokens/total": 12143536, + "tokens/train_per_sec_per_gpu": 28.66, + "tokens/trainable": 173837 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.10220319032669067, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.0007754361140541732, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00078, + "step": 381, + "tokens/total": 12175792, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 174294 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.06388501822948456, + "learning_rate": 2.5134417202396277e-05, + "loss": 2.506363671272993e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00003, + "step": 382, + "tokens/total": 12207472, + "tokens/train_per_sec_per_gpu": 25.85, + "tokens/trainable": 174701 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.006029566749930382, + "learning_rate": 2.491789760603361e-05, + "loss": 5.214785414864309e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00005, + "step": 383, + "tokens/total": 12239520, + "tokens/train_per_sec_per_gpu": 26.67, + "tokens/trainable": 175131 + }, + { + "epoch": 1.5, + "grad_norm": 0.015079713426530361, + "learning_rate": 2.4702629848741764e-05, + "loss": 0.00011988497863058001, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00012, + "step": 384, + "tokens/total": 12271600, + "tokens/train_per_sec_per_gpu": 28.23, + "tokens/trainable": 175586 + }, + { + "epoch": 1.50390625, + "grad_norm": 0.04362509027123451, + "learning_rate": 2.4488622888690785e-05, + "loss": 0.0002582473389338702, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00026, + "step": 385, + "tokens/total": 12303648, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 176026 + }, + { + "epoch": 1.5078125, + "grad_norm": 0.28308039903640747, + "learning_rate": 2.427588563158384e-05, + "loss": 0.0025924108922481537, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0026, + "step": 386, + "tokens/total": 12335728, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 176512 + }, + { + "epoch": 1.51171875, + "grad_norm": 0.21685272455215454, + "learning_rate": 2.406442693028651e-05, + "loss": 0.002609315561130643, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00261, + "step": 387, + "tokens/total": 12367824, + "tokens/train_per_sec_per_gpu": 25.55, + "tokens/trainable": 176943 + }, + { + "epoch": 1.515625, + "grad_norm": 0.0010271539213135839, + "learning_rate": 2.3854255584458547e-05, + "loss": 2.1533389372052625e-05, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00002, + "step": 388, + "tokens/total": 12400192, + "tokens/train_per_sec_per_gpu": 26.36, + "tokens/trainable": 177370 + }, + { + "epoch": 1.51953125, + "grad_norm": 0.06616228073835373, + "learning_rate": 2.3645380340187508e-05, + "loss": 0.0005350579158402979, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00054, + "step": 389, + "tokens/total": 12432224, + "tokens/train_per_sec_per_gpu": 29.11, + "tokens/trainable": 177851 + }, + { + "epoch": 1.5234375, + "grad_norm": 0.05946248397231102, + "learning_rate": 2.3437809889624914e-05, + "loss": 0.0005003588157705963, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.0005, + "step": 390, + "tokens/total": 12464048, + "tokens/train_per_sec_per_gpu": 29.09, + "tokens/trainable": 178310 + }, + { + "epoch": 1.52734375, + "grad_norm": 0.017372427508234978, + "learning_rate": 2.3231552870624487e-05, + "loss": 0.0001355188578600064, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00014, + "step": 391, + "tokens/total": 12493776, + "tokens/train_per_sec_per_gpu": 30.96, + "tokens/trainable": 178736 + }, + { + "epoch": 1.53125, + "grad_norm": 0.12941808998584747, + "learning_rate": 2.3026617866382657e-05, + "loss": 0.003045230871066451, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00305, + "step": 392, + "tokens/total": 12526000, + "tokens/train_per_sec_per_gpu": 30.37, + "tokens/trainable": 179233 + }, + { + "epoch": 1.53515625, + "grad_norm": 0.014790852554142475, + "learning_rate": 2.2823013405081507e-05, + "loss": 0.00013205324648879468, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00013, + "step": 393, + "tokens/total": 12557968, + "tokens/train_per_sec_per_gpu": 32.27, + "tokens/trainable": 179730 + }, + { + "epoch": 1.5390625, + "grad_norm": 0.02042844519019127, + "learning_rate": 2.2620747959533722e-05, + "loss": 0.00022655579959973693, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00023, + "step": 394, + "tokens/total": 12590096, + "tokens/train_per_sec_per_gpu": 31.47, + "tokens/trainable": 180227 + }, + { + "epoch": 1.54296875, + "grad_norm": 0.005440943408757448, + "learning_rate": 2.2419829946830123e-05, + "loss": 4.6342807763721794e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00005, + "step": 395, + "tokens/total": 12622160, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 180687 + }, + { + "epoch": 1.546875, + "grad_norm": 0.004878366366028786, + "learning_rate": 2.2220267727989325e-05, + "loss": 5.4212974646361545e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00005, + "step": 396, + "tokens/total": 12654336, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 181141 + }, + { + "epoch": 1.55078125, + "grad_norm": 0.20873922109603882, + "learning_rate": 2.202206960760984e-05, + "loss": 0.0026359721086919308, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00264, + "step": 397, + "tokens/total": 12686400, + "tokens/train_per_sec_per_gpu": 30.13, + "tokens/trainable": 181648 + }, + { + "epoch": 1.5546875, + "grad_norm": 0.12369472533464432, + "learning_rate": 2.182524383352446e-05, + "loss": 0.0010878165485337377, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.00109, + "step": 398, + "tokens/total": 12718144, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 182109 + }, + { + "epoch": 1.55859375, + "grad_norm": 0.0010841770563274622, + "learning_rate": 2.1629798596457056e-05, + "loss": 1.8380069377599284e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00002, + "step": 399, + "tokens/total": 12750256, + "tokens/train_per_sec_per_gpu": 26.26, + "tokens/trainable": 182559 + }, + { + "epoch": 1.5625, + "grad_norm": 0.0010006122756749392, + "learning_rate": 2.1435742029681725e-05, + "loss": 1.991226599784568e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00002, + "step": 400, + "tokens/total": 12782224, + "tokens/train_per_sec_per_gpu": 27.59, + "tokens/trainable": 182985 + }, + { + "epoch": 1.56640625, + "grad_norm": 0.0011391225270926952, + "learning_rate": 2.124308220868431e-05, + "loss": 2.2474639990832657e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00002, + "step": 401, + "tokens/total": 12814384, + "tokens/train_per_sec_per_gpu": 26.95, + "tokens/trainable": 183440 + }, + { + "epoch": 1.5703125, + "grad_norm": 0.023268507793545723, + "learning_rate": 2.105182715082638e-05, + "loss": 0.00032951770117506385, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00033, + "step": 402, + "tokens/total": 12846240, + "tokens/train_per_sec_per_gpu": 28.77, + "tokens/trainable": 183917 + }, + { + "epoch": 1.57421875, + "grad_norm": 0.11701280623674393, + "learning_rate": 2.0861984815011552e-05, + "loss": 0.001444364432245493, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00145, + "step": 403, + "tokens/total": 12878240, + "tokens/train_per_sec_per_gpu": 25.07, + "tokens/trainable": 184348 + }, + { + "epoch": 1.578125, + "grad_norm": 0.004240577574819326, + "learning_rate": 2.0673563101354323e-05, + "loss": 2.100174970109947e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00002, + "step": 404, + "tokens/total": 12910176, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 184786 + }, + { + "epoch": 1.58203125, + "grad_norm": 0.01977616548538208, + "learning_rate": 2.0486569850851317e-05, + "loss": 8.987118053482845e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00009, + "step": 405, + "tokens/total": 12942352, + "tokens/train_per_sec_per_gpu": 28.17, + "tokens/trainable": 185249 + }, + { + "epoch": 1.5859375, + "grad_norm": 0.030095387250185013, + "learning_rate": 2.0301012845054956e-05, + "loss": 0.0002793738676700741, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00028, + "step": 406, + "tokens/total": 12974368, + "tokens/train_per_sec_per_gpu": 26.3, + "tokens/trainable": 185680 + }, + { + "epoch": 1.58984375, + "grad_norm": 0.0017381443176418543, + "learning_rate": 2.011689980574966e-05, + "loss": 2.727654828049708e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00003, + "step": 407, + "tokens/total": 13006464, + "tokens/train_per_sec_per_gpu": 26.25, + "tokens/trainable": 186100 + }, + { + "epoch": 1.59375, + "grad_norm": 0.09072195738554001, + "learning_rate": 1.993423839463052e-05, + "loss": 0.0006338813109323382, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00063, + "step": 408, + "tokens/total": 13038320, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 186565 + }, + { + "epoch": 1.59765625, + "grad_norm": 0.004482298158109188, + "learning_rate": 1.975303621298445e-05, + "loss": 5.0280330469831824e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00005, + "step": 409, + "tokens/total": 13070416, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 187017 + }, + { + "epoch": 1.6015625, + "grad_norm": 0.03788210079073906, + "learning_rate": 1.957330080137385e-05, + "loss": 0.000300457701086998, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0003, + "step": 410, + "tokens/total": 13102352, + "tokens/train_per_sec_per_gpu": 27.69, + "tokens/trainable": 187468 + }, + { + "epoch": 1.60546875, + "grad_norm": 0.011773839592933655, + "learning_rate": 1.9395039639322864e-05, + "loss": 9.259363287128508e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00009, + "step": 411, + "tokens/total": 13134096, + "tokens/train_per_sec_per_gpu": 31.06, + "tokens/trainable": 187961 + }, + { + "epoch": 1.609375, + "grad_norm": 0.0031615474727004766, + "learning_rate": 1.9218260145006073e-05, + "loss": 3.231812661397271e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00003, + "step": 412, + "tokens/total": 13166256, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 188447 + }, + { + "epoch": 1.61328125, + "grad_norm": 0.07778825610876083, + "learning_rate": 1.904296967493982e-05, + "loss": 0.0005980221321806312, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.0006, + "step": 413, + "tokens/total": 13198048, + "tokens/train_per_sec_per_gpu": 29.98, + "tokens/trainable": 188903 + }, + { + "epoch": 1.6171875, + "grad_norm": 0.011902395635843277, + "learning_rate": 1.8869175523676064e-05, + "loss": 0.00013689698243979365, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.66, + "memory/max_allocated (GiB)": 36.66, + "ppl": 1.00014, + "step": 414, + "tokens/total": 13230304, + "tokens/train_per_sec_per_gpu": 30.8, + "tokens/trainable": 189398 + }, + { + "epoch": 1.62109375, + "grad_norm": 0.08490622788667679, + "learning_rate": 1.869688492349885e-05, + "loss": 0.0008053273777477443, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00081, + "step": 415, + "tokens/total": 13262288, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 189850 + }, + { + "epoch": 1.625, + "grad_norm": 0.11783187091350555, + "learning_rate": 1.85261050441233e-05, + "loss": 0.0007362872711382806, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00074, + "step": 416, + "tokens/total": 13294480, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 190306 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 9.023740471377331e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/training_args.bin b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..dec74f9f304b44cbe47132d86c4e6285e659f40b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-416/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634 +size 8273 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/README.md b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ba677470ac9c29f9770a4aacb65c644ebd671a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/deconf_wave_deconf_control/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/deconf_wave_deconf_control/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/adapter_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..21fcb5b00be777c20a2d2dcceffc539dc51679b4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/deconf_wave_deconf_control/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "gate_proj", + "v_proj", + "k_proj", + "up_proj", + "q_proj", + "o_proj", + "down_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/adapter_model.safetensors b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..ea2f7b4a1e8e3ceb0512264d6b08fb22a777b22f --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7b929c2d0258857c25571b3445b6c9eaba0da47b02edd3186326a01682142f2d +size 547777976 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/chat_template.jinja b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/optimizer.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..82cb26f718796814cbfaebe9d9c49a3796b5928c --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1f19cb3b0ad1441e4c5b22be4228b161f44427d230f8e9f9c2bc3cb773bca4d0 +size 1048106435 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/rng_state.pth b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..ca2b36b589a4bc5d6f9ddcdd63bfd147a95c1cab --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a0a9e7989950cceffe25d765593fb95049fdbb225f1463e741b44042aa9cba58 +size 14645 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/scheduler.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..2045249a1135f9f2d87c87c3e02e6227697d0cfa --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:457821a0c6da6ac211fae3b76339963abddd2e9839d13c28173a0924fca59503 +size 1465 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/tokenizer.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/tokenizer_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/tokens_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..6b7a1d3ac68e523d7436d7ab5367d44ace46dca3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/tokens_state.json @@ -0,0 +1 @@ +{"total": 14317968, "trainable": 204821} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/trainer_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..6212a813e8dd85c9a4caeb819bacceaccb4256e1 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/trainer_state.json @@ -0,0 +1,6306 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.75, + "eval_steps": 500, + "global_step": 448, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.2140090465545654, + "learning_rate": 0.0, + "loss": 0.16441112756729126, + "memory/device_reserved (GiB)": 38.08, + "memory/max_active (GiB)": 35.64, + "memory/max_allocated (GiB)": 35.64, + "ppl": 1.1787, + "step": 1, + "tokens/total": 32160, + "tokens/train_per_sec_per_gpu": 21.28, + "tokens/trainable": 470 + }, + { + "epoch": 0.0078125, + "grad_norm": 2.92620849609375, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15774038434028625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.17086, + "step": 2, + "tokens/total": 64192, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 922 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.135961890220642, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1664382815361023, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.18109, + "step": 3, + "tokens/total": 96224, + "tokens/train_per_sec_per_gpu": 28.54, + "tokens/trainable": 1396 + }, + { + "epoch": 0.015625, + "grad_norm": 0.880176305770874, + "learning_rate": 1.2e-05, + "loss": 0.15334013104438782, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.16572, + "step": 4, + "tokens/total": 128336, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 1850 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.1433289051055908, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.16352537274360657, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.17766, + "step": 5, + "tokens/total": 160784, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 2302 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9728212952613831, + "learning_rate": 2e-05, + "loss": 0.15144473314285278, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.16351, + "step": 6, + "tokens/total": 192624, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 2742 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.7796363830566406, + "learning_rate": 2.4e-05, + "loss": 0.13553467392921448, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.14515, + "step": 7, + "tokens/total": 224608, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 3208 + }, + { + "epoch": 0.03125, + "grad_norm": 1.2444697618484497, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.1077304258942604, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.11375, + "step": 8, + "tokens/total": 256592, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 3655 + }, + { + "epoch": 0.03515625, + "grad_norm": 2.7339375019073486, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.08772765100002289, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.09169, + "step": 9, + "tokens/total": 286400, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 4091 + }, + { + "epoch": 0.0390625, + "grad_norm": 4.109770774841309, + "learning_rate": 3.6e-05, + "loss": 0.130940243601799, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.1399, + "step": 10, + "tokens/total": 318352, + "tokens/train_per_sec_per_gpu": 29.13, + "tokens/trainable": 4553 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.3768608570098877, + "learning_rate": 4e-05, + "loss": 0.10059958696365356, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.10583, + "step": 11, + "tokens/total": 350336, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 4992 + }, + { + "epoch": 0.046875, + "grad_norm": 3.281402826309204, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.12574875354766846, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.134, + "step": 12, + "tokens/total": 382448, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 5494 + }, + { + "epoch": 0.05078125, + "grad_norm": 1.7738980054855347, + "learning_rate": 4.8e-05, + "loss": 0.033296722918748856, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03386, + "step": 13, + "tokens/total": 414352, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 5933 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.64290189743042, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06366822123527527, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.06574, + "step": 14, + "tokens/total": 446400, + "tokens/train_per_sec_per_gpu": 24.8, + "tokens/trainable": 6369 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.3691928386688232, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.08168518543243408, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.08511, + "step": 15, + "tokens/total": 478688, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 6820 + }, + { + "epoch": 0.0625, + "grad_norm": 2.4356393814086914, + "learning_rate": 6e-05, + "loss": 0.09031196683645248, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.09452, + "step": 16, + "tokens/total": 510512, + "tokens/train_per_sec_per_gpu": 25.7, + "tokens/trainable": 7258 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.55682373046875, + "learning_rate": 6.400000000000001e-05, + "loss": 0.07496573030948639, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.07785, + "step": 17, + "tokens/total": 542736, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 7710 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.1054646968841553, + "learning_rate": 6.800000000000001e-05, + "loss": 0.05218805745244026, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.05357, + "step": 18, + "tokens/total": 574832, + "tokens/train_per_sec_per_gpu": 25.78, + "tokens/trainable": 8174 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.7429099082946777, + "learning_rate": 7.2e-05, + "loss": 0.03790827840566635, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.03864, + "step": 19, + "tokens/total": 606976, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 8617 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9508209228515625, + "learning_rate": 7.6e-05, + "loss": 0.04993613809347153, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0512, + "step": 20, + "tokens/total": 639200, + "tokens/train_per_sec_per_gpu": 24.91, + "tokens/trainable": 9037 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.0297844409942627, + "learning_rate": 8e-05, + "loss": 0.017908263951539993, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.01807, + "step": 21, + "tokens/total": 671072, + "tokens/train_per_sec_per_gpu": 30.42, + "tokens/trainable": 9503 + }, + { + "epoch": 0.0859375, + "grad_norm": 0.9279879331588745, + "learning_rate": 8.4e-05, + "loss": 0.013478193432092667, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01357, + "step": 22, + "tokens/total": 703136, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 9972 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.526488184928894, + "learning_rate": 8.800000000000001e-05, + "loss": 0.017310388386249542, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01746, + "step": 23, + "tokens/total": 735120, + "tokens/train_per_sec_per_gpu": 31.45, + "tokens/trainable": 10447 + }, + { + "epoch": 0.09375, + "grad_norm": 3.4049417972564697, + "learning_rate": 9.200000000000001e-05, + "loss": 0.0353429913520813, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.03597, + "step": 24, + "tokens/total": 767040, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 10899 + }, + { + "epoch": 0.09765625, + "grad_norm": 7.511656761169434, + "learning_rate": 9.6e-05, + "loss": 0.1335875540971756, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.14292, + "step": 25, + "tokens/total": 798976, + "tokens/train_per_sec_per_gpu": 27.72, + "tokens/trainable": 11360 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.0069234371185303, + "learning_rate": 0.0001, + "loss": 0.04382415860891342, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0448, + "step": 26, + "tokens/total": 830880, + "tokens/train_per_sec_per_gpu": 24.61, + "tokens/trainable": 11775 + }, + { + "epoch": 0.10546875, + "grad_norm": 7.807938575744629, + "learning_rate": 9.99990636831587e-05, + "loss": 0.01210833340883255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.01218, + "step": 27, + "tokens/total": 860960, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 12242 + }, + { + "epoch": 0.109375, + "grad_norm": 2.23823881149292, + "learning_rate": 9.999625477159879e-05, + "loss": 0.0272970050573349, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02767, + "step": 28, + "tokens/total": 892816, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 12696 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.315584421157837, + "learning_rate": 9.999157338221051e-05, + "loss": 0.025516577064990997, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.02584, + "step": 29, + "tokens/total": 924736, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 13153 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.8413651585578918, + "learning_rate": 9.998501970980562e-05, + "loss": 0.007838520221412182, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00787, + "step": 30, + "tokens/total": 954592, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 13624 + }, + { + "epoch": 0.12109375, + "grad_norm": 1.6797157526016235, + "learning_rate": 9.997659402710915e-05, + "loss": 0.03002442792057991, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.03048, + "step": 31, + "tokens/total": 986592, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 14064 + }, + { + "epoch": 0.125, + "grad_norm": 1.3320434093475342, + "learning_rate": 9.996629668474818e-05, + "loss": 0.03772728145122528, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.03845, + "step": 32, + "tokens/total": 1018720, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 14565 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.6335167288780212, + "learning_rate": 9.995412811123711e-05, + "loss": 0.013970119878649712, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01407, + "step": 33, + "tokens/total": 1050624, + "tokens/train_per_sec_per_gpu": 28.18, + "tokens/trainable": 15005 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.5626921653747559, + "learning_rate": 9.994008881295999e-05, + "loss": 0.016654539853334427, + "memory/device_reserved (GiB)": 37.74, + "memory/max_active (GiB)": 36.22, + "memory/max_allocated (GiB)": 36.22, + "ppl": 1.01679, + "step": 34, + "tokens/total": 1082176, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 15459 + }, + { + "epoch": 0.13671875, + "grad_norm": 0.603219211101532, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021408583968877792, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02164, + "step": 35, + "tokens/total": 1113968, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 15960 + }, + { + "epoch": 0.140625, + "grad_norm": 0.9779674410820007, + "learning_rate": 9.99064004568618e-05, + "loss": 0.03277068957686424, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03331, + "step": 36, + "tokens/total": 1145952, + "tokens/train_per_sec_per_gpu": 28.68, + "tokens/trainable": 16428 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.8855255842208862, + "learning_rate": 9.988675280095074e-05, + "loss": 0.017048098146915436, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01719, + "step": 37, + "tokens/total": 1177872, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 16884 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.9266122579574585, + "learning_rate": 9.986523722403528e-05, + "loss": 0.045863546431064606, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.04693, + "step": 38, + "tokens/total": 1209776, + "tokens/train_per_sec_per_gpu": 26.91, + "tokens/trainable": 17341 + }, + { + "epoch": 0.15234375, + "grad_norm": 0.8280895948410034, + "learning_rate": 9.984185462146642e-05, + "loss": 0.033055759966373444, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03361, + "step": 39, + "tokens/total": 1241888, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 17786 + }, + { + "epoch": 0.15625, + "grad_norm": 5.029730319976807, + "learning_rate": 9.98166059662897e-05, + "loss": 0.025589998811483383, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02592, + "step": 40, + "tokens/total": 1274032, + "tokens/train_per_sec_per_gpu": 29.22, + "tokens/trainable": 18262 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.5688032507896423, + "learning_rate": 9.978949230920472e-05, + "loss": 0.016095872968435287, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01623, + "step": 41, + "tokens/total": 1306000, + "tokens/train_per_sec_per_gpu": 27.49, + "tokens/trainable": 18716 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8217663168907166, + "learning_rate": 9.976051477852141e-05, + "loss": 0.026798982173204422, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02716, + "step": 42, + "tokens/total": 1337936, + "tokens/train_per_sec_per_gpu": 27.51, + "tokens/trainable": 19157 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.7309132218360901, + "learning_rate": 9.972967458011312e-05, + "loss": 0.033920761197805405, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0345, + "step": 43, + "tokens/total": 1369904, + "tokens/train_per_sec_per_gpu": 30.66, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.8134022355079651, + "learning_rate": 9.96969729973664e-05, + "loss": 0.024005085229873657, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0243, + "step": 44, + "tokens/total": 1401984, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 20119 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.35084816813468933, + "learning_rate": 9.966241139112754e-05, + "loss": 0.013319254852831364, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01341, + "step": 45, + "tokens/total": 1434032, + "tokens/train_per_sec_per_gpu": 27.45, + "tokens/trainable": 20560 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.8701573014259338, + "learning_rate": 9.96259911996461e-05, + "loss": 0.01356641948223114, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01366, + "step": 46, + "tokens/total": 1465872, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 20992 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.4406442642211914, + "learning_rate": 9.958771393851491e-05, + "loss": 0.04396980255842209, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 35.86, + "memory/max_allocated (GiB)": 35.86, + "ppl": 1.04495, + "step": 47, + "tokens/total": 1495440, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 21441 + }, + { + "epoch": 0.1875, + "grad_norm": 0.4966669976711273, + "learning_rate": 9.954758120060702e-05, + "loss": 0.011179964058101177, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01124, + "step": 48, + "tokens/total": 1527776, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 21901 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.41124436259269714, + "learning_rate": 9.950559465600948e-05, + "loss": 0.0063599650748074055, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00638, + "step": 49, + "tokens/total": 1559760, + "tokens/train_per_sec_per_gpu": 27.73, + "tokens/trainable": 22341 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.829739511013031, + "learning_rate": 9.946175605195379e-05, + "loss": 0.015785304829478264, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01591, + "step": 50, + "tokens/total": 1592000, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 22833 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5181736946105957, + "learning_rate": 9.941606721274322e-05, + "loss": 0.008393109776079655, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00843, + "step": 51, + "tokens/total": 1623920, + "tokens/train_per_sec_per_gpu": 25.61, + "tokens/trainable": 23277 + }, + { + "epoch": 0.203125, + "grad_norm": 1.5778990983963013, + "learning_rate": 9.936853003967685e-05, + "loss": 0.024984844028949738, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0253, + "step": 52, + "tokens/total": 1656032, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 23759 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.4938714504241943, + "learning_rate": 9.93191465109705e-05, + "loss": 0.038523320108652115, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03927, + "step": 53, + "tokens/total": 1688368, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 24193 + }, + { + "epoch": 0.2109375, + "grad_norm": 2.1427106857299805, + "learning_rate": 9.926791868167438e-05, + "loss": 0.06302924454212189, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.06506, + "step": 54, + "tokens/total": 1720560, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 24619 + }, + { + "epoch": 0.21484375, + "grad_norm": 1.8365188837051392, + "learning_rate": 9.921484868358753e-05, + "loss": 0.052160464227199554, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.05354, + "step": 55, + "tokens/total": 1752464, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 25075 + }, + { + "epoch": 0.21875, + "grad_norm": 0.7779749631881714, + "learning_rate": 9.915993872516924e-05, + "loss": 0.021328264847397804, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.02156, + "step": 56, + "tokens/total": 1784336, + "tokens/train_per_sec_per_gpu": 26.42, + "tokens/trainable": 25519 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.4189571142196655, + "learning_rate": 9.9103191091447e-05, + "loss": 0.014279939234256744, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01438, + "step": 57, + "tokens/total": 1816304, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 25966 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.3834710717201233, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01883944310247898, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01902, + "step": 58, + "tokens/total": 1848304, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 26423 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.7693278789520264, + "learning_rate": 9.898419232046825e-05, + "loss": 0.03533369302749634, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03597, + "step": 59, + "tokens/total": 1880640, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 26934 + }, + { + "epoch": 0.234375, + "grad_norm": 0.4606446325778961, + "learning_rate": 9.892194613523633e-05, + "loss": 0.021482212468981743, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02171, + "step": 60, + "tokens/total": 1912592, + "tokens/train_per_sec_per_gpu": 27.64, + "tokens/trainable": 27393 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.46030017733573914, + "learning_rate": 9.885787217854357e-05, + "loss": 0.017044665291905403, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01719, + "step": 61, + "tokens/total": 1944544, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 27852 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.7918264269828796, + "learning_rate": 9.879197311676887e-05, + "loss": 0.016954131424427032, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0171, + "step": 62, + "tokens/total": 1976688, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 28292 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.4787926971912384, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01525965891778469, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 63, + "tokens/total": 2008928, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 28798 + }, + { + "epoch": 0.25, + "grad_norm": 0.9033918976783752, + "learning_rate": 9.865471072312528e-05, + "loss": 0.02902211993932724, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02945, + "step": 64, + "tokens/total": 2041120, + "tokens/train_per_sec_per_gpu": 29.83, + "tokens/trainable": 29283 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.39305707812309265, + "learning_rate": 9.858335310330492e-05, + "loss": 0.00901167094707489, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.32, + "memory/max_allocated (GiB)": 36.32, + "ppl": 1.00905, + "step": 65, + "tokens/total": 2073024, + "tokens/train_per_sec_per_gpu": 26.48, + "tokens/trainable": 29745 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4494345784187317, + "learning_rate": 9.851018180226185e-05, + "loss": 0.007426132448017597, + "memory/device_reserved (GiB)": 37.47, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00745, + "step": 66, + "tokens/total": 2105184, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 30186 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.629649817943573, + "learning_rate": 9.843519986495259e-05, + "loss": 0.010609081946313381, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01067, + "step": 67, + "tokens/total": 2137264, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 30622 + }, + { + "epoch": 0.265625, + "grad_norm": 0.7591162919998169, + "learning_rate": 9.835841041168162e-05, + "loss": 0.00926615484058857, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00931, + "step": 68, + "tokens/total": 2169424, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 31097 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.7288486361503601, + "learning_rate": 9.82798166379715e-05, + "loss": 0.024310583248734474, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.02461, + "step": 69, + "tokens/total": 2201424, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 31595 + }, + { + "epoch": 0.2734375, + "grad_norm": 2.134056329727173, + "learning_rate": 9.819942181443002e-05, + "loss": 0.008422967046499252, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00846, + "step": 70, + "tokens/total": 2233424, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 32036 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6362871527671814, + "learning_rate": 9.811722928661392e-05, + "loss": 0.01577262207865715, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0159, + "step": 71, + "tokens/total": 2265328, + "tokens/train_per_sec_per_gpu": 23.65, + "tokens/trainable": 32428 + }, + { + "epoch": 0.28125, + "grad_norm": 0.6091912388801575, + "learning_rate": 9.803324247488975e-05, + "loss": 0.0064800274558365345, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0065, + "step": 72, + "tokens/total": 2297248, + "tokens/train_per_sec_per_gpu": 26.86, + "tokens/trainable": 32880 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.14533838629722595, + "learning_rate": 9.794746487429161e-05, + "loss": 0.001991454279050231, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00199, + "step": 73, + "tokens/total": 2329152, + "tokens/train_per_sec_per_gpu": 27.97, + "tokens/trainable": 33323 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.5400826930999756, + "learning_rate": 9.785990005437554e-05, + "loss": 0.017662350088357925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01782, + "step": 74, + "tokens/total": 2361344, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 33792 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.6132436394691467, + "learning_rate": 9.777055165907117e-05, + "loss": 0.01787603087723255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01804, + "step": 75, + "tokens/total": 2393168, + "tokens/train_per_sec_per_gpu": 29.53, + "tokens/trainable": 34223 + }, + { + "epoch": 0.296875, + "grad_norm": 0.7287399172782898, + "learning_rate": 9.767942340652993e-05, + "loss": 0.024754449725151062, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.31, + "memory/max_allocated (GiB)": 36.31, + "ppl": 1.02506, + "step": 76, + "tokens/total": 2424880, + "tokens/train_per_sec_per_gpu": 24.92, + "tokens/trainable": 34649 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.6790545582771301, + "learning_rate": 9.758651908897035e-05, + "loss": 0.014308430254459381, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01441, + "step": 77, + "tokens/total": 2457120, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 35094 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6203456521034241, + "learning_rate": 9.749184257252033e-05, + "loss": 0.034353043884038925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03495, + "step": 78, + "tokens/total": 2489104, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 35534 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.6238825917243958, + "learning_rate": 9.739539779705614e-05, + "loss": 0.032295141369104385, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.03282, + "step": 79, + "tokens/total": 2521136, + "tokens/train_per_sec_per_gpu": 29.68, + "tokens/trainable": 36029 + }, + { + "epoch": 0.3125, + "grad_norm": 0.18977171182632446, + "learning_rate": 9.729718877603861e-05, + "loss": 0.007260813377797604, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00729, + "step": 80, + "tokens/total": 2553168, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 36469 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.8054049611091614, + "learning_rate": 9.719721959634592e-05, + "loss": 0.034894898533821106, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.03551, + "step": 81, + "tokens/total": 2585584, + "tokens/train_per_sec_per_gpu": 25.75, + "tokens/trainable": 36906 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.6884189248085022, + "learning_rate": 9.709549441810375e-05, + "loss": 0.018210047855973244, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.01838, + "step": 82, + "tokens/total": 2617808, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 37390 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.6490103602409363, + "learning_rate": 9.699201747451195e-05, + "loss": 0.02030857279896736, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.02052, + "step": 83, + "tokens/total": 2649664, + "tokens/train_per_sec_per_gpu": 25.8, + "tokens/trainable": 37838 + }, + { + "epoch": 0.328125, + "grad_norm": 0.2439805567264557, + "learning_rate": 9.688679307166854e-05, + "loss": 0.008518049493432045, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00855, + "step": 84, + "tokens/total": 2681728, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 38310 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6747375726699829, + "learning_rate": 9.677982558839042e-05, + "loss": 0.029895059764385223, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.03035, + "step": 85, + "tokens/total": 2713488, + "tokens/train_per_sec_per_gpu": 28.49, + "tokens/trainable": 38789 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.582666277885437, + "learning_rate": 9.66711194760312e-05, + "loss": 0.014998597092926502, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01511, + "step": 86, + "tokens/total": 2745536, + "tokens/train_per_sec_per_gpu": 30.27, + "tokens/trainable": 39277 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.5364654660224915, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011567167937755585, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01163, + "step": 87, + "tokens/total": 2777680, + "tokens/train_per_sec_per_gpu": 29.73, + "tokens/trainable": 39737 + }, + { + "epoch": 0.34375, + "grad_norm": 0.34024426341056824, + "learning_rate": 9.644850953105288e-05, + "loss": 0.01014852337539196, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0102, + "step": 88, + "tokens/total": 2809568, + "tokens/train_per_sec_per_gpu": 24.68, + "tokens/trainable": 40179 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.32472267746925354, + "learning_rate": 9.633461496214225e-05, + "loss": 0.01229981891810894, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01238, + "step": 89, + "tokens/total": 2841760, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 40649 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.7906899452209473, + "learning_rate": 9.621900029118195e-05, + "loss": 0.02552586793899536, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.02585, + "step": 90, + "tokens/total": 2871696, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 41080 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.7444764971733093, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0054441955871880054, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00546, + "step": 91, + "tokens/total": 2903968, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 41599 + }, + { + "epoch": 0.359375, + "grad_norm": 0.1911851465702057, + "learning_rate": 9.598262995928611e-05, + "loss": 0.004665314219892025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00468, + "step": 92, + "tokens/total": 2936048, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 42076 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.21238519251346588, + "learning_rate": 9.586188413468492e-05, + "loss": 0.003917238209396601, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00392, + "step": 93, + "tokens/total": 2968176, + "tokens/train_per_sec_per_gpu": 28.07, + "tokens/trainable": 42522 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.41729435324668884, + "learning_rate": 9.57394378802934e-05, + "loss": 0.013555807992815971, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01365, + "step": 94, + "tokens/total": 3000304, + "tokens/train_per_sec_per_gpu": 28.69, + "tokens/trainable": 42974 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.7246769070625305, + "learning_rate": 9.56152962916e-05, + "loss": 0.020221399143338203, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.02043, + "step": 95, + "tokens/total": 3032080, + "tokens/train_per_sec_per_gpu": 25.35, + "tokens/trainable": 43380 + }, + { + "epoch": 0.375, + "grad_norm": 0.16139821708202362, + "learning_rate": 9.548946453464296e-05, + "loss": 0.003853276837617159, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00386, + "step": 96, + "tokens/total": 3064320, + "tokens/train_per_sec_per_gpu": 28.47, + "tokens/trainable": 43865 + }, + { + "epoch": 0.37890625, + "grad_norm": 1.0509370565414429, + "learning_rate": 9.53619478457953e-05, + "loss": 0.015099998563528061, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01521, + "step": 97, + "tokens/total": 3096144, + "tokens/train_per_sec_per_gpu": 26.61, + "tokens/trainable": 44328 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.6859350204467773, + "learning_rate": 9.523275153154695e-05, + "loss": 0.004579808097332716, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00459, + "step": 98, + "tokens/total": 3127600, + "tokens/train_per_sec_per_gpu": 25.91, + "tokens/trainable": 44778 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5686795711517334, + "learning_rate": 9.51018809682839e-05, + "loss": 0.022520575672388077, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02278, + "step": 99, + "tokens/total": 3159520, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 45225 + }, + { + "epoch": 0.390625, + "grad_norm": 1.6308369636535645, + "learning_rate": 9.49693416020645e-05, + "loss": 0.011883002705872059, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01195, + "step": 100, + "tokens/total": 3191568, + "tokens/train_per_sec_per_gpu": 29.75, + "tokens/trainable": 45678 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.7186088562011719, + "learning_rate": 9.483513894839276e-05, + "loss": 0.0060632615350186825, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00608, + "step": 101, + "tokens/total": 3223648, + "tokens/train_per_sec_per_gpu": 27.14, + "tokens/trainable": 46125 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.12036337703466415, + "learning_rate": 9.469927859198888e-05, + "loss": 0.0019118499476462603, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00191, + "step": 102, + "tokens/total": 3255744, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 46612 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.29928353428840637, + "learning_rate": 9.456176618655689e-05, + "loss": 0.00944911316037178, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00949, + "step": 103, + "tokens/total": 3287792, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 47059 + }, + { + "epoch": 0.40625, + "grad_norm": 0.3766293227672577, + "learning_rate": 9.442260745454927e-05, + "loss": 0.008984047919511795, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.68, + "memory/max_allocated (GiB)": 36.68, + "ppl": 1.00902, + "step": 104, + "tokens/total": 3320016, + "tokens/train_per_sec_per_gpu": 28.92, + "tokens/trainable": 47536 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3481820523738861, + "learning_rate": 9.428180818692884e-05, + "loss": 0.005444999784231186, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00546, + "step": 105, + "tokens/total": 3351952, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 48037 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.8414323329925537, + "learning_rate": 9.413937424292791e-05, + "loss": 0.032438624650239944, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03297, + "step": 106, + "tokens/total": 3384032, + "tokens/train_per_sec_per_gpu": 27.11, + "tokens/trainable": 48491 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.5067655444145203, + "learning_rate": 9.399531154980424e-05, + "loss": 0.008442888967692852, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00848, + "step": 107, + "tokens/total": 3416192, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 48940 + }, + { + "epoch": 0.421875, + "grad_norm": 0.41617852449417114, + "learning_rate": 9.384962610259455e-05, + "loss": 0.008386321365833282, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00842, + "step": 108, + "tokens/total": 3448080, + "tokens/train_per_sec_per_gpu": 26.56, + "tokens/trainable": 49389 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1587515026330948, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0031663556583225727, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.03, + "memory/max_allocated (GiB)": 36.03, + "ppl": 1.00317, + "step": 109, + "tokens/total": 3477952, + "tokens/train_per_sec_per_gpu": 29.8, + "tokens/trainable": 49838 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2903148829936981, + "learning_rate": 9.355341126345868e-05, + "loss": 0.007958847098052502, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00799, + "step": 110, + "tokens/total": 3509776, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 50310 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.642983078956604, + "learning_rate": 9.340289419824107e-05, + "loss": 0.006748045329004526, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00677, + "step": 111, + "tokens/total": 3541808, + "tokens/train_per_sec_per_gpu": 28.04, + "tokens/trainable": 50755 + }, + { + "epoch": 0.4375, + "grad_norm": 0.14928282797336578, + "learning_rate": 9.325077903184159e-05, + "loss": 0.0027839094400405884, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00279, + "step": 112, + "tokens/total": 3574064, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 51213 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.05922335386276245, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0009622900979593396, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00096, + "step": 113, + "tokens/total": 3605984, + "tokens/train_per_sec_per_gpu": 27.99, + "tokens/trainable": 51660 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.44227731227874756, + "learning_rate": 9.2941779782269e-05, + "loss": 0.002763954224064946, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00277, + "step": 114, + "tokens/total": 3638080, + "tokens/train_per_sec_per_gpu": 27.03, + "tokens/trainable": 52133 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5176818370819092, + "learning_rate": 9.278490855781596e-05, + "loss": 0.015623578801751137, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01575, + "step": 115, + "tokens/total": 3670016, + "tokens/train_per_sec_per_gpu": 31.44, + "tokens/trainable": 52580 + }, + { + "epoch": 0.453125, + "grad_norm": 0.4265497922897339, + "learning_rate": 9.262646494908604e-05, + "loss": 0.00794076919555664, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00797, + "step": 116, + "tokens/total": 3701984, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 53033 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.40457987785339355, + "learning_rate": 9.246645554956457e-05, + "loss": 0.022708112373948097, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02297, + "step": 117, + "tokens/total": 3733920, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 53517 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.01414525043219328, + "learning_rate": 9.230488701789578e-05, + "loss": 0.00020209309877827764, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.0002, + "step": 118, + "tokens/total": 3765840, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 53967 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.11359831690788269, + "learning_rate": 9.214176607760577e-05, + "loss": 0.0012168455868959427, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00122, + "step": 119, + "tokens/total": 3797680, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 54430 + }, + { + "epoch": 0.46875, + "grad_norm": 0.35982534289360046, + "learning_rate": 9.197709951682268e-05, + "loss": 0.004110721405595541, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00412, + "step": 120, + "tokens/total": 3829920, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 54896 + }, + { + "epoch": 0.47265625, + "grad_norm": 1.3690382242202759, + "learning_rate": 9.181089418799428e-05, + "loss": 0.02705790475010872, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02743, + "step": 121, + "tokens/total": 3862080, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 55379 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.351891428232193, + "learning_rate": 9.164315700760271e-05, + "loss": 0.0020163573790341616, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00202, + "step": 122, + "tokens/total": 3893968, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 55803 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4747961461544037, + "learning_rate": 9.147389495587671e-05, + "loss": 0.012588979676365852, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01267, + "step": 123, + "tokens/total": 3926160, + "tokens/train_per_sec_per_gpu": 30.14, + "tokens/trainable": 56249 + }, + { + "epoch": 0.484375, + "grad_norm": 0.4064250588417053, + "learning_rate": 9.130311507650116e-05, + "loss": 0.0022941348142921925, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0023, + "step": 124, + "tokens/total": 3958208, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 56713 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.22056929767131805, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0022175831254571676, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00222, + "step": 125, + "tokens/total": 3990208, + "tokens/train_per_sec_per_gpu": 32.57, + "tokens/trainable": 57196 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5698862671852112, + "learning_rate": 9.09570303250602e-05, + "loss": 0.010205100290477276, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01026, + "step": 126, + "tokens/total": 4022224, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 57680 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.2593887448310852, + "learning_rate": 9.078173985499394e-05, + "loss": 0.0032277877908200026, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00323, + "step": 127, + "tokens/total": 4053952, + "tokens/train_per_sec_per_gpu": 26.51, + "tokens/trainable": 58110 + }, + { + "epoch": 0.5, + "grad_norm": 0.3735230565071106, + "learning_rate": 9.060496036067713e-05, + "loss": 0.01031105499714613, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01036, + "step": 128, + "tokens/total": 4086144, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 58534 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.1332484930753708, + "learning_rate": 9.042669919862615e-05, + "loss": 0.0012390739284455776, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00124, + "step": 129, + "tokens/total": 4118096, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 58998 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.20182718336582184, + "learning_rate": 9.024696378701557e-05, + "loss": 0.004501686431467533, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00451, + "step": 130, + "tokens/total": 4150272, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 59448 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.529080331325531, + "learning_rate": 9.006576160536948e-05, + "loss": 0.00763288140296936, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00766, + "step": 131, + "tokens/total": 4181904, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 59906 + }, + { + "epoch": 0.515625, + "grad_norm": 0.21625548601150513, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0024378676898777485, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00244, + "step": 132, + "tokens/total": 4214032, + "tokens/train_per_sec_per_gpu": 28.36, + "tokens/trainable": 60350 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.37116539478302, + "learning_rate": 8.969898715494506e-05, + "loss": 0.0037849934305995703, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00379, + "step": 133, + "tokens/total": 4246064, + "tokens/train_per_sec_per_gpu": 30.18, + "tokens/trainable": 60831 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5646623373031616, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006517104804515839, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00654, + "step": 134, + "tokens/total": 4278224, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 61305 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.6371963620185852, + "learning_rate": 8.932643689864568e-05, + "loss": 0.008687382563948631, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00873, + "step": 135, + "tokens/total": 4310112, + "tokens/train_per_sec_per_gpu": 31.48, + "tokens/trainable": 61792 + }, + { + "epoch": 0.53125, + "grad_norm": 0.8210524916648865, + "learning_rate": 8.913801518498845e-05, + "loss": 0.008436474949121475, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00847, + "step": 136, + "tokens/total": 4342192, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 62253 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.5692176222801208, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017321188002824783, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.01747, + "step": 137, + "tokens/total": 4374096, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.12828375399112701, + "learning_rate": 8.875691779131569e-05, + "loss": 0.0007754197577014565, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00078, + "step": 138, + "tokens/total": 4406272, + "tokens/train_per_sec_per_gpu": 24.22, + "tokens/trainable": 63145 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.3188190162181854, + "learning_rate": 8.856425797031829e-05, + "loss": 0.006707335356622934, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00673, + "step": 139, + "tokens/total": 4438112, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 63587 + }, + { + "epoch": 0.546875, + "grad_norm": 0.6216707229614258, + "learning_rate": 8.837020140354295e-05, + "loss": 0.007024695165455341, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00705, + "step": 140, + "tokens/total": 4470128, + "tokens/train_per_sec_per_gpu": 30.64, + "tokens/trainable": 64052 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.1480305939912796, + "learning_rate": 8.817475616647554e-05, + "loss": 0.00229115248657763, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00229, + "step": 141, + "tokens/total": 4502096, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 64526 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22254763543605804, + "learning_rate": 8.797793039239017e-05, + "loss": 0.0029198499396443367, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00292, + "step": 142, + "tokens/total": 4534496, + "tokens/train_per_sec_per_gpu": 29.08, + "tokens/trainable": 64983 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.6901941895484924, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013051149435341358, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01314, + "step": 143, + "tokens/total": 4566720, + "tokens/train_per_sec_per_gpu": 31.56, + "tokens/trainable": 65492 + }, + { + "epoch": 0.5625, + "grad_norm": 0.9545458555221558, + "learning_rate": 8.758017005316988e-05, + "loss": 0.02235381305217743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.02261, + "step": 144, + "tokens/total": 4598880, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 65925 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.5515892505645752, + "learning_rate": 8.737925204046629e-05, + "loss": 0.02038375660777092, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.02059, + "step": 145, + "tokens/total": 4630864, + "tokens/train_per_sec_per_gpu": 26.3, + "tokens/trainable": 66383 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3512430191040039, + "learning_rate": 8.717698659491851e-05, + "loss": 0.008233271539211273, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00827, + "step": 146, + "tokens/total": 4662864, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 66840 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.6830361485481262, + "learning_rate": 8.697338213361735e-05, + "loss": 0.013419180177152157, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01351, + "step": 147, + "tokens/total": 4694912, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 67297 + }, + { + "epoch": 0.578125, + "grad_norm": 0.2117031067609787, + "learning_rate": 8.676844712937552e-05, + "loss": 0.004134457092732191, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00414, + "step": 148, + "tokens/total": 4727120, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 67769 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.378053218126297, + "learning_rate": 8.656219011037509e-05, + "loss": 0.009747720323503017, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0098, + "step": 149, + "tokens/total": 4759168, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 68257 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.21076741814613342, + "learning_rate": 8.63546196598125e-05, + "loss": 0.005035653710365295, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00505, + "step": 150, + "tokens/total": 4791008, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 68706 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.25075697898864746, + "learning_rate": 8.614574441554145e-05, + "loss": 0.006027051247656345, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00605, + "step": 151, + "tokens/total": 4822912, + "tokens/train_per_sec_per_gpu": 27.95, + "tokens/trainable": 69131 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5774204730987549, + "learning_rate": 8.593557306971349e-05, + "loss": 0.018617186695337296, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01879, + "step": 152, + "tokens/total": 4854816, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 69586 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.1364293098449707, + "learning_rate": 8.572411436841618e-05, + "loss": 0.002049289643764496, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00205, + "step": 153, + "tokens/total": 4886656, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 70061 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.25387004017829895, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0044304742477834225, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00444, + "step": 154, + "tokens/total": 4918768, + "tokens/train_per_sec_per_gpu": 23.26, + "tokens/trainable": 70467 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3047125041484833, + "learning_rate": 8.529737015125824e-05, + "loss": 0.0043389927595853806, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00435, + "step": 155, + "tokens/total": 4951008, + "tokens/train_per_sec_per_gpu": 27.53, + "tokens/trainable": 70933 + }, + { + "epoch": 0.609375, + "grad_norm": 0.1442601978778839, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0031973645091056824, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.0032, + "step": 156, + "tokens/total": 4982976, + "tokens/train_per_sec_per_gpu": 28.86, + "tokens/trainable": 71382 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.2608455717563629, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0032951058819890022, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0033, + "step": 157, + "tokens/total": 5014640, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 71808 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.637295663356781, + "learning_rate": 8.464782037243449e-05, + "loss": 0.01436957623809576, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.01447, + "step": 158, + "tokens/total": 5046672, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 72249 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.9834056496620178, + "learning_rate": 8.442882418044202e-05, + "loss": 0.02375311404466629, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.02404, + "step": 159, + "tokens/total": 5078528, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 72726 + }, + { + "epoch": 0.625, + "grad_norm": 0.35432615876197815, + "learning_rate": 8.420860333495179e-05, + "loss": 0.006119017023593187, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00614, + "step": 160, + "tokens/total": 5110704, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 73148 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.44270065426826477, + "learning_rate": 8.398716700025208e-05, + "loss": 0.01352179329842329, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.72, + "memory/max_allocated (GiB)": 36.72, + "ppl": 1.01361, + "step": 161, + "tokens/total": 5142944, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 73600 + }, + { + "epoch": 0.6328125, + "grad_norm": 1.2181634902954102, + "learning_rate": 8.376452439121266e-05, + "loss": 0.01820964366197586, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.01838, + "step": 162, + "tokens/total": 5174592, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 74068 + }, + { + "epoch": 0.63671875, + "grad_norm": 1.9321303367614746, + "learning_rate": 8.354068477290124e-05, + "loss": 0.009742327965795994, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00979, + "step": 163, + "tokens/total": 5206432, + "tokens/train_per_sec_per_gpu": 29.89, + "tokens/trainable": 74527 + }, + { + "epoch": 0.640625, + "grad_norm": 0.3545796275138855, + "learning_rate": 8.331565746019807e-05, + "loss": 0.00975855067372322, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00981, + "step": 164, + "tokens/total": 5238384, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 74992 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.35368478298187256, + "learning_rate": 8.308945181740812e-05, + "loss": 0.010195466689765453, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01025, + "step": 165, + "tokens/total": 5270272, + "tokens/train_per_sec_per_gpu": 29.41, + "tokens/trainable": 75442 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.34957799315452576, + "learning_rate": 8.286207725787153e-05, + "loss": 0.009373681619763374, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00942, + "step": 166, + "tokens/total": 5302368, + "tokens/train_per_sec_per_gpu": 26.2, + "tokens/trainable": 75887 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.3684624135494232, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008211556822061539, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00825, + "step": 167, + "tokens/total": 5334496, + "tokens/train_per_sec_per_gpu": 28.21, + "tokens/trainable": 76331 + }, + { + "epoch": 0.65625, + "grad_norm": 0.12826858460903168, + "learning_rate": 8.240385928474219e-05, + "loss": 0.004492453299462795, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0045, + "step": 168, + "tokens/total": 5364288, + "tokens/train_per_sec_per_gpu": 30.16, + "tokens/trainable": 76745 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.31828173995018005, + "learning_rate": 8.217303493946967e-05, + "loss": 0.005733514670282602, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00575, + "step": 169, + "tokens/total": 5396512, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 77201 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6174353361129761, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01915649324655533, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01934, + "step": 170, + "tokens/total": 5428624, + "tokens/train_per_sec_per_gpu": 27.75, + "tokens/trainable": 77655 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.2902314066886902, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0036087254993617535, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00362, + "step": 171, + "tokens/total": 5460880, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 78122 + }, + { + "epoch": 0.671875, + "grad_norm": 0.3267754912376404, + "learning_rate": 8.147381587530713e-05, + "loss": 0.008090222254395485, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00812, + "step": 172, + "tokens/total": 5493088, + "tokens/train_per_sec_per_gpu": 27.86, + "tokens/trainable": 78576 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.4086189866065979, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006351984106004238, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00637, + "step": 173, + "tokens/total": 5525312, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 79059 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5261490345001221, + "learning_rate": 8.100214524900103e-05, + "loss": 0.008662862703204155, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.0087, + "step": 174, + "tokens/total": 5556880, + "tokens/train_per_sec_per_gpu": 24.62, + "tokens/trainable": 79498 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11292761564254761, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0027105510234832764, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00271, + "step": 175, + "tokens/total": 5588976, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 79922 + }, + { + "epoch": 0.6875, + "grad_norm": 0.33315548300743103, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0030926030594855547, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.0031, + "step": 176, + "tokens/total": 5618992, + "tokens/train_per_sec_per_gpu": 29.51, + "tokens/trainable": 80383 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.3136921226978302, + "learning_rate": 8.028654871074489e-05, + "loss": 0.005057850852608681, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00507, + "step": 177, + "tokens/total": 5651072, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 80824 + }, + { + "epoch": 0.6953125, + "grad_norm": 1.0526509284973145, + "learning_rate": 8.004589869885986e-05, + "loss": 0.010707498528063297, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01077, + "step": 178, + "tokens/total": 5683296, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 81243 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.1329648196697235, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0017975447699427605, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.0018, + "step": 179, + "tokens/total": 5715520, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 81716 + }, + { + "epoch": 0.703125, + "grad_norm": 1.0749201774597168, + "learning_rate": 7.95614819466576e-05, + "loss": 0.025617048144340515, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02595, + "step": 180, + "tokens/total": 5747488, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 82181 + }, + { + "epoch": 0.70703125, + "grad_norm": 1.0446465015411377, + "learning_rate": 7.931773536489872e-05, + "loss": 0.030236585065722466, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 35.9, + "memory/max_allocated (GiB)": 35.9, + "ppl": 1.0307, + "step": 181, + "tokens/total": 5777264, + "tokens/train_per_sec_per_gpu": 28.89, + "tokens/trainable": 82626 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22853964567184448, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0038214433006942272, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00383, + "step": 182, + "tokens/total": 5809264, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 83089 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.057821910828351974, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0004995397757738829, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0005, + "step": 183, + "tokens/total": 5841488, + "tokens/train_per_sec_per_gpu": 29.34, + "tokens/trainable": 83590 + }, + { + "epoch": 0.71875, + "grad_norm": 1.3264167308807373, + "learning_rate": 7.85804646415409e-05, + "loss": 0.013232000172138214, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01332, + "step": 184, + "tokens/total": 5873696, + "tokens/train_per_sec_per_gpu": 29.33, + "tokens/trainable": 84046 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.6027824878692627, + "learning_rate": 7.833273149760207e-05, + "loss": 0.012215660884976387, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01229, + "step": 185, + "tokens/total": 5905408, + "tokens/train_per_sec_per_gpu": 28.25, + "tokens/trainable": 84517 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.5848041772842407, + "learning_rate": 7.808402738346527e-05, + "loss": 0.024627480655908585, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.02493, + "step": 186, + "tokens/total": 5937392, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 84974 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.6823275089263916, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0028813849203288555, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00289, + "step": 187, + "tokens/total": 5969344, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 85460 + }, + { + "epoch": 0.734375, + "grad_norm": 0.41855958104133606, + "learning_rate": 7.758374768294647e-05, + "loss": 0.009796380065381527, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00984, + "step": 188, + "tokens/total": 6001296, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 85939 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6849080324172974, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004100777208805084, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00411, + "step": 189, + "tokens/total": 6031216, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 86377 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.15838374197483063, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003986812196671963, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00399, + "step": 190, + "tokens/total": 6063296, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 86834 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.3117486536502838, + "learning_rate": 7.682630588562518e-05, + "loss": 0.009132719598710537, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00917, + "step": 191, + "tokens/total": 6095216, + "tokens/train_per_sec_per_gpu": 26.65, + "tokens/trainable": 87265 + }, + { + "epoch": 0.75, + "grad_norm": 0.7837114334106445, + "learning_rate": 7.657199467573129e-05, + "loss": 0.007239446043968201, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00727, + "step": 192, + "tokens/total": 6127488, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 87747 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21777178347110748, + "learning_rate": 7.631678576708561e-05, + "loss": 0.002932134782895446, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00294, + "step": 193, + "tokens/total": 6159488, + "tokens/train_per_sec_per_gpu": 30.37, + "tokens/trainable": 88239 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.19159242510795593, + "learning_rate": 7.606068977997255e-05, + "loss": 0.003356864908710122, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00336, + "step": 194, + "tokens/total": 6191296, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 88718 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.3977915346622467, + "learning_rate": 7.580371737159148e-05, + "loss": 0.008549144491553307, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00859, + "step": 195, + "tokens/total": 6223120, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 89129 + }, + { + "epoch": 0.765625, + "grad_norm": 0.3875303566455841, + "learning_rate": 7.554587923561324e-05, + "loss": 0.006131038535386324, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00615, + "step": 196, + "tokens/total": 6254880, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 89567 + }, + { + "epoch": 0.76953125, + "grad_norm": 3.4133481979370117, + "learning_rate": 7.528718610173511e-05, + "loss": 0.01836150512099266, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01853, + "step": 197, + "tokens/total": 6286960, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 89988 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.2509997487068176, + "learning_rate": 7.502764873523431e-05, + "loss": 0.003821226069703698, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00383, + "step": 198, + "tokens/total": 6318880, + "tokens/train_per_sec_per_gpu": 30.29, + "tokens/trainable": 90434 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.9550195932388306, + "learning_rate": 7.476727793652011e-05, + "loss": 0.006872436031699181, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0069, + "step": 199, + "tokens/total": 6350928, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 90913 + }, + { + "epoch": 0.78125, + "grad_norm": 0.411724716424942, + "learning_rate": 7.450608454068415e-05, + "loss": 0.014243390411138535, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01435, + "step": 200, + "tokens/total": 6382976, + "tokens/train_per_sec_per_gpu": 28.29, + "tokens/trainable": 91355 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.34264692664146423, + "learning_rate": 7.424407941704987e-05, + "loss": 0.005309835076332092, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00532, + "step": 201, + "tokens/total": 6415120, + "tokens/train_per_sec_per_gpu": 31.41, + "tokens/trainable": 91856 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.41955795884132385, + "learning_rate": 7.398127346871986e-05, + "loss": 0.009196512401103973, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00924, + "step": 202, + "tokens/total": 6447216, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 92311 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.03107454814016819, + "learning_rate": 7.371767763212238e-05, + "loss": 0.0004908143309876323, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00049, + "step": 203, + "tokens/total": 6479280, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 92764 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39578482508659363, + "learning_rate": 7.345330287655617e-05, + "loss": 0.006089594680815935, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00611, + "step": 204, + "tokens/total": 6511040, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 93227 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.6444572806358337, + "learning_rate": 7.31881602037339e-05, + "loss": 0.010252210311591625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0103, + "step": 205, + "tokens/total": 6543120, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 93675 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.3283638060092926, + "learning_rate": 7.29222606473245e-05, + "loss": 0.00761406822130084, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00764, + "step": 206, + "tokens/total": 6575520, + "tokens/train_per_sec_per_gpu": 27.02, + "tokens/trainable": 94120 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.09952464699745178, + "learning_rate": 7.265561527249383e-05, + "loss": 0.001628706930205226, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00163, + "step": 207, + "tokens/total": 6607408, + "tokens/train_per_sec_per_gpu": 29.43, + "tokens/trainable": 94557 + }, + { + "epoch": 0.8125, + "grad_norm": 0.31704938411712646, + "learning_rate": 7.238823517544436e-05, + "loss": 0.005896817892789841, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00591, + "step": 208, + "tokens/total": 6639488, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 95035 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.0999862551689148, + "learning_rate": 7.212013148295333e-05, + "loss": 0.001754446537233889, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00176, + "step": 209, + "tokens/total": 6671520, + "tokens/train_per_sec_per_gpu": 31.08, + "tokens/trainable": 95517 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.1405109018087387, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0017940793186426163, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0018, + "step": 210, + "tokens/total": 6703536, + "tokens/train_per_sec_per_gpu": 30.3, + "tokens/trainable": 96026 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.09185238927602768, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0015102234901860356, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00151, + "step": 211, + "tokens/total": 6735616, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 96477 + }, + { + "epoch": 0.828125, + "grad_norm": 1.0464473962783813, + "learning_rate": 7.131159054949273e-05, + "loss": 0.016731251031160355, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01687, + "step": 212, + "tokens/total": 6767584, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 96951 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.3517152965068817, + "learning_rate": 7.104070433827139e-05, + "loss": 0.007984626106917858, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00802, + "step": 213, + "tokens/total": 6799488, + "tokens/train_per_sec_per_gpu": 25.83, + "tokens/trainable": 97350 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.28192469477653503, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00557567086070776, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00559, + "step": 214, + "tokens/total": 6831552, + "tokens/train_per_sec_per_gpu": 24.66, + "tokens/trainable": 97792 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.46135279536247253, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0170141588896513, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.01716, + "step": 215, + "tokens/total": 6863488, + "tokens/train_per_sec_per_gpu": 30.38, + "tokens/trainable": 98257 + }, + { + "epoch": 0.84375, + "grad_norm": 0.4277898073196411, + "learning_rate": 7.022408581865382e-05, + "loss": 0.009399959817528725, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.71, + "memory/max_allocated (GiB)": 36.71, + "ppl": 1.00944, + "step": 216, + "tokens/total": 6895888, + "tokens/train_per_sec_per_gpu": 27.28, + "tokens/trainable": 98731 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.34988704323768616, + "learning_rate": 6.99505974422157e-05, + "loss": 0.002909669652581215, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00291, + "step": 217, + "tokens/total": 6928000, + "tokens/train_per_sec_per_gpu": 30.25, + "tokens/trainable": 99212 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.4028005599975586, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0058855339884757996, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0059, + "step": 218, + "tokens/total": 6960096, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 99654 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.3000575304031372, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0025230362080037594, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00253, + "step": 219, + "tokens/total": 6992512, + "tokens/train_per_sec_per_gpu": 27.46, + "tokens/trainable": 100086 + }, + { + "epoch": 0.859375, + "grad_norm": 0.11351172626018524, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0021392193157225847, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00214, + "step": 220, + "tokens/total": 7024688, + "tokens/train_per_sec_per_gpu": 27.56, + "tokens/trainable": 100524 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.19512486457824707, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00650001410394907, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00652, + "step": 221, + "tokens/total": 7056928, + "tokens/train_per_sec_per_gpu": 28.05, + "tokens/trainable": 100996 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.1998104453086853, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0056140003725886345, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00563, + "step": 222, + "tokens/total": 7088944, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 101449 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.40908998250961304, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01286369375884533, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.01295, + "step": 223, + "tokens/total": 7121104, + "tokens/train_per_sec_per_gpu": 27.19, + "tokens/trainable": 101905 + }, + { + "epoch": 0.875, + "grad_norm": 0.3672473132610321, + "learning_rate": 6.801939899284132e-05, + "loss": 0.010677668265998363, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01073, + "step": 224, + "tokens/total": 7153040, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 102411 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4146183431148529, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0038945709820836782, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0039, + "step": 225, + "tokens/total": 7184912, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 102882 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.10585323721170425, + "learning_rate": 6.746257910210214e-05, + "loss": 0.00041726604104042053, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00042, + "step": 226, + "tokens/total": 7216992, + "tokens/train_per_sec_per_gpu": 28.72, + "tokens/trainable": 103332 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.36600011587142944, + "learning_rate": 6.718338543014937e-05, + "loss": 0.01526213251054287, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 227, + "tokens/total": 7249120, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 103790 + }, + { + "epoch": 0.890625, + "grad_norm": 0.16057318449020386, + "learning_rate": 6.69036847577983e-05, + "loss": 0.004024423658847809, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00403, + "step": 228, + "tokens/total": 7281072, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 104246 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.318798303604126, + "learning_rate": 6.662348872453553e-05, + "loss": 0.006663155741989613, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00669, + "step": 229, + "tokens/total": 7313120, + "tokens/train_per_sec_per_gpu": 31.3, + "tokens/trainable": 104707 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.10858234763145447, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0020512123592197895, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00205, + "step": 230, + "tokens/total": 7345184, + "tokens/train_per_sec_per_gpu": 30.81, + "tokens/trainable": 105167 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.1044008731842041, + "learning_rate": 6.60616572358065e-05, + "loss": 0.0010264476295560598, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00103, + "step": 231, + "tokens/total": 7377264, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 105576 + }, + { + "epoch": 0.90625, + "grad_norm": 0.14836947619915009, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0013845522189512849, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00139, + "step": 232, + "tokens/total": 7409424, + "tokens/train_per_sec_per_gpu": 30.32, + "tokens/trainable": 106043 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.022752461954951286, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0003245974367018789, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00032, + "step": 233, + "tokens/total": 7441456, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 106506 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.22981758415699005, + "learning_rate": 6.521548694236384e-05, + "loss": 0.002587020630016923, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00259, + "step": 234, + "tokens/total": 7473344, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 106951 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.5147526264190674, + "learning_rate": 6.493256429322259e-05, + "loss": 0.01256126631051302, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01264, + "step": 235, + "tokens/total": 7505040, + "tokens/train_per_sec_per_gpu": 25.64, + "tokens/trainable": 107382 + }, + { + "epoch": 0.921875, + "grad_norm": 0.3785130977630615, + "learning_rate": 6.464922830953799e-05, + "loss": 0.010956985875964165, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01102, + "step": 236, + "tokens/total": 7537120, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 107814 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.6193199753761292, + "learning_rate": 6.436549078207688e-05, + "loss": 0.020655140280723572, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02087, + "step": 237, + "tokens/total": 7569024, + "tokens/train_per_sec_per_gpu": 29.36, + "tokens/trainable": 108274 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.5084519982337952, + "learning_rate": 6.408136351831592e-05, + "loss": 0.011346567422151566, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01141, + "step": 238, + "tokens/total": 7600992, + "tokens/train_per_sec_per_gpu": 24.43, + "tokens/trainable": 108714 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.04367341846227646, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0004316020058467984, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00043, + "step": 239, + "tokens/total": 7633280, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 109220 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06729783862829208, + "learning_rate": 6.351198709240186e-05, + "loss": 0.000604260538239032, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.0006, + "step": 240, + "tokens/total": 7665408, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 109672 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.3010976314544678, + "learning_rate": 6.32267616243259e-05, + "loss": 0.006106264889240265, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00612, + "step": 241, + "tokens/total": 7697280, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 110135 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.11426915228366852, + "learning_rate": 6.294119380711849e-05, + "loss": 0.0014383264351636171, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00144, + "step": 242, + "tokens/total": 7729440, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 110563 + }, + { + "epoch": 0.94921875, + "grad_norm": 1.3792941570281982, + "learning_rate": 6.265529552442209e-05, + "loss": 0.003547664964571595, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00355, + "step": 243, + "tokens/total": 7761584, + "tokens/train_per_sec_per_gpu": 30.76, + "tokens/trainable": 111049 + }, + { + "epoch": 0.953125, + "grad_norm": 0.07595008611679077, + "learning_rate": 6.236907867363127e-05, + "loss": 0.0009103374904952943, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00091, + "step": 244, + "tokens/total": 7793824, + "tokens/train_per_sec_per_gpu": 28.23, + "tokens/trainable": 111495 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05425361171364784, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0013249842450022697, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00133, + "step": 245, + "tokens/total": 7825872, + "tokens/train_per_sec_per_gpu": 26.47, + "tokens/trainable": 111968 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.17112742364406586, + "learning_rate": 6.179573692313344e-05, + "loss": 0.001098912674933672, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.0011, + "step": 246, + "tokens/total": 7857840, + "tokens/train_per_sec_per_gpu": 30.09, + "tokens/trainable": 112416 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.10754083096981049, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0031517897732555866, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00316, + "step": 247, + "tokens/total": 7890000, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 112882 + }, + { + "epoch": 0.96875, + "grad_norm": 0.2477901428937912, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005755824968218803, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00577, + "step": 248, + "tokens/total": 7921984, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 113390 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.332253634929657, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.013207933865487576, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.0133, + "step": 249, + "tokens/total": 7953824, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 113904 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.18514207005500793, + "learning_rate": 6.064575550087316e-05, + "loss": 0.00523729994893074, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00525, + "step": 250, + "tokens/total": 7985984, + "tokens/train_per_sec_per_gpu": 28.41, + "tokens/trainable": 114351 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.1852743923664093, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0031474479474127293, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00315, + "step": 251, + "tokens/total": 8018256, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 114842 + }, + { + "epoch": 0.984375, + "grad_norm": 0.43020468950271606, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0053521618247032166, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00537, + "step": 252, + "tokens/total": 8050528, + "tokens/train_per_sec_per_gpu": 23.97, + "tokens/trainable": 115263 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.048135045915842056, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0005759480991400778, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00058, + "step": 253, + "tokens/total": 8082496, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 115703 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.028733640909194946, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00040354474913328886, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0004, + "step": 254, + "tokens/total": 8114592, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 116157 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.2808869481086731, + "learning_rate": 5.920308275189541e-05, + "loss": 0.006315852981060743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00634, + "step": 255, + "tokens/total": 8146752, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 116567 + }, + { + "epoch": 1.0, + "grad_norm": 0.054996367543935776, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0009099108865484595, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00091, + "step": 256, + "tokens/total": 8178768, + "tokens/train_per_sec_per_gpu": 25.98, + "tokens/trainable": 117030 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.18114623427391052, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.002796629210934043, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0028, + "step": 257, + "tokens/total": 8210800, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 117517 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.1408185362815857, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0013210453325882554, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00132, + "step": 258, + "tokens/total": 8242768, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 118015 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.36587440967559814, + "learning_rate": 5.80457242789548e-05, + "loss": 0.002160525880753994, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00216, + "step": 259, + "tokens/total": 8274784, + "tokens/train_per_sec_per_gpu": 30.13, + "tokens/trainable": 118520 + }, + { + "epoch": 1.015625, + "grad_norm": 0.02628348581492901, + "learning_rate": 5.77560376810977e-05, + "loss": 0.0003391164937056601, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00034, + "step": 260, + "tokens/total": 8306720, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 118992 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.014996240846812725, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0002124913444276899, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00021, + "step": 261, + "tokens/total": 8338784, + "tokens/train_per_sec_per_gpu": 27.96, + "tokens/trainable": 119438 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.0050656464882195, + "learning_rate": 5.717633247526522e-05, + "loss": 7.557802018709481e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00008, + "step": 262, + "tokens/total": 8370720, + "tokens/train_per_sec_per_gpu": 26.9, + "tokens/trainable": 119881 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.02459302730858326, + "learning_rate": 5.688633799118971e-05, + "loss": 0.00024908874183893204, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00025, + "step": 263, + "tokens/total": 8402736, + "tokens/train_per_sec_per_gpu": 23.2, + "tokens/trainable": 120285 + }, + { + "epoch": 1.03125, + "grad_norm": 0.017813201993703842, + "learning_rate": 5.659626500889066e-05, + "loss": 0.00021895825921092182, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00022, + "step": 264, + "tokens/total": 8434352, + "tokens/train_per_sec_per_gpu": 27.78, + "tokens/trainable": 120755 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.0038785531651228666, + "learning_rate": 5.6306125599488905e-05, + "loss": 6.556476728292182e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00007, + "step": 265, + "tokens/total": 8466208, + "tokens/train_per_sec_per_gpu": 29.86, + "tokens/trainable": 121194 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.18033447861671448, + "learning_rate": 5.601593183686955e-05, + "loss": 0.0016840343596413732, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00169, + "step": 266, + "tokens/total": 8498304, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 121670 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.22497256100177765, + "learning_rate": 5.572569579717961e-05, + "loss": 0.003026450052857399, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00303, + "step": 267, + "tokens/total": 8530400, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 122142 + }, + { + "epoch": 1.046875, + "grad_norm": 0.008878038264811039, + "learning_rate": 5.543542955832538e-05, + "loss": 8.136438555084169e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00008, + "step": 268, + "tokens/total": 8562608, + "tokens/train_per_sec_per_gpu": 27.27, + "tokens/trainable": 122585 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.19042901694774628, + "learning_rate": 5.514514519946986e-05, + "loss": 0.0020417363848537207, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00204, + "step": 269, + "tokens/total": 8594832, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 123091 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.003382593160495162, + "learning_rate": 5.485485480053015e-05, + "loss": 3.870048021781258e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00004, + "step": 270, + "tokens/total": 8627120, + "tokens/train_per_sec_per_gpu": 26.36, + "tokens/trainable": 123505 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.2004430592060089, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.002532299840822816, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.00254, + "step": 271, + "tokens/total": 8657056, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 123953 + }, + { + "epoch": 1.0625, + "grad_norm": 0.24857956171035767, + "learning_rate": 5.42743042028204e-05, + "loss": 0.004157304298132658, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00417, + "step": 272, + "tokens/total": 8689104, + "tokens/train_per_sec_per_gpu": 24.05, + "tokens/trainable": 124400 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.05064955726265907, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0006531125982291996, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00065, + "step": 273, + "tokens/total": 8721136, + "tokens/train_per_sec_per_gpu": 29.3, + "tokens/trainable": 124870 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.12664659321308136, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0018034171080216765, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00181, + "step": 274, + "tokens/total": 8753200, + "tokens/train_per_sec_per_gpu": 26.52, + "tokens/trainable": 125333 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.01593201234936714, + "learning_rate": 5.340373499110935e-05, + "loss": 0.00010935973114101216, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00011, + "step": 275, + "tokens/total": 8785168, + "tokens/train_per_sec_per_gpu": 30.45, + "tokens/trainable": 125834 + }, + { + "epoch": 1.078125, + "grad_norm": 0.0015056057600304484, + "learning_rate": 5.3113662008810304e-05, + "loss": 1.9860939573845826e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00002, + "step": 276, + "tokens/total": 8817520, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 126316 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.004457728937268257, + "learning_rate": 5.282366752473479e-05, + "loss": 2.716448398132343e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00003, + "step": 277, + "tokens/total": 8849360, + "tokens/train_per_sec_per_gpu": 29.63, + "tokens/trainable": 126798 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.03812727704644203, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.0003402164438739419, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.00034, + "step": 278, + "tokens/total": 8881152, + "tokens/train_per_sec_per_gpu": 24.5, + "tokens/trainable": 127223 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.011409996077418327, + "learning_rate": 5.224396231890232e-05, + "loss": 9.064963523996994e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.00009, + "step": 279, + "tokens/total": 8912512, + "tokens/train_per_sec_per_gpu": 29.78, + "tokens/trainable": 127672 + }, + { + "epoch": 1.09375, + "grad_norm": 0.037030577659606934, + "learning_rate": 5.195427572104522e-05, + "loss": 0.0003734154161065817, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00037, + "step": 280, + "tokens/total": 8944416, + "tokens/train_per_sec_per_gpu": 26.76, + "tokens/trainable": 128116 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.0032268385402858257, + "learning_rate": 5.166471586820751e-05, + "loss": 2.529611811041832e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00003, + "step": 281, + "tokens/total": 8976272, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 128589 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.20227248966693878, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.0024647137615829706, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00247, + "step": 282, + "tokens/total": 9008560, + "tokens/train_per_sec_per_gpu": 27.84, + "tokens/trainable": 129036 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.43708595633506775, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0028322634752839804, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00284, + "step": 283, + "tokens/total": 9040496, + "tokens/train_per_sec_per_gpu": 29.25, + "tokens/trainable": 129485 + }, + { + "epoch": 1.109375, + "grad_norm": 0.09467080235481262, + "learning_rate": 5.079691724810461e-05, + "loss": 0.001156509853899479, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00116, + "step": 284, + "tokens/total": 9072448, + "tokens/train_per_sec_per_gpu": 24.69, + "tokens/trainable": 129920 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.1778785139322281, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.0016857109731063247, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00169, + "step": 285, + "tokens/total": 9104384, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 130417 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.0003376641252543777, + "learning_rate": 5.021923930849237e-05, + "loss": 5.597693871095544e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00001, + "step": 286, + "tokens/total": 9136416, + "tokens/train_per_sec_per_gpu": 29.99, + "tokens/trainable": 130903 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.019782084971666336, + "learning_rate": 4.99306927511967e-05, + "loss": 0.00012461614096537232, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00012, + "step": 287, + "tokens/total": 9168640, + "tokens/train_per_sec_per_gpu": 29.24, + "tokens/trainable": 131343 + }, + { + "epoch": 1.125, + "grad_norm": 0.9641065001487732, + "learning_rate": 4.964235714846775e-05, + "loss": 0.011515076272189617, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01158, + "step": 288, + "tokens/total": 9200544, + "tokens/train_per_sec_per_gpu": 27.04, + "tokens/trainable": 131763 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0010062052169814706, + "learning_rate": 4.9354244499126866e-05, + "loss": 9.489545846008696e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00001, + "step": 289, + "tokens/total": 9232624, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 132229 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.002373398980125785, + "learning_rate": 4.90663667927174e-05, + "loss": 1.573777262819931e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00002, + "step": 290, + "tokens/total": 9264768, + "tokens/train_per_sec_per_gpu": 29.5, + "tokens/trainable": 132678 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.0014718567254021764, + "learning_rate": 4.877873600900581e-05, + "loss": 9.888429303828161e-06, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00001, + "step": 291, + "tokens/total": 9296944, + "tokens/train_per_sec_per_gpu": 24.47, + "tokens/trainable": 133136 + }, + { + "epoch": 1.140625, + "grad_norm": 0.0018440725980326533, + "learning_rate": 4.849136411748306e-05, + "loss": 8.275845175376162e-06, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00001, + "step": 292, + "tokens/total": 9328848, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 133608 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.12918472290039062, + "learning_rate": 4.8204263076866574e-05, + "loss": 0.0009128568926826119, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00091, + "step": 293, + "tokens/total": 9360928, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 134108 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.019195031374692917, + "learning_rate": 4.791744483460251e-05, + "loss": 9.439548011869192e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00009, + "step": 294, + "tokens/total": 9393040, + "tokens/train_per_sec_per_gpu": 25.71, + "tokens/trainable": 134541 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.1602204144001007, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0004640200058929622, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00046, + "step": 295, + "tokens/total": 9425248, + "tokens/train_per_sec_per_gpu": 24.5, + "tokens/trainable": 134966 + }, + { + "epoch": 1.15625, + "grad_norm": 0.5406288504600525, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.0034592358861118555, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00347, + "step": 296, + "tokens/total": 9457264, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 135402 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.004333238583058119, + "learning_rate": 4.705880619288153e-05, + "loss": 3.075868880841881e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00003, + "step": 297, + "tokens/total": 9489056, + "tokens/train_per_sec_per_gpu": 24.87, + "tokens/trainable": 135804 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.006247827783226967, + "learning_rate": 4.677323837567412e-05, + "loss": 2.383375249337405e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00002, + "step": 298, + "tokens/total": 9521024, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 136231 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.0779612734913826, + "learning_rate": 4.6488012907598146e-05, + "loss": 0.0004679379053413868, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00047, + "step": 299, + "tokens/total": 9553120, + "tokens/train_per_sec_per_gpu": 28.37, + "tokens/trainable": 136705 + }, + { + "epoch": 1.171875, + "grad_norm": 0.01816786825656891, + "learning_rate": 4.620314165804964e-05, + "loss": 6.801338167861104e-05, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00007, + "step": 300, + "tokens/total": 9585040, + "tokens/train_per_sec_per_gpu": 28.78, + "tokens/trainable": 137178 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.9530329704284668, + "learning_rate": 4.591863648168407e-05, + "loss": 0.008843549527227879, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00888, + "step": 301, + "tokens/total": 9617120, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 137643 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0371168851852417, + "learning_rate": 4.5634509217923135e-05, + "loss": 0.00014635953994002193, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00015, + "step": 302, + "tokens/total": 9649024, + "tokens/train_per_sec_per_gpu": 26.74, + "tokens/trainable": 138078 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.08589725196361542, + "learning_rate": 4.535077169046201e-05, + "loss": 0.00043714733328670263, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00044, + "step": 303, + "tokens/total": 9680832, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 138515 + }, + { + "epoch": 1.1875, + "grad_norm": 0.358940988779068, + "learning_rate": 4.506743570677743e-05, + "loss": 0.006240838672965765, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00626, + "step": 304, + "tokens/total": 9713040, + "tokens/train_per_sec_per_gpu": 29.29, + "tokens/trainable": 138948 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.12659867107868195, + "learning_rate": 4.478451305763618e-05, + "loss": 0.0008667556685395539, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00087, + "step": 305, + "tokens/total": 9744976, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 139408 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.08289908617734909, + "learning_rate": 4.450201551660454e-05, + "loss": 0.0006384386215358973, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.02, + "memory/max_allocated (GiB)": 36.02, + "ppl": 1.00064, + "step": 306, + "tokens/total": 9774784, + "tokens/train_per_sec_per_gpu": 27.76, + "tokens/trainable": 139840 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.19195082783699036, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.002342985011637211, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00235, + "step": 307, + "tokens/total": 9806656, + "tokens/train_per_sec_per_gpu": 27.6, + "tokens/trainable": 140281 + }, + { + "epoch": 1.203125, + "grad_norm": 0.07421132177114487, + "learning_rate": 4.393834276419352e-05, + "loss": 0.0007085398538038135, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00071, + "step": 308, + "tokens/total": 9838720, + "tokens/train_per_sec_per_gpu": 30.88, + "tokens/trainable": 140776 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.004596009384840727, + "learning_rate": 4.36571910095382e-05, + "loss": 2.8969257982680574e-05, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00003, + "step": 309, + "tokens/total": 9870784, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 141228 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.04099060595035553, + "learning_rate": 4.337651127546448e-05, + "loss": 0.00030985785997472703, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00031, + "step": 310, + "tokens/total": 9902784, + "tokens/train_per_sec_per_gpu": 30.8, + "tokens/trainable": 141679 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.18840794265270233, + "learning_rate": 4.3096315242201736e-05, + "loss": 0.0020186181645840406, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00202, + "step": 311, + "tokens/total": 9934768, + "tokens/train_per_sec_per_gpu": 26.72, + "tokens/trainable": 142122 + }, + { + "epoch": 1.21875, + "grad_norm": 0.5018841028213501, + "learning_rate": 4.2816614569850635e-05, + "loss": 0.006148474756628275, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00617, + "step": 312, + "tokens/total": 9966960, + "tokens/train_per_sec_per_gpu": 27.27, + "tokens/trainable": 142561 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.1730002760887146, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.003579454030841589, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00359, + "step": 313, + "tokens/total": 9999104, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 143046 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.6586350202560425, + "learning_rate": 4.225874584473174e-05, + "loss": 0.0015231478027999401, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00152, + "step": 314, + "tokens/total": 10031136, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 143493 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.011105087585747242, + "learning_rate": 4.19806010071587e-05, + "loss": 0.00011838486534543335, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00012, + "step": 315, + "tokens/total": 10063184, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 143956 + }, + { + "epoch": 1.234375, + "grad_norm": 0.2628834843635559, + "learning_rate": 4.170299795992081e-05, + "loss": 0.003402995876967907, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00341, + "step": 316, + "tokens/total": 10095312, + "tokens/train_per_sec_per_gpu": 28.29, + "tokens/trainable": 144411 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.13821958005428314, + "learning_rate": 4.142594825521398e-05, + "loss": 0.0016696923412382603, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00167, + "step": 317, + "tokens/total": 10127440, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 144892 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.0023418583441525698, + "learning_rate": 4.114946342220728e-05, + "loss": 2.6753859856398776e-05, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00003, + "step": 318, + "tokens/total": 10159184, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 145339 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.0008433983894065022, + "learning_rate": 4.087355496656321e-05, + "loss": 1.602486736373976e-05, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00002, + "step": 319, + "tokens/total": 10191216, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 145811 + }, + { + "epoch": 1.25, + "grad_norm": 0.12588869035243988, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0011584729654714465, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00116, + "step": 320, + "tokens/total": 10223408, + "tokens/train_per_sec_per_gpu": 32.27, + "tokens/trainable": 146314 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.028395529836416245, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.00015757072833366692, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00016, + "step": 321, + "tokens/total": 10255536, + "tokens/train_per_sec_per_gpu": 26.68, + "tokens/trainable": 146781 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.039599668234586716, + "learning_rate": 4.004940255778431e-05, + "loss": 0.000321136845741421, + "memory/device_reserved (GiB)": 37.39, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00032, + "step": 322, + "tokens/total": 10287664, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 147225 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.11097095906734467, + "learning_rate": 3.977591418134619e-05, + "loss": 0.0004549310833681375, + "memory/device_reserved (GiB)": 37.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00046, + "step": 323, + "tokens/total": 10320080, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 147673 + }, + { + "epoch": 1.265625, + "grad_norm": 0.19018623232841492, + "learning_rate": 3.95030593412612e-05, + "loss": 0.0028336881659924984, + "memory/device_reserved (GiB)": 38.53, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00284, + "step": 324, + "tokens/total": 10351984, + "tokens/train_per_sec_per_gpu": 27.33, + "tokens/trainable": 148103 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.4650349020957947, + "learning_rate": 3.923084939213296e-05, + "loss": 0.006978687364608049, + "memory/device_reserved (GiB)": 38.53, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.007, + "step": 325, + "tokens/total": 10384256, + "tokens/train_per_sec_per_gpu": 26.21, + "tokens/trainable": 148535 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.14811205863952637, + "learning_rate": 3.895929566172861e-05, + "loss": 0.0014966176822781563, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0015, + "step": 326, + "tokens/total": 10416208, + "tokens/train_per_sec_per_gpu": 28.37, + "tokens/trainable": 148999 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.11302439123392105, + "learning_rate": 3.868840945050728e-05, + "loss": 0.0011318891774863005, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00113, + "step": 327, + "tokens/total": 10448224, + "tokens/train_per_sec_per_gpu": 26.37, + "tokens/trainable": 149431 + }, + { + "epoch": 1.28125, + "grad_norm": 0.10916705429553986, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0009605163359083235, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00096, + "step": 328, + "tokens/total": 10480448, + "tokens/train_per_sec_per_gpu": 28.27, + "tokens/trainable": 149886 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.0004018457839265466, + "learning_rate": 3.814868464809027e-05, + "loss": 7.040077434794512e-06, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00001, + "step": 329, + "tokens/total": 10512096, + "tokens/train_per_sec_per_gpu": 25.92, + "tokens/trainable": 150288 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.03717898577451706, + "learning_rate": 3.787986851704667e-05, + "loss": 0.0001680658315308392, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00017, + "step": 330, + "tokens/total": 10544000, + "tokens/train_per_sec_per_gpu": 28.52, + "tokens/trainable": 150733 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.21975846588611603, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00015534088015556335, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00016, + "step": 331, + "tokens/total": 10576048, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 151207 + }, + { + "epoch": 1.296875, + "grad_norm": 0.10715529322624207, + "learning_rate": 3.734438472750619e-05, + "loss": 0.0012046336196362972, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00121, + "step": 332, + "tokens/total": 10608048, + "tokens/train_per_sec_per_gpu": 31.08, + "tokens/trainable": 151694 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.13602837920188904, + "learning_rate": 3.707773935267552e-05, + "loss": 0.00039335948531515896, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00039, + "step": 333, + "tokens/total": 10640176, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 152188 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.002963080769404769, + "learning_rate": 3.68118397962661e-05, + "loss": 6.6289721871726215e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00001, + "step": 334, + "tokens/total": 10672224, + "tokens/train_per_sec_per_gpu": 25.55, + "tokens/trainable": 152596 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.007236327510327101, + "learning_rate": 3.654669712344384e-05, + "loss": 5.341085125110112e-05, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00005, + "step": 335, + "tokens/total": 10704192, + "tokens/train_per_sec_per_gpu": 30.44, + "tokens/trainable": 153052 + }, + { + "epoch": 1.3125, + "grad_norm": 0.0897810310125351, + "learning_rate": 3.628232236787763e-05, + "loss": 0.0008687702938914299, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00087, + "step": 336, + "tokens/total": 10736272, + "tokens/train_per_sec_per_gpu": 25.33, + "tokens/trainable": 153491 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.011829929426312447, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.00011201453162357211, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00011, + "step": 337, + "tokens/total": 10768528, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 153983 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.25218063592910767, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0022679665125906467, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00227, + "step": 338, + "tokens/total": 10800464, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 154464 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.00034703267738223076, + "learning_rate": 3.549391545931585e-05, + "loss": 2.884747800635523e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.0, + "step": 339, + "tokens/total": 10832512, + "tokens/train_per_sec_per_gpu": 29.7, + "tokens/trainable": 154924 + }, + { + "epoch": 1.328125, + "grad_norm": 0.004890776239335537, + "learning_rate": 3.5232722063479914e-05, + "loss": 2.2697908207192086e-05, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00002, + "step": 340, + "tokens/total": 10864624, + "tokens/train_per_sec_per_gpu": 25.46, + "tokens/trainable": 155373 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.0013245025184005499, + "learning_rate": 3.49723512647657e-05, + "loss": 8.980642633105163e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00001, + "step": 341, + "tokens/total": 10896816, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 155873 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.03804129734635353, + "learning_rate": 3.471281389826491e-05, + "loss": 0.00021880728309042752, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00022, + "step": 342, + "tokens/total": 10928864, + "tokens/train_per_sec_per_gpu": 25.31, + "tokens/trainable": 156278 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.4880214333534241, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.005397590808570385, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00541, + "step": 343, + "tokens/total": 10960816, + "tokens/train_per_sec_per_gpu": 29.58, + "tokens/trainable": 156733 + }, + { + "epoch": 1.34375, + "grad_norm": 0.37780871987342834, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0029298821464180946, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00293, + "step": 344, + "tokens/total": 10992912, + "tokens/train_per_sec_per_gpu": 28.4, + "tokens/trainable": 157203 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.0034261250402778387, + "learning_rate": 3.3939310220027456e-05, + "loss": 2.1256186300888658e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00002, + "step": 345, + "tokens/total": 11024832, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 157707 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.0005507867899723351, + "learning_rate": 3.3683214232914404e-05, + "loss": 6.991718692006543e-06, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00001, + "step": 346, + "tokens/total": 11056992, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 158180 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.0016383324982598424, + "learning_rate": 3.342800532426873e-05, + "loss": 1.2630136552616023e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00001, + "step": 347, + "tokens/total": 11089264, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 158646 + }, + { + "epoch": 1.359375, + "grad_norm": 0.0020912184845656157, + "learning_rate": 3.317369411437484e-05, + "loss": 1.536883064545691e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00002, + "step": 348, + "tokens/total": 11120976, + "tokens/train_per_sec_per_gpu": 29.34, + "tokens/trainable": 159115 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.0016828920925036073, + "learning_rate": 3.292029118616024e-05, + "loss": 1.334734679403482e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00001, + "step": 349, + "tokens/total": 11152976, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 159538 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.02520265430212021, + "learning_rate": 3.266780708475511e-05, + "loss": 9.595962183084339e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0001, + "step": 350, + "tokens/total": 11185056, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 160016 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.009836713783442974, + "learning_rate": 3.241625231705354e-05, + "loss": 7.521865336457267e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00008, + "step": 351, + "tokens/total": 11217088, + "tokens/train_per_sec_per_gpu": 28.51, + "tokens/trainable": 160475 + }, + { + "epoch": 1.375, + "grad_norm": 0.0698060467839241, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0003666624252218753, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00037, + "step": 352, + "tokens/total": 11249088, + "tokens/train_per_sec_per_gpu": 28.31, + "tokens/trainable": 160952 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.0018428800394758582, + "learning_rate": 3.191597261653475e-05, + "loss": 1.3194729945098516e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00001, + "step": 353, + "tokens/total": 11281328, + "tokens/train_per_sec_per_gpu": 27.4, + "tokens/trainable": 161420 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.020005319267511368, + "learning_rate": 3.166726850239794e-05, + "loss": 0.00017299644241575152, + "memory/device_reserved (GiB)": 38.47, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00017, + "step": 354, + "tokens/total": 11313200, + "tokens/train_per_sec_per_gpu": 28.31, + "tokens/trainable": 161868 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.00018166302470490336, + "learning_rate": 3.141953535845912e-05, + "loss": 4.138089934713207e-06, + "memory/device_reserved (GiB)": 38.47, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0, + "step": 355, + "tokens/total": 11345008, + "tokens/train_per_sec_per_gpu": 29.74, + "tokens/trainable": 162318 + }, + { + "epoch": 1.390625, + "grad_norm": 0.009165152907371521, + "learning_rate": 3.11727834939056e-05, + "loss": 6.729022425133735e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00007, + "step": 356, + "tokens/total": 11377216, + "tokens/train_per_sec_per_gpu": 26.26, + "tokens/trainable": 162789 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.2891274690628052, + "learning_rate": 3.092702317708967e-05, + "loss": 0.005585570354014635, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.0056, + "step": 357, + "tokens/total": 11409536, + "tokens/train_per_sec_per_gpu": 30.34, + "tokens/trainable": 163254 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.0007154783816076815, + "learning_rate": 3.0682264635101276e-05, + "loss": 5.591478839050978e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00001, + "step": 358, + "tokens/total": 11441552, + "tokens/train_per_sec_per_gpu": 28.74, + "tokens/trainable": 163715 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.01731656678020954, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.0001050045175361447, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00011, + "step": 359, + "tokens/total": 11473376, + "tokens/train_per_sec_per_gpu": 28.79, + "tokens/trainable": 164197 + }, + { + "epoch": 1.40625, + "grad_norm": 0.11494546383619308, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.00048459687968716025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00048, + "step": 360, + "tokens/total": 11505312, + "tokens/train_per_sec_per_gpu": 26.54, + "tokens/trainable": 164661 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.08215854316949844, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.0002725277154240757, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00027, + "step": 361, + "tokens/total": 11537536, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 165111 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.024266909807920456, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.0001677784021012485, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 35.94, + "memory/max_allocated (GiB)": 35.94, + "ppl": 1.00017, + "step": 362, + "tokens/total": 11567216, + "tokens/train_per_sec_per_gpu": 25.3, + "tokens/trainable": 165552 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.008205167017877102, + "learning_rate": 2.9473853553877484e-05, + "loss": 3.423610905883834e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00003, + "step": 363, + "tokens/total": 11599312, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 166026 + }, + { + "epoch": 1.421875, + "grad_norm": 0.4230109453201294, + "learning_rate": 2.9235318065647e-05, + "loss": 0.005893740337342024, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00591, + "step": 364, + "tokens/total": 11631344, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 166486 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.1095249131321907, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.0006186347454786301, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00062, + "step": 365, + "tokens/total": 11663632, + "tokens/train_per_sec_per_gpu": 30.17, + "tokens/trainable": 166959 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.0014034683117642999, + "learning_rate": 2.8761473491751258e-05, + "loss": 1.2728614819934592e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.27, + "memory/max_allocated (GiB)": 36.27, + "ppl": 1.00001, + "step": 366, + "tokens/total": 11695360, + "tokens/train_per_sec_per_gpu": 28.19, + "tokens/trainable": 167394 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.044087354093790054, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.0001466910180170089, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00015, + "step": 367, + "tokens/total": 11727184, + "tokens/train_per_sec_per_gpu": 23.64, + "tokens/trainable": 167832 + }, + { + "epoch": 1.4375, + "grad_norm": 0.0005008528823964298, + "learning_rate": 2.829199644117484e-05, + "loss": 1.1319095392536838e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00001, + "step": 368, + "tokens/total": 11759312, + "tokens/train_per_sec_per_gpu": 25.79, + "tokens/trainable": 168242 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.02015153504908085, + "learning_rate": 2.8058920186702553e-05, + "loss": 0.00011158352572238073, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00011, + "step": 369, + "tokens/total": 11791536, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 168715 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.00031807494815438986, + "learning_rate": 2.782696506053033e-05, + "loss": 6.533119631058071e-06, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00001, + "step": 370, + "tokens/total": 11823632, + "tokens/train_per_sec_per_gpu": 26.68, + "tokens/trainable": 169147 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.4510185122489929, + "learning_rate": 2.7596140715257824e-05, + "loss": 0.004431337118148804, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00444, + "step": 371, + "tokens/total": 11855392, + "tokens/train_per_sec_per_gpu": 28.61, + "tokens/trainable": 169628 + }, + { + "epoch": 1.453125, + "grad_norm": 0.0034806979820132256, + "learning_rate": 2.7366456756428184e-05, + "loss": 2.722550561884418e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00003, + "step": 372, + "tokens/total": 11887424, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 170085 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.018526704981923103, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.00013849925016984344, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00014, + "step": 373, + "tokens/total": 11919280, + "tokens/train_per_sec_per_gpu": 29.39, + "tokens/trainable": 170584 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.010734346695244312, + "learning_rate": 2.691054818259188e-05, + "loss": 1.963317481568083e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00002, + "step": 374, + "tokens/total": 11951488, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 171058 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.004906357266008854, + "learning_rate": 2.6684342539801933e-05, + "loss": 4.44888137280941e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00004, + "step": 375, + "tokens/total": 11983504, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 171518 + }, + { + "epoch": 1.46875, + "grad_norm": 0.0035627244506031275, + "learning_rate": 2.645931522709877e-05, + "loss": 1.879796946013812e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00002, + "step": 376, + "tokens/total": 12015360, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 172007 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.014949376694858074, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.0001050513528753072, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00011, + "step": 377, + "tokens/total": 12047296, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 172468 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.024432353675365448, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.00010428359382785857, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0001, + "step": 378, + "tokens/total": 12079392, + "tokens/train_per_sec_per_gpu": 27.91, + "tokens/trainable": 172923 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.5905564427375793, + "learning_rate": 2.579139666504821e-05, + "loss": 0.019045401364564896, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01923, + "step": 379, + "tokens/total": 12111616, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 173370 + }, + { + "epoch": 1.484375, + "grad_norm": 0.0027490397915244102, + "learning_rate": 2.557117581955798e-05, + "loss": 3.106705116806552e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00003, + "step": 380, + "tokens/total": 12143536, + "tokens/train_per_sec_per_gpu": 28.66, + "tokens/trainable": 173837 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.10220319032669067, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.0007754361140541732, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00078, + "step": 381, + "tokens/total": 12175792, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 174294 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.06388501822948456, + "learning_rate": 2.5134417202396277e-05, + "loss": 2.506363671272993e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00003, + "step": 382, + "tokens/total": 12207472, + "tokens/train_per_sec_per_gpu": 25.85, + "tokens/trainable": 174701 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.006029566749930382, + "learning_rate": 2.491789760603361e-05, + "loss": 5.214785414864309e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00005, + "step": 383, + "tokens/total": 12239520, + "tokens/train_per_sec_per_gpu": 26.67, + "tokens/trainable": 175131 + }, + { + "epoch": 1.5, + "grad_norm": 0.015079713426530361, + "learning_rate": 2.4702629848741764e-05, + "loss": 0.00011988497863058001, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00012, + "step": 384, + "tokens/total": 12271600, + "tokens/train_per_sec_per_gpu": 28.23, + "tokens/trainable": 175586 + }, + { + "epoch": 1.50390625, + "grad_norm": 0.04362509027123451, + "learning_rate": 2.4488622888690785e-05, + "loss": 0.0002582473389338702, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00026, + "step": 385, + "tokens/total": 12303648, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 176026 + }, + { + "epoch": 1.5078125, + "grad_norm": 0.28308039903640747, + "learning_rate": 2.427588563158384e-05, + "loss": 0.0025924108922481537, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0026, + "step": 386, + "tokens/total": 12335728, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 176512 + }, + { + "epoch": 1.51171875, + "grad_norm": 0.21685272455215454, + "learning_rate": 2.406442693028651e-05, + "loss": 0.002609315561130643, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00261, + "step": 387, + "tokens/total": 12367824, + "tokens/train_per_sec_per_gpu": 25.55, + "tokens/trainable": 176943 + }, + { + "epoch": 1.515625, + "grad_norm": 0.0010271539213135839, + "learning_rate": 2.3854255584458547e-05, + "loss": 2.1533389372052625e-05, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00002, + "step": 388, + "tokens/total": 12400192, + "tokens/train_per_sec_per_gpu": 26.36, + "tokens/trainable": 177370 + }, + { + "epoch": 1.51953125, + "grad_norm": 0.06616228073835373, + "learning_rate": 2.3645380340187508e-05, + "loss": 0.0005350579158402979, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00054, + "step": 389, + "tokens/total": 12432224, + "tokens/train_per_sec_per_gpu": 29.11, + "tokens/trainable": 177851 + }, + { + "epoch": 1.5234375, + "grad_norm": 0.05946248397231102, + "learning_rate": 2.3437809889624914e-05, + "loss": 0.0005003588157705963, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.0005, + "step": 390, + "tokens/total": 12464048, + "tokens/train_per_sec_per_gpu": 29.09, + "tokens/trainable": 178310 + }, + { + "epoch": 1.52734375, + "grad_norm": 0.017372427508234978, + "learning_rate": 2.3231552870624487e-05, + "loss": 0.0001355188578600064, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00014, + "step": 391, + "tokens/total": 12493776, + "tokens/train_per_sec_per_gpu": 30.96, + "tokens/trainable": 178736 + }, + { + "epoch": 1.53125, + "grad_norm": 0.12941808998584747, + "learning_rate": 2.3026617866382657e-05, + "loss": 0.003045230871066451, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00305, + "step": 392, + "tokens/total": 12526000, + "tokens/train_per_sec_per_gpu": 30.37, + "tokens/trainable": 179233 + }, + { + "epoch": 1.53515625, + "grad_norm": 0.014790852554142475, + "learning_rate": 2.2823013405081507e-05, + "loss": 0.00013205324648879468, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00013, + "step": 393, + "tokens/total": 12557968, + "tokens/train_per_sec_per_gpu": 32.27, + "tokens/trainable": 179730 + }, + { + "epoch": 1.5390625, + "grad_norm": 0.02042844519019127, + "learning_rate": 2.2620747959533722e-05, + "loss": 0.00022655579959973693, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00023, + "step": 394, + "tokens/total": 12590096, + "tokens/train_per_sec_per_gpu": 31.47, + "tokens/trainable": 180227 + }, + { + "epoch": 1.54296875, + "grad_norm": 0.005440943408757448, + "learning_rate": 2.2419829946830123e-05, + "loss": 4.6342807763721794e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00005, + "step": 395, + "tokens/total": 12622160, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 180687 + }, + { + "epoch": 1.546875, + "grad_norm": 0.004878366366028786, + "learning_rate": 2.2220267727989325e-05, + "loss": 5.4212974646361545e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00005, + "step": 396, + "tokens/total": 12654336, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 181141 + }, + { + "epoch": 1.55078125, + "grad_norm": 0.20873922109603882, + "learning_rate": 2.202206960760984e-05, + "loss": 0.0026359721086919308, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00264, + "step": 397, + "tokens/total": 12686400, + "tokens/train_per_sec_per_gpu": 30.13, + "tokens/trainable": 181648 + }, + { + "epoch": 1.5546875, + "grad_norm": 0.12369472533464432, + "learning_rate": 2.182524383352446e-05, + "loss": 0.0010878165485337377, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.00109, + "step": 398, + "tokens/total": 12718144, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 182109 + }, + { + "epoch": 1.55859375, + "grad_norm": 0.0010841770563274622, + "learning_rate": 2.1629798596457056e-05, + "loss": 1.8380069377599284e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00002, + "step": 399, + "tokens/total": 12750256, + "tokens/train_per_sec_per_gpu": 26.26, + "tokens/trainable": 182559 + }, + { + "epoch": 1.5625, + "grad_norm": 0.0010006122756749392, + "learning_rate": 2.1435742029681725e-05, + "loss": 1.991226599784568e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00002, + "step": 400, + "tokens/total": 12782224, + "tokens/train_per_sec_per_gpu": 27.59, + "tokens/trainable": 182985 + }, + { + "epoch": 1.56640625, + "grad_norm": 0.0011391225270926952, + "learning_rate": 2.124308220868431e-05, + "loss": 2.2474639990832657e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00002, + "step": 401, + "tokens/total": 12814384, + "tokens/train_per_sec_per_gpu": 26.95, + "tokens/trainable": 183440 + }, + { + "epoch": 1.5703125, + "grad_norm": 0.023268507793545723, + "learning_rate": 2.105182715082638e-05, + "loss": 0.00032951770117506385, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00033, + "step": 402, + "tokens/total": 12846240, + "tokens/train_per_sec_per_gpu": 28.77, + "tokens/trainable": 183917 + }, + { + "epoch": 1.57421875, + "grad_norm": 0.11701280623674393, + "learning_rate": 2.0861984815011552e-05, + "loss": 0.001444364432245493, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00145, + "step": 403, + "tokens/total": 12878240, + "tokens/train_per_sec_per_gpu": 25.07, + "tokens/trainable": 184348 + }, + { + "epoch": 1.578125, + "grad_norm": 0.004240577574819326, + "learning_rate": 2.0673563101354323e-05, + "loss": 2.100174970109947e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00002, + "step": 404, + "tokens/total": 12910176, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 184786 + }, + { + "epoch": 1.58203125, + "grad_norm": 0.01977616548538208, + "learning_rate": 2.0486569850851317e-05, + "loss": 8.987118053482845e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00009, + "step": 405, + "tokens/total": 12942352, + "tokens/train_per_sec_per_gpu": 28.17, + "tokens/trainable": 185249 + }, + { + "epoch": 1.5859375, + "grad_norm": 0.030095387250185013, + "learning_rate": 2.0301012845054956e-05, + "loss": 0.0002793738676700741, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00028, + "step": 406, + "tokens/total": 12974368, + "tokens/train_per_sec_per_gpu": 26.3, + "tokens/trainable": 185680 + }, + { + "epoch": 1.58984375, + "grad_norm": 0.0017381443176418543, + "learning_rate": 2.011689980574966e-05, + "loss": 2.727654828049708e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00003, + "step": 407, + "tokens/total": 13006464, + "tokens/train_per_sec_per_gpu": 26.25, + "tokens/trainable": 186100 + }, + { + "epoch": 1.59375, + "grad_norm": 0.09072195738554001, + "learning_rate": 1.993423839463052e-05, + "loss": 0.0006338813109323382, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00063, + "step": 408, + "tokens/total": 13038320, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 186565 + }, + { + "epoch": 1.59765625, + "grad_norm": 0.004482298158109188, + "learning_rate": 1.975303621298445e-05, + "loss": 5.0280330469831824e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00005, + "step": 409, + "tokens/total": 13070416, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 187017 + }, + { + "epoch": 1.6015625, + "grad_norm": 0.03788210079073906, + "learning_rate": 1.957330080137385e-05, + "loss": 0.000300457701086998, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0003, + "step": 410, + "tokens/total": 13102352, + "tokens/train_per_sec_per_gpu": 27.69, + "tokens/trainable": 187468 + }, + { + "epoch": 1.60546875, + "grad_norm": 0.011773839592933655, + "learning_rate": 1.9395039639322864e-05, + "loss": 9.259363287128508e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00009, + "step": 411, + "tokens/total": 13134096, + "tokens/train_per_sec_per_gpu": 31.06, + "tokens/trainable": 187961 + }, + { + "epoch": 1.609375, + "grad_norm": 0.0031615474727004766, + "learning_rate": 1.9218260145006073e-05, + "loss": 3.231812661397271e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00003, + "step": 412, + "tokens/total": 13166256, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 188447 + }, + { + "epoch": 1.61328125, + "grad_norm": 0.07778825610876083, + "learning_rate": 1.904296967493982e-05, + "loss": 0.0005980221321806312, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.0006, + "step": 413, + "tokens/total": 13198048, + "tokens/train_per_sec_per_gpu": 29.98, + "tokens/trainable": 188903 + }, + { + "epoch": 1.6171875, + "grad_norm": 0.011902395635843277, + "learning_rate": 1.8869175523676064e-05, + "loss": 0.00013689698243979365, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.66, + "memory/max_allocated (GiB)": 36.66, + "ppl": 1.00014, + "step": 414, + "tokens/total": 13230304, + "tokens/train_per_sec_per_gpu": 30.8, + "tokens/trainable": 189398 + }, + { + "epoch": 1.62109375, + "grad_norm": 0.08490622788667679, + "learning_rate": 1.869688492349885e-05, + "loss": 0.0008053273777477443, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00081, + "step": 415, + "tokens/total": 13262288, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 189850 + }, + { + "epoch": 1.625, + "grad_norm": 0.11783187091350555, + "learning_rate": 1.85261050441233e-05, + "loss": 0.0007362872711382806, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00074, + "step": 416, + "tokens/total": 13294480, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 190306 + }, + { + "epoch": 1.62890625, + "grad_norm": 0.26603376865386963, + "learning_rate": 1.8356842992397304e-05, + "loss": 0.002245941199362278, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00225, + "step": 417, + "tokens/total": 13326608, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 190746 + }, + { + "epoch": 1.6328125, + "grad_norm": 0.003691659774631262, + "learning_rate": 1.8189105812005714e-05, + "loss": 4.167412407696247e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00004, + "step": 418, + "tokens/total": 13358592, + "tokens/train_per_sec_per_gpu": 26.1, + "tokens/trainable": 191156 + }, + { + "epoch": 1.63671875, + "grad_norm": 0.021083569154143333, + "learning_rate": 1.802290048317732e-05, + "loss": 0.00017740413022693247, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00018, + "step": 419, + "tokens/total": 13390592, + "tokens/train_per_sec_per_gpu": 24.49, + "tokens/trainable": 191573 + }, + { + "epoch": 1.640625, + "grad_norm": 0.003719380358234048, + "learning_rate": 1.785823392239424e-05, + "loss": 4.420262484927662e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00004, + "step": 420, + "tokens/total": 13422672, + "tokens/train_per_sec_per_gpu": 27.22, + "tokens/trainable": 192014 + }, + { + "epoch": 1.64453125, + "grad_norm": 0.044042862951755524, + "learning_rate": 1.7695112982104225e-05, + "loss": 0.0004268632619641721, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00043, + "step": 421, + "tokens/total": 13454624, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 192453 + }, + { + "epoch": 1.6484375, + "grad_norm": 0.0021698337513953447, + "learning_rate": 1.7533544450435433e-05, + "loss": 2.580507134553045e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.31, + "memory/max_allocated (GiB)": 36.31, + "ppl": 1.00003, + "step": 422, + "tokens/total": 13486208, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 192886 + }, + { + "epoch": 1.65234375, + "grad_norm": 0.019405441358685493, + "learning_rate": 1.7373535050913946e-05, + "loss": 0.00013853301061317325, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.71, + "memory/max_allocated (GiB)": 36.71, + "ppl": 1.00014, + "step": 423, + "tokens/total": 13518496, + "tokens/train_per_sec_per_gpu": 25.15, + "tokens/trainable": 193323 + }, + { + "epoch": 1.65625, + "grad_norm": 0.005207414738833904, + "learning_rate": 1.721509144218405e-05, + "loss": 6.985871004872024e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00007, + "step": 424, + "tokens/total": 13550768, + "tokens/train_per_sec_per_gpu": 27.59, + "tokens/trainable": 193753 + }, + { + "epoch": 1.66015625, + "grad_norm": 0.01700798235833645, + "learning_rate": 1.705822021773101e-05, + "loss": 0.00024273117014672607, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00024, + "step": 425, + "tokens/total": 13582960, + "tokens/train_per_sec_per_gpu": 30.03, + "tokens/trainable": 194240 + }, + { + "epoch": 1.6640625, + "grad_norm": 0.2864547073841095, + "learning_rate": 1.69029279056068e-05, + "loss": 0.004108238499611616, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00412, + "step": 426, + "tokens/total": 13615056, + "tokens/train_per_sec_per_gpu": 29.6, + "tokens/trainable": 194680 + }, + { + "epoch": 1.66796875, + "grad_norm": 0.012734112329781055, + "learning_rate": 1.6749220968158415e-05, + "loss": 9.277237404603511e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00009, + "step": 427, + "tokens/total": 13647088, + "tokens/train_per_sec_per_gpu": 32.17, + "tokens/trainable": 195186 + }, + { + "epoch": 1.671875, + "grad_norm": 0.04110024869441986, + "learning_rate": 1.659710580175893e-05, + "loss": 0.00021035004465375096, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00021, + "step": 428, + "tokens/total": 13679168, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 195693 + }, + { + "epoch": 1.67578125, + "grad_norm": 0.03119366057217121, + "learning_rate": 1.644658873654133e-05, + "loss": 0.00022089436242822558, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00022, + "step": 429, + "tokens/total": 13711120, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 196130 + }, + { + "epoch": 1.6796875, + "grad_norm": 0.0020074129570275545, + "learning_rate": 1.629767603613508e-05, + "loss": 1.550407614558935e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00002, + "step": 430, + "tokens/total": 13743088, + "tokens/train_per_sec_per_gpu": 24.67, + "tokens/trainable": 196530 + }, + { + "epoch": 1.68359375, + "grad_norm": 0.040013547986745834, + "learning_rate": 1.615037389740547e-05, + "loss": 0.0006465734331868589, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00065, + "step": 431, + "tokens/total": 13774912, + "tokens/train_per_sec_per_gpu": 25.06, + "tokens/trainable": 196974 + }, + { + "epoch": 1.6875, + "grad_norm": 0.0517231747508049, + "learning_rate": 1.600468845019576e-05, + "loss": 0.00047026947140693665, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00047, + "step": 432, + "tokens/total": 13806800, + "tokens/train_per_sec_per_gpu": 29.38, + "tokens/trainable": 197454 + }, + { + "epoch": 1.69140625, + "grad_norm": 0.0004246715398039669, + "learning_rate": 1.5860625757072092e-05, + "loss": 7.613366506120656e-06, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00001, + "step": 433, + "tokens/total": 13838704, + "tokens/train_per_sec_per_gpu": 26.36, + "tokens/trainable": 197902 + }, + { + "epoch": 1.6953125, + "grad_norm": 0.0011862553656101227, + "learning_rate": 1.571819181307116e-05, + "loss": 1.5171106497291476e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00002, + "step": 434, + "tokens/total": 13870640, + "tokens/train_per_sec_per_gpu": 29.63, + "tokens/trainable": 198374 + }, + { + "epoch": 1.69921875, + "grad_norm": 0.0019507558317855, + "learning_rate": 1.557739254545075e-05, + "loss": 1.7753134670783766e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00002, + "step": 435, + "tokens/total": 13902672, + "tokens/train_per_sec_per_gpu": 31.27, + "tokens/trainable": 198846 + }, + { + "epoch": 1.703125, + "grad_norm": 0.014270462095737457, + "learning_rate": 1.543823381344311e-05, + "loss": 0.0001223236322402954, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00012, + "step": 436, + "tokens/total": 13935008, + "tokens/train_per_sec_per_gpu": 31.41, + "tokens/trainable": 199320 + }, + { + "epoch": 1.70703125, + "grad_norm": 0.020011236891150475, + "learning_rate": 1.5300721408011114e-05, + "loss": 0.00015811943740118295, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00016, + "step": 437, + "tokens/total": 13967024, + "tokens/train_per_sec_per_gpu": 31.19, + "tokens/trainable": 199777 + }, + { + "epoch": 1.7109375, + "grad_norm": 0.0006321229157038033, + "learning_rate": 1.5164861051607254e-05, + "loss": 9.932198736350983e-06, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00001, + "step": 438, + "tokens/total": 13998688, + "tokens/train_per_sec_per_gpu": 25.77, + "tokens/trainable": 200194 + }, + { + "epoch": 1.71484375, + "grad_norm": 0.002675980795174837, + "learning_rate": 1.5030658397935521e-05, + "loss": 2.0016837879666127e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00002, + "step": 439, + "tokens/total": 14030992, + "tokens/train_per_sec_per_gpu": 28.45, + "tokens/trainable": 200663 + }, + { + "epoch": 1.71875, + "grad_norm": 0.005228503607213497, + "learning_rate": 1.4898119031716104e-05, + "loss": 4.561378591461107e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00005, + "step": 440, + "tokens/total": 14063440, + "tokens/train_per_sec_per_gpu": 28.52, + "tokens/trainable": 201121 + }, + { + "epoch": 1.72265625, + "grad_norm": 0.015322903171181679, + "learning_rate": 1.476724846845306e-05, + "loss": 0.0001199191392515786, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00012, + "step": 441, + "tokens/total": 14095536, + "tokens/train_per_sec_per_gpu": 33.0, + "tokens/trainable": 201598 + }, + { + "epoch": 1.7265625, + "grad_norm": 0.05436247959733009, + "learning_rate": 1.463805215420471e-05, + "loss": 0.0005605737096630037, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00056, + "step": 442, + "tokens/total": 14127488, + "tokens/train_per_sec_per_gpu": 29.37, + "tokens/trainable": 202068 + }, + { + "epoch": 1.73046875, + "grad_norm": 0.005114950239658356, + "learning_rate": 1.451053546535705e-05, + "loss": 2.6398556656204164e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00003, + "step": 443, + "tokens/total": 14159472, + "tokens/train_per_sec_per_gpu": 29.43, + "tokens/trainable": 202524 + }, + { + "epoch": 1.734375, + "grad_norm": 0.000681015953887254, + "learning_rate": 1.438470370840001e-05, + "loss": 6.228779966477305e-06, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00001, + "step": 444, + "tokens/total": 14191520, + "tokens/train_per_sec_per_gpu": 30.27, + "tokens/trainable": 202996 + }, + { + "epoch": 1.73828125, + "grad_norm": 0.03594636172056198, + "learning_rate": 1.4260562119706606e-05, + "loss": 0.0003221426741220057, + "memory/device_reserved (GiB)": 39.16, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00032, + "step": 445, + "tokens/total": 14221680, + "tokens/train_per_sec_per_gpu": 32.89, + "tokens/trainable": 203470 + }, + { + "epoch": 1.7421875, + "grad_norm": 0.16816683113574982, + "learning_rate": 1.413811586531508e-05, + "loss": 0.0016490641282871366, + "memory/device_reserved (GiB)": 39.16, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00165, + "step": 446, + "tokens/total": 14253680, + "tokens/train_per_sec_per_gpu": 29.37, + "tokens/trainable": 203919 + }, + { + "epoch": 1.74609375, + "grad_norm": 0.0071723381988704205, + "learning_rate": 1.4017370040713884e-05, + "loss": 6.93315378157422e-05, + "memory/device_reserved (GiB)": 39.16, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00007, + "step": 447, + "tokens/total": 14285824, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 204376 + }, + { + "epoch": 1.75, + "grad_norm": 0.02812999114394188, + "learning_rate": 1.3898329670629645e-05, + "loss": 0.0001953901955857873, + "memory/device_reserved (GiB)": 39.16, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.0002, + "step": 448, + "tokens/total": 14317968, + "tokens/train_per_sec_per_gpu": 26.64, + "tokens/trainable": 204821 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 9.718441586995922e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/training_args.bin b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..dec74f9f304b44cbe47132d86c4e6285e659f40b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-448/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634 +size 8273 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/README.md b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ba677470ac9c29f9770a4aacb65c644ebd671a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/deconf_wave_deconf_control/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/deconf_wave_deconf_control/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/adapter_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..21fcb5b00be777c20a2d2dcceffc539dc51679b4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/deconf_wave_deconf_control/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "gate_proj", + "v_proj", + "k_proj", + "up_proj", + "q_proj", + "o_proj", + "down_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/adapter_model.safetensors b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..f2846837d189cfe009e42c94bf5a645e5c9809a5 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:80fa45f2312362444efbfe66a98b1a99e7297e5e6632daee7de0dd4b2d65aa79 +size 547777976 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/chat_template.jinja b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/optimizer.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..cedcba527ed50e646fa112e8f305c2925f69aca7 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4831686a1617a595f1da6b4673e86d834057bff05132ff21d54565ca6aa99ca1 +size 1048106435 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/rng_state.pth b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..6eea6132d2a10abc328da2d04edb8f4ff77f15ae --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:922c150f699a9e55675384172d3e3fade02afcaddf7ce293fab68f44630bc453 +size 14645 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/scheduler.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..ce1c4e63a6c6f8be8c137d3add4eda184d59a043 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1b68903e42777cffa699a4ceb5e57a17a53fd50d8cdf046737093de36719dc01 +size 1465 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/tokenizer.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/tokenizer_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/tokens_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..bd6e35173331e52bcce9178dbd09f23a39777c3f --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/tokens_state.json @@ -0,0 +1 @@ +{"total": 15343360, "trainable": 219412} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/trainer_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..1afc1364f803d2184ead3c9d6b5635747f5782b8 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/trainer_state.json @@ -0,0 +1,6754 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.875, + "eval_steps": 500, + "global_step": 480, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.2140090465545654, + "learning_rate": 0.0, + "loss": 0.16441112756729126, + "memory/device_reserved (GiB)": 38.08, + "memory/max_active (GiB)": 35.64, + "memory/max_allocated (GiB)": 35.64, + "ppl": 1.1787, + "step": 1, + "tokens/total": 32160, + "tokens/train_per_sec_per_gpu": 21.28, + "tokens/trainable": 470 + }, + { + "epoch": 0.0078125, + "grad_norm": 2.92620849609375, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15774038434028625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.17086, + "step": 2, + "tokens/total": 64192, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 922 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.135961890220642, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1664382815361023, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.18109, + "step": 3, + "tokens/total": 96224, + "tokens/train_per_sec_per_gpu": 28.54, + "tokens/trainable": 1396 + }, + { + "epoch": 0.015625, + "grad_norm": 0.880176305770874, + "learning_rate": 1.2e-05, + "loss": 0.15334013104438782, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.16572, + "step": 4, + "tokens/total": 128336, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 1850 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.1433289051055908, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.16352537274360657, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.17766, + "step": 5, + "tokens/total": 160784, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 2302 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9728212952613831, + "learning_rate": 2e-05, + "loss": 0.15144473314285278, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.16351, + "step": 6, + "tokens/total": 192624, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 2742 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.7796363830566406, + "learning_rate": 2.4e-05, + "loss": 0.13553467392921448, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.14515, + "step": 7, + "tokens/total": 224608, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 3208 + }, + { + "epoch": 0.03125, + "grad_norm": 1.2444697618484497, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.1077304258942604, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.11375, + "step": 8, + "tokens/total": 256592, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 3655 + }, + { + "epoch": 0.03515625, + "grad_norm": 2.7339375019073486, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.08772765100002289, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.09169, + "step": 9, + "tokens/total": 286400, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 4091 + }, + { + "epoch": 0.0390625, + "grad_norm": 4.109770774841309, + "learning_rate": 3.6e-05, + "loss": 0.130940243601799, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.1399, + "step": 10, + "tokens/total": 318352, + "tokens/train_per_sec_per_gpu": 29.13, + "tokens/trainable": 4553 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.3768608570098877, + "learning_rate": 4e-05, + "loss": 0.10059958696365356, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.10583, + "step": 11, + "tokens/total": 350336, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 4992 + }, + { + "epoch": 0.046875, + "grad_norm": 3.281402826309204, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.12574875354766846, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.134, + "step": 12, + "tokens/total": 382448, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 5494 + }, + { + "epoch": 0.05078125, + "grad_norm": 1.7738980054855347, + "learning_rate": 4.8e-05, + "loss": 0.033296722918748856, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03386, + "step": 13, + "tokens/total": 414352, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 5933 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.64290189743042, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06366822123527527, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.06574, + "step": 14, + "tokens/total": 446400, + "tokens/train_per_sec_per_gpu": 24.8, + "tokens/trainable": 6369 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.3691928386688232, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.08168518543243408, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.08511, + "step": 15, + "tokens/total": 478688, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 6820 + }, + { + "epoch": 0.0625, + "grad_norm": 2.4356393814086914, + "learning_rate": 6e-05, + "loss": 0.09031196683645248, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.09452, + "step": 16, + "tokens/total": 510512, + "tokens/train_per_sec_per_gpu": 25.7, + "tokens/trainable": 7258 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.55682373046875, + "learning_rate": 6.400000000000001e-05, + "loss": 0.07496573030948639, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.07785, + "step": 17, + "tokens/total": 542736, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 7710 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.1054646968841553, + "learning_rate": 6.800000000000001e-05, + "loss": 0.05218805745244026, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.05357, + "step": 18, + "tokens/total": 574832, + "tokens/train_per_sec_per_gpu": 25.78, + "tokens/trainable": 8174 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.7429099082946777, + "learning_rate": 7.2e-05, + "loss": 0.03790827840566635, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.03864, + "step": 19, + "tokens/total": 606976, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 8617 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9508209228515625, + "learning_rate": 7.6e-05, + "loss": 0.04993613809347153, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0512, + "step": 20, + "tokens/total": 639200, + "tokens/train_per_sec_per_gpu": 24.91, + "tokens/trainable": 9037 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.0297844409942627, + "learning_rate": 8e-05, + "loss": 0.017908263951539993, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.01807, + "step": 21, + "tokens/total": 671072, + "tokens/train_per_sec_per_gpu": 30.42, + "tokens/trainable": 9503 + }, + { + "epoch": 0.0859375, + "grad_norm": 0.9279879331588745, + "learning_rate": 8.4e-05, + "loss": 0.013478193432092667, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01357, + "step": 22, + "tokens/total": 703136, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 9972 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.526488184928894, + "learning_rate": 8.800000000000001e-05, + "loss": 0.017310388386249542, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01746, + "step": 23, + "tokens/total": 735120, + "tokens/train_per_sec_per_gpu": 31.45, + "tokens/trainable": 10447 + }, + { + "epoch": 0.09375, + "grad_norm": 3.4049417972564697, + "learning_rate": 9.200000000000001e-05, + "loss": 0.0353429913520813, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.03597, + "step": 24, + "tokens/total": 767040, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 10899 + }, + { + "epoch": 0.09765625, + "grad_norm": 7.511656761169434, + "learning_rate": 9.6e-05, + "loss": 0.1335875540971756, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.14292, + "step": 25, + "tokens/total": 798976, + "tokens/train_per_sec_per_gpu": 27.72, + "tokens/trainable": 11360 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.0069234371185303, + "learning_rate": 0.0001, + "loss": 0.04382415860891342, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0448, + "step": 26, + "tokens/total": 830880, + "tokens/train_per_sec_per_gpu": 24.61, + "tokens/trainable": 11775 + }, + { + "epoch": 0.10546875, + "grad_norm": 7.807938575744629, + "learning_rate": 9.99990636831587e-05, + "loss": 0.01210833340883255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.01218, + "step": 27, + "tokens/total": 860960, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 12242 + }, + { + "epoch": 0.109375, + "grad_norm": 2.23823881149292, + "learning_rate": 9.999625477159879e-05, + "loss": 0.0272970050573349, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02767, + "step": 28, + "tokens/total": 892816, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 12696 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.315584421157837, + "learning_rate": 9.999157338221051e-05, + "loss": 0.025516577064990997, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.02584, + "step": 29, + "tokens/total": 924736, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 13153 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.8413651585578918, + "learning_rate": 9.998501970980562e-05, + "loss": 0.007838520221412182, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00787, + "step": 30, + "tokens/total": 954592, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 13624 + }, + { + "epoch": 0.12109375, + "grad_norm": 1.6797157526016235, + "learning_rate": 9.997659402710915e-05, + "loss": 0.03002442792057991, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.03048, + "step": 31, + "tokens/total": 986592, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 14064 + }, + { + "epoch": 0.125, + "grad_norm": 1.3320434093475342, + "learning_rate": 9.996629668474818e-05, + "loss": 0.03772728145122528, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.03845, + "step": 32, + "tokens/total": 1018720, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 14565 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.6335167288780212, + "learning_rate": 9.995412811123711e-05, + "loss": 0.013970119878649712, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01407, + "step": 33, + "tokens/total": 1050624, + "tokens/train_per_sec_per_gpu": 28.18, + "tokens/trainable": 15005 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.5626921653747559, + "learning_rate": 9.994008881295999e-05, + "loss": 0.016654539853334427, + "memory/device_reserved (GiB)": 37.74, + "memory/max_active (GiB)": 36.22, + "memory/max_allocated (GiB)": 36.22, + "ppl": 1.01679, + "step": 34, + "tokens/total": 1082176, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 15459 + }, + { + "epoch": 0.13671875, + "grad_norm": 0.603219211101532, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021408583968877792, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02164, + "step": 35, + "tokens/total": 1113968, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 15960 + }, + { + "epoch": 0.140625, + "grad_norm": 0.9779674410820007, + "learning_rate": 9.99064004568618e-05, + "loss": 0.03277068957686424, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03331, + "step": 36, + "tokens/total": 1145952, + "tokens/train_per_sec_per_gpu": 28.68, + "tokens/trainable": 16428 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.8855255842208862, + "learning_rate": 9.988675280095074e-05, + "loss": 0.017048098146915436, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01719, + "step": 37, + "tokens/total": 1177872, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 16884 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.9266122579574585, + "learning_rate": 9.986523722403528e-05, + "loss": 0.045863546431064606, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.04693, + "step": 38, + "tokens/total": 1209776, + "tokens/train_per_sec_per_gpu": 26.91, + "tokens/trainable": 17341 + }, + { + "epoch": 0.15234375, + "grad_norm": 0.8280895948410034, + "learning_rate": 9.984185462146642e-05, + "loss": 0.033055759966373444, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03361, + "step": 39, + "tokens/total": 1241888, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 17786 + }, + { + "epoch": 0.15625, + "grad_norm": 5.029730319976807, + "learning_rate": 9.98166059662897e-05, + "loss": 0.025589998811483383, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02592, + "step": 40, + "tokens/total": 1274032, + "tokens/train_per_sec_per_gpu": 29.22, + "tokens/trainable": 18262 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.5688032507896423, + "learning_rate": 9.978949230920472e-05, + "loss": 0.016095872968435287, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01623, + "step": 41, + "tokens/total": 1306000, + "tokens/train_per_sec_per_gpu": 27.49, + "tokens/trainable": 18716 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8217663168907166, + "learning_rate": 9.976051477852141e-05, + "loss": 0.026798982173204422, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02716, + "step": 42, + "tokens/total": 1337936, + "tokens/train_per_sec_per_gpu": 27.51, + "tokens/trainable": 19157 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.7309132218360901, + "learning_rate": 9.972967458011312e-05, + "loss": 0.033920761197805405, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0345, + "step": 43, + "tokens/total": 1369904, + "tokens/train_per_sec_per_gpu": 30.66, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.8134022355079651, + "learning_rate": 9.96969729973664e-05, + "loss": 0.024005085229873657, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0243, + "step": 44, + "tokens/total": 1401984, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 20119 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.35084816813468933, + "learning_rate": 9.966241139112754e-05, + "loss": 0.013319254852831364, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01341, + "step": 45, + "tokens/total": 1434032, + "tokens/train_per_sec_per_gpu": 27.45, + "tokens/trainable": 20560 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.8701573014259338, + "learning_rate": 9.96259911996461e-05, + "loss": 0.01356641948223114, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01366, + "step": 46, + "tokens/total": 1465872, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 20992 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.4406442642211914, + "learning_rate": 9.958771393851491e-05, + "loss": 0.04396980255842209, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 35.86, + "memory/max_allocated (GiB)": 35.86, + "ppl": 1.04495, + "step": 47, + "tokens/total": 1495440, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 21441 + }, + { + "epoch": 0.1875, + "grad_norm": 0.4966669976711273, + "learning_rate": 9.954758120060702e-05, + "loss": 0.011179964058101177, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01124, + "step": 48, + "tokens/total": 1527776, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 21901 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.41124436259269714, + "learning_rate": 9.950559465600948e-05, + "loss": 0.0063599650748074055, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00638, + "step": 49, + "tokens/total": 1559760, + "tokens/train_per_sec_per_gpu": 27.73, + "tokens/trainable": 22341 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.829739511013031, + "learning_rate": 9.946175605195379e-05, + "loss": 0.015785304829478264, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01591, + "step": 50, + "tokens/total": 1592000, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 22833 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5181736946105957, + "learning_rate": 9.941606721274322e-05, + "loss": 0.008393109776079655, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00843, + "step": 51, + "tokens/total": 1623920, + "tokens/train_per_sec_per_gpu": 25.61, + "tokens/trainable": 23277 + }, + { + "epoch": 0.203125, + "grad_norm": 1.5778990983963013, + "learning_rate": 9.936853003967685e-05, + "loss": 0.024984844028949738, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0253, + "step": 52, + "tokens/total": 1656032, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 23759 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.4938714504241943, + "learning_rate": 9.93191465109705e-05, + "loss": 0.038523320108652115, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03927, + "step": 53, + "tokens/total": 1688368, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 24193 + }, + { + "epoch": 0.2109375, + "grad_norm": 2.1427106857299805, + "learning_rate": 9.926791868167438e-05, + "loss": 0.06302924454212189, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.06506, + "step": 54, + "tokens/total": 1720560, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 24619 + }, + { + "epoch": 0.21484375, + "grad_norm": 1.8365188837051392, + "learning_rate": 9.921484868358753e-05, + "loss": 0.052160464227199554, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.05354, + "step": 55, + "tokens/total": 1752464, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 25075 + }, + { + "epoch": 0.21875, + "grad_norm": 0.7779749631881714, + "learning_rate": 9.915993872516924e-05, + "loss": 0.021328264847397804, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.02156, + "step": 56, + "tokens/total": 1784336, + "tokens/train_per_sec_per_gpu": 26.42, + "tokens/trainable": 25519 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.4189571142196655, + "learning_rate": 9.9103191091447e-05, + "loss": 0.014279939234256744, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01438, + "step": 57, + "tokens/total": 1816304, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 25966 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.3834710717201233, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01883944310247898, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01902, + "step": 58, + "tokens/total": 1848304, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 26423 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.7693278789520264, + "learning_rate": 9.898419232046825e-05, + "loss": 0.03533369302749634, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03597, + "step": 59, + "tokens/total": 1880640, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 26934 + }, + { + "epoch": 0.234375, + "grad_norm": 0.4606446325778961, + "learning_rate": 9.892194613523633e-05, + "loss": 0.021482212468981743, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02171, + "step": 60, + "tokens/total": 1912592, + "tokens/train_per_sec_per_gpu": 27.64, + "tokens/trainable": 27393 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.46030017733573914, + "learning_rate": 9.885787217854357e-05, + "loss": 0.017044665291905403, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01719, + "step": 61, + "tokens/total": 1944544, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 27852 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.7918264269828796, + "learning_rate": 9.879197311676887e-05, + "loss": 0.016954131424427032, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0171, + "step": 62, + "tokens/total": 1976688, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 28292 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.4787926971912384, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01525965891778469, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 63, + "tokens/total": 2008928, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 28798 + }, + { + "epoch": 0.25, + "grad_norm": 0.9033918976783752, + "learning_rate": 9.865471072312528e-05, + "loss": 0.02902211993932724, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02945, + "step": 64, + "tokens/total": 2041120, + "tokens/train_per_sec_per_gpu": 29.83, + "tokens/trainable": 29283 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.39305707812309265, + "learning_rate": 9.858335310330492e-05, + "loss": 0.00901167094707489, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.32, + "memory/max_allocated (GiB)": 36.32, + "ppl": 1.00905, + "step": 65, + "tokens/total": 2073024, + "tokens/train_per_sec_per_gpu": 26.48, + "tokens/trainable": 29745 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4494345784187317, + "learning_rate": 9.851018180226185e-05, + "loss": 0.007426132448017597, + "memory/device_reserved (GiB)": 37.47, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00745, + "step": 66, + "tokens/total": 2105184, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 30186 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.629649817943573, + "learning_rate": 9.843519986495259e-05, + "loss": 0.010609081946313381, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01067, + "step": 67, + "tokens/total": 2137264, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 30622 + }, + { + "epoch": 0.265625, + "grad_norm": 0.7591162919998169, + "learning_rate": 9.835841041168162e-05, + "loss": 0.00926615484058857, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00931, + "step": 68, + "tokens/total": 2169424, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 31097 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.7288486361503601, + "learning_rate": 9.82798166379715e-05, + "loss": 0.024310583248734474, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.02461, + "step": 69, + "tokens/total": 2201424, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 31595 + }, + { + "epoch": 0.2734375, + "grad_norm": 2.134056329727173, + "learning_rate": 9.819942181443002e-05, + "loss": 0.008422967046499252, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00846, + "step": 70, + "tokens/total": 2233424, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 32036 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6362871527671814, + "learning_rate": 9.811722928661392e-05, + "loss": 0.01577262207865715, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0159, + "step": 71, + "tokens/total": 2265328, + "tokens/train_per_sec_per_gpu": 23.65, + "tokens/trainable": 32428 + }, + { + "epoch": 0.28125, + "grad_norm": 0.6091912388801575, + "learning_rate": 9.803324247488975e-05, + "loss": 0.0064800274558365345, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0065, + "step": 72, + "tokens/total": 2297248, + "tokens/train_per_sec_per_gpu": 26.86, + "tokens/trainable": 32880 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.14533838629722595, + "learning_rate": 9.794746487429161e-05, + "loss": 0.001991454279050231, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00199, + "step": 73, + "tokens/total": 2329152, + "tokens/train_per_sec_per_gpu": 27.97, + "tokens/trainable": 33323 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.5400826930999756, + "learning_rate": 9.785990005437554e-05, + "loss": 0.017662350088357925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01782, + "step": 74, + "tokens/total": 2361344, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 33792 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.6132436394691467, + "learning_rate": 9.777055165907117e-05, + "loss": 0.01787603087723255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01804, + "step": 75, + "tokens/total": 2393168, + "tokens/train_per_sec_per_gpu": 29.53, + "tokens/trainable": 34223 + }, + { + "epoch": 0.296875, + "grad_norm": 0.7287399172782898, + "learning_rate": 9.767942340652993e-05, + "loss": 0.024754449725151062, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.31, + "memory/max_allocated (GiB)": 36.31, + "ppl": 1.02506, + "step": 76, + "tokens/total": 2424880, + "tokens/train_per_sec_per_gpu": 24.92, + "tokens/trainable": 34649 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.6790545582771301, + "learning_rate": 9.758651908897035e-05, + "loss": 0.014308430254459381, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01441, + "step": 77, + "tokens/total": 2457120, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 35094 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6203456521034241, + "learning_rate": 9.749184257252033e-05, + "loss": 0.034353043884038925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03495, + "step": 78, + "tokens/total": 2489104, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 35534 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.6238825917243958, + "learning_rate": 9.739539779705614e-05, + "loss": 0.032295141369104385, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.03282, + "step": 79, + "tokens/total": 2521136, + "tokens/train_per_sec_per_gpu": 29.68, + "tokens/trainable": 36029 + }, + { + "epoch": 0.3125, + "grad_norm": 0.18977171182632446, + "learning_rate": 9.729718877603861e-05, + "loss": 0.007260813377797604, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00729, + "step": 80, + "tokens/total": 2553168, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 36469 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.8054049611091614, + "learning_rate": 9.719721959634592e-05, + "loss": 0.034894898533821106, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.03551, + "step": 81, + "tokens/total": 2585584, + "tokens/train_per_sec_per_gpu": 25.75, + "tokens/trainable": 36906 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.6884189248085022, + "learning_rate": 9.709549441810375e-05, + "loss": 0.018210047855973244, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.01838, + "step": 82, + "tokens/total": 2617808, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 37390 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.6490103602409363, + "learning_rate": 9.699201747451195e-05, + "loss": 0.02030857279896736, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.02052, + "step": 83, + "tokens/total": 2649664, + "tokens/train_per_sec_per_gpu": 25.8, + "tokens/trainable": 37838 + }, + { + "epoch": 0.328125, + "grad_norm": 0.2439805567264557, + "learning_rate": 9.688679307166854e-05, + "loss": 0.008518049493432045, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00855, + "step": 84, + "tokens/total": 2681728, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 38310 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6747375726699829, + "learning_rate": 9.677982558839042e-05, + "loss": 0.029895059764385223, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.03035, + "step": 85, + "tokens/total": 2713488, + "tokens/train_per_sec_per_gpu": 28.49, + "tokens/trainable": 38789 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.582666277885437, + "learning_rate": 9.66711194760312e-05, + "loss": 0.014998597092926502, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01511, + "step": 86, + "tokens/total": 2745536, + "tokens/train_per_sec_per_gpu": 30.27, + "tokens/trainable": 39277 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.5364654660224915, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011567167937755585, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01163, + "step": 87, + "tokens/total": 2777680, + "tokens/train_per_sec_per_gpu": 29.73, + "tokens/trainable": 39737 + }, + { + "epoch": 0.34375, + "grad_norm": 0.34024426341056824, + "learning_rate": 9.644850953105288e-05, + "loss": 0.01014852337539196, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0102, + "step": 88, + "tokens/total": 2809568, + "tokens/train_per_sec_per_gpu": 24.68, + "tokens/trainable": 40179 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.32472267746925354, + "learning_rate": 9.633461496214225e-05, + "loss": 0.01229981891810894, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01238, + "step": 89, + "tokens/total": 2841760, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 40649 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.7906899452209473, + "learning_rate": 9.621900029118195e-05, + "loss": 0.02552586793899536, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.02585, + "step": 90, + "tokens/total": 2871696, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 41080 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.7444764971733093, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0054441955871880054, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00546, + "step": 91, + "tokens/total": 2903968, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 41599 + }, + { + "epoch": 0.359375, + "grad_norm": 0.1911851465702057, + "learning_rate": 9.598262995928611e-05, + "loss": 0.004665314219892025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00468, + "step": 92, + "tokens/total": 2936048, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 42076 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.21238519251346588, + "learning_rate": 9.586188413468492e-05, + "loss": 0.003917238209396601, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00392, + "step": 93, + "tokens/total": 2968176, + "tokens/train_per_sec_per_gpu": 28.07, + "tokens/trainable": 42522 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.41729435324668884, + "learning_rate": 9.57394378802934e-05, + "loss": 0.013555807992815971, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01365, + "step": 94, + "tokens/total": 3000304, + "tokens/train_per_sec_per_gpu": 28.69, + "tokens/trainable": 42974 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.7246769070625305, + "learning_rate": 9.56152962916e-05, + "loss": 0.020221399143338203, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.02043, + "step": 95, + "tokens/total": 3032080, + "tokens/train_per_sec_per_gpu": 25.35, + "tokens/trainable": 43380 + }, + { + "epoch": 0.375, + "grad_norm": 0.16139821708202362, + "learning_rate": 9.548946453464296e-05, + "loss": 0.003853276837617159, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00386, + "step": 96, + "tokens/total": 3064320, + "tokens/train_per_sec_per_gpu": 28.47, + "tokens/trainable": 43865 + }, + { + "epoch": 0.37890625, + "grad_norm": 1.0509370565414429, + "learning_rate": 9.53619478457953e-05, + "loss": 0.015099998563528061, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01521, + "step": 97, + "tokens/total": 3096144, + "tokens/train_per_sec_per_gpu": 26.61, + "tokens/trainable": 44328 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.6859350204467773, + "learning_rate": 9.523275153154695e-05, + "loss": 0.004579808097332716, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00459, + "step": 98, + "tokens/total": 3127600, + "tokens/train_per_sec_per_gpu": 25.91, + "tokens/trainable": 44778 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5686795711517334, + "learning_rate": 9.51018809682839e-05, + "loss": 0.022520575672388077, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02278, + "step": 99, + "tokens/total": 3159520, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 45225 + }, + { + "epoch": 0.390625, + "grad_norm": 1.6308369636535645, + "learning_rate": 9.49693416020645e-05, + "loss": 0.011883002705872059, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01195, + "step": 100, + "tokens/total": 3191568, + "tokens/train_per_sec_per_gpu": 29.75, + "tokens/trainable": 45678 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.7186088562011719, + "learning_rate": 9.483513894839276e-05, + "loss": 0.0060632615350186825, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00608, + "step": 101, + "tokens/total": 3223648, + "tokens/train_per_sec_per_gpu": 27.14, + "tokens/trainable": 46125 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.12036337703466415, + "learning_rate": 9.469927859198888e-05, + "loss": 0.0019118499476462603, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00191, + "step": 102, + "tokens/total": 3255744, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 46612 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.29928353428840637, + "learning_rate": 9.456176618655689e-05, + "loss": 0.00944911316037178, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00949, + "step": 103, + "tokens/total": 3287792, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 47059 + }, + { + "epoch": 0.40625, + "grad_norm": 0.3766293227672577, + "learning_rate": 9.442260745454927e-05, + "loss": 0.008984047919511795, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.68, + "memory/max_allocated (GiB)": 36.68, + "ppl": 1.00902, + "step": 104, + "tokens/total": 3320016, + "tokens/train_per_sec_per_gpu": 28.92, + "tokens/trainable": 47536 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3481820523738861, + "learning_rate": 9.428180818692884e-05, + "loss": 0.005444999784231186, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00546, + "step": 105, + "tokens/total": 3351952, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 48037 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.8414323329925537, + "learning_rate": 9.413937424292791e-05, + "loss": 0.032438624650239944, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03297, + "step": 106, + "tokens/total": 3384032, + "tokens/train_per_sec_per_gpu": 27.11, + "tokens/trainable": 48491 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.5067655444145203, + "learning_rate": 9.399531154980424e-05, + "loss": 0.008442888967692852, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00848, + "step": 107, + "tokens/total": 3416192, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 48940 + }, + { + "epoch": 0.421875, + "grad_norm": 0.41617852449417114, + "learning_rate": 9.384962610259455e-05, + "loss": 0.008386321365833282, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00842, + "step": 108, + "tokens/total": 3448080, + "tokens/train_per_sec_per_gpu": 26.56, + "tokens/trainable": 49389 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1587515026330948, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0031663556583225727, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.03, + "memory/max_allocated (GiB)": 36.03, + "ppl": 1.00317, + "step": 109, + "tokens/total": 3477952, + "tokens/train_per_sec_per_gpu": 29.8, + "tokens/trainable": 49838 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2903148829936981, + "learning_rate": 9.355341126345868e-05, + "loss": 0.007958847098052502, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00799, + "step": 110, + "tokens/total": 3509776, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 50310 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.642983078956604, + "learning_rate": 9.340289419824107e-05, + "loss": 0.006748045329004526, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00677, + "step": 111, + "tokens/total": 3541808, + "tokens/train_per_sec_per_gpu": 28.04, + "tokens/trainable": 50755 + }, + { + "epoch": 0.4375, + "grad_norm": 0.14928282797336578, + "learning_rate": 9.325077903184159e-05, + "loss": 0.0027839094400405884, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00279, + "step": 112, + "tokens/total": 3574064, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 51213 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.05922335386276245, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0009622900979593396, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00096, + "step": 113, + "tokens/total": 3605984, + "tokens/train_per_sec_per_gpu": 27.99, + "tokens/trainable": 51660 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.44227731227874756, + "learning_rate": 9.2941779782269e-05, + "loss": 0.002763954224064946, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00277, + "step": 114, + "tokens/total": 3638080, + "tokens/train_per_sec_per_gpu": 27.03, + "tokens/trainable": 52133 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5176818370819092, + "learning_rate": 9.278490855781596e-05, + "loss": 0.015623578801751137, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01575, + "step": 115, + "tokens/total": 3670016, + "tokens/train_per_sec_per_gpu": 31.44, + "tokens/trainable": 52580 + }, + { + "epoch": 0.453125, + "grad_norm": 0.4265497922897339, + "learning_rate": 9.262646494908604e-05, + "loss": 0.00794076919555664, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00797, + "step": 116, + "tokens/total": 3701984, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 53033 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.40457987785339355, + "learning_rate": 9.246645554956457e-05, + "loss": 0.022708112373948097, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02297, + "step": 117, + "tokens/total": 3733920, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 53517 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.01414525043219328, + "learning_rate": 9.230488701789578e-05, + "loss": 0.00020209309877827764, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.0002, + "step": 118, + "tokens/total": 3765840, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 53967 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.11359831690788269, + "learning_rate": 9.214176607760577e-05, + "loss": 0.0012168455868959427, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00122, + "step": 119, + "tokens/total": 3797680, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 54430 + }, + { + "epoch": 0.46875, + "grad_norm": 0.35982534289360046, + "learning_rate": 9.197709951682268e-05, + "loss": 0.004110721405595541, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00412, + "step": 120, + "tokens/total": 3829920, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 54896 + }, + { + "epoch": 0.47265625, + "grad_norm": 1.3690382242202759, + "learning_rate": 9.181089418799428e-05, + "loss": 0.02705790475010872, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02743, + "step": 121, + "tokens/total": 3862080, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 55379 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.351891428232193, + "learning_rate": 9.164315700760271e-05, + "loss": 0.0020163573790341616, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00202, + "step": 122, + "tokens/total": 3893968, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 55803 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4747961461544037, + "learning_rate": 9.147389495587671e-05, + "loss": 0.012588979676365852, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01267, + "step": 123, + "tokens/total": 3926160, + "tokens/train_per_sec_per_gpu": 30.14, + "tokens/trainable": 56249 + }, + { + "epoch": 0.484375, + "grad_norm": 0.4064250588417053, + "learning_rate": 9.130311507650116e-05, + "loss": 0.0022941348142921925, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0023, + "step": 124, + "tokens/total": 3958208, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 56713 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.22056929767131805, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0022175831254571676, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00222, + "step": 125, + "tokens/total": 3990208, + "tokens/train_per_sec_per_gpu": 32.57, + "tokens/trainable": 57196 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5698862671852112, + "learning_rate": 9.09570303250602e-05, + "loss": 0.010205100290477276, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01026, + "step": 126, + "tokens/total": 4022224, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 57680 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.2593887448310852, + "learning_rate": 9.078173985499394e-05, + "loss": 0.0032277877908200026, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00323, + "step": 127, + "tokens/total": 4053952, + "tokens/train_per_sec_per_gpu": 26.51, + "tokens/trainable": 58110 + }, + { + "epoch": 0.5, + "grad_norm": 0.3735230565071106, + "learning_rate": 9.060496036067713e-05, + "loss": 0.01031105499714613, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01036, + "step": 128, + "tokens/total": 4086144, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 58534 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.1332484930753708, + "learning_rate": 9.042669919862615e-05, + "loss": 0.0012390739284455776, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00124, + "step": 129, + "tokens/total": 4118096, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 58998 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.20182718336582184, + "learning_rate": 9.024696378701557e-05, + "loss": 0.004501686431467533, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00451, + "step": 130, + "tokens/total": 4150272, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 59448 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.529080331325531, + "learning_rate": 9.006576160536948e-05, + "loss": 0.00763288140296936, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00766, + "step": 131, + "tokens/total": 4181904, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 59906 + }, + { + "epoch": 0.515625, + "grad_norm": 0.21625548601150513, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0024378676898777485, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00244, + "step": 132, + "tokens/total": 4214032, + "tokens/train_per_sec_per_gpu": 28.36, + "tokens/trainable": 60350 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.37116539478302, + "learning_rate": 8.969898715494506e-05, + "loss": 0.0037849934305995703, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00379, + "step": 133, + "tokens/total": 4246064, + "tokens/train_per_sec_per_gpu": 30.18, + "tokens/trainable": 60831 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5646623373031616, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006517104804515839, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00654, + "step": 134, + "tokens/total": 4278224, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 61305 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.6371963620185852, + "learning_rate": 8.932643689864568e-05, + "loss": 0.008687382563948631, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00873, + "step": 135, + "tokens/total": 4310112, + "tokens/train_per_sec_per_gpu": 31.48, + "tokens/trainable": 61792 + }, + { + "epoch": 0.53125, + "grad_norm": 0.8210524916648865, + "learning_rate": 8.913801518498845e-05, + "loss": 0.008436474949121475, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00847, + "step": 136, + "tokens/total": 4342192, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 62253 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.5692176222801208, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017321188002824783, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.01747, + "step": 137, + "tokens/total": 4374096, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.12828375399112701, + "learning_rate": 8.875691779131569e-05, + "loss": 0.0007754197577014565, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00078, + "step": 138, + "tokens/total": 4406272, + "tokens/train_per_sec_per_gpu": 24.22, + "tokens/trainable": 63145 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.3188190162181854, + "learning_rate": 8.856425797031829e-05, + "loss": 0.006707335356622934, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00673, + "step": 139, + "tokens/total": 4438112, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 63587 + }, + { + "epoch": 0.546875, + "grad_norm": 0.6216707229614258, + "learning_rate": 8.837020140354295e-05, + "loss": 0.007024695165455341, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00705, + "step": 140, + "tokens/total": 4470128, + "tokens/train_per_sec_per_gpu": 30.64, + "tokens/trainable": 64052 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.1480305939912796, + "learning_rate": 8.817475616647554e-05, + "loss": 0.00229115248657763, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00229, + "step": 141, + "tokens/total": 4502096, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 64526 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22254763543605804, + "learning_rate": 8.797793039239017e-05, + "loss": 0.0029198499396443367, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00292, + "step": 142, + "tokens/total": 4534496, + "tokens/train_per_sec_per_gpu": 29.08, + "tokens/trainable": 64983 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.6901941895484924, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013051149435341358, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01314, + "step": 143, + "tokens/total": 4566720, + "tokens/train_per_sec_per_gpu": 31.56, + "tokens/trainable": 65492 + }, + { + "epoch": 0.5625, + "grad_norm": 0.9545458555221558, + "learning_rate": 8.758017005316988e-05, + "loss": 0.02235381305217743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.02261, + "step": 144, + "tokens/total": 4598880, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 65925 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.5515892505645752, + "learning_rate": 8.737925204046629e-05, + "loss": 0.02038375660777092, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.02059, + "step": 145, + "tokens/total": 4630864, + "tokens/train_per_sec_per_gpu": 26.3, + "tokens/trainable": 66383 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3512430191040039, + "learning_rate": 8.717698659491851e-05, + "loss": 0.008233271539211273, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00827, + "step": 146, + "tokens/total": 4662864, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 66840 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.6830361485481262, + "learning_rate": 8.697338213361735e-05, + "loss": 0.013419180177152157, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01351, + "step": 147, + "tokens/total": 4694912, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 67297 + }, + { + "epoch": 0.578125, + "grad_norm": 0.2117031067609787, + "learning_rate": 8.676844712937552e-05, + "loss": 0.004134457092732191, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00414, + "step": 148, + "tokens/total": 4727120, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 67769 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.378053218126297, + "learning_rate": 8.656219011037509e-05, + "loss": 0.009747720323503017, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0098, + "step": 149, + "tokens/total": 4759168, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 68257 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.21076741814613342, + "learning_rate": 8.63546196598125e-05, + "loss": 0.005035653710365295, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00505, + "step": 150, + "tokens/total": 4791008, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 68706 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.25075697898864746, + "learning_rate": 8.614574441554145e-05, + "loss": 0.006027051247656345, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00605, + "step": 151, + "tokens/total": 4822912, + "tokens/train_per_sec_per_gpu": 27.95, + "tokens/trainable": 69131 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5774204730987549, + "learning_rate": 8.593557306971349e-05, + "loss": 0.018617186695337296, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01879, + "step": 152, + "tokens/total": 4854816, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 69586 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.1364293098449707, + "learning_rate": 8.572411436841618e-05, + "loss": 0.002049289643764496, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00205, + "step": 153, + "tokens/total": 4886656, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 70061 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.25387004017829895, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0044304742477834225, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00444, + "step": 154, + "tokens/total": 4918768, + "tokens/train_per_sec_per_gpu": 23.26, + "tokens/trainable": 70467 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3047125041484833, + "learning_rate": 8.529737015125824e-05, + "loss": 0.0043389927595853806, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00435, + "step": 155, + "tokens/total": 4951008, + "tokens/train_per_sec_per_gpu": 27.53, + "tokens/trainable": 70933 + }, + { + "epoch": 0.609375, + "grad_norm": 0.1442601978778839, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0031973645091056824, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.0032, + "step": 156, + "tokens/total": 4982976, + "tokens/train_per_sec_per_gpu": 28.86, + "tokens/trainable": 71382 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.2608455717563629, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0032951058819890022, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0033, + "step": 157, + "tokens/total": 5014640, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 71808 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.637295663356781, + "learning_rate": 8.464782037243449e-05, + "loss": 0.01436957623809576, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.01447, + "step": 158, + "tokens/total": 5046672, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 72249 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.9834056496620178, + "learning_rate": 8.442882418044202e-05, + "loss": 0.02375311404466629, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.02404, + "step": 159, + "tokens/total": 5078528, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 72726 + }, + { + "epoch": 0.625, + "grad_norm": 0.35432615876197815, + "learning_rate": 8.420860333495179e-05, + "loss": 0.006119017023593187, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00614, + "step": 160, + "tokens/total": 5110704, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 73148 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.44270065426826477, + "learning_rate": 8.398716700025208e-05, + "loss": 0.01352179329842329, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.72, + "memory/max_allocated (GiB)": 36.72, + "ppl": 1.01361, + "step": 161, + "tokens/total": 5142944, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 73600 + }, + { + "epoch": 0.6328125, + "grad_norm": 1.2181634902954102, + "learning_rate": 8.376452439121266e-05, + "loss": 0.01820964366197586, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.01838, + "step": 162, + "tokens/total": 5174592, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 74068 + }, + { + "epoch": 0.63671875, + "grad_norm": 1.9321303367614746, + "learning_rate": 8.354068477290124e-05, + "loss": 0.009742327965795994, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00979, + "step": 163, + "tokens/total": 5206432, + "tokens/train_per_sec_per_gpu": 29.89, + "tokens/trainable": 74527 + }, + { + "epoch": 0.640625, + "grad_norm": 0.3545796275138855, + "learning_rate": 8.331565746019807e-05, + "loss": 0.00975855067372322, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00981, + "step": 164, + "tokens/total": 5238384, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 74992 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.35368478298187256, + "learning_rate": 8.308945181740812e-05, + "loss": 0.010195466689765453, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01025, + "step": 165, + "tokens/total": 5270272, + "tokens/train_per_sec_per_gpu": 29.41, + "tokens/trainable": 75442 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.34957799315452576, + "learning_rate": 8.286207725787153e-05, + "loss": 0.009373681619763374, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00942, + "step": 166, + "tokens/total": 5302368, + "tokens/train_per_sec_per_gpu": 26.2, + "tokens/trainable": 75887 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.3684624135494232, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008211556822061539, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00825, + "step": 167, + "tokens/total": 5334496, + "tokens/train_per_sec_per_gpu": 28.21, + "tokens/trainable": 76331 + }, + { + "epoch": 0.65625, + "grad_norm": 0.12826858460903168, + "learning_rate": 8.240385928474219e-05, + "loss": 0.004492453299462795, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0045, + "step": 168, + "tokens/total": 5364288, + "tokens/train_per_sec_per_gpu": 30.16, + "tokens/trainable": 76745 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.31828173995018005, + "learning_rate": 8.217303493946967e-05, + "loss": 0.005733514670282602, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00575, + "step": 169, + "tokens/total": 5396512, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 77201 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6174353361129761, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01915649324655533, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01934, + "step": 170, + "tokens/total": 5428624, + "tokens/train_per_sec_per_gpu": 27.75, + "tokens/trainable": 77655 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.2902314066886902, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0036087254993617535, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00362, + "step": 171, + "tokens/total": 5460880, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 78122 + }, + { + "epoch": 0.671875, + "grad_norm": 0.3267754912376404, + "learning_rate": 8.147381587530713e-05, + "loss": 0.008090222254395485, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00812, + "step": 172, + "tokens/total": 5493088, + "tokens/train_per_sec_per_gpu": 27.86, + "tokens/trainable": 78576 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.4086189866065979, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006351984106004238, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00637, + "step": 173, + "tokens/total": 5525312, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 79059 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5261490345001221, + "learning_rate": 8.100214524900103e-05, + "loss": 0.008662862703204155, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.0087, + "step": 174, + "tokens/total": 5556880, + "tokens/train_per_sec_per_gpu": 24.62, + "tokens/trainable": 79498 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11292761564254761, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0027105510234832764, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00271, + "step": 175, + "tokens/total": 5588976, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 79922 + }, + { + "epoch": 0.6875, + "grad_norm": 0.33315548300743103, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0030926030594855547, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.0031, + "step": 176, + "tokens/total": 5618992, + "tokens/train_per_sec_per_gpu": 29.51, + "tokens/trainable": 80383 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.3136921226978302, + "learning_rate": 8.028654871074489e-05, + "loss": 0.005057850852608681, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00507, + "step": 177, + "tokens/total": 5651072, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 80824 + }, + { + "epoch": 0.6953125, + "grad_norm": 1.0526509284973145, + "learning_rate": 8.004589869885986e-05, + "loss": 0.010707498528063297, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01077, + "step": 178, + "tokens/total": 5683296, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 81243 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.1329648196697235, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0017975447699427605, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.0018, + "step": 179, + "tokens/total": 5715520, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 81716 + }, + { + "epoch": 0.703125, + "grad_norm": 1.0749201774597168, + "learning_rate": 7.95614819466576e-05, + "loss": 0.025617048144340515, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02595, + "step": 180, + "tokens/total": 5747488, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 82181 + }, + { + "epoch": 0.70703125, + "grad_norm": 1.0446465015411377, + "learning_rate": 7.931773536489872e-05, + "loss": 0.030236585065722466, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 35.9, + "memory/max_allocated (GiB)": 35.9, + "ppl": 1.0307, + "step": 181, + "tokens/total": 5777264, + "tokens/train_per_sec_per_gpu": 28.89, + "tokens/trainable": 82626 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22853964567184448, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0038214433006942272, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00383, + "step": 182, + "tokens/total": 5809264, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 83089 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.057821910828351974, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0004995397757738829, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0005, + "step": 183, + "tokens/total": 5841488, + "tokens/train_per_sec_per_gpu": 29.34, + "tokens/trainable": 83590 + }, + { + "epoch": 0.71875, + "grad_norm": 1.3264167308807373, + "learning_rate": 7.85804646415409e-05, + "loss": 0.013232000172138214, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01332, + "step": 184, + "tokens/total": 5873696, + "tokens/train_per_sec_per_gpu": 29.33, + "tokens/trainable": 84046 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.6027824878692627, + "learning_rate": 7.833273149760207e-05, + "loss": 0.012215660884976387, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01229, + "step": 185, + "tokens/total": 5905408, + "tokens/train_per_sec_per_gpu": 28.25, + "tokens/trainable": 84517 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.5848041772842407, + "learning_rate": 7.808402738346527e-05, + "loss": 0.024627480655908585, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.02493, + "step": 186, + "tokens/total": 5937392, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 84974 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.6823275089263916, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0028813849203288555, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00289, + "step": 187, + "tokens/total": 5969344, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 85460 + }, + { + "epoch": 0.734375, + "grad_norm": 0.41855958104133606, + "learning_rate": 7.758374768294647e-05, + "loss": 0.009796380065381527, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00984, + "step": 188, + "tokens/total": 6001296, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 85939 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6849080324172974, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004100777208805084, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00411, + "step": 189, + "tokens/total": 6031216, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 86377 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.15838374197483063, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003986812196671963, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00399, + "step": 190, + "tokens/total": 6063296, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 86834 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.3117486536502838, + "learning_rate": 7.682630588562518e-05, + "loss": 0.009132719598710537, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00917, + "step": 191, + "tokens/total": 6095216, + "tokens/train_per_sec_per_gpu": 26.65, + "tokens/trainable": 87265 + }, + { + "epoch": 0.75, + "grad_norm": 0.7837114334106445, + "learning_rate": 7.657199467573129e-05, + "loss": 0.007239446043968201, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00727, + "step": 192, + "tokens/total": 6127488, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 87747 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21777178347110748, + "learning_rate": 7.631678576708561e-05, + "loss": 0.002932134782895446, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00294, + "step": 193, + "tokens/total": 6159488, + "tokens/train_per_sec_per_gpu": 30.37, + "tokens/trainable": 88239 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.19159242510795593, + "learning_rate": 7.606068977997255e-05, + "loss": 0.003356864908710122, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00336, + "step": 194, + "tokens/total": 6191296, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 88718 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.3977915346622467, + "learning_rate": 7.580371737159148e-05, + "loss": 0.008549144491553307, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00859, + "step": 195, + "tokens/total": 6223120, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 89129 + }, + { + "epoch": 0.765625, + "grad_norm": 0.3875303566455841, + "learning_rate": 7.554587923561324e-05, + "loss": 0.006131038535386324, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00615, + "step": 196, + "tokens/total": 6254880, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 89567 + }, + { + "epoch": 0.76953125, + "grad_norm": 3.4133481979370117, + "learning_rate": 7.528718610173511e-05, + "loss": 0.01836150512099266, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01853, + "step": 197, + "tokens/total": 6286960, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 89988 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.2509997487068176, + "learning_rate": 7.502764873523431e-05, + "loss": 0.003821226069703698, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00383, + "step": 198, + "tokens/total": 6318880, + "tokens/train_per_sec_per_gpu": 30.29, + "tokens/trainable": 90434 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.9550195932388306, + "learning_rate": 7.476727793652011e-05, + "loss": 0.006872436031699181, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0069, + "step": 199, + "tokens/total": 6350928, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 90913 + }, + { + "epoch": 0.78125, + "grad_norm": 0.411724716424942, + "learning_rate": 7.450608454068415e-05, + "loss": 0.014243390411138535, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01435, + "step": 200, + "tokens/total": 6382976, + "tokens/train_per_sec_per_gpu": 28.29, + "tokens/trainable": 91355 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.34264692664146423, + "learning_rate": 7.424407941704987e-05, + "loss": 0.005309835076332092, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00532, + "step": 201, + "tokens/total": 6415120, + "tokens/train_per_sec_per_gpu": 31.41, + "tokens/trainable": 91856 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.41955795884132385, + "learning_rate": 7.398127346871986e-05, + "loss": 0.009196512401103973, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00924, + "step": 202, + "tokens/total": 6447216, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 92311 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.03107454814016819, + "learning_rate": 7.371767763212238e-05, + "loss": 0.0004908143309876323, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00049, + "step": 203, + "tokens/total": 6479280, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 92764 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39578482508659363, + "learning_rate": 7.345330287655617e-05, + "loss": 0.006089594680815935, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00611, + "step": 204, + "tokens/total": 6511040, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 93227 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.6444572806358337, + "learning_rate": 7.31881602037339e-05, + "loss": 0.010252210311591625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0103, + "step": 205, + "tokens/total": 6543120, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 93675 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.3283638060092926, + "learning_rate": 7.29222606473245e-05, + "loss": 0.00761406822130084, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00764, + "step": 206, + "tokens/total": 6575520, + "tokens/train_per_sec_per_gpu": 27.02, + "tokens/trainable": 94120 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.09952464699745178, + "learning_rate": 7.265561527249383e-05, + "loss": 0.001628706930205226, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00163, + "step": 207, + "tokens/total": 6607408, + "tokens/train_per_sec_per_gpu": 29.43, + "tokens/trainable": 94557 + }, + { + "epoch": 0.8125, + "grad_norm": 0.31704938411712646, + "learning_rate": 7.238823517544436e-05, + "loss": 0.005896817892789841, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00591, + "step": 208, + "tokens/total": 6639488, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 95035 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.0999862551689148, + "learning_rate": 7.212013148295333e-05, + "loss": 0.001754446537233889, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00176, + "step": 209, + "tokens/total": 6671520, + "tokens/train_per_sec_per_gpu": 31.08, + "tokens/trainable": 95517 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.1405109018087387, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0017940793186426163, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0018, + "step": 210, + "tokens/total": 6703536, + "tokens/train_per_sec_per_gpu": 30.3, + "tokens/trainable": 96026 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.09185238927602768, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0015102234901860356, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00151, + "step": 211, + "tokens/total": 6735616, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 96477 + }, + { + "epoch": 0.828125, + "grad_norm": 1.0464473962783813, + "learning_rate": 7.131159054949273e-05, + "loss": 0.016731251031160355, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01687, + "step": 212, + "tokens/total": 6767584, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 96951 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.3517152965068817, + "learning_rate": 7.104070433827139e-05, + "loss": 0.007984626106917858, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00802, + "step": 213, + "tokens/total": 6799488, + "tokens/train_per_sec_per_gpu": 25.83, + "tokens/trainable": 97350 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.28192469477653503, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00557567086070776, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00559, + "step": 214, + "tokens/total": 6831552, + "tokens/train_per_sec_per_gpu": 24.66, + "tokens/trainable": 97792 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.46135279536247253, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0170141588896513, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.01716, + "step": 215, + "tokens/total": 6863488, + "tokens/train_per_sec_per_gpu": 30.38, + "tokens/trainable": 98257 + }, + { + "epoch": 0.84375, + "grad_norm": 0.4277898073196411, + "learning_rate": 7.022408581865382e-05, + "loss": 0.009399959817528725, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.71, + "memory/max_allocated (GiB)": 36.71, + "ppl": 1.00944, + "step": 216, + "tokens/total": 6895888, + "tokens/train_per_sec_per_gpu": 27.28, + "tokens/trainable": 98731 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.34988704323768616, + "learning_rate": 6.99505974422157e-05, + "loss": 0.002909669652581215, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00291, + "step": 217, + "tokens/total": 6928000, + "tokens/train_per_sec_per_gpu": 30.25, + "tokens/trainable": 99212 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.4028005599975586, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0058855339884757996, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0059, + "step": 218, + "tokens/total": 6960096, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 99654 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.3000575304031372, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0025230362080037594, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00253, + "step": 219, + "tokens/total": 6992512, + "tokens/train_per_sec_per_gpu": 27.46, + "tokens/trainable": 100086 + }, + { + "epoch": 0.859375, + "grad_norm": 0.11351172626018524, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0021392193157225847, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00214, + "step": 220, + "tokens/total": 7024688, + "tokens/train_per_sec_per_gpu": 27.56, + "tokens/trainable": 100524 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.19512486457824707, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00650001410394907, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00652, + "step": 221, + "tokens/total": 7056928, + "tokens/train_per_sec_per_gpu": 28.05, + "tokens/trainable": 100996 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.1998104453086853, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0056140003725886345, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00563, + "step": 222, + "tokens/total": 7088944, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 101449 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.40908998250961304, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01286369375884533, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.01295, + "step": 223, + "tokens/total": 7121104, + "tokens/train_per_sec_per_gpu": 27.19, + "tokens/trainable": 101905 + }, + { + "epoch": 0.875, + "grad_norm": 0.3672473132610321, + "learning_rate": 6.801939899284132e-05, + "loss": 0.010677668265998363, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01073, + "step": 224, + "tokens/total": 7153040, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 102411 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4146183431148529, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0038945709820836782, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0039, + "step": 225, + "tokens/total": 7184912, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 102882 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.10585323721170425, + "learning_rate": 6.746257910210214e-05, + "loss": 0.00041726604104042053, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00042, + "step": 226, + "tokens/total": 7216992, + "tokens/train_per_sec_per_gpu": 28.72, + "tokens/trainable": 103332 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.36600011587142944, + "learning_rate": 6.718338543014937e-05, + "loss": 0.01526213251054287, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 227, + "tokens/total": 7249120, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 103790 + }, + { + "epoch": 0.890625, + "grad_norm": 0.16057318449020386, + "learning_rate": 6.69036847577983e-05, + "loss": 0.004024423658847809, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00403, + "step": 228, + "tokens/total": 7281072, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 104246 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.318798303604126, + "learning_rate": 6.662348872453553e-05, + "loss": 0.006663155741989613, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00669, + "step": 229, + "tokens/total": 7313120, + "tokens/train_per_sec_per_gpu": 31.3, + "tokens/trainable": 104707 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.10858234763145447, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0020512123592197895, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00205, + "step": 230, + "tokens/total": 7345184, + "tokens/train_per_sec_per_gpu": 30.81, + "tokens/trainable": 105167 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.1044008731842041, + "learning_rate": 6.60616572358065e-05, + "loss": 0.0010264476295560598, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00103, + "step": 231, + "tokens/total": 7377264, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 105576 + }, + { + "epoch": 0.90625, + "grad_norm": 0.14836947619915009, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0013845522189512849, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00139, + "step": 232, + "tokens/total": 7409424, + "tokens/train_per_sec_per_gpu": 30.32, + "tokens/trainable": 106043 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.022752461954951286, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0003245974367018789, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00032, + "step": 233, + "tokens/total": 7441456, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 106506 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.22981758415699005, + "learning_rate": 6.521548694236384e-05, + "loss": 0.002587020630016923, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00259, + "step": 234, + "tokens/total": 7473344, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 106951 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.5147526264190674, + "learning_rate": 6.493256429322259e-05, + "loss": 0.01256126631051302, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01264, + "step": 235, + "tokens/total": 7505040, + "tokens/train_per_sec_per_gpu": 25.64, + "tokens/trainable": 107382 + }, + { + "epoch": 0.921875, + "grad_norm": 0.3785130977630615, + "learning_rate": 6.464922830953799e-05, + "loss": 0.010956985875964165, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01102, + "step": 236, + "tokens/total": 7537120, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 107814 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.6193199753761292, + "learning_rate": 6.436549078207688e-05, + "loss": 0.020655140280723572, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02087, + "step": 237, + "tokens/total": 7569024, + "tokens/train_per_sec_per_gpu": 29.36, + "tokens/trainable": 108274 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.5084519982337952, + "learning_rate": 6.408136351831592e-05, + "loss": 0.011346567422151566, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01141, + "step": 238, + "tokens/total": 7600992, + "tokens/train_per_sec_per_gpu": 24.43, + "tokens/trainable": 108714 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.04367341846227646, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0004316020058467984, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00043, + "step": 239, + "tokens/total": 7633280, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 109220 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06729783862829208, + "learning_rate": 6.351198709240186e-05, + "loss": 0.000604260538239032, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.0006, + "step": 240, + "tokens/total": 7665408, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 109672 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.3010976314544678, + "learning_rate": 6.32267616243259e-05, + "loss": 0.006106264889240265, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00612, + "step": 241, + "tokens/total": 7697280, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 110135 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.11426915228366852, + "learning_rate": 6.294119380711849e-05, + "loss": 0.0014383264351636171, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00144, + "step": 242, + "tokens/total": 7729440, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 110563 + }, + { + "epoch": 0.94921875, + "grad_norm": 1.3792941570281982, + "learning_rate": 6.265529552442209e-05, + "loss": 0.003547664964571595, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00355, + "step": 243, + "tokens/total": 7761584, + "tokens/train_per_sec_per_gpu": 30.76, + "tokens/trainable": 111049 + }, + { + "epoch": 0.953125, + "grad_norm": 0.07595008611679077, + "learning_rate": 6.236907867363127e-05, + "loss": 0.0009103374904952943, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00091, + "step": 244, + "tokens/total": 7793824, + "tokens/train_per_sec_per_gpu": 28.23, + "tokens/trainable": 111495 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05425361171364784, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0013249842450022697, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00133, + "step": 245, + "tokens/total": 7825872, + "tokens/train_per_sec_per_gpu": 26.47, + "tokens/trainable": 111968 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.17112742364406586, + "learning_rate": 6.179573692313344e-05, + "loss": 0.001098912674933672, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.0011, + "step": 246, + "tokens/total": 7857840, + "tokens/train_per_sec_per_gpu": 30.09, + "tokens/trainable": 112416 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.10754083096981049, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0031517897732555866, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00316, + "step": 247, + "tokens/total": 7890000, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 112882 + }, + { + "epoch": 0.96875, + "grad_norm": 0.2477901428937912, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005755824968218803, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00577, + "step": 248, + "tokens/total": 7921984, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 113390 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.332253634929657, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.013207933865487576, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.0133, + "step": 249, + "tokens/total": 7953824, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 113904 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.18514207005500793, + "learning_rate": 6.064575550087316e-05, + "loss": 0.00523729994893074, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00525, + "step": 250, + "tokens/total": 7985984, + "tokens/train_per_sec_per_gpu": 28.41, + "tokens/trainable": 114351 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.1852743923664093, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0031474479474127293, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00315, + "step": 251, + "tokens/total": 8018256, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 114842 + }, + { + "epoch": 0.984375, + "grad_norm": 0.43020468950271606, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0053521618247032166, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00537, + "step": 252, + "tokens/total": 8050528, + "tokens/train_per_sec_per_gpu": 23.97, + "tokens/trainable": 115263 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.048135045915842056, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0005759480991400778, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00058, + "step": 253, + "tokens/total": 8082496, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 115703 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.028733640909194946, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00040354474913328886, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0004, + "step": 254, + "tokens/total": 8114592, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 116157 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.2808869481086731, + "learning_rate": 5.920308275189541e-05, + "loss": 0.006315852981060743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00634, + "step": 255, + "tokens/total": 8146752, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 116567 + }, + { + "epoch": 1.0, + "grad_norm": 0.054996367543935776, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0009099108865484595, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00091, + "step": 256, + "tokens/total": 8178768, + "tokens/train_per_sec_per_gpu": 25.98, + "tokens/trainable": 117030 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.18114623427391052, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.002796629210934043, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0028, + "step": 257, + "tokens/total": 8210800, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 117517 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.1408185362815857, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0013210453325882554, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00132, + "step": 258, + "tokens/total": 8242768, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 118015 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.36587440967559814, + "learning_rate": 5.80457242789548e-05, + "loss": 0.002160525880753994, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00216, + "step": 259, + "tokens/total": 8274784, + "tokens/train_per_sec_per_gpu": 30.13, + "tokens/trainable": 118520 + }, + { + "epoch": 1.015625, + "grad_norm": 0.02628348581492901, + "learning_rate": 5.77560376810977e-05, + "loss": 0.0003391164937056601, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00034, + "step": 260, + "tokens/total": 8306720, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 118992 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.014996240846812725, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0002124913444276899, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00021, + "step": 261, + "tokens/total": 8338784, + "tokens/train_per_sec_per_gpu": 27.96, + "tokens/trainable": 119438 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.0050656464882195, + "learning_rate": 5.717633247526522e-05, + "loss": 7.557802018709481e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00008, + "step": 262, + "tokens/total": 8370720, + "tokens/train_per_sec_per_gpu": 26.9, + "tokens/trainable": 119881 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.02459302730858326, + "learning_rate": 5.688633799118971e-05, + "loss": 0.00024908874183893204, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00025, + "step": 263, + "tokens/total": 8402736, + "tokens/train_per_sec_per_gpu": 23.2, + "tokens/trainable": 120285 + }, + { + "epoch": 1.03125, + "grad_norm": 0.017813201993703842, + "learning_rate": 5.659626500889066e-05, + "loss": 0.00021895825921092182, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00022, + "step": 264, + "tokens/total": 8434352, + "tokens/train_per_sec_per_gpu": 27.78, + "tokens/trainable": 120755 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.0038785531651228666, + "learning_rate": 5.6306125599488905e-05, + "loss": 6.556476728292182e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00007, + "step": 265, + "tokens/total": 8466208, + "tokens/train_per_sec_per_gpu": 29.86, + "tokens/trainable": 121194 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.18033447861671448, + "learning_rate": 5.601593183686955e-05, + "loss": 0.0016840343596413732, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00169, + "step": 266, + "tokens/total": 8498304, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 121670 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.22497256100177765, + "learning_rate": 5.572569579717961e-05, + "loss": 0.003026450052857399, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00303, + "step": 267, + "tokens/total": 8530400, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 122142 + }, + { + "epoch": 1.046875, + "grad_norm": 0.008878038264811039, + "learning_rate": 5.543542955832538e-05, + "loss": 8.136438555084169e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00008, + "step": 268, + "tokens/total": 8562608, + "tokens/train_per_sec_per_gpu": 27.27, + "tokens/trainable": 122585 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.19042901694774628, + "learning_rate": 5.514514519946986e-05, + "loss": 0.0020417363848537207, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00204, + "step": 269, + "tokens/total": 8594832, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 123091 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.003382593160495162, + "learning_rate": 5.485485480053015e-05, + "loss": 3.870048021781258e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00004, + "step": 270, + "tokens/total": 8627120, + "tokens/train_per_sec_per_gpu": 26.36, + "tokens/trainable": 123505 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.2004430592060089, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.002532299840822816, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.00254, + "step": 271, + "tokens/total": 8657056, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 123953 + }, + { + "epoch": 1.0625, + "grad_norm": 0.24857956171035767, + "learning_rate": 5.42743042028204e-05, + "loss": 0.004157304298132658, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00417, + "step": 272, + "tokens/total": 8689104, + "tokens/train_per_sec_per_gpu": 24.05, + "tokens/trainable": 124400 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.05064955726265907, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0006531125982291996, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00065, + "step": 273, + "tokens/total": 8721136, + "tokens/train_per_sec_per_gpu": 29.3, + "tokens/trainable": 124870 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.12664659321308136, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0018034171080216765, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00181, + "step": 274, + "tokens/total": 8753200, + "tokens/train_per_sec_per_gpu": 26.52, + "tokens/trainable": 125333 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.01593201234936714, + "learning_rate": 5.340373499110935e-05, + "loss": 0.00010935973114101216, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00011, + "step": 275, + "tokens/total": 8785168, + "tokens/train_per_sec_per_gpu": 30.45, + "tokens/trainable": 125834 + }, + { + "epoch": 1.078125, + "grad_norm": 0.0015056057600304484, + "learning_rate": 5.3113662008810304e-05, + "loss": 1.9860939573845826e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00002, + "step": 276, + "tokens/total": 8817520, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 126316 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.004457728937268257, + "learning_rate": 5.282366752473479e-05, + "loss": 2.716448398132343e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00003, + "step": 277, + "tokens/total": 8849360, + "tokens/train_per_sec_per_gpu": 29.63, + "tokens/trainable": 126798 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.03812727704644203, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.0003402164438739419, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.00034, + "step": 278, + "tokens/total": 8881152, + "tokens/train_per_sec_per_gpu": 24.5, + "tokens/trainable": 127223 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.011409996077418327, + "learning_rate": 5.224396231890232e-05, + "loss": 9.064963523996994e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.00009, + "step": 279, + "tokens/total": 8912512, + "tokens/train_per_sec_per_gpu": 29.78, + "tokens/trainable": 127672 + }, + { + "epoch": 1.09375, + "grad_norm": 0.037030577659606934, + "learning_rate": 5.195427572104522e-05, + "loss": 0.0003734154161065817, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00037, + "step": 280, + "tokens/total": 8944416, + "tokens/train_per_sec_per_gpu": 26.76, + "tokens/trainable": 128116 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.0032268385402858257, + "learning_rate": 5.166471586820751e-05, + "loss": 2.529611811041832e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00003, + "step": 281, + "tokens/total": 8976272, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 128589 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.20227248966693878, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.0024647137615829706, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00247, + "step": 282, + "tokens/total": 9008560, + "tokens/train_per_sec_per_gpu": 27.84, + "tokens/trainable": 129036 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.43708595633506775, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0028322634752839804, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00284, + "step": 283, + "tokens/total": 9040496, + "tokens/train_per_sec_per_gpu": 29.25, + "tokens/trainable": 129485 + }, + { + "epoch": 1.109375, + "grad_norm": 0.09467080235481262, + "learning_rate": 5.079691724810461e-05, + "loss": 0.001156509853899479, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00116, + "step": 284, + "tokens/total": 9072448, + "tokens/train_per_sec_per_gpu": 24.69, + "tokens/trainable": 129920 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.1778785139322281, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.0016857109731063247, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00169, + "step": 285, + "tokens/total": 9104384, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 130417 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.0003376641252543777, + "learning_rate": 5.021923930849237e-05, + "loss": 5.597693871095544e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00001, + "step": 286, + "tokens/total": 9136416, + "tokens/train_per_sec_per_gpu": 29.99, + "tokens/trainable": 130903 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.019782084971666336, + "learning_rate": 4.99306927511967e-05, + "loss": 0.00012461614096537232, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00012, + "step": 287, + "tokens/total": 9168640, + "tokens/train_per_sec_per_gpu": 29.24, + "tokens/trainable": 131343 + }, + { + "epoch": 1.125, + "grad_norm": 0.9641065001487732, + "learning_rate": 4.964235714846775e-05, + "loss": 0.011515076272189617, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01158, + "step": 288, + "tokens/total": 9200544, + "tokens/train_per_sec_per_gpu": 27.04, + "tokens/trainable": 131763 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0010062052169814706, + "learning_rate": 4.9354244499126866e-05, + "loss": 9.489545846008696e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00001, + "step": 289, + "tokens/total": 9232624, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 132229 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.002373398980125785, + "learning_rate": 4.90663667927174e-05, + "loss": 1.573777262819931e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00002, + "step": 290, + "tokens/total": 9264768, + "tokens/train_per_sec_per_gpu": 29.5, + "tokens/trainable": 132678 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.0014718567254021764, + "learning_rate": 4.877873600900581e-05, + "loss": 9.888429303828161e-06, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00001, + "step": 291, + "tokens/total": 9296944, + "tokens/train_per_sec_per_gpu": 24.47, + "tokens/trainable": 133136 + }, + { + "epoch": 1.140625, + "grad_norm": 0.0018440725980326533, + "learning_rate": 4.849136411748306e-05, + "loss": 8.275845175376162e-06, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00001, + "step": 292, + "tokens/total": 9328848, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 133608 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.12918472290039062, + "learning_rate": 4.8204263076866574e-05, + "loss": 0.0009128568926826119, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00091, + "step": 293, + "tokens/total": 9360928, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 134108 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.019195031374692917, + "learning_rate": 4.791744483460251e-05, + "loss": 9.439548011869192e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00009, + "step": 294, + "tokens/total": 9393040, + "tokens/train_per_sec_per_gpu": 25.71, + "tokens/trainable": 134541 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.1602204144001007, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0004640200058929622, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00046, + "step": 295, + "tokens/total": 9425248, + "tokens/train_per_sec_per_gpu": 24.5, + "tokens/trainable": 134966 + }, + { + "epoch": 1.15625, + "grad_norm": 0.5406288504600525, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.0034592358861118555, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00347, + "step": 296, + "tokens/total": 9457264, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 135402 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.004333238583058119, + "learning_rate": 4.705880619288153e-05, + "loss": 3.075868880841881e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00003, + "step": 297, + "tokens/total": 9489056, + "tokens/train_per_sec_per_gpu": 24.87, + "tokens/trainable": 135804 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.006247827783226967, + "learning_rate": 4.677323837567412e-05, + "loss": 2.383375249337405e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00002, + "step": 298, + "tokens/total": 9521024, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 136231 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.0779612734913826, + "learning_rate": 4.6488012907598146e-05, + "loss": 0.0004679379053413868, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00047, + "step": 299, + "tokens/total": 9553120, + "tokens/train_per_sec_per_gpu": 28.37, + "tokens/trainable": 136705 + }, + { + "epoch": 1.171875, + "grad_norm": 0.01816786825656891, + "learning_rate": 4.620314165804964e-05, + "loss": 6.801338167861104e-05, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00007, + "step": 300, + "tokens/total": 9585040, + "tokens/train_per_sec_per_gpu": 28.78, + "tokens/trainable": 137178 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.9530329704284668, + "learning_rate": 4.591863648168407e-05, + "loss": 0.008843549527227879, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00888, + "step": 301, + "tokens/total": 9617120, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 137643 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0371168851852417, + "learning_rate": 4.5634509217923135e-05, + "loss": 0.00014635953994002193, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00015, + "step": 302, + "tokens/total": 9649024, + "tokens/train_per_sec_per_gpu": 26.74, + "tokens/trainable": 138078 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.08589725196361542, + "learning_rate": 4.535077169046201e-05, + "loss": 0.00043714733328670263, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00044, + "step": 303, + "tokens/total": 9680832, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 138515 + }, + { + "epoch": 1.1875, + "grad_norm": 0.358940988779068, + "learning_rate": 4.506743570677743e-05, + "loss": 0.006240838672965765, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00626, + "step": 304, + "tokens/total": 9713040, + "tokens/train_per_sec_per_gpu": 29.29, + "tokens/trainable": 138948 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.12659867107868195, + "learning_rate": 4.478451305763618e-05, + "loss": 0.0008667556685395539, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00087, + "step": 305, + "tokens/total": 9744976, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 139408 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.08289908617734909, + "learning_rate": 4.450201551660454e-05, + "loss": 0.0006384386215358973, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.02, + "memory/max_allocated (GiB)": 36.02, + "ppl": 1.00064, + "step": 306, + "tokens/total": 9774784, + "tokens/train_per_sec_per_gpu": 27.76, + "tokens/trainable": 139840 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.19195082783699036, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.002342985011637211, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00235, + "step": 307, + "tokens/total": 9806656, + "tokens/train_per_sec_per_gpu": 27.6, + "tokens/trainable": 140281 + }, + { + "epoch": 1.203125, + "grad_norm": 0.07421132177114487, + "learning_rate": 4.393834276419352e-05, + "loss": 0.0007085398538038135, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00071, + "step": 308, + "tokens/total": 9838720, + "tokens/train_per_sec_per_gpu": 30.88, + "tokens/trainable": 140776 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.004596009384840727, + "learning_rate": 4.36571910095382e-05, + "loss": 2.8969257982680574e-05, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00003, + "step": 309, + "tokens/total": 9870784, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 141228 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.04099060595035553, + "learning_rate": 4.337651127546448e-05, + "loss": 0.00030985785997472703, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00031, + "step": 310, + "tokens/total": 9902784, + "tokens/train_per_sec_per_gpu": 30.8, + "tokens/trainable": 141679 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.18840794265270233, + "learning_rate": 4.3096315242201736e-05, + "loss": 0.0020186181645840406, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00202, + "step": 311, + "tokens/total": 9934768, + "tokens/train_per_sec_per_gpu": 26.72, + "tokens/trainable": 142122 + }, + { + "epoch": 1.21875, + "grad_norm": 0.5018841028213501, + "learning_rate": 4.2816614569850635e-05, + "loss": 0.006148474756628275, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00617, + "step": 312, + "tokens/total": 9966960, + "tokens/train_per_sec_per_gpu": 27.27, + "tokens/trainable": 142561 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.1730002760887146, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.003579454030841589, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00359, + "step": 313, + "tokens/total": 9999104, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 143046 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.6586350202560425, + "learning_rate": 4.225874584473174e-05, + "loss": 0.0015231478027999401, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00152, + "step": 314, + "tokens/total": 10031136, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 143493 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.011105087585747242, + "learning_rate": 4.19806010071587e-05, + "loss": 0.00011838486534543335, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00012, + "step": 315, + "tokens/total": 10063184, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 143956 + }, + { + "epoch": 1.234375, + "grad_norm": 0.2628834843635559, + "learning_rate": 4.170299795992081e-05, + "loss": 0.003402995876967907, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00341, + "step": 316, + "tokens/total": 10095312, + "tokens/train_per_sec_per_gpu": 28.29, + "tokens/trainable": 144411 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.13821958005428314, + "learning_rate": 4.142594825521398e-05, + "loss": 0.0016696923412382603, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00167, + "step": 317, + "tokens/total": 10127440, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 144892 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.0023418583441525698, + "learning_rate": 4.114946342220728e-05, + "loss": 2.6753859856398776e-05, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00003, + "step": 318, + "tokens/total": 10159184, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 145339 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.0008433983894065022, + "learning_rate": 4.087355496656321e-05, + "loss": 1.602486736373976e-05, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00002, + "step": 319, + "tokens/total": 10191216, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 145811 + }, + { + "epoch": 1.25, + "grad_norm": 0.12588869035243988, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0011584729654714465, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00116, + "step": 320, + "tokens/total": 10223408, + "tokens/train_per_sec_per_gpu": 32.27, + "tokens/trainable": 146314 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.028395529836416245, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.00015757072833366692, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00016, + "step": 321, + "tokens/total": 10255536, + "tokens/train_per_sec_per_gpu": 26.68, + "tokens/trainable": 146781 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.039599668234586716, + "learning_rate": 4.004940255778431e-05, + "loss": 0.000321136845741421, + "memory/device_reserved (GiB)": 37.39, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00032, + "step": 322, + "tokens/total": 10287664, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 147225 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.11097095906734467, + "learning_rate": 3.977591418134619e-05, + "loss": 0.0004549310833681375, + "memory/device_reserved (GiB)": 37.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00046, + "step": 323, + "tokens/total": 10320080, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 147673 + }, + { + "epoch": 1.265625, + "grad_norm": 0.19018623232841492, + "learning_rate": 3.95030593412612e-05, + "loss": 0.0028336881659924984, + "memory/device_reserved (GiB)": 38.53, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00284, + "step": 324, + "tokens/total": 10351984, + "tokens/train_per_sec_per_gpu": 27.33, + "tokens/trainable": 148103 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.4650349020957947, + "learning_rate": 3.923084939213296e-05, + "loss": 0.006978687364608049, + "memory/device_reserved (GiB)": 38.53, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.007, + "step": 325, + "tokens/total": 10384256, + "tokens/train_per_sec_per_gpu": 26.21, + "tokens/trainable": 148535 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.14811205863952637, + "learning_rate": 3.895929566172861e-05, + "loss": 0.0014966176822781563, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0015, + "step": 326, + "tokens/total": 10416208, + "tokens/train_per_sec_per_gpu": 28.37, + "tokens/trainable": 148999 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.11302439123392105, + "learning_rate": 3.868840945050728e-05, + "loss": 0.0011318891774863005, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00113, + "step": 327, + "tokens/total": 10448224, + "tokens/train_per_sec_per_gpu": 26.37, + "tokens/trainable": 149431 + }, + { + "epoch": 1.28125, + "grad_norm": 0.10916705429553986, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0009605163359083235, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00096, + "step": 328, + "tokens/total": 10480448, + "tokens/train_per_sec_per_gpu": 28.27, + "tokens/trainable": 149886 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.0004018457839265466, + "learning_rate": 3.814868464809027e-05, + "loss": 7.040077434794512e-06, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00001, + "step": 329, + "tokens/total": 10512096, + "tokens/train_per_sec_per_gpu": 25.92, + "tokens/trainable": 150288 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.03717898577451706, + "learning_rate": 3.787986851704667e-05, + "loss": 0.0001680658315308392, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00017, + "step": 330, + "tokens/total": 10544000, + "tokens/train_per_sec_per_gpu": 28.52, + "tokens/trainable": 150733 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.21975846588611603, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00015534088015556335, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00016, + "step": 331, + "tokens/total": 10576048, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 151207 + }, + { + "epoch": 1.296875, + "grad_norm": 0.10715529322624207, + "learning_rate": 3.734438472750619e-05, + "loss": 0.0012046336196362972, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00121, + "step": 332, + "tokens/total": 10608048, + "tokens/train_per_sec_per_gpu": 31.08, + "tokens/trainable": 151694 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.13602837920188904, + "learning_rate": 3.707773935267552e-05, + "loss": 0.00039335948531515896, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00039, + "step": 333, + "tokens/total": 10640176, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 152188 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.002963080769404769, + "learning_rate": 3.68118397962661e-05, + "loss": 6.6289721871726215e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00001, + "step": 334, + "tokens/total": 10672224, + "tokens/train_per_sec_per_gpu": 25.55, + "tokens/trainable": 152596 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.007236327510327101, + "learning_rate": 3.654669712344384e-05, + "loss": 5.341085125110112e-05, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00005, + "step": 335, + "tokens/total": 10704192, + "tokens/train_per_sec_per_gpu": 30.44, + "tokens/trainable": 153052 + }, + { + "epoch": 1.3125, + "grad_norm": 0.0897810310125351, + "learning_rate": 3.628232236787763e-05, + "loss": 0.0008687702938914299, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00087, + "step": 336, + "tokens/total": 10736272, + "tokens/train_per_sec_per_gpu": 25.33, + "tokens/trainable": 153491 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.011829929426312447, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.00011201453162357211, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00011, + "step": 337, + "tokens/total": 10768528, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 153983 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.25218063592910767, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0022679665125906467, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00227, + "step": 338, + "tokens/total": 10800464, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 154464 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.00034703267738223076, + "learning_rate": 3.549391545931585e-05, + "loss": 2.884747800635523e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.0, + "step": 339, + "tokens/total": 10832512, + "tokens/train_per_sec_per_gpu": 29.7, + "tokens/trainable": 154924 + }, + { + "epoch": 1.328125, + "grad_norm": 0.004890776239335537, + "learning_rate": 3.5232722063479914e-05, + "loss": 2.2697908207192086e-05, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00002, + "step": 340, + "tokens/total": 10864624, + "tokens/train_per_sec_per_gpu": 25.46, + "tokens/trainable": 155373 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.0013245025184005499, + "learning_rate": 3.49723512647657e-05, + "loss": 8.980642633105163e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00001, + "step": 341, + "tokens/total": 10896816, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 155873 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.03804129734635353, + "learning_rate": 3.471281389826491e-05, + "loss": 0.00021880728309042752, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00022, + "step": 342, + "tokens/total": 10928864, + "tokens/train_per_sec_per_gpu": 25.31, + "tokens/trainable": 156278 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.4880214333534241, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.005397590808570385, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00541, + "step": 343, + "tokens/total": 10960816, + "tokens/train_per_sec_per_gpu": 29.58, + "tokens/trainable": 156733 + }, + { + "epoch": 1.34375, + "grad_norm": 0.37780871987342834, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0029298821464180946, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00293, + "step": 344, + "tokens/total": 10992912, + "tokens/train_per_sec_per_gpu": 28.4, + "tokens/trainable": 157203 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.0034261250402778387, + "learning_rate": 3.3939310220027456e-05, + "loss": 2.1256186300888658e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00002, + "step": 345, + "tokens/total": 11024832, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 157707 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.0005507867899723351, + "learning_rate": 3.3683214232914404e-05, + "loss": 6.991718692006543e-06, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00001, + "step": 346, + "tokens/total": 11056992, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 158180 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.0016383324982598424, + "learning_rate": 3.342800532426873e-05, + "loss": 1.2630136552616023e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00001, + "step": 347, + "tokens/total": 11089264, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 158646 + }, + { + "epoch": 1.359375, + "grad_norm": 0.0020912184845656157, + "learning_rate": 3.317369411437484e-05, + "loss": 1.536883064545691e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00002, + "step": 348, + "tokens/total": 11120976, + "tokens/train_per_sec_per_gpu": 29.34, + "tokens/trainable": 159115 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.0016828920925036073, + "learning_rate": 3.292029118616024e-05, + "loss": 1.334734679403482e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00001, + "step": 349, + "tokens/total": 11152976, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 159538 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.02520265430212021, + "learning_rate": 3.266780708475511e-05, + "loss": 9.595962183084339e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0001, + "step": 350, + "tokens/total": 11185056, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 160016 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.009836713783442974, + "learning_rate": 3.241625231705354e-05, + "loss": 7.521865336457267e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00008, + "step": 351, + "tokens/total": 11217088, + "tokens/train_per_sec_per_gpu": 28.51, + "tokens/trainable": 160475 + }, + { + "epoch": 1.375, + "grad_norm": 0.0698060467839241, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0003666624252218753, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00037, + "step": 352, + "tokens/total": 11249088, + "tokens/train_per_sec_per_gpu": 28.31, + "tokens/trainable": 160952 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.0018428800394758582, + "learning_rate": 3.191597261653475e-05, + "loss": 1.3194729945098516e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00001, + "step": 353, + "tokens/total": 11281328, + "tokens/train_per_sec_per_gpu": 27.4, + "tokens/trainable": 161420 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.020005319267511368, + "learning_rate": 3.166726850239794e-05, + "loss": 0.00017299644241575152, + "memory/device_reserved (GiB)": 38.47, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00017, + "step": 354, + "tokens/total": 11313200, + "tokens/train_per_sec_per_gpu": 28.31, + "tokens/trainable": 161868 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.00018166302470490336, + "learning_rate": 3.141953535845912e-05, + "loss": 4.138089934713207e-06, + "memory/device_reserved (GiB)": 38.47, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0, + "step": 355, + "tokens/total": 11345008, + "tokens/train_per_sec_per_gpu": 29.74, + "tokens/trainable": 162318 + }, + { + "epoch": 1.390625, + "grad_norm": 0.009165152907371521, + "learning_rate": 3.11727834939056e-05, + "loss": 6.729022425133735e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00007, + "step": 356, + "tokens/total": 11377216, + "tokens/train_per_sec_per_gpu": 26.26, + "tokens/trainable": 162789 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.2891274690628052, + "learning_rate": 3.092702317708967e-05, + "loss": 0.005585570354014635, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.0056, + "step": 357, + "tokens/total": 11409536, + "tokens/train_per_sec_per_gpu": 30.34, + "tokens/trainable": 163254 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.0007154783816076815, + "learning_rate": 3.0682264635101276e-05, + "loss": 5.591478839050978e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00001, + "step": 358, + "tokens/total": 11441552, + "tokens/train_per_sec_per_gpu": 28.74, + "tokens/trainable": 163715 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.01731656678020954, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.0001050045175361447, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00011, + "step": 359, + "tokens/total": 11473376, + "tokens/train_per_sec_per_gpu": 28.79, + "tokens/trainable": 164197 + }, + { + "epoch": 1.40625, + "grad_norm": 0.11494546383619308, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.00048459687968716025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00048, + "step": 360, + "tokens/total": 11505312, + "tokens/train_per_sec_per_gpu": 26.54, + "tokens/trainable": 164661 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.08215854316949844, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.0002725277154240757, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00027, + "step": 361, + "tokens/total": 11537536, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 165111 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.024266909807920456, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.0001677784021012485, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 35.94, + "memory/max_allocated (GiB)": 35.94, + "ppl": 1.00017, + "step": 362, + "tokens/total": 11567216, + "tokens/train_per_sec_per_gpu": 25.3, + "tokens/trainable": 165552 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.008205167017877102, + "learning_rate": 2.9473853553877484e-05, + "loss": 3.423610905883834e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00003, + "step": 363, + "tokens/total": 11599312, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 166026 + }, + { + "epoch": 1.421875, + "grad_norm": 0.4230109453201294, + "learning_rate": 2.9235318065647e-05, + "loss": 0.005893740337342024, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00591, + "step": 364, + "tokens/total": 11631344, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 166486 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.1095249131321907, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.0006186347454786301, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00062, + "step": 365, + "tokens/total": 11663632, + "tokens/train_per_sec_per_gpu": 30.17, + "tokens/trainable": 166959 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.0014034683117642999, + "learning_rate": 2.8761473491751258e-05, + "loss": 1.2728614819934592e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.27, + "memory/max_allocated (GiB)": 36.27, + "ppl": 1.00001, + "step": 366, + "tokens/total": 11695360, + "tokens/train_per_sec_per_gpu": 28.19, + "tokens/trainable": 167394 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.044087354093790054, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.0001466910180170089, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00015, + "step": 367, + "tokens/total": 11727184, + "tokens/train_per_sec_per_gpu": 23.64, + "tokens/trainable": 167832 + }, + { + "epoch": 1.4375, + "grad_norm": 0.0005008528823964298, + "learning_rate": 2.829199644117484e-05, + "loss": 1.1319095392536838e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00001, + "step": 368, + "tokens/total": 11759312, + "tokens/train_per_sec_per_gpu": 25.79, + "tokens/trainable": 168242 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.02015153504908085, + "learning_rate": 2.8058920186702553e-05, + "loss": 0.00011158352572238073, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00011, + "step": 369, + "tokens/total": 11791536, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 168715 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.00031807494815438986, + "learning_rate": 2.782696506053033e-05, + "loss": 6.533119631058071e-06, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00001, + "step": 370, + "tokens/total": 11823632, + "tokens/train_per_sec_per_gpu": 26.68, + "tokens/trainable": 169147 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.4510185122489929, + "learning_rate": 2.7596140715257824e-05, + "loss": 0.004431337118148804, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00444, + "step": 371, + "tokens/total": 11855392, + "tokens/train_per_sec_per_gpu": 28.61, + "tokens/trainable": 169628 + }, + { + "epoch": 1.453125, + "grad_norm": 0.0034806979820132256, + "learning_rate": 2.7366456756428184e-05, + "loss": 2.722550561884418e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00003, + "step": 372, + "tokens/total": 11887424, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 170085 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.018526704981923103, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.00013849925016984344, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00014, + "step": 373, + "tokens/total": 11919280, + "tokens/train_per_sec_per_gpu": 29.39, + "tokens/trainable": 170584 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.010734346695244312, + "learning_rate": 2.691054818259188e-05, + "loss": 1.963317481568083e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00002, + "step": 374, + "tokens/total": 11951488, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 171058 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.004906357266008854, + "learning_rate": 2.6684342539801933e-05, + "loss": 4.44888137280941e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00004, + "step": 375, + "tokens/total": 11983504, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 171518 + }, + { + "epoch": 1.46875, + "grad_norm": 0.0035627244506031275, + "learning_rate": 2.645931522709877e-05, + "loss": 1.879796946013812e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00002, + "step": 376, + "tokens/total": 12015360, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 172007 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.014949376694858074, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.0001050513528753072, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00011, + "step": 377, + "tokens/total": 12047296, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 172468 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.024432353675365448, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.00010428359382785857, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0001, + "step": 378, + "tokens/total": 12079392, + "tokens/train_per_sec_per_gpu": 27.91, + "tokens/trainable": 172923 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.5905564427375793, + "learning_rate": 2.579139666504821e-05, + "loss": 0.019045401364564896, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01923, + "step": 379, + "tokens/total": 12111616, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 173370 + }, + { + "epoch": 1.484375, + "grad_norm": 0.0027490397915244102, + "learning_rate": 2.557117581955798e-05, + "loss": 3.106705116806552e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00003, + "step": 380, + "tokens/total": 12143536, + "tokens/train_per_sec_per_gpu": 28.66, + "tokens/trainable": 173837 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.10220319032669067, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.0007754361140541732, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00078, + "step": 381, + "tokens/total": 12175792, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 174294 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.06388501822948456, + "learning_rate": 2.5134417202396277e-05, + "loss": 2.506363671272993e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00003, + "step": 382, + "tokens/total": 12207472, + "tokens/train_per_sec_per_gpu": 25.85, + "tokens/trainable": 174701 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.006029566749930382, + "learning_rate": 2.491789760603361e-05, + "loss": 5.214785414864309e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00005, + "step": 383, + "tokens/total": 12239520, + "tokens/train_per_sec_per_gpu": 26.67, + "tokens/trainable": 175131 + }, + { + "epoch": 1.5, + "grad_norm": 0.015079713426530361, + "learning_rate": 2.4702629848741764e-05, + "loss": 0.00011988497863058001, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00012, + "step": 384, + "tokens/total": 12271600, + "tokens/train_per_sec_per_gpu": 28.23, + "tokens/trainable": 175586 + }, + { + "epoch": 1.50390625, + "grad_norm": 0.04362509027123451, + "learning_rate": 2.4488622888690785e-05, + "loss": 0.0002582473389338702, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00026, + "step": 385, + "tokens/total": 12303648, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 176026 + }, + { + "epoch": 1.5078125, + "grad_norm": 0.28308039903640747, + "learning_rate": 2.427588563158384e-05, + "loss": 0.0025924108922481537, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0026, + "step": 386, + "tokens/total": 12335728, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 176512 + }, + { + "epoch": 1.51171875, + "grad_norm": 0.21685272455215454, + "learning_rate": 2.406442693028651e-05, + "loss": 0.002609315561130643, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00261, + "step": 387, + "tokens/total": 12367824, + "tokens/train_per_sec_per_gpu": 25.55, + "tokens/trainable": 176943 + }, + { + "epoch": 1.515625, + "grad_norm": 0.0010271539213135839, + "learning_rate": 2.3854255584458547e-05, + "loss": 2.1533389372052625e-05, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00002, + "step": 388, + "tokens/total": 12400192, + "tokens/train_per_sec_per_gpu": 26.36, + "tokens/trainable": 177370 + }, + { + "epoch": 1.51953125, + "grad_norm": 0.06616228073835373, + "learning_rate": 2.3645380340187508e-05, + "loss": 0.0005350579158402979, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00054, + "step": 389, + "tokens/total": 12432224, + "tokens/train_per_sec_per_gpu": 29.11, + "tokens/trainable": 177851 + }, + { + "epoch": 1.5234375, + "grad_norm": 0.05946248397231102, + "learning_rate": 2.3437809889624914e-05, + "loss": 0.0005003588157705963, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.0005, + "step": 390, + "tokens/total": 12464048, + "tokens/train_per_sec_per_gpu": 29.09, + "tokens/trainable": 178310 + }, + { + "epoch": 1.52734375, + "grad_norm": 0.017372427508234978, + "learning_rate": 2.3231552870624487e-05, + "loss": 0.0001355188578600064, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00014, + "step": 391, + "tokens/total": 12493776, + "tokens/train_per_sec_per_gpu": 30.96, + "tokens/trainable": 178736 + }, + { + "epoch": 1.53125, + "grad_norm": 0.12941808998584747, + "learning_rate": 2.3026617866382657e-05, + "loss": 0.003045230871066451, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00305, + "step": 392, + "tokens/total": 12526000, + "tokens/train_per_sec_per_gpu": 30.37, + "tokens/trainable": 179233 + }, + { + "epoch": 1.53515625, + "grad_norm": 0.014790852554142475, + "learning_rate": 2.2823013405081507e-05, + "loss": 0.00013205324648879468, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00013, + "step": 393, + "tokens/total": 12557968, + "tokens/train_per_sec_per_gpu": 32.27, + "tokens/trainable": 179730 + }, + { + "epoch": 1.5390625, + "grad_norm": 0.02042844519019127, + "learning_rate": 2.2620747959533722e-05, + "loss": 0.00022655579959973693, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00023, + "step": 394, + "tokens/total": 12590096, + "tokens/train_per_sec_per_gpu": 31.47, + "tokens/trainable": 180227 + }, + { + "epoch": 1.54296875, + "grad_norm": 0.005440943408757448, + "learning_rate": 2.2419829946830123e-05, + "loss": 4.6342807763721794e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00005, + "step": 395, + "tokens/total": 12622160, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 180687 + }, + { + "epoch": 1.546875, + "grad_norm": 0.004878366366028786, + "learning_rate": 2.2220267727989325e-05, + "loss": 5.4212974646361545e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00005, + "step": 396, + "tokens/total": 12654336, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 181141 + }, + { + "epoch": 1.55078125, + "grad_norm": 0.20873922109603882, + "learning_rate": 2.202206960760984e-05, + "loss": 0.0026359721086919308, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00264, + "step": 397, + "tokens/total": 12686400, + "tokens/train_per_sec_per_gpu": 30.13, + "tokens/trainable": 181648 + }, + { + "epoch": 1.5546875, + "grad_norm": 0.12369472533464432, + "learning_rate": 2.182524383352446e-05, + "loss": 0.0010878165485337377, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.00109, + "step": 398, + "tokens/total": 12718144, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 182109 + }, + { + "epoch": 1.55859375, + "grad_norm": 0.0010841770563274622, + "learning_rate": 2.1629798596457056e-05, + "loss": 1.8380069377599284e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00002, + "step": 399, + "tokens/total": 12750256, + "tokens/train_per_sec_per_gpu": 26.26, + "tokens/trainable": 182559 + }, + { + "epoch": 1.5625, + "grad_norm": 0.0010006122756749392, + "learning_rate": 2.1435742029681725e-05, + "loss": 1.991226599784568e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00002, + "step": 400, + "tokens/total": 12782224, + "tokens/train_per_sec_per_gpu": 27.59, + "tokens/trainable": 182985 + }, + { + "epoch": 1.56640625, + "grad_norm": 0.0011391225270926952, + "learning_rate": 2.124308220868431e-05, + "loss": 2.2474639990832657e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00002, + "step": 401, + "tokens/total": 12814384, + "tokens/train_per_sec_per_gpu": 26.95, + "tokens/trainable": 183440 + }, + { + "epoch": 1.5703125, + "grad_norm": 0.023268507793545723, + "learning_rate": 2.105182715082638e-05, + "loss": 0.00032951770117506385, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00033, + "step": 402, + "tokens/total": 12846240, + "tokens/train_per_sec_per_gpu": 28.77, + "tokens/trainable": 183917 + }, + { + "epoch": 1.57421875, + "grad_norm": 0.11701280623674393, + "learning_rate": 2.0861984815011552e-05, + "loss": 0.001444364432245493, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00145, + "step": 403, + "tokens/total": 12878240, + "tokens/train_per_sec_per_gpu": 25.07, + "tokens/trainable": 184348 + }, + { + "epoch": 1.578125, + "grad_norm": 0.004240577574819326, + "learning_rate": 2.0673563101354323e-05, + "loss": 2.100174970109947e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00002, + "step": 404, + "tokens/total": 12910176, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 184786 + }, + { + "epoch": 1.58203125, + "grad_norm": 0.01977616548538208, + "learning_rate": 2.0486569850851317e-05, + "loss": 8.987118053482845e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00009, + "step": 405, + "tokens/total": 12942352, + "tokens/train_per_sec_per_gpu": 28.17, + "tokens/trainable": 185249 + }, + { + "epoch": 1.5859375, + "grad_norm": 0.030095387250185013, + "learning_rate": 2.0301012845054956e-05, + "loss": 0.0002793738676700741, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00028, + "step": 406, + "tokens/total": 12974368, + "tokens/train_per_sec_per_gpu": 26.3, + "tokens/trainable": 185680 + }, + { + "epoch": 1.58984375, + "grad_norm": 0.0017381443176418543, + "learning_rate": 2.011689980574966e-05, + "loss": 2.727654828049708e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00003, + "step": 407, + "tokens/total": 13006464, + "tokens/train_per_sec_per_gpu": 26.25, + "tokens/trainable": 186100 + }, + { + "epoch": 1.59375, + "grad_norm": 0.09072195738554001, + "learning_rate": 1.993423839463052e-05, + "loss": 0.0006338813109323382, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00063, + "step": 408, + "tokens/total": 13038320, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 186565 + }, + { + "epoch": 1.59765625, + "grad_norm": 0.004482298158109188, + "learning_rate": 1.975303621298445e-05, + "loss": 5.0280330469831824e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00005, + "step": 409, + "tokens/total": 13070416, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 187017 + }, + { + "epoch": 1.6015625, + "grad_norm": 0.03788210079073906, + "learning_rate": 1.957330080137385e-05, + "loss": 0.000300457701086998, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0003, + "step": 410, + "tokens/total": 13102352, + "tokens/train_per_sec_per_gpu": 27.69, + "tokens/trainable": 187468 + }, + { + "epoch": 1.60546875, + "grad_norm": 0.011773839592933655, + "learning_rate": 1.9395039639322864e-05, + "loss": 9.259363287128508e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00009, + "step": 411, + "tokens/total": 13134096, + "tokens/train_per_sec_per_gpu": 31.06, + "tokens/trainable": 187961 + }, + { + "epoch": 1.609375, + "grad_norm": 0.0031615474727004766, + "learning_rate": 1.9218260145006073e-05, + "loss": 3.231812661397271e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00003, + "step": 412, + "tokens/total": 13166256, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 188447 + }, + { + "epoch": 1.61328125, + "grad_norm": 0.07778825610876083, + "learning_rate": 1.904296967493982e-05, + "loss": 0.0005980221321806312, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.0006, + "step": 413, + "tokens/total": 13198048, + "tokens/train_per_sec_per_gpu": 29.98, + "tokens/trainable": 188903 + }, + { + "epoch": 1.6171875, + "grad_norm": 0.011902395635843277, + "learning_rate": 1.8869175523676064e-05, + "loss": 0.00013689698243979365, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.66, + "memory/max_allocated (GiB)": 36.66, + "ppl": 1.00014, + "step": 414, + "tokens/total": 13230304, + "tokens/train_per_sec_per_gpu": 30.8, + "tokens/trainable": 189398 + }, + { + "epoch": 1.62109375, + "grad_norm": 0.08490622788667679, + "learning_rate": 1.869688492349885e-05, + "loss": 0.0008053273777477443, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00081, + "step": 415, + "tokens/total": 13262288, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 189850 + }, + { + "epoch": 1.625, + "grad_norm": 0.11783187091350555, + "learning_rate": 1.85261050441233e-05, + "loss": 0.0007362872711382806, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00074, + "step": 416, + "tokens/total": 13294480, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 190306 + }, + { + "epoch": 1.62890625, + "grad_norm": 0.26603376865386963, + "learning_rate": 1.8356842992397304e-05, + "loss": 0.002245941199362278, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00225, + "step": 417, + "tokens/total": 13326608, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 190746 + }, + { + "epoch": 1.6328125, + "grad_norm": 0.003691659774631262, + "learning_rate": 1.8189105812005714e-05, + "loss": 4.167412407696247e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00004, + "step": 418, + "tokens/total": 13358592, + "tokens/train_per_sec_per_gpu": 26.1, + "tokens/trainable": 191156 + }, + { + "epoch": 1.63671875, + "grad_norm": 0.021083569154143333, + "learning_rate": 1.802290048317732e-05, + "loss": 0.00017740413022693247, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00018, + "step": 419, + "tokens/total": 13390592, + "tokens/train_per_sec_per_gpu": 24.49, + "tokens/trainable": 191573 + }, + { + "epoch": 1.640625, + "grad_norm": 0.003719380358234048, + "learning_rate": 1.785823392239424e-05, + "loss": 4.420262484927662e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00004, + "step": 420, + "tokens/total": 13422672, + "tokens/train_per_sec_per_gpu": 27.22, + "tokens/trainable": 192014 + }, + { + "epoch": 1.64453125, + "grad_norm": 0.044042862951755524, + "learning_rate": 1.7695112982104225e-05, + "loss": 0.0004268632619641721, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00043, + "step": 421, + "tokens/total": 13454624, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 192453 + }, + { + "epoch": 1.6484375, + "grad_norm": 0.0021698337513953447, + "learning_rate": 1.7533544450435433e-05, + "loss": 2.580507134553045e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.31, + "memory/max_allocated (GiB)": 36.31, + "ppl": 1.00003, + "step": 422, + "tokens/total": 13486208, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 192886 + }, + { + "epoch": 1.65234375, + "grad_norm": 0.019405441358685493, + "learning_rate": 1.7373535050913946e-05, + "loss": 0.00013853301061317325, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.71, + "memory/max_allocated (GiB)": 36.71, + "ppl": 1.00014, + "step": 423, + "tokens/total": 13518496, + "tokens/train_per_sec_per_gpu": 25.15, + "tokens/trainable": 193323 + }, + { + "epoch": 1.65625, + "grad_norm": 0.005207414738833904, + "learning_rate": 1.721509144218405e-05, + "loss": 6.985871004872024e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00007, + "step": 424, + "tokens/total": 13550768, + "tokens/train_per_sec_per_gpu": 27.59, + "tokens/trainable": 193753 + }, + { + "epoch": 1.66015625, + "grad_norm": 0.01700798235833645, + "learning_rate": 1.705822021773101e-05, + "loss": 0.00024273117014672607, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00024, + "step": 425, + "tokens/total": 13582960, + "tokens/train_per_sec_per_gpu": 30.03, + "tokens/trainable": 194240 + }, + { + "epoch": 1.6640625, + "grad_norm": 0.2864547073841095, + "learning_rate": 1.69029279056068e-05, + "loss": 0.004108238499611616, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00412, + "step": 426, + "tokens/total": 13615056, + "tokens/train_per_sec_per_gpu": 29.6, + "tokens/trainable": 194680 + }, + { + "epoch": 1.66796875, + "grad_norm": 0.012734112329781055, + "learning_rate": 1.6749220968158415e-05, + "loss": 9.277237404603511e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00009, + "step": 427, + "tokens/total": 13647088, + "tokens/train_per_sec_per_gpu": 32.17, + "tokens/trainable": 195186 + }, + { + "epoch": 1.671875, + "grad_norm": 0.04110024869441986, + "learning_rate": 1.659710580175893e-05, + "loss": 0.00021035004465375096, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00021, + "step": 428, + "tokens/total": 13679168, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 195693 + }, + { + "epoch": 1.67578125, + "grad_norm": 0.03119366057217121, + "learning_rate": 1.644658873654133e-05, + "loss": 0.00022089436242822558, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00022, + "step": 429, + "tokens/total": 13711120, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 196130 + }, + { + "epoch": 1.6796875, + "grad_norm": 0.0020074129570275545, + "learning_rate": 1.629767603613508e-05, + "loss": 1.550407614558935e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00002, + "step": 430, + "tokens/total": 13743088, + "tokens/train_per_sec_per_gpu": 24.67, + "tokens/trainable": 196530 + }, + { + "epoch": 1.68359375, + "grad_norm": 0.040013547986745834, + "learning_rate": 1.615037389740547e-05, + "loss": 0.0006465734331868589, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00065, + "step": 431, + "tokens/total": 13774912, + "tokens/train_per_sec_per_gpu": 25.06, + "tokens/trainable": 196974 + }, + { + "epoch": 1.6875, + "grad_norm": 0.0517231747508049, + "learning_rate": 1.600468845019576e-05, + "loss": 0.00047026947140693665, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00047, + "step": 432, + "tokens/total": 13806800, + "tokens/train_per_sec_per_gpu": 29.38, + "tokens/trainable": 197454 + }, + { + "epoch": 1.69140625, + "grad_norm": 0.0004246715398039669, + "learning_rate": 1.5860625757072092e-05, + "loss": 7.613366506120656e-06, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00001, + "step": 433, + "tokens/total": 13838704, + "tokens/train_per_sec_per_gpu": 26.36, + "tokens/trainable": 197902 + }, + { + "epoch": 1.6953125, + "grad_norm": 0.0011862553656101227, + "learning_rate": 1.571819181307116e-05, + "loss": 1.5171106497291476e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00002, + "step": 434, + "tokens/total": 13870640, + "tokens/train_per_sec_per_gpu": 29.63, + "tokens/trainable": 198374 + }, + { + "epoch": 1.69921875, + "grad_norm": 0.0019507558317855, + "learning_rate": 1.557739254545075e-05, + "loss": 1.7753134670783766e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00002, + "step": 435, + "tokens/total": 13902672, + "tokens/train_per_sec_per_gpu": 31.27, + "tokens/trainable": 198846 + }, + { + "epoch": 1.703125, + "grad_norm": 0.014270462095737457, + "learning_rate": 1.543823381344311e-05, + "loss": 0.0001223236322402954, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00012, + "step": 436, + "tokens/total": 13935008, + "tokens/train_per_sec_per_gpu": 31.41, + "tokens/trainable": 199320 + }, + { + "epoch": 1.70703125, + "grad_norm": 0.020011236891150475, + "learning_rate": 1.5300721408011114e-05, + "loss": 0.00015811943740118295, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00016, + "step": 437, + "tokens/total": 13967024, + "tokens/train_per_sec_per_gpu": 31.19, + "tokens/trainable": 199777 + }, + { + "epoch": 1.7109375, + "grad_norm": 0.0006321229157038033, + "learning_rate": 1.5164861051607254e-05, + "loss": 9.932198736350983e-06, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00001, + "step": 438, + "tokens/total": 13998688, + "tokens/train_per_sec_per_gpu": 25.77, + "tokens/trainable": 200194 + }, + { + "epoch": 1.71484375, + "grad_norm": 0.002675980795174837, + "learning_rate": 1.5030658397935521e-05, + "loss": 2.0016837879666127e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00002, + "step": 439, + "tokens/total": 14030992, + "tokens/train_per_sec_per_gpu": 28.45, + "tokens/trainable": 200663 + }, + { + "epoch": 1.71875, + "grad_norm": 0.005228503607213497, + "learning_rate": 1.4898119031716104e-05, + "loss": 4.561378591461107e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00005, + "step": 440, + "tokens/total": 14063440, + "tokens/train_per_sec_per_gpu": 28.52, + "tokens/trainable": 201121 + }, + { + "epoch": 1.72265625, + "grad_norm": 0.015322903171181679, + "learning_rate": 1.476724846845306e-05, + "loss": 0.0001199191392515786, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00012, + "step": 441, + "tokens/total": 14095536, + "tokens/train_per_sec_per_gpu": 33.0, + "tokens/trainable": 201598 + }, + { + "epoch": 1.7265625, + "grad_norm": 0.05436247959733009, + "learning_rate": 1.463805215420471e-05, + "loss": 0.0005605737096630037, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00056, + "step": 442, + "tokens/total": 14127488, + "tokens/train_per_sec_per_gpu": 29.37, + "tokens/trainable": 202068 + }, + { + "epoch": 1.73046875, + "grad_norm": 0.005114950239658356, + "learning_rate": 1.451053546535705e-05, + "loss": 2.6398556656204164e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00003, + "step": 443, + "tokens/total": 14159472, + "tokens/train_per_sec_per_gpu": 29.43, + "tokens/trainable": 202524 + }, + { + "epoch": 1.734375, + "grad_norm": 0.000681015953887254, + "learning_rate": 1.438470370840001e-05, + "loss": 6.228779966477305e-06, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00001, + "step": 444, + "tokens/total": 14191520, + "tokens/train_per_sec_per_gpu": 30.27, + "tokens/trainable": 202996 + }, + { + "epoch": 1.73828125, + "grad_norm": 0.03594636172056198, + "learning_rate": 1.4260562119706606e-05, + "loss": 0.0003221426741220057, + "memory/device_reserved (GiB)": 39.16, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00032, + "step": 445, + "tokens/total": 14221680, + "tokens/train_per_sec_per_gpu": 32.89, + "tokens/trainable": 203470 + }, + { + "epoch": 1.7421875, + "grad_norm": 0.16816683113574982, + "learning_rate": 1.413811586531508e-05, + "loss": 0.0016490641282871366, + "memory/device_reserved (GiB)": 39.16, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00165, + "step": 446, + "tokens/total": 14253680, + "tokens/train_per_sec_per_gpu": 29.37, + "tokens/trainable": 203919 + }, + { + "epoch": 1.74609375, + "grad_norm": 0.0071723381988704205, + "learning_rate": 1.4017370040713884e-05, + "loss": 6.93315378157422e-05, + "memory/device_reserved (GiB)": 39.16, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00007, + "step": 447, + "tokens/total": 14285824, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 204376 + }, + { + "epoch": 1.75, + "grad_norm": 0.02812999114394188, + "learning_rate": 1.3898329670629645e-05, + "loss": 0.0001953901955857873, + "memory/device_reserved (GiB)": 39.16, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.0002, + "step": 448, + "tokens/total": 14317968, + "tokens/train_per_sec_per_gpu": 26.64, + "tokens/trainable": 204821 + }, + { + "epoch": 1.75390625, + "grad_norm": 0.003404787741601467, + "learning_rate": 1.3780999708818058e-05, + "loss": 7.17312059350661e-06, + "memory/device_reserved (GiB)": 39.16, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00001, + "step": 449, + "tokens/total": 14350080, + "tokens/train_per_sec_per_gpu": 28.3, + "tokens/trainable": 205271 + }, + { + "epoch": 1.7578125, + "grad_norm": 0.0005480491672642529, + "learning_rate": 1.3665385037857758e-05, + "loss": 4.81248343930929e-06, + "memory/device_reserved (GiB)": 37.41, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0, + "step": 450, + "tokens/total": 14382240, + "tokens/train_per_sec_per_gpu": 25.5, + "tokens/trainable": 205723 + }, + { + "epoch": 1.76171875, + "grad_norm": 0.014193670824170113, + "learning_rate": 1.3551490468947126e-05, + "loss": 5.97102043684572e-05, + "memory/device_reserved (GiB)": 37.63, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00006, + "step": 451, + "tokens/total": 14414320, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 206169 + }, + { + "epoch": 1.765625, + "grad_norm": 0.02763630636036396, + "learning_rate": 1.3439320741704075e-05, + "loss": 0.00014929058670531958, + "memory/device_reserved (GiB)": 37.76, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00015, + "step": 452, + "tokens/total": 14446464, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 206592 + }, + { + "epoch": 1.76953125, + "grad_norm": 0.010347527451813221, + "learning_rate": 1.3328880523968808e-05, + "loss": 5.998069536872208e-05, + "memory/device_reserved (GiB)": 38.45, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00006, + "step": 453, + "tokens/total": 14478624, + "tokens/train_per_sec_per_gpu": 27.59, + "tokens/trainable": 207055 + }, + { + "epoch": 1.7734375, + "grad_norm": 0.4689827561378479, + "learning_rate": 1.3220174411609587e-05, + "loss": 0.0033441532868891954, + "memory/device_reserved (GiB)": 38.45, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.00335, + "step": 454, + "tokens/total": 14510192, + "tokens/train_per_sec_per_gpu": 28.86, + "tokens/trainable": 207494 + }, + { + "epoch": 1.77734375, + "grad_norm": 0.04199036583304405, + "learning_rate": 1.3113206928331471e-05, + "loss": 0.0002628727233968675, + "memory/device_reserved (GiB)": 38.45, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00026, + "step": 455, + "tokens/total": 14541920, + "tokens/train_per_sec_per_gpu": 27.5, + "tokens/trainable": 207956 + }, + { + "epoch": 1.78125, + "grad_norm": 0.0015001439023762941, + "learning_rate": 1.300798252548806e-05, + "loss": 1.346761200693436e-05, + "memory/device_reserved (GiB)": 38.45, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00001, + "step": 456, + "tokens/total": 14573904, + "tokens/train_per_sec_per_gpu": 29.6, + "tokens/trainable": 208408 + }, + { + "epoch": 1.78515625, + "grad_norm": 0.013151494786143303, + "learning_rate": 1.2904505581896265e-05, + "loss": 3.6732359149027616e-05, + "memory/device_reserved (GiB)": 38.45, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00004, + "step": 457, + "tokens/total": 14606032, + "tokens/train_per_sec_per_gpu": 29.71, + "tokens/trainable": 208889 + }, + { + "epoch": 1.7890625, + "grad_norm": 0.0006910350639373064, + "learning_rate": 1.2802780403654082e-05, + "loss": 1.0416523764433805e-05, + "memory/device_reserved (GiB)": 38.45, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00001, + "step": 458, + "tokens/total": 14637728, + "tokens/train_per_sec_per_gpu": 28.36, + "tokens/trainable": 209336 + }, + { + "epoch": 1.79296875, + "grad_norm": 0.003835468553006649, + "learning_rate": 1.2702811223961408e-05, + "loss": 3.105392897850834e-05, + "memory/device_reserved (GiB)": 38.63, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00003, + "step": 459, + "tokens/total": 14669984, + "tokens/train_per_sec_per_gpu": 27.49, + "tokens/trainable": 209797 + }, + { + "epoch": 1.796875, + "grad_norm": 0.0003044170734938234, + "learning_rate": 1.2604602202943861e-05, + "loss": 6.697504431940615e-06, + "memory/device_reserved (GiB)": 38.67, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00001, + "step": 460, + "tokens/total": 14702064, + "tokens/train_per_sec_per_gpu": 30.05, + "tokens/trainable": 210240 + }, + { + "epoch": 1.80078125, + "grad_norm": 0.01640794239938259, + "learning_rate": 1.2508157427479686e-05, + "loss": 7.829760579625145e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00008, + "step": 461, + "tokens/total": 14734144, + "tokens/train_per_sec_per_gpu": 29.07, + "tokens/trainable": 210702 + }, + { + "epoch": 1.8046875, + "grad_norm": 0.04974092170596123, + "learning_rate": 1.2413480911029655e-05, + "loss": 0.00036814718623645604, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00037, + "step": 462, + "tokens/total": 14766416, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 211140 + }, + { + "epoch": 1.80859375, + "grad_norm": 0.0007990560261532664, + "learning_rate": 1.2320576593470082e-05, + "loss": 7.1813856266089715e-06, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00001, + "step": 463, + "tokens/total": 14798224, + "tokens/train_per_sec_per_gpu": 27.29, + "tokens/trainable": 211575 + }, + { + "epoch": 1.8125, + "grad_norm": 0.08440390229225159, + "learning_rate": 1.2229448340928828e-05, + "loss": 0.00048587226774543524, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00049, + "step": 464, + "tokens/total": 14830336, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 212083 + }, + { + "epoch": 1.81640625, + "grad_norm": 0.0018794891657307744, + "learning_rate": 1.2140099945624458e-05, + "loss": 1.08363010440371e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00001, + "step": 465, + "tokens/total": 14862560, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 212558 + }, + { + "epoch": 1.8203125, + "grad_norm": 0.008357501588761806, + "learning_rate": 1.205253512570841e-05, + "loss": 6.44918909529224e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00006, + "step": 466, + "tokens/total": 14894432, + "tokens/train_per_sec_per_gpu": 27.29, + "tokens/trainable": 213014 + }, + { + "epoch": 1.82421875, + "grad_norm": 0.003019345458596945, + "learning_rate": 1.1966757525110255e-05, + "loss": 2.4649680199217983e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00002, + "step": 467, + "tokens/total": 14926624, + "tokens/train_per_sec_per_gpu": 28.39, + "tokens/trainable": 213465 + }, + { + "epoch": 1.828125, + "grad_norm": 0.0832735076546669, + "learning_rate": 1.1882770713386095e-05, + "loss": 0.0008976737735792994, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0009, + "step": 468, + "tokens/total": 14958640, + "tokens/train_per_sec_per_gpu": 30.58, + "tokens/trainable": 213940 + }, + { + "epoch": 1.83203125, + "grad_norm": 0.0038668843917548656, + "learning_rate": 1.180057818556998e-05, + "loss": 1.873084511316847e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00002, + "step": 469, + "tokens/total": 14990864, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 214415 + }, + { + "epoch": 1.8359375, + "grad_norm": 0.0006158083560876548, + "learning_rate": 1.1720183362028494e-05, + "loss": 5.440972017822787e-06, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00001, + "step": 470, + "tokens/total": 15022800, + "tokens/train_per_sec_per_gpu": 28.77, + "tokens/trainable": 214880 + }, + { + "epoch": 1.83984375, + "grad_norm": 0.0015613064169883728, + "learning_rate": 1.1641589588318387e-05, + "loss": 1.487641657149652e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00001, + "step": 471, + "tokens/total": 15054784, + "tokens/train_per_sec_per_gpu": 27.81, + "tokens/trainable": 215334 + }, + { + "epoch": 1.84375, + "grad_norm": 0.00033517941483296454, + "learning_rate": 1.1564800135047418e-05, + "loss": 4.4488651838037185e-06, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0, + "step": 472, + "tokens/total": 15086800, + "tokens/train_per_sec_per_gpu": 29.18, + "tokens/trainable": 215822 + }, + { + "epoch": 1.84765625, + "grad_norm": 0.014701228588819504, + "learning_rate": 1.148981819773816e-05, + "loss": 0.00012692881864495575, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00013, + "step": 473, + "tokens/total": 15118736, + "tokens/train_per_sec_per_gpu": 25.65, + "tokens/trainable": 216239 + }, + { + "epoch": 1.8515625, + "grad_norm": 0.03242962434887886, + "learning_rate": 1.1416646896695086e-05, + "loss": 0.00016460703045595437, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00016, + "step": 474, + "tokens/total": 15150896, + "tokens/train_per_sec_per_gpu": 27.53, + "tokens/trainable": 216691 + }, + { + "epoch": 1.85546875, + "grad_norm": 0.022134000435471535, + "learning_rate": 1.1345289276874717e-05, + "loss": 0.00015468102355953306, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00015, + "step": 475, + "tokens/total": 15183120, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 217162 + }, + { + "epoch": 1.859375, + "grad_norm": 0.00101348920725286, + "learning_rate": 1.1275748307758873e-05, + "loss": 1.263963622477604e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00001, + "step": 476, + "tokens/total": 15215216, + "tokens/train_per_sec_per_gpu": 30.11, + "tokens/trainable": 217638 + }, + { + "epoch": 1.86328125, + "grad_norm": 0.09451806545257568, + "learning_rate": 1.1208026883231147e-05, + "loss": 0.0005900258547626436, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00059, + "step": 477, + "tokens/total": 15247328, + "tokens/train_per_sec_per_gpu": 27.45, + "tokens/trainable": 218053 + }, + { + "epoch": 1.8671875, + "grad_norm": 0.0023683386389166117, + "learning_rate": 1.1142127821456433e-05, + "loss": 1.8595857909531333e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00002, + "step": 478, + "tokens/total": 15279472, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 218503 + }, + { + "epoch": 1.87109375, + "grad_norm": 0.37432360649108887, + "learning_rate": 1.1078053864763674e-05, + "loss": 0.0017199859721586108, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00172, + "step": 479, + "tokens/total": 15311536, + "tokens/train_per_sec_per_gpu": 30.18, + "tokens/trainable": 218964 + }, + { + "epoch": 1.875, + "grad_norm": 0.006793392356485128, + "learning_rate": 1.1015807679531756e-05, + "loss": 2.6950599931296892e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00003, + "step": 480, + "tokens/total": 15343360, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 219412 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.0414435058679398e+18, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/training_args.bin b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..dec74f9f304b44cbe47132d86c4e6285e659f40b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-480/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634 +size 8273 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/README.md b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ba677470ac9c29f9770a4aacb65c644ebd671a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/deconf_wave_deconf_control/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/deconf_wave_deconf_control/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/adapter_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..21fcb5b00be777c20a2d2dcceffc539dc51679b4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/deconf_wave_deconf_control/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "gate_proj", + "v_proj", + "k_proj", + "up_proj", + "q_proj", + "o_proj", + "down_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/adapter_model.safetensors b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..3c097a56f43f97ee848d5ab0592509805bc87dd8 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cb902f917f4594e641254c1053d1e99022840e561efc592be774f6f17040ba38 +size 547777976 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/chat_template.jinja b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/optimizer.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..573f1645db5e509efc0cd687cb9bd948a16cd5b3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:eef61f92ba0f4e2022cfb6ce24e1c8fb0704c2c5300ba67589f60a6038f2b0dd +size 1048106435 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/rng_state.pth b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..60e7fe4958809948c6438bdeaea1a079577fab82 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0167bd44db439d036c339ec7676c03866d5a37ac97d68de38616699306db0a21 +size 14645 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/scheduler.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..22325b3b3d08e0697784fe21b8c321485a51f7b5 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:697fe8894f7795467da8b1a7ccebf8570b28357dc457e2ba8ccdb525507cfef4 +size 1465 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/tokenizer.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/tokenizer_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/tokens_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..415f604472aab780d8c16a48970486f9d7a11469 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/tokens_state.json @@ -0,0 +1 @@ +{"total": 16362544, "trainable": 234060} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/trainer_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..b913befa879c512a9b970c3a9b0401e2952ed65b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/trainer_state.json @@ -0,0 +1,7202 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 2.0, + "eval_steps": 500, + "global_step": 512, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.2140090465545654, + "learning_rate": 0.0, + "loss": 0.16441112756729126, + "memory/device_reserved (GiB)": 38.08, + "memory/max_active (GiB)": 35.64, + "memory/max_allocated (GiB)": 35.64, + "ppl": 1.1787, + "step": 1, + "tokens/total": 32160, + "tokens/train_per_sec_per_gpu": 21.28, + "tokens/trainable": 470 + }, + { + "epoch": 0.0078125, + "grad_norm": 2.92620849609375, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15774038434028625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.17086, + "step": 2, + "tokens/total": 64192, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 922 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.135961890220642, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1664382815361023, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.18109, + "step": 3, + "tokens/total": 96224, + "tokens/train_per_sec_per_gpu": 28.54, + "tokens/trainable": 1396 + }, + { + "epoch": 0.015625, + "grad_norm": 0.880176305770874, + "learning_rate": 1.2e-05, + "loss": 0.15334013104438782, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.16572, + "step": 4, + "tokens/total": 128336, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 1850 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.1433289051055908, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.16352537274360657, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.17766, + "step": 5, + "tokens/total": 160784, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 2302 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9728212952613831, + "learning_rate": 2e-05, + "loss": 0.15144473314285278, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.16351, + "step": 6, + "tokens/total": 192624, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 2742 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.7796363830566406, + "learning_rate": 2.4e-05, + "loss": 0.13553467392921448, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.14515, + "step": 7, + "tokens/total": 224608, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 3208 + }, + { + "epoch": 0.03125, + "grad_norm": 1.2444697618484497, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.1077304258942604, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.11375, + "step": 8, + "tokens/total": 256592, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 3655 + }, + { + "epoch": 0.03515625, + "grad_norm": 2.7339375019073486, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.08772765100002289, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.09169, + "step": 9, + "tokens/total": 286400, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 4091 + }, + { + "epoch": 0.0390625, + "grad_norm": 4.109770774841309, + "learning_rate": 3.6e-05, + "loss": 0.130940243601799, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.1399, + "step": 10, + "tokens/total": 318352, + "tokens/train_per_sec_per_gpu": 29.13, + "tokens/trainable": 4553 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.3768608570098877, + "learning_rate": 4e-05, + "loss": 0.10059958696365356, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.10583, + "step": 11, + "tokens/total": 350336, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 4992 + }, + { + "epoch": 0.046875, + "grad_norm": 3.281402826309204, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.12574875354766846, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.134, + "step": 12, + "tokens/total": 382448, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 5494 + }, + { + "epoch": 0.05078125, + "grad_norm": 1.7738980054855347, + "learning_rate": 4.8e-05, + "loss": 0.033296722918748856, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03386, + "step": 13, + "tokens/total": 414352, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 5933 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.64290189743042, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06366822123527527, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.06574, + "step": 14, + "tokens/total": 446400, + "tokens/train_per_sec_per_gpu": 24.8, + "tokens/trainable": 6369 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.3691928386688232, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.08168518543243408, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.08511, + "step": 15, + "tokens/total": 478688, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 6820 + }, + { + "epoch": 0.0625, + "grad_norm": 2.4356393814086914, + "learning_rate": 6e-05, + "loss": 0.09031196683645248, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.09452, + "step": 16, + "tokens/total": 510512, + "tokens/train_per_sec_per_gpu": 25.7, + "tokens/trainable": 7258 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.55682373046875, + "learning_rate": 6.400000000000001e-05, + "loss": 0.07496573030948639, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.07785, + "step": 17, + "tokens/total": 542736, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 7710 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.1054646968841553, + "learning_rate": 6.800000000000001e-05, + "loss": 0.05218805745244026, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.05357, + "step": 18, + "tokens/total": 574832, + "tokens/train_per_sec_per_gpu": 25.78, + "tokens/trainable": 8174 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.7429099082946777, + "learning_rate": 7.2e-05, + "loss": 0.03790827840566635, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.03864, + "step": 19, + "tokens/total": 606976, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 8617 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9508209228515625, + "learning_rate": 7.6e-05, + "loss": 0.04993613809347153, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0512, + "step": 20, + "tokens/total": 639200, + "tokens/train_per_sec_per_gpu": 24.91, + "tokens/trainable": 9037 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.0297844409942627, + "learning_rate": 8e-05, + "loss": 0.017908263951539993, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.01807, + "step": 21, + "tokens/total": 671072, + "tokens/train_per_sec_per_gpu": 30.42, + "tokens/trainable": 9503 + }, + { + "epoch": 0.0859375, + "grad_norm": 0.9279879331588745, + "learning_rate": 8.4e-05, + "loss": 0.013478193432092667, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01357, + "step": 22, + "tokens/total": 703136, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 9972 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.526488184928894, + "learning_rate": 8.800000000000001e-05, + "loss": 0.017310388386249542, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01746, + "step": 23, + "tokens/total": 735120, + "tokens/train_per_sec_per_gpu": 31.45, + "tokens/trainable": 10447 + }, + { + "epoch": 0.09375, + "grad_norm": 3.4049417972564697, + "learning_rate": 9.200000000000001e-05, + "loss": 0.0353429913520813, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.03597, + "step": 24, + "tokens/total": 767040, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 10899 + }, + { + "epoch": 0.09765625, + "grad_norm": 7.511656761169434, + "learning_rate": 9.6e-05, + "loss": 0.1335875540971756, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.14292, + "step": 25, + "tokens/total": 798976, + "tokens/train_per_sec_per_gpu": 27.72, + "tokens/trainable": 11360 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.0069234371185303, + "learning_rate": 0.0001, + "loss": 0.04382415860891342, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0448, + "step": 26, + "tokens/total": 830880, + "tokens/train_per_sec_per_gpu": 24.61, + "tokens/trainable": 11775 + }, + { + "epoch": 0.10546875, + "grad_norm": 7.807938575744629, + "learning_rate": 9.99990636831587e-05, + "loss": 0.01210833340883255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.01218, + "step": 27, + "tokens/total": 860960, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 12242 + }, + { + "epoch": 0.109375, + "grad_norm": 2.23823881149292, + "learning_rate": 9.999625477159879e-05, + "loss": 0.0272970050573349, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02767, + "step": 28, + "tokens/total": 892816, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 12696 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.315584421157837, + "learning_rate": 9.999157338221051e-05, + "loss": 0.025516577064990997, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.02584, + "step": 29, + "tokens/total": 924736, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 13153 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.8413651585578918, + "learning_rate": 9.998501970980562e-05, + "loss": 0.007838520221412182, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00787, + "step": 30, + "tokens/total": 954592, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 13624 + }, + { + "epoch": 0.12109375, + "grad_norm": 1.6797157526016235, + "learning_rate": 9.997659402710915e-05, + "loss": 0.03002442792057991, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.03048, + "step": 31, + "tokens/total": 986592, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 14064 + }, + { + "epoch": 0.125, + "grad_norm": 1.3320434093475342, + "learning_rate": 9.996629668474818e-05, + "loss": 0.03772728145122528, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.03845, + "step": 32, + "tokens/total": 1018720, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 14565 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.6335167288780212, + "learning_rate": 9.995412811123711e-05, + "loss": 0.013970119878649712, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01407, + "step": 33, + "tokens/total": 1050624, + "tokens/train_per_sec_per_gpu": 28.18, + "tokens/trainable": 15005 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.5626921653747559, + "learning_rate": 9.994008881295999e-05, + "loss": 0.016654539853334427, + "memory/device_reserved (GiB)": 37.74, + "memory/max_active (GiB)": 36.22, + "memory/max_allocated (GiB)": 36.22, + "ppl": 1.01679, + "step": 34, + "tokens/total": 1082176, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 15459 + }, + { + "epoch": 0.13671875, + "grad_norm": 0.603219211101532, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021408583968877792, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02164, + "step": 35, + "tokens/total": 1113968, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 15960 + }, + { + "epoch": 0.140625, + "grad_norm": 0.9779674410820007, + "learning_rate": 9.99064004568618e-05, + "loss": 0.03277068957686424, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03331, + "step": 36, + "tokens/total": 1145952, + "tokens/train_per_sec_per_gpu": 28.68, + "tokens/trainable": 16428 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.8855255842208862, + "learning_rate": 9.988675280095074e-05, + "loss": 0.017048098146915436, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01719, + "step": 37, + "tokens/total": 1177872, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 16884 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.9266122579574585, + "learning_rate": 9.986523722403528e-05, + "loss": 0.045863546431064606, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.04693, + "step": 38, + "tokens/total": 1209776, + "tokens/train_per_sec_per_gpu": 26.91, + "tokens/trainable": 17341 + }, + { + "epoch": 0.15234375, + "grad_norm": 0.8280895948410034, + "learning_rate": 9.984185462146642e-05, + "loss": 0.033055759966373444, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03361, + "step": 39, + "tokens/total": 1241888, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 17786 + }, + { + "epoch": 0.15625, + "grad_norm": 5.029730319976807, + "learning_rate": 9.98166059662897e-05, + "loss": 0.025589998811483383, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02592, + "step": 40, + "tokens/total": 1274032, + "tokens/train_per_sec_per_gpu": 29.22, + "tokens/trainable": 18262 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.5688032507896423, + "learning_rate": 9.978949230920472e-05, + "loss": 0.016095872968435287, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01623, + "step": 41, + "tokens/total": 1306000, + "tokens/train_per_sec_per_gpu": 27.49, + "tokens/trainable": 18716 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8217663168907166, + "learning_rate": 9.976051477852141e-05, + "loss": 0.026798982173204422, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02716, + "step": 42, + "tokens/total": 1337936, + "tokens/train_per_sec_per_gpu": 27.51, + "tokens/trainable": 19157 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.7309132218360901, + "learning_rate": 9.972967458011312e-05, + "loss": 0.033920761197805405, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0345, + "step": 43, + "tokens/total": 1369904, + "tokens/train_per_sec_per_gpu": 30.66, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.8134022355079651, + "learning_rate": 9.96969729973664e-05, + "loss": 0.024005085229873657, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0243, + "step": 44, + "tokens/total": 1401984, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 20119 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.35084816813468933, + "learning_rate": 9.966241139112754e-05, + "loss": 0.013319254852831364, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01341, + "step": 45, + "tokens/total": 1434032, + "tokens/train_per_sec_per_gpu": 27.45, + "tokens/trainable": 20560 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.8701573014259338, + "learning_rate": 9.96259911996461e-05, + "loss": 0.01356641948223114, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01366, + "step": 46, + "tokens/total": 1465872, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 20992 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.4406442642211914, + "learning_rate": 9.958771393851491e-05, + "loss": 0.04396980255842209, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 35.86, + "memory/max_allocated (GiB)": 35.86, + "ppl": 1.04495, + "step": 47, + "tokens/total": 1495440, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 21441 + }, + { + "epoch": 0.1875, + "grad_norm": 0.4966669976711273, + "learning_rate": 9.954758120060702e-05, + "loss": 0.011179964058101177, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01124, + "step": 48, + "tokens/total": 1527776, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 21901 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.41124436259269714, + "learning_rate": 9.950559465600948e-05, + "loss": 0.0063599650748074055, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00638, + "step": 49, + "tokens/total": 1559760, + "tokens/train_per_sec_per_gpu": 27.73, + "tokens/trainable": 22341 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.829739511013031, + "learning_rate": 9.946175605195379e-05, + "loss": 0.015785304829478264, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01591, + "step": 50, + "tokens/total": 1592000, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 22833 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5181736946105957, + "learning_rate": 9.941606721274322e-05, + "loss": 0.008393109776079655, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00843, + "step": 51, + "tokens/total": 1623920, + "tokens/train_per_sec_per_gpu": 25.61, + "tokens/trainable": 23277 + }, + { + "epoch": 0.203125, + "grad_norm": 1.5778990983963013, + "learning_rate": 9.936853003967685e-05, + "loss": 0.024984844028949738, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0253, + "step": 52, + "tokens/total": 1656032, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 23759 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.4938714504241943, + "learning_rate": 9.93191465109705e-05, + "loss": 0.038523320108652115, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03927, + "step": 53, + "tokens/total": 1688368, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 24193 + }, + { + "epoch": 0.2109375, + "grad_norm": 2.1427106857299805, + "learning_rate": 9.926791868167438e-05, + "loss": 0.06302924454212189, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.06506, + "step": 54, + "tokens/total": 1720560, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 24619 + }, + { + "epoch": 0.21484375, + "grad_norm": 1.8365188837051392, + "learning_rate": 9.921484868358753e-05, + "loss": 0.052160464227199554, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.05354, + "step": 55, + "tokens/total": 1752464, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 25075 + }, + { + "epoch": 0.21875, + "grad_norm": 0.7779749631881714, + "learning_rate": 9.915993872516924e-05, + "loss": 0.021328264847397804, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.02156, + "step": 56, + "tokens/total": 1784336, + "tokens/train_per_sec_per_gpu": 26.42, + "tokens/trainable": 25519 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.4189571142196655, + "learning_rate": 9.9103191091447e-05, + "loss": 0.014279939234256744, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01438, + "step": 57, + "tokens/total": 1816304, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 25966 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.3834710717201233, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01883944310247898, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01902, + "step": 58, + "tokens/total": 1848304, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 26423 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.7693278789520264, + "learning_rate": 9.898419232046825e-05, + "loss": 0.03533369302749634, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03597, + "step": 59, + "tokens/total": 1880640, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 26934 + }, + { + "epoch": 0.234375, + "grad_norm": 0.4606446325778961, + "learning_rate": 9.892194613523633e-05, + "loss": 0.021482212468981743, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02171, + "step": 60, + "tokens/total": 1912592, + "tokens/train_per_sec_per_gpu": 27.64, + "tokens/trainable": 27393 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.46030017733573914, + "learning_rate": 9.885787217854357e-05, + "loss": 0.017044665291905403, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01719, + "step": 61, + "tokens/total": 1944544, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 27852 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.7918264269828796, + "learning_rate": 9.879197311676887e-05, + "loss": 0.016954131424427032, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0171, + "step": 62, + "tokens/total": 1976688, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 28292 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.4787926971912384, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01525965891778469, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 63, + "tokens/total": 2008928, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 28798 + }, + { + "epoch": 0.25, + "grad_norm": 0.9033918976783752, + "learning_rate": 9.865471072312528e-05, + "loss": 0.02902211993932724, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02945, + "step": 64, + "tokens/total": 2041120, + "tokens/train_per_sec_per_gpu": 29.83, + "tokens/trainable": 29283 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.39305707812309265, + "learning_rate": 9.858335310330492e-05, + "loss": 0.00901167094707489, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.32, + "memory/max_allocated (GiB)": 36.32, + "ppl": 1.00905, + "step": 65, + "tokens/total": 2073024, + "tokens/train_per_sec_per_gpu": 26.48, + "tokens/trainable": 29745 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4494345784187317, + "learning_rate": 9.851018180226185e-05, + "loss": 0.007426132448017597, + "memory/device_reserved (GiB)": 37.47, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00745, + "step": 66, + "tokens/total": 2105184, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 30186 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.629649817943573, + "learning_rate": 9.843519986495259e-05, + "loss": 0.010609081946313381, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01067, + "step": 67, + "tokens/total": 2137264, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 30622 + }, + { + "epoch": 0.265625, + "grad_norm": 0.7591162919998169, + "learning_rate": 9.835841041168162e-05, + "loss": 0.00926615484058857, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00931, + "step": 68, + "tokens/total": 2169424, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 31097 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.7288486361503601, + "learning_rate": 9.82798166379715e-05, + "loss": 0.024310583248734474, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.02461, + "step": 69, + "tokens/total": 2201424, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 31595 + }, + { + "epoch": 0.2734375, + "grad_norm": 2.134056329727173, + "learning_rate": 9.819942181443002e-05, + "loss": 0.008422967046499252, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00846, + "step": 70, + "tokens/total": 2233424, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 32036 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6362871527671814, + "learning_rate": 9.811722928661392e-05, + "loss": 0.01577262207865715, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0159, + "step": 71, + "tokens/total": 2265328, + "tokens/train_per_sec_per_gpu": 23.65, + "tokens/trainable": 32428 + }, + { + "epoch": 0.28125, + "grad_norm": 0.6091912388801575, + "learning_rate": 9.803324247488975e-05, + "loss": 0.0064800274558365345, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0065, + "step": 72, + "tokens/total": 2297248, + "tokens/train_per_sec_per_gpu": 26.86, + "tokens/trainable": 32880 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.14533838629722595, + "learning_rate": 9.794746487429161e-05, + "loss": 0.001991454279050231, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00199, + "step": 73, + "tokens/total": 2329152, + "tokens/train_per_sec_per_gpu": 27.97, + "tokens/trainable": 33323 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.5400826930999756, + "learning_rate": 9.785990005437554e-05, + "loss": 0.017662350088357925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01782, + "step": 74, + "tokens/total": 2361344, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 33792 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.6132436394691467, + "learning_rate": 9.777055165907117e-05, + "loss": 0.01787603087723255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01804, + "step": 75, + "tokens/total": 2393168, + "tokens/train_per_sec_per_gpu": 29.53, + "tokens/trainable": 34223 + }, + { + "epoch": 0.296875, + "grad_norm": 0.7287399172782898, + "learning_rate": 9.767942340652993e-05, + "loss": 0.024754449725151062, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.31, + "memory/max_allocated (GiB)": 36.31, + "ppl": 1.02506, + "step": 76, + "tokens/total": 2424880, + "tokens/train_per_sec_per_gpu": 24.92, + "tokens/trainable": 34649 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.6790545582771301, + "learning_rate": 9.758651908897035e-05, + "loss": 0.014308430254459381, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01441, + "step": 77, + "tokens/total": 2457120, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 35094 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6203456521034241, + "learning_rate": 9.749184257252033e-05, + "loss": 0.034353043884038925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03495, + "step": 78, + "tokens/total": 2489104, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 35534 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.6238825917243958, + "learning_rate": 9.739539779705614e-05, + "loss": 0.032295141369104385, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.03282, + "step": 79, + "tokens/total": 2521136, + "tokens/train_per_sec_per_gpu": 29.68, + "tokens/trainable": 36029 + }, + { + "epoch": 0.3125, + "grad_norm": 0.18977171182632446, + "learning_rate": 9.729718877603861e-05, + "loss": 0.007260813377797604, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00729, + "step": 80, + "tokens/total": 2553168, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 36469 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.8054049611091614, + "learning_rate": 9.719721959634592e-05, + "loss": 0.034894898533821106, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.03551, + "step": 81, + "tokens/total": 2585584, + "tokens/train_per_sec_per_gpu": 25.75, + "tokens/trainable": 36906 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.6884189248085022, + "learning_rate": 9.709549441810375e-05, + "loss": 0.018210047855973244, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.01838, + "step": 82, + "tokens/total": 2617808, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 37390 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.6490103602409363, + "learning_rate": 9.699201747451195e-05, + "loss": 0.02030857279896736, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.02052, + "step": 83, + "tokens/total": 2649664, + "tokens/train_per_sec_per_gpu": 25.8, + "tokens/trainable": 37838 + }, + { + "epoch": 0.328125, + "grad_norm": 0.2439805567264557, + "learning_rate": 9.688679307166854e-05, + "loss": 0.008518049493432045, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00855, + "step": 84, + "tokens/total": 2681728, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 38310 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6747375726699829, + "learning_rate": 9.677982558839042e-05, + "loss": 0.029895059764385223, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.03035, + "step": 85, + "tokens/total": 2713488, + "tokens/train_per_sec_per_gpu": 28.49, + "tokens/trainable": 38789 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.582666277885437, + "learning_rate": 9.66711194760312e-05, + "loss": 0.014998597092926502, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01511, + "step": 86, + "tokens/total": 2745536, + "tokens/train_per_sec_per_gpu": 30.27, + "tokens/trainable": 39277 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.5364654660224915, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011567167937755585, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01163, + "step": 87, + "tokens/total": 2777680, + "tokens/train_per_sec_per_gpu": 29.73, + "tokens/trainable": 39737 + }, + { + "epoch": 0.34375, + "grad_norm": 0.34024426341056824, + "learning_rate": 9.644850953105288e-05, + "loss": 0.01014852337539196, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0102, + "step": 88, + "tokens/total": 2809568, + "tokens/train_per_sec_per_gpu": 24.68, + "tokens/trainable": 40179 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.32472267746925354, + "learning_rate": 9.633461496214225e-05, + "loss": 0.01229981891810894, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01238, + "step": 89, + "tokens/total": 2841760, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 40649 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.7906899452209473, + "learning_rate": 9.621900029118195e-05, + "loss": 0.02552586793899536, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.02585, + "step": 90, + "tokens/total": 2871696, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 41080 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.7444764971733093, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0054441955871880054, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00546, + "step": 91, + "tokens/total": 2903968, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 41599 + }, + { + "epoch": 0.359375, + "grad_norm": 0.1911851465702057, + "learning_rate": 9.598262995928611e-05, + "loss": 0.004665314219892025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00468, + "step": 92, + "tokens/total": 2936048, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 42076 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.21238519251346588, + "learning_rate": 9.586188413468492e-05, + "loss": 0.003917238209396601, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00392, + "step": 93, + "tokens/total": 2968176, + "tokens/train_per_sec_per_gpu": 28.07, + "tokens/trainable": 42522 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.41729435324668884, + "learning_rate": 9.57394378802934e-05, + "loss": 0.013555807992815971, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01365, + "step": 94, + "tokens/total": 3000304, + "tokens/train_per_sec_per_gpu": 28.69, + "tokens/trainable": 42974 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.7246769070625305, + "learning_rate": 9.56152962916e-05, + "loss": 0.020221399143338203, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.02043, + "step": 95, + "tokens/total": 3032080, + "tokens/train_per_sec_per_gpu": 25.35, + "tokens/trainable": 43380 + }, + { + "epoch": 0.375, + "grad_norm": 0.16139821708202362, + "learning_rate": 9.548946453464296e-05, + "loss": 0.003853276837617159, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00386, + "step": 96, + "tokens/total": 3064320, + "tokens/train_per_sec_per_gpu": 28.47, + "tokens/trainable": 43865 + }, + { + "epoch": 0.37890625, + "grad_norm": 1.0509370565414429, + "learning_rate": 9.53619478457953e-05, + "loss": 0.015099998563528061, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01521, + "step": 97, + "tokens/total": 3096144, + "tokens/train_per_sec_per_gpu": 26.61, + "tokens/trainable": 44328 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.6859350204467773, + "learning_rate": 9.523275153154695e-05, + "loss": 0.004579808097332716, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00459, + "step": 98, + "tokens/total": 3127600, + "tokens/train_per_sec_per_gpu": 25.91, + "tokens/trainable": 44778 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5686795711517334, + "learning_rate": 9.51018809682839e-05, + "loss": 0.022520575672388077, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02278, + "step": 99, + "tokens/total": 3159520, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 45225 + }, + { + "epoch": 0.390625, + "grad_norm": 1.6308369636535645, + "learning_rate": 9.49693416020645e-05, + "loss": 0.011883002705872059, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01195, + "step": 100, + "tokens/total": 3191568, + "tokens/train_per_sec_per_gpu": 29.75, + "tokens/trainable": 45678 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.7186088562011719, + "learning_rate": 9.483513894839276e-05, + "loss": 0.0060632615350186825, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00608, + "step": 101, + "tokens/total": 3223648, + "tokens/train_per_sec_per_gpu": 27.14, + "tokens/trainable": 46125 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.12036337703466415, + "learning_rate": 9.469927859198888e-05, + "loss": 0.0019118499476462603, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00191, + "step": 102, + "tokens/total": 3255744, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 46612 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.29928353428840637, + "learning_rate": 9.456176618655689e-05, + "loss": 0.00944911316037178, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00949, + "step": 103, + "tokens/total": 3287792, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 47059 + }, + { + "epoch": 0.40625, + "grad_norm": 0.3766293227672577, + "learning_rate": 9.442260745454927e-05, + "loss": 0.008984047919511795, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.68, + "memory/max_allocated (GiB)": 36.68, + "ppl": 1.00902, + "step": 104, + "tokens/total": 3320016, + "tokens/train_per_sec_per_gpu": 28.92, + "tokens/trainable": 47536 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3481820523738861, + "learning_rate": 9.428180818692884e-05, + "loss": 0.005444999784231186, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00546, + "step": 105, + "tokens/total": 3351952, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 48037 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.8414323329925537, + "learning_rate": 9.413937424292791e-05, + "loss": 0.032438624650239944, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03297, + "step": 106, + "tokens/total": 3384032, + "tokens/train_per_sec_per_gpu": 27.11, + "tokens/trainable": 48491 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.5067655444145203, + "learning_rate": 9.399531154980424e-05, + "loss": 0.008442888967692852, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00848, + "step": 107, + "tokens/total": 3416192, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 48940 + }, + { + "epoch": 0.421875, + "grad_norm": 0.41617852449417114, + "learning_rate": 9.384962610259455e-05, + "loss": 0.008386321365833282, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00842, + "step": 108, + "tokens/total": 3448080, + "tokens/train_per_sec_per_gpu": 26.56, + "tokens/trainable": 49389 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1587515026330948, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0031663556583225727, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.03, + "memory/max_allocated (GiB)": 36.03, + "ppl": 1.00317, + "step": 109, + "tokens/total": 3477952, + "tokens/train_per_sec_per_gpu": 29.8, + "tokens/trainable": 49838 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2903148829936981, + "learning_rate": 9.355341126345868e-05, + "loss": 0.007958847098052502, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00799, + "step": 110, + "tokens/total": 3509776, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 50310 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.642983078956604, + "learning_rate": 9.340289419824107e-05, + "loss": 0.006748045329004526, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00677, + "step": 111, + "tokens/total": 3541808, + "tokens/train_per_sec_per_gpu": 28.04, + "tokens/trainable": 50755 + }, + { + "epoch": 0.4375, + "grad_norm": 0.14928282797336578, + "learning_rate": 9.325077903184159e-05, + "loss": 0.0027839094400405884, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00279, + "step": 112, + "tokens/total": 3574064, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 51213 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.05922335386276245, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0009622900979593396, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00096, + "step": 113, + "tokens/total": 3605984, + "tokens/train_per_sec_per_gpu": 27.99, + "tokens/trainable": 51660 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.44227731227874756, + "learning_rate": 9.2941779782269e-05, + "loss": 0.002763954224064946, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00277, + "step": 114, + "tokens/total": 3638080, + "tokens/train_per_sec_per_gpu": 27.03, + "tokens/trainable": 52133 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5176818370819092, + "learning_rate": 9.278490855781596e-05, + "loss": 0.015623578801751137, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01575, + "step": 115, + "tokens/total": 3670016, + "tokens/train_per_sec_per_gpu": 31.44, + "tokens/trainable": 52580 + }, + { + "epoch": 0.453125, + "grad_norm": 0.4265497922897339, + "learning_rate": 9.262646494908604e-05, + "loss": 0.00794076919555664, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00797, + "step": 116, + "tokens/total": 3701984, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 53033 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.40457987785339355, + "learning_rate": 9.246645554956457e-05, + "loss": 0.022708112373948097, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02297, + "step": 117, + "tokens/total": 3733920, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 53517 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.01414525043219328, + "learning_rate": 9.230488701789578e-05, + "loss": 0.00020209309877827764, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.0002, + "step": 118, + "tokens/total": 3765840, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 53967 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.11359831690788269, + "learning_rate": 9.214176607760577e-05, + "loss": 0.0012168455868959427, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00122, + "step": 119, + "tokens/total": 3797680, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 54430 + }, + { + "epoch": 0.46875, + "grad_norm": 0.35982534289360046, + "learning_rate": 9.197709951682268e-05, + "loss": 0.004110721405595541, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00412, + "step": 120, + "tokens/total": 3829920, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 54896 + }, + { + "epoch": 0.47265625, + "grad_norm": 1.3690382242202759, + "learning_rate": 9.181089418799428e-05, + "loss": 0.02705790475010872, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02743, + "step": 121, + "tokens/total": 3862080, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 55379 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.351891428232193, + "learning_rate": 9.164315700760271e-05, + "loss": 0.0020163573790341616, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00202, + "step": 122, + "tokens/total": 3893968, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 55803 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4747961461544037, + "learning_rate": 9.147389495587671e-05, + "loss": 0.012588979676365852, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01267, + "step": 123, + "tokens/total": 3926160, + "tokens/train_per_sec_per_gpu": 30.14, + "tokens/trainable": 56249 + }, + { + "epoch": 0.484375, + "grad_norm": 0.4064250588417053, + "learning_rate": 9.130311507650116e-05, + "loss": 0.0022941348142921925, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0023, + "step": 124, + "tokens/total": 3958208, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 56713 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.22056929767131805, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0022175831254571676, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00222, + "step": 125, + "tokens/total": 3990208, + "tokens/train_per_sec_per_gpu": 32.57, + "tokens/trainable": 57196 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5698862671852112, + "learning_rate": 9.09570303250602e-05, + "loss": 0.010205100290477276, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01026, + "step": 126, + "tokens/total": 4022224, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 57680 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.2593887448310852, + "learning_rate": 9.078173985499394e-05, + "loss": 0.0032277877908200026, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00323, + "step": 127, + "tokens/total": 4053952, + "tokens/train_per_sec_per_gpu": 26.51, + "tokens/trainable": 58110 + }, + { + "epoch": 0.5, + "grad_norm": 0.3735230565071106, + "learning_rate": 9.060496036067713e-05, + "loss": 0.01031105499714613, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01036, + "step": 128, + "tokens/total": 4086144, + "tokens/train_per_sec_per_gpu": 26.5, + "tokens/trainable": 58534 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.1332484930753708, + "learning_rate": 9.042669919862615e-05, + "loss": 0.0012390739284455776, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00124, + "step": 129, + "tokens/total": 4118096, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 58998 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.20182718336582184, + "learning_rate": 9.024696378701557e-05, + "loss": 0.004501686431467533, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00451, + "step": 130, + "tokens/total": 4150272, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 59448 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.529080331325531, + "learning_rate": 9.006576160536948e-05, + "loss": 0.00763288140296936, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00766, + "step": 131, + "tokens/total": 4181904, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 59906 + }, + { + "epoch": 0.515625, + "grad_norm": 0.21625548601150513, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0024378676898777485, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00244, + "step": 132, + "tokens/total": 4214032, + "tokens/train_per_sec_per_gpu": 28.36, + "tokens/trainable": 60350 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.37116539478302, + "learning_rate": 8.969898715494506e-05, + "loss": 0.0037849934305995703, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00379, + "step": 133, + "tokens/total": 4246064, + "tokens/train_per_sec_per_gpu": 30.18, + "tokens/trainable": 60831 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5646623373031616, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006517104804515839, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00654, + "step": 134, + "tokens/total": 4278224, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 61305 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.6371963620185852, + "learning_rate": 8.932643689864568e-05, + "loss": 0.008687382563948631, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00873, + "step": 135, + "tokens/total": 4310112, + "tokens/train_per_sec_per_gpu": 31.48, + "tokens/trainable": 61792 + }, + { + "epoch": 0.53125, + "grad_norm": 0.8210524916648865, + "learning_rate": 8.913801518498845e-05, + "loss": 0.008436474949121475, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00847, + "step": 136, + "tokens/total": 4342192, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 62253 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.5692176222801208, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017321188002824783, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.01747, + "step": 137, + "tokens/total": 4374096, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.12828375399112701, + "learning_rate": 8.875691779131569e-05, + "loss": 0.0007754197577014565, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00078, + "step": 138, + "tokens/total": 4406272, + "tokens/train_per_sec_per_gpu": 24.22, + "tokens/trainable": 63145 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.3188190162181854, + "learning_rate": 8.856425797031829e-05, + "loss": 0.006707335356622934, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00673, + "step": 139, + "tokens/total": 4438112, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 63587 + }, + { + "epoch": 0.546875, + "grad_norm": 0.6216707229614258, + "learning_rate": 8.837020140354295e-05, + "loss": 0.007024695165455341, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00705, + "step": 140, + "tokens/total": 4470128, + "tokens/train_per_sec_per_gpu": 30.64, + "tokens/trainable": 64052 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.1480305939912796, + "learning_rate": 8.817475616647554e-05, + "loss": 0.00229115248657763, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00229, + "step": 141, + "tokens/total": 4502096, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 64526 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22254763543605804, + "learning_rate": 8.797793039239017e-05, + "loss": 0.0029198499396443367, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.67, + "memory/max_allocated (GiB)": 36.67, + "ppl": 1.00292, + "step": 142, + "tokens/total": 4534496, + "tokens/train_per_sec_per_gpu": 29.08, + "tokens/trainable": 64983 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.6901941895484924, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013051149435341358, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01314, + "step": 143, + "tokens/total": 4566720, + "tokens/train_per_sec_per_gpu": 31.56, + "tokens/trainable": 65492 + }, + { + "epoch": 0.5625, + "grad_norm": 0.9545458555221558, + "learning_rate": 8.758017005316988e-05, + "loss": 0.02235381305217743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.02261, + "step": 144, + "tokens/total": 4598880, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 65925 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.5515892505645752, + "learning_rate": 8.737925204046629e-05, + "loss": 0.02038375660777092, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.02059, + "step": 145, + "tokens/total": 4630864, + "tokens/train_per_sec_per_gpu": 26.3, + "tokens/trainable": 66383 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3512430191040039, + "learning_rate": 8.717698659491851e-05, + "loss": 0.008233271539211273, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00827, + "step": 146, + "tokens/total": 4662864, + "tokens/train_per_sec_per_gpu": 26.23, + "tokens/trainable": 66840 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.6830361485481262, + "learning_rate": 8.697338213361735e-05, + "loss": 0.013419180177152157, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01351, + "step": 147, + "tokens/total": 4694912, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 67297 + }, + { + "epoch": 0.578125, + "grad_norm": 0.2117031067609787, + "learning_rate": 8.676844712937552e-05, + "loss": 0.004134457092732191, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00414, + "step": 148, + "tokens/total": 4727120, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 67769 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.378053218126297, + "learning_rate": 8.656219011037509e-05, + "loss": 0.009747720323503017, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0098, + "step": 149, + "tokens/total": 4759168, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 68257 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.21076741814613342, + "learning_rate": 8.63546196598125e-05, + "loss": 0.005035653710365295, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00505, + "step": 150, + "tokens/total": 4791008, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 68706 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.25075697898864746, + "learning_rate": 8.614574441554145e-05, + "loss": 0.006027051247656345, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00605, + "step": 151, + "tokens/total": 4822912, + "tokens/train_per_sec_per_gpu": 27.95, + "tokens/trainable": 69131 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5774204730987549, + "learning_rate": 8.593557306971349e-05, + "loss": 0.018617186695337296, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.01879, + "step": 152, + "tokens/total": 4854816, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 69586 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.1364293098449707, + "learning_rate": 8.572411436841618e-05, + "loss": 0.002049289643764496, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00205, + "step": 153, + "tokens/total": 4886656, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 70061 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.25387004017829895, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0044304742477834225, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00444, + "step": 154, + "tokens/total": 4918768, + "tokens/train_per_sec_per_gpu": 23.26, + "tokens/trainable": 70467 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3047125041484833, + "learning_rate": 8.529737015125824e-05, + "loss": 0.0043389927595853806, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00435, + "step": 155, + "tokens/total": 4951008, + "tokens/train_per_sec_per_gpu": 27.53, + "tokens/trainable": 70933 + }, + { + "epoch": 0.609375, + "grad_norm": 0.1442601978778839, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0031973645091056824, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.0032, + "step": 156, + "tokens/total": 4982976, + "tokens/train_per_sec_per_gpu": 28.86, + "tokens/trainable": 71382 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.2608455717563629, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0032951058819890022, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0033, + "step": 157, + "tokens/total": 5014640, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 71808 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.637295663356781, + "learning_rate": 8.464782037243449e-05, + "loss": 0.01436957623809576, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.01447, + "step": 158, + "tokens/total": 5046672, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 72249 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.9834056496620178, + "learning_rate": 8.442882418044202e-05, + "loss": 0.02375311404466629, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.02404, + "step": 159, + "tokens/total": 5078528, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 72726 + }, + { + "epoch": 0.625, + "grad_norm": 0.35432615876197815, + "learning_rate": 8.420860333495179e-05, + "loss": 0.006119017023593187, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00614, + "step": 160, + "tokens/total": 5110704, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 73148 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.44270065426826477, + "learning_rate": 8.398716700025208e-05, + "loss": 0.01352179329842329, + "memory/device_reserved (GiB)": 39.12, + "memory/max_active (GiB)": 36.72, + "memory/max_allocated (GiB)": 36.72, + "ppl": 1.01361, + "step": 161, + "tokens/total": 5142944, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 73600 + }, + { + "epoch": 0.6328125, + "grad_norm": 1.2181634902954102, + "learning_rate": 8.376452439121266e-05, + "loss": 0.01820964366197586, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.01838, + "step": 162, + "tokens/total": 5174592, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 74068 + }, + { + "epoch": 0.63671875, + "grad_norm": 1.9321303367614746, + "learning_rate": 8.354068477290124e-05, + "loss": 0.009742327965795994, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00979, + "step": 163, + "tokens/total": 5206432, + "tokens/train_per_sec_per_gpu": 29.89, + "tokens/trainable": 74527 + }, + { + "epoch": 0.640625, + "grad_norm": 0.3545796275138855, + "learning_rate": 8.331565746019807e-05, + "loss": 0.00975855067372322, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00981, + "step": 164, + "tokens/total": 5238384, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 74992 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.35368478298187256, + "learning_rate": 8.308945181740812e-05, + "loss": 0.010195466689765453, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01025, + "step": 165, + "tokens/total": 5270272, + "tokens/train_per_sec_per_gpu": 29.41, + "tokens/trainable": 75442 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.34957799315452576, + "learning_rate": 8.286207725787153e-05, + "loss": 0.009373681619763374, + "memory/device_reserved (GiB)": 38.62, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00942, + "step": 166, + "tokens/total": 5302368, + "tokens/train_per_sec_per_gpu": 26.2, + "tokens/trainable": 75887 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.3684624135494232, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008211556822061539, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00825, + "step": 167, + "tokens/total": 5334496, + "tokens/train_per_sec_per_gpu": 28.21, + "tokens/trainable": 76331 + }, + { + "epoch": 0.65625, + "grad_norm": 0.12826858460903168, + "learning_rate": 8.240385928474219e-05, + "loss": 0.004492453299462795, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0045, + "step": 168, + "tokens/total": 5364288, + "tokens/train_per_sec_per_gpu": 30.16, + "tokens/trainable": 76745 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.31828173995018005, + "learning_rate": 8.217303493946967e-05, + "loss": 0.005733514670282602, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00575, + "step": 169, + "tokens/total": 5396512, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 77201 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6174353361129761, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01915649324655533, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01934, + "step": 170, + "tokens/total": 5428624, + "tokens/train_per_sec_per_gpu": 27.75, + "tokens/trainable": 77655 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.2902314066886902, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0036087254993617535, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00362, + "step": 171, + "tokens/total": 5460880, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 78122 + }, + { + "epoch": 0.671875, + "grad_norm": 0.3267754912376404, + "learning_rate": 8.147381587530713e-05, + "loss": 0.008090222254395485, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00812, + "step": 172, + "tokens/total": 5493088, + "tokens/train_per_sec_per_gpu": 27.86, + "tokens/trainable": 78576 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.4086189866065979, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006351984106004238, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00637, + "step": 173, + "tokens/total": 5525312, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 79059 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5261490345001221, + "learning_rate": 8.100214524900103e-05, + "loss": 0.008662862703204155, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.0087, + "step": 174, + "tokens/total": 5556880, + "tokens/train_per_sec_per_gpu": 24.62, + "tokens/trainable": 79498 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11292761564254761, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0027105510234832764, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00271, + "step": 175, + "tokens/total": 5588976, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 79922 + }, + { + "epoch": 0.6875, + "grad_norm": 0.33315548300743103, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0030926030594855547, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.0031, + "step": 176, + "tokens/total": 5618992, + "tokens/train_per_sec_per_gpu": 29.51, + "tokens/trainable": 80383 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.3136921226978302, + "learning_rate": 8.028654871074489e-05, + "loss": 0.005057850852608681, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00507, + "step": 177, + "tokens/total": 5651072, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 80824 + }, + { + "epoch": 0.6953125, + "grad_norm": 1.0526509284973145, + "learning_rate": 8.004589869885986e-05, + "loss": 0.010707498528063297, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01077, + "step": 178, + "tokens/total": 5683296, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 81243 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.1329648196697235, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0017975447699427605, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.0018, + "step": 179, + "tokens/total": 5715520, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 81716 + }, + { + "epoch": 0.703125, + "grad_norm": 1.0749201774597168, + "learning_rate": 7.95614819466576e-05, + "loss": 0.025617048144340515, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02595, + "step": 180, + "tokens/total": 5747488, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 82181 + }, + { + "epoch": 0.70703125, + "grad_norm": 1.0446465015411377, + "learning_rate": 7.931773536489872e-05, + "loss": 0.030236585065722466, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 35.9, + "memory/max_allocated (GiB)": 35.9, + "ppl": 1.0307, + "step": 181, + "tokens/total": 5777264, + "tokens/train_per_sec_per_gpu": 28.89, + "tokens/trainable": 82626 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22853964567184448, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0038214433006942272, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00383, + "step": 182, + "tokens/total": 5809264, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 83089 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.057821910828351974, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0004995397757738829, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0005, + "step": 183, + "tokens/total": 5841488, + "tokens/train_per_sec_per_gpu": 29.34, + "tokens/trainable": 83590 + }, + { + "epoch": 0.71875, + "grad_norm": 1.3264167308807373, + "learning_rate": 7.85804646415409e-05, + "loss": 0.013232000172138214, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01332, + "step": 184, + "tokens/total": 5873696, + "tokens/train_per_sec_per_gpu": 29.33, + "tokens/trainable": 84046 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.6027824878692627, + "learning_rate": 7.833273149760207e-05, + "loss": 0.012215660884976387, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01229, + "step": 185, + "tokens/total": 5905408, + "tokens/train_per_sec_per_gpu": 28.25, + "tokens/trainable": 84517 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.5848041772842407, + "learning_rate": 7.808402738346527e-05, + "loss": 0.024627480655908585, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.02493, + "step": 186, + "tokens/total": 5937392, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 84974 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.6823275089263916, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0028813849203288555, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00289, + "step": 187, + "tokens/total": 5969344, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 85460 + }, + { + "epoch": 0.734375, + "grad_norm": 0.41855958104133606, + "learning_rate": 7.758374768294647e-05, + "loss": 0.009796380065381527, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00984, + "step": 188, + "tokens/total": 6001296, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 85939 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6849080324172974, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004100777208805084, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00411, + "step": 189, + "tokens/total": 6031216, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 86377 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.15838374197483063, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003986812196671963, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00399, + "step": 190, + "tokens/total": 6063296, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 86834 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.3117486536502838, + "learning_rate": 7.682630588562518e-05, + "loss": 0.009132719598710537, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00917, + "step": 191, + "tokens/total": 6095216, + "tokens/train_per_sec_per_gpu": 26.65, + "tokens/trainable": 87265 + }, + { + "epoch": 0.75, + "grad_norm": 0.7837114334106445, + "learning_rate": 7.657199467573129e-05, + "loss": 0.007239446043968201, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00727, + "step": 192, + "tokens/total": 6127488, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 87747 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21777178347110748, + "learning_rate": 7.631678576708561e-05, + "loss": 0.002932134782895446, + "memory/device_reserved (GiB)": 40.04, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00294, + "step": 193, + "tokens/total": 6159488, + "tokens/train_per_sec_per_gpu": 30.37, + "tokens/trainable": 88239 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.19159242510795593, + "learning_rate": 7.606068977997255e-05, + "loss": 0.003356864908710122, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00336, + "step": 194, + "tokens/total": 6191296, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 88718 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.3977915346622467, + "learning_rate": 7.580371737159148e-05, + "loss": 0.008549144491553307, + "memory/device_reserved (GiB)": 38.19, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00859, + "step": 195, + "tokens/total": 6223120, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 89129 + }, + { + "epoch": 0.765625, + "grad_norm": 0.3875303566455841, + "learning_rate": 7.554587923561324e-05, + "loss": 0.006131038535386324, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00615, + "step": 196, + "tokens/total": 6254880, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 89567 + }, + { + "epoch": 0.76953125, + "grad_norm": 3.4133481979370117, + "learning_rate": 7.528718610173511e-05, + "loss": 0.01836150512099266, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01853, + "step": 197, + "tokens/total": 6286960, + "tokens/train_per_sec_per_gpu": 25.23, + "tokens/trainable": 89988 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.2509997487068176, + "learning_rate": 7.502764873523431e-05, + "loss": 0.003821226069703698, + "memory/device_reserved (GiB)": 38.2, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00383, + "step": 198, + "tokens/total": 6318880, + "tokens/train_per_sec_per_gpu": 30.29, + "tokens/trainable": 90434 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.9550195932388306, + "learning_rate": 7.476727793652011e-05, + "loss": 0.006872436031699181, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0069, + "step": 199, + "tokens/total": 6350928, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 90913 + }, + { + "epoch": 0.78125, + "grad_norm": 0.411724716424942, + "learning_rate": 7.450608454068415e-05, + "loss": 0.014243390411138535, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01435, + "step": 200, + "tokens/total": 6382976, + "tokens/train_per_sec_per_gpu": 28.29, + "tokens/trainable": 91355 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.34264692664146423, + "learning_rate": 7.424407941704987e-05, + "loss": 0.005309835076332092, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00532, + "step": 201, + "tokens/total": 6415120, + "tokens/train_per_sec_per_gpu": 31.41, + "tokens/trainable": 91856 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.41955795884132385, + "learning_rate": 7.398127346871986e-05, + "loss": 0.009196512401103973, + "memory/device_reserved (GiB)": 38.76, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00924, + "step": 202, + "tokens/total": 6447216, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 92311 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.03107454814016819, + "learning_rate": 7.371767763212238e-05, + "loss": 0.0004908143309876323, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00049, + "step": 203, + "tokens/total": 6479280, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 92764 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39578482508659363, + "learning_rate": 7.345330287655617e-05, + "loss": 0.006089594680815935, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00611, + "step": 204, + "tokens/total": 6511040, + "tokens/train_per_sec_per_gpu": 27.48, + "tokens/trainable": 93227 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.6444572806358337, + "learning_rate": 7.31881602037339e-05, + "loss": 0.010252210311591625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0103, + "step": 205, + "tokens/total": 6543120, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 93675 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.3283638060092926, + "learning_rate": 7.29222606473245e-05, + "loss": 0.00761406822130084, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00764, + "step": 206, + "tokens/total": 6575520, + "tokens/train_per_sec_per_gpu": 27.02, + "tokens/trainable": 94120 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.09952464699745178, + "learning_rate": 7.265561527249383e-05, + "loss": 0.001628706930205226, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00163, + "step": 207, + "tokens/total": 6607408, + "tokens/train_per_sec_per_gpu": 29.43, + "tokens/trainable": 94557 + }, + { + "epoch": 0.8125, + "grad_norm": 0.31704938411712646, + "learning_rate": 7.238823517544436e-05, + "loss": 0.005896817892789841, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00591, + "step": 208, + "tokens/total": 6639488, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 95035 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.0999862551689148, + "learning_rate": 7.212013148295333e-05, + "loss": 0.001754446537233889, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00176, + "step": 209, + "tokens/total": 6671520, + "tokens/train_per_sec_per_gpu": 31.08, + "tokens/trainable": 95517 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.1405109018087387, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0017940793186426163, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0018, + "step": 210, + "tokens/total": 6703536, + "tokens/train_per_sec_per_gpu": 30.3, + "tokens/trainable": 96026 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.09185238927602768, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0015102234901860356, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00151, + "step": 211, + "tokens/total": 6735616, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 96477 + }, + { + "epoch": 0.828125, + "grad_norm": 1.0464473962783813, + "learning_rate": 7.131159054949273e-05, + "loss": 0.016731251031160355, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01687, + "step": 212, + "tokens/total": 6767584, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 96951 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.3517152965068817, + "learning_rate": 7.104070433827139e-05, + "loss": 0.007984626106917858, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00802, + "step": 213, + "tokens/total": 6799488, + "tokens/train_per_sec_per_gpu": 25.83, + "tokens/trainable": 97350 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.28192469477653503, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00557567086070776, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00559, + "step": 214, + "tokens/total": 6831552, + "tokens/train_per_sec_per_gpu": 24.66, + "tokens/trainable": 97792 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.46135279536247253, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0170141588896513, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.01716, + "step": 215, + "tokens/total": 6863488, + "tokens/train_per_sec_per_gpu": 30.38, + "tokens/trainable": 98257 + }, + { + "epoch": 0.84375, + "grad_norm": 0.4277898073196411, + "learning_rate": 7.022408581865382e-05, + "loss": 0.009399959817528725, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.71, + "memory/max_allocated (GiB)": 36.71, + "ppl": 1.00944, + "step": 216, + "tokens/total": 6895888, + "tokens/train_per_sec_per_gpu": 27.28, + "tokens/trainable": 98731 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.34988704323768616, + "learning_rate": 6.99505974422157e-05, + "loss": 0.002909669652581215, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00291, + "step": 217, + "tokens/total": 6928000, + "tokens/train_per_sec_per_gpu": 30.25, + "tokens/trainable": 99212 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.4028005599975586, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0058855339884757996, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0059, + "step": 218, + "tokens/total": 6960096, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 99654 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.3000575304031372, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0025230362080037594, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00253, + "step": 219, + "tokens/total": 6992512, + "tokens/train_per_sec_per_gpu": 27.46, + "tokens/trainable": 100086 + }, + { + "epoch": 0.859375, + "grad_norm": 0.11351172626018524, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0021392193157225847, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00214, + "step": 220, + "tokens/total": 7024688, + "tokens/train_per_sec_per_gpu": 27.56, + "tokens/trainable": 100524 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.19512486457824707, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00650001410394907, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00652, + "step": 221, + "tokens/total": 7056928, + "tokens/train_per_sec_per_gpu": 28.05, + "tokens/trainable": 100996 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.1998104453086853, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0056140003725886345, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00563, + "step": 222, + "tokens/total": 7088944, + "tokens/train_per_sec_per_gpu": 26.49, + "tokens/trainable": 101449 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.40908998250961304, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01286369375884533, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.01295, + "step": 223, + "tokens/total": 7121104, + "tokens/train_per_sec_per_gpu": 27.19, + "tokens/trainable": 101905 + }, + { + "epoch": 0.875, + "grad_norm": 0.3672473132610321, + "learning_rate": 6.801939899284132e-05, + "loss": 0.010677668265998363, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01073, + "step": 224, + "tokens/total": 7153040, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 102411 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4146183431148529, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0038945709820836782, + "memory/device_reserved (GiB)": 39.19, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0039, + "step": 225, + "tokens/total": 7184912, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 102882 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.10585323721170425, + "learning_rate": 6.746257910210214e-05, + "loss": 0.00041726604104042053, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00042, + "step": 226, + "tokens/total": 7216992, + "tokens/train_per_sec_per_gpu": 28.72, + "tokens/trainable": 103332 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.36600011587142944, + "learning_rate": 6.718338543014937e-05, + "loss": 0.01526213251054287, + "memory/device_reserved (GiB)": 38.33, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 227, + "tokens/total": 7249120, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 103790 + }, + { + "epoch": 0.890625, + "grad_norm": 0.16057318449020386, + "learning_rate": 6.69036847577983e-05, + "loss": 0.004024423658847809, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00403, + "step": 228, + "tokens/total": 7281072, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 104246 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.318798303604126, + "learning_rate": 6.662348872453553e-05, + "loss": 0.006663155741989613, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00669, + "step": 229, + "tokens/total": 7313120, + "tokens/train_per_sec_per_gpu": 31.3, + "tokens/trainable": 104707 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.10858234763145447, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0020512123592197895, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00205, + "step": 230, + "tokens/total": 7345184, + "tokens/train_per_sec_per_gpu": 30.81, + "tokens/trainable": 105167 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.1044008731842041, + "learning_rate": 6.60616572358065e-05, + "loss": 0.0010264476295560598, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00103, + "step": 231, + "tokens/total": 7377264, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 105576 + }, + { + "epoch": 0.90625, + "grad_norm": 0.14836947619915009, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0013845522189512849, + "memory/device_reserved (GiB)": 38.37, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00139, + "step": 232, + "tokens/total": 7409424, + "tokens/train_per_sec_per_gpu": 30.32, + "tokens/trainable": 106043 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.022752461954951286, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0003245974367018789, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00032, + "step": 233, + "tokens/total": 7441456, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 106506 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.22981758415699005, + "learning_rate": 6.521548694236384e-05, + "loss": 0.002587020630016923, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00259, + "step": 234, + "tokens/total": 7473344, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 106951 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.5147526264190674, + "learning_rate": 6.493256429322259e-05, + "loss": 0.01256126631051302, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01264, + "step": 235, + "tokens/total": 7505040, + "tokens/train_per_sec_per_gpu": 25.64, + "tokens/trainable": 107382 + }, + { + "epoch": 0.921875, + "grad_norm": 0.3785130977630615, + "learning_rate": 6.464922830953799e-05, + "loss": 0.010956985875964165, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01102, + "step": 236, + "tokens/total": 7537120, + "tokens/train_per_sec_per_gpu": 27.41, + "tokens/trainable": 107814 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.6193199753761292, + "learning_rate": 6.436549078207688e-05, + "loss": 0.020655140280723572, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02087, + "step": 237, + "tokens/total": 7569024, + "tokens/train_per_sec_per_gpu": 29.36, + "tokens/trainable": 108274 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.5084519982337952, + "learning_rate": 6.408136351831592e-05, + "loss": 0.011346567422151566, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.01141, + "step": 238, + "tokens/total": 7600992, + "tokens/train_per_sec_per_gpu": 24.43, + "tokens/trainable": 108714 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.04367341846227646, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0004316020058467984, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00043, + "step": 239, + "tokens/total": 7633280, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 109220 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06729783862829208, + "learning_rate": 6.351198709240186e-05, + "loss": 0.000604260538239032, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.0006, + "step": 240, + "tokens/total": 7665408, + "tokens/train_per_sec_per_gpu": 26.58, + "tokens/trainable": 109672 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.3010976314544678, + "learning_rate": 6.32267616243259e-05, + "loss": 0.006106264889240265, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00612, + "step": 241, + "tokens/total": 7697280, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 110135 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.11426915228366852, + "learning_rate": 6.294119380711849e-05, + "loss": 0.0014383264351636171, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00144, + "step": 242, + "tokens/total": 7729440, + "tokens/train_per_sec_per_gpu": 25.66, + "tokens/trainable": 110563 + }, + { + "epoch": 0.94921875, + "grad_norm": 1.3792941570281982, + "learning_rate": 6.265529552442209e-05, + "loss": 0.003547664964571595, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00355, + "step": 243, + "tokens/total": 7761584, + "tokens/train_per_sec_per_gpu": 30.76, + "tokens/trainable": 111049 + }, + { + "epoch": 0.953125, + "grad_norm": 0.07595008611679077, + "learning_rate": 6.236907867363127e-05, + "loss": 0.0009103374904952943, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00091, + "step": 244, + "tokens/total": 7793824, + "tokens/train_per_sec_per_gpu": 28.23, + "tokens/trainable": 111495 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05425361171364784, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0013249842450022697, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00133, + "step": 245, + "tokens/total": 7825872, + "tokens/train_per_sec_per_gpu": 26.47, + "tokens/trainable": 111968 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.17112742364406586, + "learning_rate": 6.179573692313344e-05, + "loss": 0.001098912674933672, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.0011, + "step": 246, + "tokens/total": 7857840, + "tokens/train_per_sec_per_gpu": 30.09, + "tokens/trainable": 112416 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.10754083096981049, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0031517897732555866, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00316, + "step": 247, + "tokens/total": 7890000, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 112882 + }, + { + "epoch": 0.96875, + "grad_norm": 0.2477901428937912, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005755824968218803, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00577, + "step": 248, + "tokens/total": 7921984, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 113390 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.332253634929657, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.013207933865487576, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.0133, + "step": 249, + "tokens/total": 7953824, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 113904 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.18514207005500793, + "learning_rate": 6.064575550087316e-05, + "loss": 0.00523729994893074, + "memory/device_reserved (GiB)": 38.86, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00525, + "step": 250, + "tokens/total": 7985984, + "tokens/train_per_sec_per_gpu": 28.41, + "tokens/trainable": 114351 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.1852743923664093, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0031474479474127293, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00315, + "step": 251, + "tokens/total": 8018256, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 114842 + }, + { + "epoch": 0.984375, + "grad_norm": 0.43020468950271606, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0053521618247032166, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00537, + "step": 252, + "tokens/total": 8050528, + "tokens/train_per_sec_per_gpu": 23.97, + "tokens/trainable": 115263 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.048135045915842056, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0005759480991400778, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00058, + "step": 253, + "tokens/total": 8082496, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 115703 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.028733640909194946, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00040354474913328886, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0004, + "step": 254, + "tokens/total": 8114592, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 116157 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.2808869481086731, + "learning_rate": 5.920308275189541e-05, + "loss": 0.006315852981060743, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00634, + "step": 255, + "tokens/total": 8146752, + "tokens/train_per_sec_per_gpu": 26.44, + "tokens/trainable": 116567 + }, + { + "epoch": 1.0, + "grad_norm": 0.054996367543935776, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0009099108865484595, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00091, + "step": 256, + "tokens/total": 8178768, + "tokens/train_per_sec_per_gpu": 25.98, + "tokens/trainable": 117030 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.18114623427391052, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.002796629210934043, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0028, + "step": 257, + "tokens/total": 8210800, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 117517 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.1408185362815857, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0013210453325882554, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00132, + "step": 258, + "tokens/total": 8242768, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 118015 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.36587440967559814, + "learning_rate": 5.80457242789548e-05, + "loss": 0.002160525880753994, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00216, + "step": 259, + "tokens/total": 8274784, + "tokens/train_per_sec_per_gpu": 30.13, + "tokens/trainable": 118520 + }, + { + "epoch": 1.015625, + "grad_norm": 0.02628348581492901, + "learning_rate": 5.77560376810977e-05, + "loss": 0.0003391164937056601, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00034, + "step": 260, + "tokens/total": 8306720, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 118992 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.014996240846812725, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0002124913444276899, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00021, + "step": 261, + "tokens/total": 8338784, + "tokens/train_per_sec_per_gpu": 27.96, + "tokens/trainable": 119438 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.0050656464882195, + "learning_rate": 5.717633247526522e-05, + "loss": 7.557802018709481e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00008, + "step": 262, + "tokens/total": 8370720, + "tokens/train_per_sec_per_gpu": 26.9, + "tokens/trainable": 119881 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.02459302730858326, + "learning_rate": 5.688633799118971e-05, + "loss": 0.00024908874183893204, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00025, + "step": 263, + "tokens/total": 8402736, + "tokens/train_per_sec_per_gpu": 23.2, + "tokens/trainable": 120285 + }, + { + "epoch": 1.03125, + "grad_norm": 0.017813201993703842, + "learning_rate": 5.659626500889066e-05, + "loss": 0.00021895825921092182, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.23, + "memory/max_allocated (GiB)": 36.23, + "ppl": 1.00022, + "step": 264, + "tokens/total": 8434352, + "tokens/train_per_sec_per_gpu": 27.78, + "tokens/trainable": 120755 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.0038785531651228666, + "learning_rate": 5.6306125599488905e-05, + "loss": 6.556476728292182e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00007, + "step": 265, + "tokens/total": 8466208, + "tokens/train_per_sec_per_gpu": 29.86, + "tokens/trainable": 121194 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.18033447861671448, + "learning_rate": 5.601593183686955e-05, + "loss": 0.0016840343596413732, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00169, + "step": 266, + "tokens/total": 8498304, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 121670 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.22497256100177765, + "learning_rate": 5.572569579717961e-05, + "loss": 0.003026450052857399, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00303, + "step": 267, + "tokens/total": 8530400, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 122142 + }, + { + "epoch": 1.046875, + "grad_norm": 0.008878038264811039, + "learning_rate": 5.543542955832538e-05, + "loss": 8.136438555084169e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00008, + "step": 268, + "tokens/total": 8562608, + "tokens/train_per_sec_per_gpu": 27.27, + "tokens/trainable": 122585 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.19042901694774628, + "learning_rate": 5.514514519946986e-05, + "loss": 0.0020417363848537207, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00204, + "step": 269, + "tokens/total": 8594832, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 123091 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.003382593160495162, + "learning_rate": 5.485485480053015e-05, + "loss": 3.870048021781258e-05, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00004, + "step": 270, + "tokens/total": 8627120, + "tokens/train_per_sec_per_gpu": 26.36, + "tokens/trainable": 123505 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.2004430592060089, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.002532299840822816, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.00254, + "step": 271, + "tokens/total": 8657056, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 123953 + }, + { + "epoch": 1.0625, + "grad_norm": 0.24857956171035767, + "learning_rate": 5.42743042028204e-05, + "loss": 0.004157304298132658, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00417, + "step": 272, + "tokens/total": 8689104, + "tokens/train_per_sec_per_gpu": 24.05, + "tokens/trainable": 124400 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.05064955726265907, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0006531125982291996, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00065, + "step": 273, + "tokens/total": 8721136, + "tokens/train_per_sec_per_gpu": 29.3, + "tokens/trainable": 124870 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.12664659321308136, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0018034171080216765, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00181, + "step": 274, + "tokens/total": 8753200, + "tokens/train_per_sec_per_gpu": 26.52, + "tokens/trainable": 125333 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.01593201234936714, + "learning_rate": 5.340373499110935e-05, + "loss": 0.00010935973114101216, + "memory/device_reserved (GiB)": 38.84, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00011, + "step": 275, + "tokens/total": 8785168, + "tokens/train_per_sec_per_gpu": 30.45, + "tokens/trainable": 125834 + }, + { + "epoch": 1.078125, + "grad_norm": 0.0015056057600304484, + "learning_rate": 5.3113662008810304e-05, + "loss": 1.9860939573845826e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00002, + "step": 276, + "tokens/total": 8817520, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 126316 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.004457728937268257, + "learning_rate": 5.282366752473479e-05, + "loss": 2.716448398132343e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00003, + "step": 277, + "tokens/total": 8849360, + "tokens/train_per_sec_per_gpu": 29.63, + "tokens/trainable": 126798 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.03812727704644203, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.0003402164438739419, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.00034, + "step": 278, + "tokens/total": 8881152, + "tokens/train_per_sec_per_gpu": 24.5, + "tokens/trainable": 127223 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.011409996077418327, + "learning_rate": 5.224396231890232e-05, + "loss": 9.064963523996994e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.25, + "memory/max_allocated (GiB)": 36.25, + "ppl": 1.00009, + "step": 279, + "tokens/total": 8912512, + "tokens/train_per_sec_per_gpu": 29.78, + "tokens/trainable": 127672 + }, + { + "epoch": 1.09375, + "grad_norm": 0.037030577659606934, + "learning_rate": 5.195427572104522e-05, + "loss": 0.0003734154161065817, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.28, + "memory/max_allocated (GiB)": 36.28, + "ppl": 1.00037, + "step": 280, + "tokens/total": 8944416, + "tokens/train_per_sec_per_gpu": 26.76, + "tokens/trainable": 128116 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.0032268385402858257, + "learning_rate": 5.166471586820751e-05, + "loss": 2.529611811041832e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00003, + "step": 281, + "tokens/total": 8976272, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 128589 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.20227248966693878, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.0024647137615829706, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00247, + "step": 282, + "tokens/total": 9008560, + "tokens/train_per_sec_per_gpu": 27.84, + "tokens/trainable": 129036 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.43708595633506775, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0028322634752839804, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00284, + "step": 283, + "tokens/total": 9040496, + "tokens/train_per_sec_per_gpu": 29.25, + "tokens/trainable": 129485 + }, + { + "epoch": 1.109375, + "grad_norm": 0.09467080235481262, + "learning_rate": 5.079691724810461e-05, + "loss": 0.001156509853899479, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00116, + "step": 284, + "tokens/total": 9072448, + "tokens/train_per_sec_per_gpu": 24.69, + "tokens/trainable": 129920 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.1778785139322281, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.0016857109731063247, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00169, + "step": 285, + "tokens/total": 9104384, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 130417 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.0003376641252543777, + "learning_rate": 5.021923930849237e-05, + "loss": 5.597693871095544e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00001, + "step": 286, + "tokens/total": 9136416, + "tokens/train_per_sec_per_gpu": 29.99, + "tokens/trainable": 130903 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.019782084971666336, + "learning_rate": 4.99306927511967e-05, + "loss": 0.00012461614096537232, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00012, + "step": 287, + "tokens/total": 9168640, + "tokens/train_per_sec_per_gpu": 29.24, + "tokens/trainable": 131343 + }, + { + "epoch": 1.125, + "grad_norm": 0.9641065001487732, + "learning_rate": 4.964235714846775e-05, + "loss": 0.011515076272189617, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01158, + "step": 288, + "tokens/total": 9200544, + "tokens/train_per_sec_per_gpu": 27.04, + "tokens/trainable": 131763 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0010062052169814706, + "learning_rate": 4.9354244499126866e-05, + "loss": 9.489545846008696e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00001, + "step": 289, + "tokens/total": 9232624, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 132229 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.002373398980125785, + "learning_rate": 4.90663667927174e-05, + "loss": 1.573777262819931e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00002, + "step": 290, + "tokens/total": 9264768, + "tokens/train_per_sec_per_gpu": 29.5, + "tokens/trainable": 132678 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.0014718567254021764, + "learning_rate": 4.877873600900581e-05, + "loss": 9.888429303828161e-06, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00001, + "step": 291, + "tokens/total": 9296944, + "tokens/train_per_sec_per_gpu": 24.47, + "tokens/trainable": 133136 + }, + { + "epoch": 1.140625, + "grad_norm": 0.0018440725980326533, + "learning_rate": 4.849136411748306e-05, + "loss": 8.275845175376162e-06, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00001, + "step": 292, + "tokens/total": 9328848, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 133608 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.12918472290039062, + "learning_rate": 4.8204263076866574e-05, + "loss": 0.0009128568926826119, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00091, + "step": 293, + "tokens/total": 9360928, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 134108 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.019195031374692917, + "learning_rate": 4.791744483460251e-05, + "loss": 9.439548011869192e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00009, + "step": 294, + "tokens/total": 9393040, + "tokens/train_per_sec_per_gpu": 25.71, + "tokens/trainable": 134541 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.1602204144001007, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0004640200058929622, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00046, + "step": 295, + "tokens/total": 9425248, + "tokens/train_per_sec_per_gpu": 24.5, + "tokens/trainable": 134966 + }, + { + "epoch": 1.15625, + "grad_norm": 0.5406288504600525, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.0034592358861118555, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00347, + "step": 296, + "tokens/total": 9457264, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 135402 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.004333238583058119, + "learning_rate": 4.705880619288153e-05, + "loss": 3.075868880841881e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00003, + "step": 297, + "tokens/total": 9489056, + "tokens/train_per_sec_per_gpu": 24.87, + "tokens/trainable": 135804 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.006247827783226967, + "learning_rate": 4.677323837567412e-05, + "loss": 2.383375249337405e-05, + "memory/device_reserved (GiB)": 38.7, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00002, + "step": 298, + "tokens/total": 9521024, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 136231 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.0779612734913826, + "learning_rate": 4.6488012907598146e-05, + "loss": 0.0004679379053413868, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00047, + "step": 299, + "tokens/total": 9553120, + "tokens/train_per_sec_per_gpu": 28.37, + "tokens/trainable": 136705 + }, + { + "epoch": 1.171875, + "grad_norm": 0.01816786825656891, + "learning_rate": 4.620314165804964e-05, + "loss": 6.801338167861104e-05, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00007, + "step": 300, + "tokens/total": 9585040, + "tokens/train_per_sec_per_gpu": 28.78, + "tokens/trainable": 137178 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.9530329704284668, + "learning_rate": 4.591863648168407e-05, + "loss": 0.008843549527227879, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00888, + "step": 301, + "tokens/total": 9617120, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 137643 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0371168851852417, + "learning_rate": 4.5634509217923135e-05, + "loss": 0.00014635953994002193, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00015, + "step": 302, + "tokens/total": 9649024, + "tokens/train_per_sec_per_gpu": 26.74, + "tokens/trainable": 138078 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.08589725196361542, + "learning_rate": 4.535077169046201e-05, + "loss": 0.00043714733328670263, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00044, + "step": 303, + "tokens/total": 9680832, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 138515 + }, + { + "epoch": 1.1875, + "grad_norm": 0.358940988779068, + "learning_rate": 4.506743570677743e-05, + "loss": 0.006240838672965765, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00626, + "step": 304, + "tokens/total": 9713040, + "tokens/train_per_sec_per_gpu": 29.29, + "tokens/trainable": 138948 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.12659867107868195, + "learning_rate": 4.478451305763618e-05, + "loss": 0.0008667556685395539, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00087, + "step": 305, + "tokens/total": 9744976, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 139408 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.08289908617734909, + "learning_rate": 4.450201551660454e-05, + "loss": 0.0006384386215358973, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.02, + "memory/max_allocated (GiB)": 36.02, + "ppl": 1.00064, + "step": 306, + "tokens/total": 9774784, + "tokens/train_per_sec_per_gpu": 27.76, + "tokens/trainable": 139840 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.19195082783699036, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.002342985011637211, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00235, + "step": 307, + "tokens/total": 9806656, + "tokens/train_per_sec_per_gpu": 27.6, + "tokens/trainable": 140281 + }, + { + "epoch": 1.203125, + "grad_norm": 0.07421132177114487, + "learning_rate": 4.393834276419352e-05, + "loss": 0.0007085398538038135, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00071, + "step": 308, + "tokens/total": 9838720, + "tokens/train_per_sec_per_gpu": 30.88, + "tokens/trainable": 140776 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.004596009384840727, + "learning_rate": 4.36571910095382e-05, + "loss": 2.8969257982680574e-05, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00003, + "step": 309, + "tokens/total": 9870784, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 141228 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.04099060595035553, + "learning_rate": 4.337651127546448e-05, + "loss": 0.00030985785997472703, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00031, + "step": 310, + "tokens/total": 9902784, + "tokens/train_per_sec_per_gpu": 30.8, + "tokens/trainable": 141679 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.18840794265270233, + "learning_rate": 4.3096315242201736e-05, + "loss": 0.0020186181645840406, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00202, + "step": 311, + "tokens/total": 9934768, + "tokens/train_per_sec_per_gpu": 26.72, + "tokens/trainable": 142122 + }, + { + "epoch": 1.21875, + "grad_norm": 0.5018841028213501, + "learning_rate": 4.2816614569850635e-05, + "loss": 0.006148474756628275, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00617, + "step": 312, + "tokens/total": 9966960, + "tokens/train_per_sec_per_gpu": 27.27, + "tokens/trainable": 142561 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.1730002760887146, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.003579454030841589, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00359, + "step": 313, + "tokens/total": 9999104, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 143046 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.6586350202560425, + "learning_rate": 4.225874584473174e-05, + "loss": 0.0015231478027999401, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00152, + "step": 314, + "tokens/total": 10031136, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 143493 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.011105087585747242, + "learning_rate": 4.19806010071587e-05, + "loss": 0.00011838486534543335, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00012, + "step": 315, + "tokens/total": 10063184, + "tokens/train_per_sec_per_gpu": 29.1, + "tokens/trainable": 143956 + }, + { + "epoch": 1.234375, + "grad_norm": 0.2628834843635559, + "learning_rate": 4.170299795992081e-05, + "loss": 0.003402995876967907, + "memory/device_reserved (GiB)": 38.82, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00341, + "step": 316, + "tokens/total": 10095312, + "tokens/train_per_sec_per_gpu": 28.29, + "tokens/trainable": 144411 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.13821958005428314, + "learning_rate": 4.142594825521398e-05, + "loss": 0.0016696923412382603, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00167, + "step": 317, + "tokens/total": 10127440, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 144892 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.0023418583441525698, + "learning_rate": 4.114946342220728e-05, + "loss": 2.6753859856398776e-05, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00003, + "step": 318, + "tokens/total": 10159184, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 145339 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.0008433983894065022, + "learning_rate": 4.087355496656321e-05, + "loss": 1.602486736373976e-05, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00002, + "step": 319, + "tokens/total": 10191216, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 145811 + }, + { + "epoch": 1.25, + "grad_norm": 0.12588869035243988, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0011584729654714465, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00116, + "step": 320, + "tokens/total": 10223408, + "tokens/train_per_sec_per_gpu": 32.27, + "tokens/trainable": 146314 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.028395529836416245, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.00015757072833366692, + "memory/device_reserved (GiB)": 38.94, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00016, + "step": 321, + "tokens/total": 10255536, + "tokens/train_per_sec_per_gpu": 26.68, + "tokens/trainable": 146781 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.039599668234586716, + "learning_rate": 4.004940255778431e-05, + "loss": 0.000321136845741421, + "memory/device_reserved (GiB)": 37.39, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00032, + "step": 322, + "tokens/total": 10287664, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 147225 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.11097095906734467, + "learning_rate": 3.977591418134619e-05, + "loss": 0.0004549310833681375, + "memory/device_reserved (GiB)": 37.77, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00046, + "step": 323, + "tokens/total": 10320080, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 147673 + }, + { + "epoch": 1.265625, + "grad_norm": 0.19018623232841492, + "learning_rate": 3.95030593412612e-05, + "loss": 0.0028336881659924984, + "memory/device_reserved (GiB)": 38.53, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00284, + "step": 324, + "tokens/total": 10351984, + "tokens/train_per_sec_per_gpu": 27.33, + "tokens/trainable": 148103 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.4650349020957947, + "learning_rate": 3.923084939213296e-05, + "loss": 0.006978687364608049, + "memory/device_reserved (GiB)": 38.53, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.007, + "step": 325, + "tokens/total": 10384256, + "tokens/train_per_sec_per_gpu": 26.21, + "tokens/trainable": 148535 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.14811205863952637, + "learning_rate": 3.895929566172861e-05, + "loss": 0.0014966176822781563, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0015, + "step": 326, + "tokens/total": 10416208, + "tokens/train_per_sec_per_gpu": 28.37, + "tokens/trainable": 148999 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.11302439123392105, + "learning_rate": 3.868840945050728e-05, + "loss": 0.0011318891774863005, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00113, + "step": 327, + "tokens/total": 10448224, + "tokens/train_per_sec_per_gpu": 26.37, + "tokens/trainable": 149431 + }, + { + "epoch": 1.28125, + "grad_norm": 0.10916705429553986, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0009605163359083235, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00096, + "step": 328, + "tokens/total": 10480448, + "tokens/train_per_sec_per_gpu": 28.27, + "tokens/trainable": 149886 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.0004018457839265466, + "learning_rate": 3.814868464809027e-05, + "loss": 7.040077434794512e-06, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00001, + "step": 329, + "tokens/total": 10512096, + "tokens/train_per_sec_per_gpu": 25.92, + "tokens/trainable": 150288 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.03717898577451706, + "learning_rate": 3.787986851704667e-05, + "loss": 0.0001680658315308392, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00017, + "step": 330, + "tokens/total": 10544000, + "tokens/train_per_sec_per_gpu": 28.52, + "tokens/trainable": 150733 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.21975846588611603, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00015534088015556335, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00016, + "step": 331, + "tokens/total": 10576048, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 151207 + }, + { + "epoch": 1.296875, + "grad_norm": 0.10715529322624207, + "learning_rate": 3.734438472750619e-05, + "loss": 0.0012046336196362972, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00121, + "step": 332, + "tokens/total": 10608048, + "tokens/train_per_sec_per_gpu": 31.08, + "tokens/trainable": 151694 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.13602837920188904, + "learning_rate": 3.707773935267552e-05, + "loss": 0.00039335948531515896, + "memory/device_reserved (GiB)": 38.65, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00039, + "step": 333, + "tokens/total": 10640176, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 152188 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.002963080769404769, + "learning_rate": 3.68118397962661e-05, + "loss": 6.6289721871726215e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00001, + "step": 334, + "tokens/total": 10672224, + "tokens/train_per_sec_per_gpu": 25.55, + "tokens/trainable": 152596 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.007236327510327101, + "learning_rate": 3.654669712344384e-05, + "loss": 5.341085125110112e-05, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00005, + "step": 335, + "tokens/total": 10704192, + "tokens/train_per_sec_per_gpu": 30.44, + "tokens/trainable": 153052 + }, + { + "epoch": 1.3125, + "grad_norm": 0.0897810310125351, + "learning_rate": 3.628232236787763e-05, + "loss": 0.0008687702938914299, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00087, + "step": 336, + "tokens/total": 10736272, + "tokens/train_per_sec_per_gpu": 25.33, + "tokens/trainable": 153491 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.011829929426312447, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.00011201453162357211, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00011, + "step": 337, + "tokens/total": 10768528, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 153983 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.25218063592910767, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0022679665125906467, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00227, + "step": 338, + "tokens/total": 10800464, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 154464 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.00034703267738223076, + "learning_rate": 3.549391545931585e-05, + "loss": 2.884747800635523e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.0, + "step": 339, + "tokens/total": 10832512, + "tokens/train_per_sec_per_gpu": 29.7, + "tokens/trainable": 154924 + }, + { + "epoch": 1.328125, + "grad_norm": 0.004890776239335537, + "learning_rate": 3.5232722063479914e-05, + "loss": 2.2697908207192086e-05, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00002, + "step": 340, + "tokens/total": 10864624, + "tokens/train_per_sec_per_gpu": 25.46, + "tokens/trainable": 155373 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.0013245025184005499, + "learning_rate": 3.49723512647657e-05, + "loss": 8.980642633105163e-06, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00001, + "step": 341, + "tokens/total": 10896816, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 155873 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.03804129734635353, + "learning_rate": 3.471281389826491e-05, + "loss": 0.00021880728309042752, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00022, + "step": 342, + "tokens/total": 10928864, + "tokens/train_per_sec_per_gpu": 25.31, + "tokens/trainable": 156278 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.4880214333534241, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.005397590808570385, + "memory/device_reserved (GiB)": 38.75, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00541, + "step": 343, + "tokens/total": 10960816, + "tokens/train_per_sec_per_gpu": 29.58, + "tokens/trainable": 156733 + }, + { + "epoch": 1.34375, + "grad_norm": 0.37780871987342834, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0029298821464180946, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00293, + "step": 344, + "tokens/total": 10992912, + "tokens/train_per_sec_per_gpu": 28.4, + "tokens/trainable": 157203 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.0034261250402778387, + "learning_rate": 3.3939310220027456e-05, + "loss": 2.1256186300888658e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00002, + "step": 345, + "tokens/total": 11024832, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 157707 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.0005507867899723351, + "learning_rate": 3.3683214232914404e-05, + "loss": 6.991718692006543e-06, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.65, + "memory/max_allocated (GiB)": 36.65, + "ppl": 1.00001, + "step": 346, + "tokens/total": 11056992, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 158180 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.0016383324982598424, + "learning_rate": 3.342800532426873e-05, + "loss": 1.2630136552616023e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00001, + "step": 347, + "tokens/total": 11089264, + "tokens/train_per_sec_per_gpu": 28.55, + "tokens/trainable": 158646 + }, + { + "epoch": 1.359375, + "grad_norm": 0.0020912184845656157, + "learning_rate": 3.317369411437484e-05, + "loss": 1.536883064545691e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00002, + "step": 348, + "tokens/total": 11120976, + "tokens/train_per_sec_per_gpu": 29.34, + "tokens/trainable": 159115 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.0016828920925036073, + "learning_rate": 3.292029118616024e-05, + "loss": 1.334734679403482e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00001, + "step": 349, + "tokens/total": 11152976, + "tokens/train_per_sec_per_gpu": 26.34, + "tokens/trainable": 159538 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.02520265430212021, + "learning_rate": 3.266780708475511e-05, + "loss": 9.595962183084339e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0001, + "step": 350, + "tokens/total": 11185056, + "tokens/train_per_sec_per_gpu": 27.61, + "tokens/trainable": 160016 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.009836713783442974, + "learning_rate": 3.241625231705354e-05, + "loss": 7.521865336457267e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00008, + "step": 351, + "tokens/total": 11217088, + "tokens/train_per_sec_per_gpu": 28.51, + "tokens/trainable": 160475 + }, + { + "epoch": 1.375, + "grad_norm": 0.0698060467839241, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0003666624252218753, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00037, + "step": 352, + "tokens/total": 11249088, + "tokens/train_per_sec_per_gpu": 28.31, + "tokens/trainable": 160952 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.0018428800394758582, + "learning_rate": 3.191597261653475e-05, + "loss": 1.3194729945098516e-05, + "memory/device_reserved (GiB)": 38.77, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00001, + "step": 353, + "tokens/total": 11281328, + "tokens/train_per_sec_per_gpu": 27.4, + "tokens/trainable": 161420 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.020005319267511368, + "learning_rate": 3.166726850239794e-05, + "loss": 0.00017299644241575152, + "memory/device_reserved (GiB)": 38.47, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00017, + "step": 354, + "tokens/total": 11313200, + "tokens/train_per_sec_per_gpu": 28.31, + "tokens/trainable": 161868 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.00018166302470490336, + "learning_rate": 3.141953535845912e-05, + "loss": 4.138089934713207e-06, + "memory/device_reserved (GiB)": 38.47, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0, + "step": 355, + "tokens/total": 11345008, + "tokens/train_per_sec_per_gpu": 29.74, + "tokens/trainable": 162318 + }, + { + "epoch": 1.390625, + "grad_norm": 0.009165152907371521, + "learning_rate": 3.11727834939056e-05, + "loss": 6.729022425133735e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00007, + "step": 356, + "tokens/total": 11377216, + "tokens/train_per_sec_per_gpu": 26.26, + "tokens/trainable": 162789 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.2891274690628052, + "learning_rate": 3.092702317708967e-05, + "loss": 0.005585570354014635, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.0056, + "step": 357, + "tokens/total": 11409536, + "tokens/train_per_sec_per_gpu": 30.34, + "tokens/trainable": 163254 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.0007154783816076815, + "learning_rate": 3.0682264635101276e-05, + "loss": 5.591478839050978e-06, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00001, + "step": 358, + "tokens/total": 11441552, + "tokens/train_per_sec_per_gpu": 28.74, + "tokens/trainable": 163715 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.01731656678020954, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.0001050045175361447, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00011, + "step": 359, + "tokens/total": 11473376, + "tokens/train_per_sec_per_gpu": 28.79, + "tokens/trainable": 164197 + }, + { + "epoch": 1.40625, + "grad_norm": 0.11494546383619308, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.00048459687968716025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00048, + "step": 360, + "tokens/total": 11505312, + "tokens/train_per_sec_per_gpu": 26.54, + "tokens/trainable": 164661 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.08215854316949844, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.0002725277154240757, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00027, + "step": 361, + "tokens/total": 11537536, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 165111 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.024266909807920456, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.0001677784021012485, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 35.94, + "memory/max_allocated (GiB)": 35.94, + "ppl": 1.00017, + "step": 362, + "tokens/total": 11567216, + "tokens/train_per_sec_per_gpu": 25.3, + "tokens/trainable": 165552 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.008205167017877102, + "learning_rate": 2.9473853553877484e-05, + "loss": 3.423610905883834e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00003, + "step": 363, + "tokens/total": 11599312, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 166026 + }, + { + "epoch": 1.421875, + "grad_norm": 0.4230109453201294, + "learning_rate": 2.9235318065647e-05, + "loss": 0.005893740337342024, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00591, + "step": 364, + "tokens/total": 11631344, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 166486 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.1095249131321907, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.0006186347454786301, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00062, + "step": 365, + "tokens/total": 11663632, + "tokens/train_per_sec_per_gpu": 30.17, + "tokens/trainable": 166959 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.0014034683117642999, + "learning_rate": 2.8761473491751258e-05, + "loss": 1.2728614819934592e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.27, + "memory/max_allocated (GiB)": 36.27, + "ppl": 1.00001, + "step": 366, + "tokens/total": 11695360, + "tokens/train_per_sec_per_gpu": 28.19, + "tokens/trainable": 167394 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.044087354093790054, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.0001466910180170089, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00015, + "step": 367, + "tokens/total": 11727184, + "tokens/train_per_sec_per_gpu": 23.64, + "tokens/trainable": 167832 + }, + { + "epoch": 1.4375, + "grad_norm": 0.0005008528823964298, + "learning_rate": 2.829199644117484e-05, + "loss": 1.1319095392536838e-05, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00001, + "step": 368, + "tokens/total": 11759312, + "tokens/train_per_sec_per_gpu": 25.79, + "tokens/trainable": 168242 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.02015153504908085, + "learning_rate": 2.8058920186702553e-05, + "loss": 0.00011158352572238073, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00011, + "step": 369, + "tokens/total": 11791536, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 168715 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.00031807494815438986, + "learning_rate": 2.782696506053033e-05, + "loss": 6.533119631058071e-06, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00001, + "step": 370, + "tokens/total": 11823632, + "tokens/train_per_sec_per_gpu": 26.68, + "tokens/trainable": 169147 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.4510185122489929, + "learning_rate": 2.7596140715257824e-05, + "loss": 0.004431337118148804, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00444, + "step": 371, + "tokens/total": 11855392, + "tokens/train_per_sec_per_gpu": 28.61, + "tokens/trainable": 169628 + }, + { + "epoch": 1.453125, + "grad_norm": 0.0034806979820132256, + "learning_rate": 2.7366456756428184e-05, + "loss": 2.722550561884418e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00003, + "step": 372, + "tokens/total": 11887424, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 170085 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.018526704981923103, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.00013849925016984344, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00014, + "step": 373, + "tokens/total": 11919280, + "tokens/train_per_sec_per_gpu": 29.39, + "tokens/trainable": 170584 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.010734346695244312, + "learning_rate": 2.691054818259188e-05, + "loss": 1.963317481568083e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00002, + "step": 374, + "tokens/total": 11951488, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 171058 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.004906357266008854, + "learning_rate": 2.6684342539801933e-05, + "loss": 4.44888137280941e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00004, + "step": 375, + "tokens/total": 11983504, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 171518 + }, + { + "epoch": 1.46875, + "grad_norm": 0.0035627244506031275, + "learning_rate": 2.645931522709877e-05, + "loss": 1.879796946013812e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00002, + "step": 376, + "tokens/total": 12015360, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 172007 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.014949376694858074, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.0001050513528753072, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00011, + "step": 377, + "tokens/total": 12047296, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 172468 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.024432353675365448, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.00010428359382785857, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0001, + "step": 378, + "tokens/total": 12079392, + "tokens/train_per_sec_per_gpu": 27.91, + "tokens/trainable": 172923 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.5905564427375793, + "learning_rate": 2.579139666504821e-05, + "loss": 0.019045401364564896, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01923, + "step": 379, + "tokens/total": 12111616, + "tokens/train_per_sec_per_gpu": 28.26, + "tokens/trainable": 173370 + }, + { + "epoch": 1.484375, + "grad_norm": 0.0027490397915244102, + "learning_rate": 2.557117581955798e-05, + "loss": 3.106705116806552e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00003, + "step": 380, + "tokens/total": 12143536, + "tokens/train_per_sec_per_gpu": 28.66, + "tokens/trainable": 173837 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.10220319032669067, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.0007754361140541732, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00078, + "step": 381, + "tokens/total": 12175792, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 174294 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.06388501822948456, + "learning_rate": 2.5134417202396277e-05, + "loss": 2.506363671272993e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00003, + "step": 382, + "tokens/total": 12207472, + "tokens/train_per_sec_per_gpu": 25.85, + "tokens/trainable": 174701 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.006029566749930382, + "learning_rate": 2.491789760603361e-05, + "loss": 5.214785414864309e-05, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00005, + "step": 383, + "tokens/total": 12239520, + "tokens/train_per_sec_per_gpu": 26.67, + "tokens/trainable": 175131 + }, + { + "epoch": 1.5, + "grad_norm": 0.015079713426530361, + "learning_rate": 2.4702629848741764e-05, + "loss": 0.00011988497863058001, + "memory/device_reserved (GiB)": 39.17, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00012, + "step": 384, + "tokens/total": 12271600, + "tokens/train_per_sec_per_gpu": 28.23, + "tokens/trainable": 175586 + }, + { + "epoch": 1.50390625, + "grad_norm": 0.04362509027123451, + "learning_rate": 2.4488622888690785e-05, + "loss": 0.0002582473389338702, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00026, + "step": 385, + "tokens/total": 12303648, + "tokens/train_per_sec_per_gpu": 28.44, + "tokens/trainable": 176026 + }, + { + "epoch": 1.5078125, + "grad_norm": 0.28308039903640747, + "learning_rate": 2.427588563158384e-05, + "loss": 0.0025924108922481537, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0026, + "step": 386, + "tokens/total": 12335728, + "tokens/train_per_sec_per_gpu": 28.2, + "tokens/trainable": 176512 + }, + { + "epoch": 1.51171875, + "grad_norm": 0.21685272455215454, + "learning_rate": 2.406442693028651e-05, + "loss": 0.002609315561130643, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00261, + "step": 387, + "tokens/total": 12367824, + "tokens/train_per_sec_per_gpu": 25.55, + "tokens/trainable": 176943 + }, + { + "epoch": 1.515625, + "grad_norm": 0.0010271539213135839, + "learning_rate": 2.3854255584458547e-05, + "loss": 2.1533389372052625e-05, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00002, + "step": 388, + "tokens/total": 12400192, + "tokens/train_per_sec_per_gpu": 26.36, + "tokens/trainable": 177370 + }, + { + "epoch": 1.51953125, + "grad_norm": 0.06616228073835373, + "learning_rate": 2.3645380340187508e-05, + "loss": 0.0005350579158402979, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00054, + "step": 389, + "tokens/total": 12432224, + "tokens/train_per_sec_per_gpu": 29.11, + "tokens/trainable": 177851 + }, + { + "epoch": 1.5234375, + "grad_norm": 0.05946248397231102, + "learning_rate": 2.3437809889624914e-05, + "loss": 0.0005003588157705963, + "memory/device_reserved (GiB)": 39.52, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.0005, + "step": 390, + "tokens/total": 12464048, + "tokens/train_per_sec_per_gpu": 29.09, + "tokens/trainable": 178310 + }, + { + "epoch": 1.52734375, + "grad_norm": 0.017372427508234978, + "learning_rate": 2.3231552870624487e-05, + "loss": 0.0001355188578600064, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00014, + "step": 391, + "tokens/total": 12493776, + "tokens/train_per_sec_per_gpu": 30.96, + "tokens/trainable": 178736 + }, + { + "epoch": 1.53125, + "grad_norm": 0.12941808998584747, + "learning_rate": 2.3026617866382657e-05, + "loss": 0.003045230871066451, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00305, + "step": 392, + "tokens/total": 12526000, + "tokens/train_per_sec_per_gpu": 30.37, + "tokens/trainable": 179233 + }, + { + "epoch": 1.53515625, + "grad_norm": 0.014790852554142475, + "learning_rate": 2.2823013405081507e-05, + "loss": 0.00013205324648879468, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00013, + "step": 393, + "tokens/total": 12557968, + "tokens/train_per_sec_per_gpu": 32.27, + "tokens/trainable": 179730 + }, + { + "epoch": 1.5390625, + "grad_norm": 0.02042844519019127, + "learning_rate": 2.2620747959533722e-05, + "loss": 0.00022655579959973693, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00023, + "step": 394, + "tokens/total": 12590096, + "tokens/train_per_sec_per_gpu": 31.47, + "tokens/trainable": 180227 + }, + { + "epoch": 1.54296875, + "grad_norm": 0.005440943408757448, + "learning_rate": 2.2419829946830123e-05, + "loss": 4.6342807763721794e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00005, + "step": 395, + "tokens/total": 12622160, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 180687 + }, + { + "epoch": 1.546875, + "grad_norm": 0.004878366366028786, + "learning_rate": 2.2220267727989325e-05, + "loss": 5.4212974646361545e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00005, + "step": 396, + "tokens/total": 12654336, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 181141 + }, + { + "epoch": 1.55078125, + "grad_norm": 0.20873922109603882, + "learning_rate": 2.202206960760984e-05, + "loss": 0.0026359721086919308, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00264, + "step": 397, + "tokens/total": 12686400, + "tokens/train_per_sec_per_gpu": 30.13, + "tokens/trainable": 181648 + }, + { + "epoch": 1.5546875, + "grad_norm": 0.12369472533464432, + "learning_rate": 2.182524383352446e-05, + "loss": 0.0010878165485337377, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.00109, + "step": 398, + "tokens/total": 12718144, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 182109 + }, + { + "epoch": 1.55859375, + "grad_norm": 0.0010841770563274622, + "learning_rate": 2.1629798596457056e-05, + "loss": 1.8380069377599284e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00002, + "step": 399, + "tokens/total": 12750256, + "tokens/train_per_sec_per_gpu": 26.26, + "tokens/trainable": 182559 + }, + { + "epoch": 1.5625, + "grad_norm": 0.0010006122756749392, + "learning_rate": 2.1435742029681725e-05, + "loss": 1.991226599784568e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00002, + "step": 400, + "tokens/total": 12782224, + "tokens/train_per_sec_per_gpu": 27.59, + "tokens/trainable": 182985 + }, + { + "epoch": 1.56640625, + "grad_norm": 0.0011391225270926952, + "learning_rate": 2.124308220868431e-05, + "loss": 2.2474639990832657e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00002, + "step": 401, + "tokens/total": 12814384, + "tokens/train_per_sec_per_gpu": 26.95, + "tokens/trainable": 183440 + }, + { + "epoch": 1.5703125, + "grad_norm": 0.023268507793545723, + "learning_rate": 2.105182715082638e-05, + "loss": 0.00032951770117506385, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00033, + "step": 402, + "tokens/total": 12846240, + "tokens/train_per_sec_per_gpu": 28.77, + "tokens/trainable": 183917 + }, + { + "epoch": 1.57421875, + "grad_norm": 0.11701280623674393, + "learning_rate": 2.0861984815011552e-05, + "loss": 0.001444364432245493, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00145, + "step": 403, + "tokens/total": 12878240, + "tokens/train_per_sec_per_gpu": 25.07, + "tokens/trainable": 184348 + }, + { + "epoch": 1.578125, + "grad_norm": 0.004240577574819326, + "learning_rate": 2.0673563101354323e-05, + "loss": 2.100174970109947e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00002, + "step": 404, + "tokens/total": 12910176, + "tokens/train_per_sec_per_gpu": 26.38, + "tokens/trainable": 184786 + }, + { + "epoch": 1.58203125, + "grad_norm": 0.01977616548538208, + "learning_rate": 2.0486569850851317e-05, + "loss": 8.987118053482845e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00009, + "step": 405, + "tokens/total": 12942352, + "tokens/train_per_sec_per_gpu": 28.17, + "tokens/trainable": 185249 + }, + { + "epoch": 1.5859375, + "grad_norm": 0.030095387250185013, + "learning_rate": 2.0301012845054956e-05, + "loss": 0.0002793738676700741, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00028, + "step": 406, + "tokens/total": 12974368, + "tokens/train_per_sec_per_gpu": 26.3, + "tokens/trainable": 185680 + }, + { + "epoch": 1.58984375, + "grad_norm": 0.0017381443176418543, + "learning_rate": 2.011689980574966e-05, + "loss": 2.727654828049708e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00003, + "step": 407, + "tokens/total": 13006464, + "tokens/train_per_sec_per_gpu": 26.25, + "tokens/trainable": 186100 + }, + { + "epoch": 1.59375, + "grad_norm": 0.09072195738554001, + "learning_rate": 1.993423839463052e-05, + "loss": 0.0006338813109323382, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00063, + "step": 408, + "tokens/total": 13038320, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 186565 + }, + { + "epoch": 1.59765625, + "grad_norm": 0.004482298158109188, + "learning_rate": 1.975303621298445e-05, + "loss": 5.0280330469831824e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00005, + "step": 409, + "tokens/total": 13070416, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 187017 + }, + { + "epoch": 1.6015625, + "grad_norm": 0.03788210079073906, + "learning_rate": 1.957330080137385e-05, + "loss": 0.000300457701086998, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0003, + "step": 410, + "tokens/total": 13102352, + "tokens/train_per_sec_per_gpu": 27.69, + "tokens/trainable": 187468 + }, + { + "epoch": 1.60546875, + "grad_norm": 0.011773839592933655, + "learning_rate": 1.9395039639322864e-05, + "loss": 9.259363287128508e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00009, + "step": 411, + "tokens/total": 13134096, + "tokens/train_per_sec_per_gpu": 31.06, + "tokens/trainable": 187961 + }, + { + "epoch": 1.609375, + "grad_norm": 0.0031615474727004766, + "learning_rate": 1.9218260145006073e-05, + "loss": 3.231812661397271e-05, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00003, + "step": 412, + "tokens/total": 13166256, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 188447 + }, + { + "epoch": 1.61328125, + "grad_norm": 0.07778825610876083, + "learning_rate": 1.904296967493982e-05, + "loss": 0.0005980221321806312, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.0006, + "step": 413, + "tokens/total": 13198048, + "tokens/train_per_sec_per_gpu": 29.98, + "tokens/trainable": 188903 + }, + { + "epoch": 1.6171875, + "grad_norm": 0.011902395635843277, + "learning_rate": 1.8869175523676064e-05, + "loss": 0.00013689698243979365, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.66, + "memory/max_allocated (GiB)": 36.66, + "ppl": 1.00014, + "step": 414, + "tokens/total": 13230304, + "tokens/train_per_sec_per_gpu": 30.8, + "tokens/trainable": 189398 + }, + { + "epoch": 1.62109375, + "grad_norm": 0.08490622788667679, + "learning_rate": 1.869688492349885e-05, + "loss": 0.0008053273777477443, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00081, + "step": 415, + "tokens/total": 13262288, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 189850 + }, + { + "epoch": 1.625, + "grad_norm": 0.11783187091350555, + "learning_rate": 1.85261050441233e-05, + "loss": 0.0007362872711382806, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00074, + "step": 416, + "tokens/total": 13294480, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 190306 + }, + { + "epoch": 1.62890625, + "grad_norm": 0.26603376865386963, + "learning_rate": 1.8356842992397304e-05, + "loss": 0.002245941199362278, + "memory/device_reserved (GiB)": 39.53, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00225, + "step": 417, + "tokens/total": 13326608, + "tokens/train_per_sec_per_gpu": 28.38, + "tokens/trainable": 190746 + }, + { + "epoch": 1.6328125, + "grad_norm": 0.003691659774631262, + "learning_rate": 1.8189105812005714e-05, + "loss": 4.167412407696247e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00004, + "step": 418, + "tokens/total": 13358592, + "tokens/train_per_sec_per_gpu": 26.1, + "tokens/trainable": 191156 + }, + { + "epoch": 1.63671875, + "grad_norm": 0.021083569154143333, + "learning_rate": 1.802290048317732e-05, + "loss": 0.00017740413022693247, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00018, + "step": 419, + "tokens/total": 13390592, + "tokens/train_per_sec_per_gpu": 24.49, + "tokens/trainable": 191573 + }, + { + "epoch": 1.640625, + "grad_norm": 0.003719380358234048, + "learning_rate": 1.785823392239424e-05, + "loss": 4.420262484927662e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00004, + "step": 420, + "tokens/total": 13422672, + "tokens/train_per_sec_per_gpu": 27.22, + "tokens/trainable": 192014 + }, + { + "epoch": 1.64453125, + "grad_norm": 0.044042862951755524, + "learning_rate": 1.7695112982104225e-05, + "loss": 0.0004268632619641721, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00043, + "step": 421, + "tokens/total": 13454624, + "tokens/train_per_sec_per_gpu": 27.25, + "tokens/trainable": 192453 + }, + { + "epoch": 1.6484375, + "grad_norm": 0.0021698337513953447, + "learning_rate": 1.7533544450435433e-05, + "loss": 2.580507134553045e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.31, + "memory/max_allocated (GiB)": 36.31, + "ppl": 1.00003, + "step": 422, + "tokens/total": 13486208, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 192886 + }, + { + "epoch": 1.65234375, + "grad_norm": 0.019405441358685493, + "learning_rate": 1.7373535050913946e-05, + "loss": 0.00013853301061317325, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.71, + "memory/max_allocated (GiB)": 36.71, + "ppl": 1.00014, + "step": 423, + "tokens/total": 13518496, + "tokens/train_per_sec_per_gpu": 25.15, + "tokens/trainable": 193323 + }, + { + "epoch": 1.65625, + "grad_norm": 0.005207414738833904, + "learning_rate": 1.721509144218405e-05, + "loss": 6.985871004872024e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00007, + "step": 424, + "tokens/total": 13550768, + "tokens/train_per_sec_per_gpu": 27.59, + "tokens/trainable": 193753 + }, + { + "epoch": 1.66015625, + "grad_norm": 0.01700798235833645, + "learning_rate": 1.705822021773101e-05, + "loss": 0.00024273117014672607, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00024, + "step": 425, + "tokens/total": 13582960, + "tokens/train_per_sec_per_gpu": 30.03, + "tokens/trainable": 194240 + }, + { + "epoch": 1.6640625, + "grad_norm": 0.2864547073841095, + "learning_rate": 1.69029279056068e-05, + "loss": 0.004108238499611616, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00412, + "step": 426, + "tokens/total": 13615056, + "tokens/train_per_sec_per_gpu": 29.6, + "tokens/trainable": 194680 + }, + { + "epoch": 1.66796875, + "grad_norm": 0.012734112329781055, + "learning_rate": 1.6749220968158415e-05, + "loss": 9.277237404603511e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00009, + "step": 427, + "tokens/total": 13647088, + "tokens/train_per_sec_per_gpu": 32.17, + "tokens/trainable": 195186 + }, + { + "epoch": 1.671875, + "grad_norm": 0.04110024869441986, + "learning_rate": 1.659710580175893e-05, + "loss": 0.00021035004465375096, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00021, + "step": 428, + "tokens/total": 13679168, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 195693 + }, + { + "epoch": 1.67578125, + "grad_norm": 0.03119366057217121, + "learning_rate": 1.644658873654133e-05, + "loss": 0.00022089436242822558, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00022, + "step": 429, + "tokens/total": 13711120, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 196130 + }, + { + "epoch": 1.6796875, + "grad_norm": 0.0020074129570275545, + "learning_rate": 1.629767603613508e-05, + "loss": 1.550407614558935e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00002, + "step": 430, + "tokens/total": 13743088, + "tokens/train_per_sec_per_gpu": 24.67, + "tokens/trainable": 196530 + }, + { + "epoch": 1.68359375, + "grad_norm": 0.040013547986745834, + "learning_rate": 1.615037389740547e-05, + "loss": 0.0006465734331868589, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00065, + "step": 431, + "tokens/total": 13774912, + "tokens/train_per_sec_per_gpu": 25.06, + "tokens/trainable": 196974 + }, + { + "epoch": 1.6875, + "grad_norm": 0.0517231747508049, + "learning_rate": 1.600468845019576e-05, + "loss": 0.00047026947140693665, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00047, + "step": 432, + "tokens/total": 13806800, + "tokens/train_per_sec_per_gpu": 29.38, + "tokens/trainable": 197454 + }, + { + "epoch": 1.69140625, + "grad_norm": 0.0004246715398039669, + "learning_rate": 1.5860625757072092e-05, + "loss": 7.613366506120656e-06, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00001, + "step": 433, + "tokens/total": 13838704, + "tokens/train_per_sec_per_gpu": 26.36, + "tokens/trainable": 197902 + }, + { + "epoch": 1.6953125, + "grad_norm": 0.0011862553656101227, + "learning_rate": 1.571819181307116e-05, + "loss": 1.5171106497291476e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00002, + "step": 434, + "tokens/total": 13870640, + "tokens/train_per_sec_per_gpu": 29.63, + "tokens/trainable": 198374 + }, + { + "epoch": 1.69921875, + "grad_norm": 0.0019507558317855, + "learning_rate": 1.557739254545075e-05, + "loss": 1.7753134670783766e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00002, + "step": 435, + "tokens/total": 13902672, + "tokens/train_per_sec_per_gpu": 31.27, + "tokens/trainable": 198846 + }, + { + "epoch": 1.703125, + "grad_norm": 0.014270462095737457, + "learning_rate": 1.543823381344311e-05, + "loss": 0.0001223236322402954, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00012, + "step": 436, + "tokens/total": 13935008, + "tokens/train_per_sec_per_gpu": 31.41, + "tokens/trainable": 199320 + }, + { + "epoch": 1.70703125, + "grad_norm": 0.020011236891150475, + "learning_rate": 1.5300721408011114e-05, + "loss": 0.00015811943740118295, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00016, + "step": 437, + "tokens/total": 13967024, + "tokens/train_per_sec_per_gpu": 31.19, + "tokens/trainable": 199777 + }, + { + "epoch": 1.7109375, + "grad_norm": 0.0006321229157038033, + "learning_rate": 1.5164861051607254e-05, + "loss": 9.932198736350983e-06, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00001, + "step": 438, + "tokens/total": 13998688, + "tokens/train_per_sec_per_gpu": 25.77, + "tokens/trainable": 200194 + }, + { + "epoch": 1.71484375, + "grad_norm": 0.002675980795174837, + "learning_rate": 1.5030658397935521e-05, + "loss": 2.0016837879666127e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00002, + "step": 439, + "tokens/total": 14030992, + "tokens/train_per_sec_per_gpu": 28.45, + "tokens/trainable": 200663 + }, + { + "epoch": 1.71875, + "grad_norm": 0.005228503607213497, + "learning_rate": 1.4898119031716104e-05, + "loss": 4.561378591461107e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.00005, + "step": 440, + "tokens/total": 14063440, + "tokens/train_per_sec_per_gpu": 28.52, + "tokens/trainable": 201121 + }, + { + "epoch": 1.72265625, + "grad_norm": 0.015322903171181679, + "learning_rate": 1.476724846845306e-05, + "loss": 0.0001199191392515786, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00012, + "step": 441, + "tokens/total": 14095536, + "tokens/train_per_sec_per_gpu": 33.0, + "tokens/trainable": 201598 + }, + { + "epoch": 1.7265625, + "grad_norm": 0.05436247959733009, + "learning_rate": 1.463805215420471e-05, + "loss": 0.0005605737096630037, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00056, + "step": 442, + "tokens/total": 14127488, + "tokens/train_per_sec_per_gpu": 29.37, + "tokens/trainable": 202068 + }, + { + "epoch": 1.73046875, + "grad_norm": 0.005114950239658356, + "learning_rate": 1.451053546535705e-05, + "loss": 2.6398556656204164e-05, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.00003, + "step": 443, + "tokens/total": 14159472, + "tokens/train_per_sec_per_gpu": 29.43, + "tokens/trainable": 202524 + }, + { + "epoch": 1.734375, + "grad_norm": 0.000681015953887254, + "learning_rate": 1.438470370840001e-05, + "loss": 6.228779966477305e-06, + "memory/device_reserved (GiB)": 39.15, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00001, + "step": 444, + "tokens/total": 14191520, + "tokens/train_per_sec_per_gpu": 30.27, + "tokens/trainable": 202996 + }, + { + "epoch": 1.73828125, + "grad_norm": 0.03594636172056198, + "learning_rate": 1.4260562119706606e-05, + "loss": 0.0003221426741220057, + "memory/device_reserved (GiB)": 39.16, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00032, + "step": 445, + "tokens/total": 14221680, + "tokens/train_per_sec_per_gpu": 32.89, + "tokens/trainable": 203470 + }, + { + "epoch": 1.7421875, + "grad_norm": 0.16816683113574982, + "learning_rate": 1.413811586531508e-05, + "loss": 0.0016490641282871366, + "memory/device_reserved (GiB)": 39.16, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00165, + "step": 446, + "tokens/total": 14253680, + "tokens/train_per_sec_per_gpu": 29.37, + "tokens/trainable": 203919 + }, + { + "epoch": 1.74609375, + "grad_norm": 0.0071723381988704205, + "learning_rate": 1.4017370040713884e-05, + "loss": 6.93315378157422e-05, + "memory/device_reserved (GiB)": 39.16, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00007, + "step": 447, + "tokens/total": 14285824, + "tokens/train_per_sec_per_gpu": 27.58, + "tokens/trainable": 204376 + }, + { + "epoch": 1.75, + "grad_norm": 0.02812999114394188, + "learning_rate": 1.3898329670629645e-05, + "loss": 0.0001953901955857873, + "memory/device_reserved (GiB)": 39.16, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.0002, + "step": 448, + "tokens/total": 14317968, + "tokens/train_per_sec_per_gpu": 26.64, + "tokens/trainable": 204821 + }, + { + "epoch": 1.75390625, + "grad_norm": 0.003404787741601467, + "learning_rate": 1.3780999708818058e-05, + "loss": 7.17312059350661e-06, + "memory/device_reserved (GiB)": 39.16, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00001, + "step": 449, + "tokens/total": 14350080, + "tokens/train_per_sec_per_gpu": 28.3, + "tokens/trainable": 205271 + }, + { + "epoch": 1.7578125, + "grad_norm": 0.0005480491672642529, + "learning_rate": 1.3665385037857758e-05, + "loss": 4.81248343930929e-06, + "memory/device_reserved (GiB)": 37.41, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.0, + "step": 450, + "tokens/total": 14382240, + "tokens/train_per_sec_per_gpu": 25.5, + "tokens/trainable": 205723 + }, + { + "epoch": 1.76171875, + "grad_norm": 0.014193670824170113, + "learning_rate": 1.3551490468947126e-05, + "loss": 5.97102043684572e-05, + "memory/device_reserved (GiB)": 37.63, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00006, + "step": 451, + "tokens/total": 14414320, + "tokens/train_per_sec_per_gpu": 27.12, + "tokens/trainable": 206169 + }, + { + "epoch": 1.765625, + "grad_norm": 0.02763630636036396, + "learning_rate": 1.3439320741704075e-05, + "loss": 0.00014929058670531958, + "memory/device_reserved (GiB)": 37.76, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00015, + "step": 452, + "tokens/total": 14446464, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 206592 + }, + { + "epoch": 1.76953125, + "grad_norm": 0.010347527451813221, + "learning_rate": 1.3328880523968808e-05, + "loss": 5.998069536872208e-05, + "memory/device_reserved (GiB)": 38.45, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00006, + "step": 453, + "tokens/total": 14478624, + "tokens/train_per_sec_per_gpu": 27.59, + "tokens/trainable": 207055 + }, + { + "epoch": 1.7734375, + "grad_norm": 0.4689827561378479, + "learning_rate": 1.3220174411609587e-05, + "loss": 0.0033441532868891954, + "memory/device_reserved (GiB)": 38.45, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.00335, + "step": 454, + "tokens/total": 14510192, + "tokens/train_per_sec_per_gpu": 28.86, + "tokens/trainable": 207494 + }, + { + "epoch": 1.77734375, + "grad_norm": 0.04199036583304405, + "learning_rate": 1.3113206928331471e-05, + "loss": 0.0002628727233968675, + "memory/device_reserved (GiB)": 38.45, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00026, + "step": 455, + "tokens/total": 14541920, + "tokens/train_per_sec_per_gpu": 27.5, + "tokens/trainable": 207956 + }, + { + "epoch": 1.78125, + "grad_norm": 0.0015001439023762941, + "learning_rate": 1.300798252548806e-05, + "loss": 1.346761200693436e-05, + "memory/device_reserved (GiB)": 38.45, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00001, + "step": 456, + "tokens/total": 14573904, + "tokens/train_per_sec_per_gpu": 29.6, + "tokens/trainable": 208408 + }, + { + "epoch": 1.78515625, + "grad_norm": 0.013151494786143303, + "learning_rate": 1.2904505581896265e-05, + "loss": 3.6732359149027616e-05, + "memory/device_reserved (GiB)": 38.45, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00004, + "step": 457, + "tokens/total": 14606032, + "tokens/train_per_sec_per_gpu": 29.71, + "tokens/trainable": 208889 + }, + { + "epoch": 1.7890625, + "grad_norm": 0.0006910350639373064, + "learning_rate": 1.2802780403654082e-05, + "loss": 1.0416523764433805e-05, + "memory/device_reserved (GiB)": 38.45, + "memory/max_active (GiB)": 36.34, + "memory/max_allocated (GiB)": 36.34, + "ppl": 1.00001, + "step": 458, + "tokens/total": 14637728, + "tokens/train_per_sec_per_gpu": 28.36, + "tokens/trainable": 209336 + }, + { + "epoch": 1.79296875, + "grad_norm": 0.003835468553006649, + "learning_rate": 1.2702811223961408e-05, + "loss": 3.105392897850834e-05, + "memory/device_reserved (GiB)": 38.63, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00003, + "step": 459, + "tokens/total": 14669984, + "tokens/train_per_sec_per_gpu": 27.49, + "tokens/trainable": 209797 + }, + { + "epoch": 1.796875, + "grad_norm": 0.0003044170734938234, + "learning_rate": 1.2604602202943861e-05, + "loss": 6.697504431940615e-06, + "memory/device_reserved (GiB)": 38.67, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00001, + "step": 460, + "tokens/total": 14702064, + "tokens/train_per_sec_per_gpu": 30.05, + "tokens/trainable": 210240 + }, + { + "epoch": 1.80078125, + "grad_norm": 0.01640794239938259, + "learning_rate": 1.2508157427479686e-05, + "loss": 7.829760579625145e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00008, + "step": 461, + "tokens/total": 14734144, + "tokens/train_per_sec_per_gpu": 29.07, + "tokens/trainable": 210702 + }, + { + "epoch": 1.8046875, + "grad_norm": 0.04974092170596123, + "learning_rate": 1.2413480911029655e-05, + "loss": 0.00036814718623645604, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00037, + "step": 462, + "tokens/total": 14766416, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 211140 + }, + { + "epoch": 1.80859375, + "grad_norm": 0.0007990560261532664, + "learning_rate": 1.2320576593470082e-05, + "loss": 7.1813856266089715e-06, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.00001, + "step": 463, + "tokens/total": 14798224, + "tokens/train_per_sec_per_gpu": 27.29, + "tokens/trainable": 211575 + }, + { + "epoch": 1.8125, + "grad_norm": 0.08440390229225159, + "learning_rate": 1.2229448340928828e-05, + "loss": 0.00048587226774543524, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00049, + "step": 464, + "tokens/total": 14830336, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 212083 + }, + { + "epoch": 1.81640625, + "grad_norm": 0.0018794891657307744, + "learning_rate": 1.2140099945624458e-05, + "loss": 1.08363010440371e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00001, + "step": 465, + "tokens/total": 14862560, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 212558 + }, + { + "epoch": 1.8203125, + "grad_norm": 0.008357501588761806, + "learning_rate": 1.205253512570841e-05, + "loss": 6.44918909529224e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00006, + "step": 466, + "tokens/total": 14894432, + "tokens/train_per_sec_per_gpu": 27.29, + "tokens/trainable": 213014 + }, + { + "epoch": 1.82421875, + "grad_norm": 0.003019345458596945, + "learning_rate": 1.1966757525110255e-05, + "loss": 2.4649680199217983e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00002, + "step": 467, + "tokens/total": 14926624, + "tokens/train_per_sec_per_gpu": 28.39, + "tokens/trainable": 213465 + }, + { + "epoch": 1.828125, + "grad_norm": 0.0832735076546669, + "learning_rate": 1.1882770713386095e-05, + "loss": 0.0008976737735792994, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0009, + "step": 468, + "tokens/total": 14958640, + "tokens/train_per_sec_per_gpu": 30.58, + "tokens/trainable": 213940 + }, + { + "epoch": 1.83203125, + "grad_norm": 0.0038668843917548656, + "learning_rate": 1.180057818556998e-05, + "loss": 1.873084511316847e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.00002, + "step": 469, + "tokens/total": 14990864, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 214415 + }, + { + "epoch": 1.8359375, + "grad_norm": 0.0006158083560876548, + "learning_rate": 1.1720183362028494e-05, + "loss": 5.440972017822787e-06, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.00001, + "step": 470, + "tokens/total": 15022800, + "tokens/train_per_sec_per_gpu": 28.77, + "tokens/trainable": 214880 + }, + { + "epoch": 1.83984375, + "grad_norm": 0.0015613064169883728, + "learning_rate": 1.1641589588318387e-05, + "loss": 1.487641657149652e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00001, + "step": 471, + "tokens/total": 15054784, + "tokens/train_per_sec_per_gpu": 27.81, + "tokens/trainable": 215334 + }, + { + "epoch": 1.84375, + "grad_norm": 0.00033517941483296454, + "learning_rate": 1.1564800135047418e-05, + "loss": 4.4488651838037185e-06, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0, + "step": 472, + "tokens/total": 15086800, + "tokens/train_per_sec_per_gpu": 29.18, + "tokens/trainable": 215822 + }, + { + "epoch": 1.84765625, + "grad_norm": 0.014701228588819504, + "learning_rate": 1.148981819773816e-05, + "loss": 0.00012692881864495575, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00013, + "step": 473, + "tokens/total": 15118736, + "tokens/train_per_sec_per_gpu": 25.65, + "tokens/trainable": 216239 + }, + { + "epoch": 1.8515625, + "grad_norm": 0.03242962434887886, + "learning_rate": 1.1416646896695086e-05, + "loss": 0.00016460703045595437, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00016, + "step": 474, + "tokens/total": 15150896, + "tokens/train_per_sec_per_gpu": 27.53, + "tokens/trainable": 216691 + }, + { + "epoch": 1.85546875, + "grad_norm": 0.022134000435471535, + "learning_rate": 1.1345289276874717e-05, + "loss": 0.00015468102355953306, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.00015, + "step": 475, + "tokens/total": 15183120, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 217162 + }, + { + "epoch": 1.859375, + "grad_norm": 0.00101348920725286, + "learning_rate": 1.1275748307758873e-05, + "loss": 1.263963622477604e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00001, + "step": 476, + "tokens/total": 15215216, + "tokens/train_per_sec_per_gpu": 30.11, + "tokens/trainable": 217638 + }, + { + "epoch": 1.86328125, + "grad_norm": 0.09451806545257568, + "learning_rate": 1.1208026883231147e-05, + "loss": 0.0005900258547626436, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00059, + "step": 477, + "tokens/total": 15247328, + "tokens/train_per_sec_per_gpu": 27.45, + "tokens/trainable": 218053 + }, + { + "epoch": 1.8671875, + "grad_norm": 0.0023683386389166117, + "learning_rate": 1.1142127821456433e-05, + "loss": 1.8595857909531333e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00002, + "step": 478, + "tokens/total": 15279472, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 218503 + }, + { + "epoch": 1.87109375, + "grad_norm": 0.37432360649108887, + "learning_rate": 1.1078053864763674e-05, + "loss": 0.0017199859721586108, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00172, + "step": 479, + "tokens/total": 15311536, + "tokens/train_per_sec_per_gpu": 30.18, + "tokens/trainable": 218964 + }, + { + "epoch": 1.875, + "grad_norm": 0.006793392356485128, + "learning_rate": 1.1015807679531756e-05, + "loss": 2.6950599931296892e-05, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.00003, + "step": 480, + "tokens/total": 15343360, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 219412 + }, + { + "epoch": 1.87890625, + "grad_norm": 0.07207302004098892, + "learning_rate": 1.0955391856078528e-05, + "loss": 0.0004579552623908967, + "memory/device_reserved (GiB)": 38.74, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.00046, + "step": 481, + "tokens/total": 15375008, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 219886 + }, + { + "epoch": 1.8828125, + "grad_norm": 0.0035383424255996943, + "learning_rate": 1.0896808908553007e-05, + "loss": 2.970332934637554e-05, + "memory/device_reserved (GiB)": 37.51, + "memory/max_active (GiB)": 36.33, + "memory/max_allocated (GiB)": 36.33, + "ppl": 1.00003, + "step": 482, + "tokens/total": 15406816, + "tokens/train_per_sec_per_gpu": 28.98, + "tokens/trainable": 220343 + }, + { + "epoch": 1.88671875, + "grad_norm": 0.005184710957109928, + "learning_rate": 1.0840061274830763e-05, + "loss": 3.948260928154923e-05, + "memory/device_reserved (GiB)": 38.63, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00004, + "step": 483, + "tokens/total": 15438816, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 220825 + }, + { + "epoch": 1.890625, + "grad_norm": 0.0017558662220835686, + "learning_rate": 1.0785151316412473e-05, + "loss": 1.1586276741581969e-05, + "memory/device_reserved (GiB)": 38.63, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00001, + "step": 484, + "tokens/total": 15471008, + "tokens/train_per_sec_per_gpu": 27.13, + "tokens/trainable": 221282 + }, + { + "epoch": 1.89453125, + "grad_norm": 0.0002366721018915996, + "learning_rate": 1.0732081318325639e-05, + "loss": 3.967344127886463e-06, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.0, + "step": 485, + "tokens/total": 15503248, + "tokens/train_per_sec_per_gpu": 31.3, + "tokens/trainable": 221780 + }, + { + "epoch": 1.8984375, + "grad_norm": 0.18045902252197266, + "learning_rate": 1.0680853489029501e-05, + "loss": 0.000754694570787251, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.00075, + "step": 486, + "tokens/total": 15535504, + "tokens/train_per_sec_per_gpu": 26.37, + "tokens/trainable": 222210 + }, + { + "epoch": 1.90234375, + "grad_norm": 0.0010426960652694106, + "learning_rate": 1.0631469960323152e-05, + "loss": 1.0654634934326168e-05, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00001, + "step": 487, + "tokens/total": 15567568, + "tokens/train_per_sec_per_gpu": 24.42, + "tokens/trainable": 222646 + }, + { + "epoch": 1.90625, + "grad_norm": 0.018443502485752106, + "learning_rate": 1.0583932787256783e-05, + "loss": 0.0001445821108063683, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.00014, + "step": 488, + "tokens/total": 15599456, + "tokens/train_per_sec_per_gpu": 27.75, + "tokens/trainable": 223066 + }, + { + "epoch": 1.91015625, + "grad_norm": 0.08126334846019745, + "learning_rate": 1.0538243948046206e-05, + "loss": 0.00040054236887954175, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.0004, + "step": 489, + "tokens/total": 15631168, + "tokens/train_per_sec_per_gpu": 28.8, + "tokens/trainable": 223512 + }, + { + "epoch": 1.9140625, + "grad_norm": 0.01388081070035696, + "learning_rate": 1.0494405343990523e-05, + "loss": 5.4693471611244604e-05, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00005, + "step": 490, + "tokens/total": 15661104, + "tokens/train_per_sec_per_gpu": 32.49, + "tokens/trainable": 223956 + }, + { + "epoch": 1.91796875, + "grad_norm": 0.014524613507091999, + "learning_rate": 1.0452418799392985e-05, + "loss": 9.905001206789166e-05, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.09, + "memory/max_allocated (GiB)": 36.09, + "ppl": 1.0001, + "step": 491, + "tokens/total": 15691120, + "tokens/train_per_sec_per_gpu": 30.04, + "tokens/trainable": 224416 + }, + { + "epoch": 1.921875, + "grad_norm": 0.016004936769604683, + "learning_rate": 1.0412286061485102e-05, + "loss": 7.77841269155033e-05, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.00008, + "step": 492, + "tokens/total": 15723104, + "tokens/train_per_sec_per_gpu": 28.42, + "tokens/trainable": 224883 + }, + { + "epoch": 1.92578125, + "grad_norm": 0.00028383161406964064, + "learning_rate": 1.03740088003539e-05, + "loss": 4.6946679503889754e-06, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0, + "step": 493, + "tokens/total": 15754960, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 225324 + }, + { + "epoch": 1.9296875, + "grad_norm": 0.003787089604884386, + "learning_rate": 1.0337588608872463e-05, + "loss": 2.9438704586937092e-05, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.00003, + "step": 494, + "tokens/total": 15787008, + "tokens/train_per_sec_per_gpu": 29.03, + "tokens/trainable": 225783 + }, + { + "epoch": 1.93359375, + "grad_norm": 0.0024508354254066944, + "learning_rate": 1.0303027002633622e-05, + "loss": 1.2276758752705064e-05, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00001, + "step": 495, + "tokens/total": 15818896, + "tokens/train_per_sec_per_gpu": 30.48, + "tokens/trainable": 226281 + }, + { + "epoch": 1.9375, + "grad_norm": 0.0002896689693443477, + "learning_rate": 1.0270325419886884e-05, + "loss": 4.6574341467930935e-06, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.0, + "step": 496, + "tokens/total": 15850624, + "tokens/train_per_sec_per_gpu": 30.32, + "tokens/trainable": 226735 + }, + { + "epoch": 1.94140625, + "grad_norm": 0.002518043853342533, + "learning_rate": 1.0239485221478599e-05, + "loss": 2.2658114176010713e-05, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00002, + "step": 497, + "tokens/total": 15882752, + "tokens/train_per_sec_per_gpu": 29.98, + "tokens/trainable": 227198 + }, + { + "epoch": 1.9453125, + "grad_norm": 0.00016003627388272434, + "learning_rate": 1.0210507690795292e-05, + "loss": 3.791648168771644e-06, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.27, + "memory/max_allocated (GiB)": 36.27, + "ppl": 1.0, + "step": 498, + "tokens/total": 15914416, + "tokens/train_per_sec_per_gpu": 25.94, + "tokens/trainable": 227623 + }, + { + "epoch": 1.94921875, + "grad_norm": 0.00043029585503973067, + "learning_rate": 1.0183394033710305e-05, + "loss": 6.0818256315542385e-06, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00001, + "step": 499, + "tokens/total": 15946320, + "tokens/train_per_sec_per_gpu": 26.99, + "tokens/trainable": 228074 + }, + { + "epoch": 1.953125, + "grad_norm": 0.03700384125113487, + "learning_rate": 1.0158145378533583e-05, + "loss": 0.00016803065955173224, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00017, + "step": 500, + "tokens/total": 15978384, + "tokens/train_per_sec_per_gpu": 29.58, + "tokens/trainable": 228541 + }, + { + "epoch": 1.95703125, + "grad_norm": 0.0032133073545992374, + "learning_rate": 1.0134762775964726e-05, + "loss": 7.3915689426939934e-06, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00001, + "step": 501, + "tokens/total": 16010672, + "tokens/train_per_sec_per_gpu": 31.11, + "tokens/trainable": 229031 + }, + { + "epoch": 1.9609375, + "grad_norm": 0.00048456323565915227, + "learning_rate": 1.0113247199049278e-05, + "loss": 7.050684871501289e-06, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00001, + "step": 502, + "tokens/total": 16042672, + "tokens/train_per_sec_per_gpu": 32.29, + "tokens/trainable": 229512 + }, + { + "epoch": 1.96484375, + "grad_norm": 0.0932794138789177, + "learning_rate": 1.0093599543138205e-05, + "loss": 0.0006240660441108048, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00062, + "step": 503, + "tokens/total": 16074592, + "tokens/train_per_sec_per_gpu": 28.41, + "tokens/trainable": 229989 + }, + { + "epoch": 1.96875, + "grad_norm": 0.012090178206562996, + "learning_rate": 1.0075820625850675e-05, + "loss": 8.021057874429971e-05, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00008, + "step": 504, + "tokens/total": 16106944, + "tokens/train_per_sec_per_gpu": 29.66, + "tokens/trainable": 230447 + }, + { + "epoch": 1.97265625, + "grad_norm": 0.010548670776188374, + "learning_rate": 1.0059911187040013e-05, + "loss": 1.3973164641356561e-05, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00001, + "step": 505, + "tokens/total": 16138800, + "tokens/train_per_sec_per_gpu": 28.63, + "tokens/trainable": 230879 + }, + { + "epoch": 1.9765625, + "grad_norm": 0.004747380968183279, + "learning_rate": 1.0045871888762893e-05, + "loss": 4.118381184525788e-05, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.00004, + "step": 506, + "tokens/total": 16170976, + "tokens/train_per_sec_per_gpu": 28.35, + "tokens/trainable": 231348 + }, + { + "epoch": 1.98046875, + "grad_norm": 0.0008630049414932728, + "learning_rate": 1.003370331525184e-05, + "loss": 7.490401458198903e-06, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.00001, + "step": 507, + "tokens/total": 16202976, + "tokens/train_per_sec_per_gpu": 29.33, + "tokens/trainable": 231819 + }, + { + "epoch": 1.984375, + "grad_norm": 0.0002915443910751492, + "learning_rate": 1.002340597289085e-05, + "loss": 5.796592631668318e-06, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.00001, + "step": 508, + "tokens/total": 16234816, + "tokens/train_per_sec_per_gpu": 28.79, + "tokens/trainable": 232269 + }, + { + "epoch": 1.98828125, + "grad_norm": 0.00034642108948901296, + "learning_rate": 1.0014980290194387e-05, + "loss": 3.7650847843906377e-06, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.0, + "step": 509, + "tokens/total": 16266960, + "tokens/train_per_sec_per_gpu": 28.22, + "tokens/trainable": 232716 + }, + { + "epoch": 1.9921875, + "grad_norm": 0.17613324522972107, + "learning_rate": 1.0008426617789489e-05, + "loss": 0.0010949716670438647, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.0011, + "step": 510, + "tokens/total": 16298912, + "tokens/train_per_sec_per_gpu": 30.16, + "tokens/trainable": 233178 + }, + { + "epoch": 1.99609375, + "grad_norm": 0.00045643217163160443, + "learning_rate": 1.0003745228401215e-05, + "loss": 5.581615369010251e-06, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00001, + "step": 511, + "tokens/total": 16330944, + "tokens/train_per_sec_per_gpu": 29.97, + "tokens/trainable": 233647 + }, + { + "epoch": 2.0, + "grad_norm": 0.05987092852592468, + "learning_rate": 1.0000936316841296e-05, + "loss": 0.00014668369840364903, + "memory/device_reserved (GiB)": 39.0, + "memory/max_active (GiB)": 36.26, + "memory/max_allocated (GiB)": 36.26, + "ppl": 1.00015, + "step": 512, + "tokens/total": 16362544, + "tokens/train_per_sec_per_gpu": 24.67, + "tokens/trainable": 234060 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": true + }, + "attributes": {} + } + }, + "total_flos": 1.1106214797983247e+18, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/training_args.bin b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..dec74f9f304b44cbe47132d86c4e6285e659f40b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-512/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634 +size 8273 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/README.md b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ba677470ac9c29f9770a4aacb65c644ebd671a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/deconf_wave_deconf_control/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/deconf_wave_deconf_control/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/adapter_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..21fcb5b00be777c20a2d2dcceffc539dc51679b4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/deconf_wave_deconf_control/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "gate_proj", + "v_proj", + "k_proj", + "up_proj", + "q_proj", + "o_proj", + "down_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/adapter_model.safetensors b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..583be8bd32bce7eaf50a382d175453ed55ed6e61 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:91215613ea1ede42eb43ed93ac9b2b911d7ec04725d5993513c054d39e33a9f0 +size 547777976 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/chat_template.jinja b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/optimizer.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..e9fc678da92e84f2b09692e418fb1ebdeef77a01 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:48458ccef9531f08594780e501750c3bd9fa19b613c3f4925edc61129c8605fb +size 1048106435 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/rng_state.pth b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..6e0215c1ab78f483fc47d8db5fc26d40c3188c73 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8fceeb869af3ebdb70932f5d864119bc217c0be1c007e6c15d68521eb402e36d +size 14645 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/scheduler.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..34b7fc400f1006f136b1c89d6c58cf3d17830d72 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:15b31b2361cee4c0a1206aa5d9efeb71d8dc96ceaff5b2fe054baf0716df3503 +size 1465 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/tokenizer.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/tokenizer_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/tokens_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..53895333ee9e37b1dedc4c8820fedd9dd0e129fc --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/tokens_state.json @@ -0,0 +1 @@ +{"total": 2041120, "trainable": 29283} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/trainer_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..7529a0f5a2a0dad15976f1867d67586af35a4319 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/trainer_state.json @@ -0,0 +1,930 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.25, + "eval_steps": 500, + "global_step": 64, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.2140090465545654, + "learning_rate": 0.0, + "loss": 0.16441112756729126, + "memory/device_reserved (GiB)": 38.08, + "memory/max_active (GiB)": 35.64, + "memory/max_allocated (GiB)": 35.64, + "ppl": 1.1787, + "step": 1, + "tokens/total": 32160, + "tokens/train_per_sec_per_gpu": 21.28, + "tokens/trainable": 470 + }, + { + "epoch": 0.0078125, + "grad_norm": 2.92620849609375, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15774038434028625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.17086, + "step": 2, + "tokens/total": 64192, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 922 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.135961890220642, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1664382815361023, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.18109, + "step": 3, + "tokens/total": 96224, + "tokens/train_per_sec_per_gpu": 28.54, + "tokens/trainable": 1396 + }, + { + "epoch": 0.015625, + "grad_norm": 0.880176305770874, + "learning_rate": 1.2e-05, + "loss": 0.15334013104438782, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.16572, + "step": 4, + "tokens/total": 128336, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 1850 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.1433289051055908, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.16352537274360657, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.17766, + "step": 5, + "tokens/total": 160784, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 2302 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9728212952613831, + "learning_rate": 2e-05, + "loss": 0.15144473314285278, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.16351, + "step": 6, + "tokens/total": 192624, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 2742 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.7796363830566406, + "learning_rate": 2.4e-05, + "loss": 0.13553467392921448, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.14515, + "step": 7, + "tokens/total": 224608, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 3208 + }, + { + "epoch": 0.03125, + "grad_norm": 1.2444697618484497, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.1077304258942604, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.11375, + "step": 8, + "tokens/total": 256592, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 3655 + }, + { + "epoch": 0.03515625, + "grad_norm": 2.7339375019073486, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.08772765100002289, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.09169, + "step": 9, + "tokens/total": 286400, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 4091 + }, + { + "epoch": 0.0390625, + "grad_norm": 4.109770774841309, + "learning_rate": 3.6e-05, + "loss": 0.130940243601799, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.1399, + "step": 10, + "tokens/total": 318352, + "tokens/train_per_sec_per_gpu": 29.13, + "tokens/trainable": 4553 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.3768608570098877, + "learning_rate": 4e-05, + "loss": 0.10059958696365356, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.10583, + "step": 11, + "tokens/total": 350336, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 4992 + }, + { + "epoch": 0.046875, + "grad_norm": 3.281402826309204, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.12574875354766846, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.134, + "step": 12, + "tokens/total": 382448, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 5494 + }, + { + "epoch": 0.05078125, + "grad_norm": 1.7738980054855347, + "learning_rate": 4.8e-05, + "loss": 0.033296722918748856, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03386, + "step": 13, + "tokens/total": 414352, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 5933 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.64290189743042, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06366822123527527, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.06574, + "step": 14, + "tokens/total": 446400, + "tokens/train_per_sec_per_gpu": 24.8, + "tokens/trainable": 6369 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.3691928386688232, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.08168518543243408, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.08511, + "step": 15, + "tokens/total": 478688, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 6820 + }, + { + "epoch": 0.0625, + "grad_norm": 2.4356393814086914, + "learning_rate": 6e-05, + "loss": 0.09031196683645248, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.09452, + "step": 16, + "tokens/total": 510512, + "tokens/train_per_sec_per_gpu": 25.7, + "tokens/trainable": 7258 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.55682373046875, + "learning_rate": 6.400000000000001e-05, + "loss": 0.07496573030948639, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.07785, + "step": 17, + "tokens/total": 542736, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 7710 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.1054646968841553, + "learning_rate": 6.800000000000001e-05, + "loss": 0.05218805745244026, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.05357, + "step": 18, + "tokens/total": 574832, + "tokens/train_per_sec_per_gpu": 25.78, + "tokens/trainable": 8174 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.7429099082946777, + "learning_rate": 7.2e-05, + "loss": 0.03790827840566635, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.03864, + "step": 19, + "tokens/total": 606976, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 8617 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9508209228515625, + "learning_rate": 7.6e-05, + "loss": 0.04993613809347153, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0512, + "step": 20, + "tokens/total": 639200, + "tokens/train_per_sec_per_gpu": 24.91, + "tokens/trainable": 9037 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.0297844409942627, + "learning_rate": 8e-05, + "loss": 0.017908263951539993, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.01807, + "step": 21, + "tokens/total": 671072, + "tokens/train_per_sec_per_gpu": 30.42, + "tokens/trainable": 9503 + }, + { + "epoch": 0.0859375, + "grad_norm": 0.9279879331588745, + "learning_rate": 8.4e-05, + "loss": 0.013478193432092667, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01357, + "step": 22, + "tokens/total": 703136, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 9972 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.526488184928894, + "learning_rate": 8.800000000000001e-05, + "loss": 0.017310388386249542, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01746, + "step": 23, + "tokens/total": 735120, + "tokens/train_per_sec_per_gpu": 31.45, + "tokens/trainable": 10447 + }, + { + "epoch": 0.09375, + "grad_norm": 3.4049417972564697, + "learning_rate": 9.200000000000001e-05, + "loss": 0.0353429913520813, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.03597, + "step": 24, + "tokens/total": 767040, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 10899 + }, + { + "epoch": 0.09765625, + "grad_norm": 7.511656761169434, + "learning_rate": 9.6e-05, + "loss": 0.1335875540971756, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.14292, + "step": 25, + "tokens/total": 798976, + "tokens/train_per_sec_per_gpu": 27.72, + "tokens/trainable": 11360 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.0069234371185303, + "learning_rate": 0.0001, + "loss": 0.04382415860891342, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0448, + "step": 26, + "tokens/total": 830880, + "tokens/train_per_sec_per_gpu": 24.61, + "tokens/trainable": 11775 + }, + { + "epoch": 0.10546875, + "grad_norm": 7.807938575744629, + "learning_rate": 9.99990636831587e-05, + "loss": 0.01210833340883255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.01218, + "step": 27, + "tokens/total": 860960, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 12242 + }, + { + "epoch": 0.109375, + "grad_norm": 2.23823881149292, + "learning_rate": 9.999625477159879e-05, + "loss": 0.0272970050573349, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02767, + "step": 28, + "tokens/total": 892816, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 12696 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.315584421157837, + "learning_rate": 9.999157338221051e-05, + "loss": 0.025516577064990997, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.02584, + "step": 29, + "tokens/total": 924736, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 13153 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.8413651585578918, + "learning_rate": 9.998501970980562e-05, + "loss": 0.007838520221412182, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00787, + "step": 30, + "tokens/total": 954592, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 13624 + }, + { + "epoch": 0.12109375, + "grad_norm": 1.6797157526016235, + "learning_rate": 9.997659402710915e-05, + "loss": 0.03002442792057991, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.03048, + "step": 31, + "tokens/total": 986592, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 14064 + }, + { + "epoch": 0.125, + "grad_norm": 1.3320434093475342, + "learning_rate": 9.996629668474818e-05, + "loss": 0.03772728145122528, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.03845, + "step": 32, + "tokens/total": 1018720, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 14565 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.6335167288780212, + "learning_rate": 9.995412811123711e-05, + "loss": 0.013970119878649712, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01407, + "step": 33, + "tokens/total": 1050624, + "tokens/train_per_sec_per_gpu": 28.18, + "tokens/trainable": 15005 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.5626921653747559, + "learning_rate": 9.994008881295999e-05, + "loss": 0.016654539853334427, + "memory/device_reserved (GiB)": 37.74, + "memory/max_active (GiB)": 36.22, + "memory/max_allocated (GiB)": 36.22, + "ppl": 1.01679, + "step": 34, + "tokens/total": 1082176, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 15459 + }, + { + "epoch": 0.13671875, + "grad_norm": 0.603219211101532, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021408583968877792, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02164, + "step": 35, + "tokens/total": 1113968, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 15960 + }, + { + "epoch": 0.140625, + "grad_norm": 0.9779674410820007, + "learning_rate": 9.99064004568618e-05, + "loss": 0.03277068957686424, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03331, + "step": 36, + "tokens/total": 1145952, + "tokens/train_per_sec_per_gpu": 28.68, + "tokens/trainable": 16428 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.8855255842208862, + "learning_rate": 9.988675280095074e-05, + "loss": 0.017048098146915436, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01719, + "step": 37, + "tokens/total": 1177872, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 16884 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.9266122579574585, + "learning_rate": 9.986523722403528e-05, + "loss": 0.045863546431064606, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.04693, + "step": 38, + "tokens/total": 1209776, + "tokens/train_per_sec_per_gpu": 26.91, + "tokens/trainable": 17341 + }, + { + "epoch": 0.15234375, + "grad_norm": 0.8280895948410034, + "learning_rate": 9.984185462146642e-05, + "loss": 0.033055759966373444, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03361, + "step": 39, + "tokens/total": 1241888, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 17786 + }, + { + "epoch": 0.15625, + "grad_norm": 5.029730319976807, + "learning_rate": 9.98166059662897e-05, + "loss": 0.025589998811483383, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02592, + "step": 40, + "tokens/total": 1274032, + "tokens/train_per_sec_per_gpu": 29.22, + "tokens/trainable": 18262 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.5688032507896423, + "learning_rate": 9.978949230920472e-05, + "loss": 0.016095872968435287, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01623, + "step": 41, + "tokens/total": 1306000, + "tokens/train_per_sec_per_gpu": 27.49, + "tokens/trainable": 18716 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8217663168907166, + "learning_rate": 9.976051477852141e-05, + "loss": 0.026798982173204422, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02716, + "step": 42, + "tokens/total": 1337936, + "tokens/train_per_sec_per_gpu": 27.51, + "tokens/trainable": 19157 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.7309132218360901, + "learning_rate": 9.972967458011312e-05, + "loss": 0.033920761197805405, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0345, + "step": 43, + "tokens/total": 1369904, + "tokens/train_per_sec_per_gpu": 30.66, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.8134022355079651, + "learning_rate": 9.96969729973664e-05, + "loss": 0.024005085229873657, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0243, + "step": 44, + "tokens/total": 1401984, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 20119 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.35084816813468933, + "learning_rate": 9.966241139112754e-05, + "loss": 0.013319254852831364, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01341, + "step": 45, + "tokens/total": 1434032, + "tokens/train_per_sec_per_gpu": 27.45, + "tokens/trainable": 20560 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.8701573014259338, + "learning_rate": 9.96259911996461e-05, + "loss": 0.01356641948223114, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01366, + "step": 46, + "tokens/total": 1465872, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 20992 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.4406442642211914, + "learning_rate": 9.958771393851491e-05, + "loss": 0.04396980255842209, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 35.86, + "memory/max_allocated (GiB)": 35.86, + "ppl": 1.04495, + "step": 47, + "tokens/total": 1495440, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 21441 + }, + { + "epoch": 0.1875, + "grad_norm": 0.4966669976711273, + "learning_rate": 9.954758120060702e-05, + "loss": 0.011179964058101177, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01124, + "step": 48, + "tokens/total": 1527776, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 21901 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.41124436259269714, + "learning_rate": 9.950559465600948e-05, + "loss": 0.0063599650748074055, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00638, + "step": 49, + "tokens/total": 1559760, + "tokens/train_per_sec_per_gpu": 27.73, + "tokens/trainable": 22341 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.829739511013031, + "learning_rate": 9.946175605195379e-05, + "loss": 0.015785304829478264, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01591, + "step": 50, + "tokens/total": 1592000, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 22833 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5181736946105957, + "learning_rate": 9.941606721274322e-05, + "loss": 0.008393109776079655, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00843, + "step": 51, + "tokens/total": 1623920, + "tokens/train_per_sec_per_gpu": 25.61, + "tokens/trainable": 23277 + }, + { + "epoch": 0.203125, + "grad_norm": 1.5778990983963013, + "learning_rate": 9.936853003967685e-05, + "loss": 0.024984844028949738, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0253, + "step": 52, + "tokens/total": 1656032, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 23759 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.4938714504241943, + "learning_rate": 9.93191465109705e-05, + "loss": 0.038523320108652115, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03927, + "step": 53, + "tokens/total": 1688368, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 24193 + }, + { + "epoch": 0.2109375, + "grad_norm": 2.1427106857299805, + "learning_rate": 9.926791868167438e-05, + "loss": 0.06302924454212189, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.06506, + "step": 54, + "tokens/total": 1720560, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 24619 + }, + { + "epoch": 0.21484375, + "grad_norm": 1.8365188837051392, + "learning_rate": 9.921484868358753e-05, + "loss": 0.052160464227199554, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.05354, + "step": 55, + "tokens/total": 1752464, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 25075 + }, + { + "epoch": 0.21875, + "grad_norm": 0.7779749631881714, + "learning_rate": 9.915993872516924e-05, + "loss": 0.021328264847397804, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.02156, + "step": 56, + "tokens/total": 1784336, + "tokens/train_per_sec_per_gpu": 26.42, + "tokens/trainable": 25519 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.4189571142196655, + "learning_rate": 9.9103191091447e-05, + "loss": 0.014279939234256744, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01438, + "step": 57, + "tokens/total": 1816304, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 25966 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.3834710717201233, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01883944310247898, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01902, + "step": 58, + "tokens/total": 1848304, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 26423 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.7693278789520264, + "learning_rate": 9.898419232046825e-05, + "loss": 0.03533369302749634, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03597, + "step": 59, + "tokens/total": 1880640, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 26934 + }, + { + "epoch": 0.234375, + "grad_norm": 0.4606446325778961, + "learning_rate": 9.892194613523633e-05, + "loss": 0.021482212468981743, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02171, + "step": 60, + "tokens/total": 1912592, + "tokens/train_per_sec_per_gpu": 27.64, + "tokens/trainable": 27393 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.46030017733573914, + "learning_rate": 9.885787217854357e-05, + "loss": 0.017044665291905403, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01719, + "step": 61, + "tokens/total": 1944544, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 27852 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.7918264269828796, + "learning_rate": 9.879197311676887e-05, + "loss": 0.016954131424427032, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0171, + "step": 62, + "tokens/total": 1976688, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 28292 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.4787926971912384, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01525965891778469, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 63, + "tokens/total": 2008928, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 28798 + }, + { + "epoch": 0.25, + "grad_norm": 0.9033918976783752, + "learning_rate": 9.865471072312528e-05, + "loss": 0.02902211993932724, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02945, + "step": 64, + "tokens/total": 2041120, + "tokens/train_per_sec_per_gpu": 29.83, + "tokens/trainable": 29283 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.3854274218275328e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/training_args.bin b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..dec74f9f304b44cbe47132d86c4e6285e659f40b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-64/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634 +size 8273 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/README.md b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/README.md new file mode 100644 index 0000000000000000000000000000000000000000..7ba677470ac9c29f9770a4aacb65c644ebd671a3 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/deconf_wave_deconf_control/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/deconf_wave_deconf_control/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/adapter_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..21fcb5b00be777c20a2d2dcceffc539dc51679b4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/deconf_wave_deconf_control/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "gate_proj", + "v_proj", + "k_proj", + "up_proj", + "q_proj", + "o_proj", + "down_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/adapter_model.safetensors b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..ec649cc0f91a442762d824189f63da4bad0eaf1e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:36ee949a99df8fadec22a9d7518ae0c690c37dfcec59bcfd8a19cd0110a88d0a +size 547777976 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/chat_template.jinja b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/optimizer.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..0cea17efd42a02341ac1ec4590a83469b78852e9 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:51ae84e88af6abcfa89a77c89ebde8c1edad063d6c5c612fe7bc547a7c3b7dc1 +size 1048106435 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/rng_state.pth b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..dc7d8a691391cf7203d8b55b2c7219b798471eaf --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b684b305d9e7b60f772479d8acbd0a25afb20f096dcba946a80a0a66dc2064de +size 14645 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/scheduler.pt b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..433f82b14836de77d366e2c961bc5bdacdc9eb63 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:786444fedf73fac372c74a0ffd25119b2bb7107a3f00f8bfd9a8a46a6f2f4ff0 +size 1465 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/tokenizer.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/tokenizer_config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/tokens_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..282b8575c617a84782a4e2e017fb876894083fe4 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/tokens_state.json @@ -0,0 +1 @@ +{"total": 3064320, "trainable": 43865} \ No newline at end of file diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/trainer_state.json b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..8630d53c9943c9e25d63fc248acdf663fe22c9ad --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/trainer_state.json @@ -0,0 +1,1378 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.375, + "eval_steps": 500, + "global_step": 96, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.2140090465545654, + "learning_rate": 0.0, + "loss": 0.16441112756729126, + "memory/device_reserved (GiB)": 38.08, + "memory/max_active (GiB)": 35.64, + "memory/max_allocated (GiB)": 35.64, + "ppl": 1.1787, + "step": 1, + "tokens/total": 32160, + "tokens/train_per_sec_per_gpu": 21.28, + "tokens/trainable": 470 + }, + { + "epoch": 0.0078125, + "grad_norm": 2.92620849609375, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15774038434028625, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.17086, + "step": 2, + "tokens/total": 64192, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 922 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.135961890220642, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1664382815361023, + "memory/device_reserved (GiB)": 38.78, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.18109, + "step": 3, + "tokens/total": 96224, + "tokens/train_per_sec_per_gpu": 28.54, + "tokens/trainable": 1396 + }, + { + "epoch": 0.015625, + "grad_norm": 0.880176305770874, + "learning_rate": 1.2e-05, + "loss": 0.15334013104438782, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.16572, + "step": 4, + "tokens/total": 128336, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 1850 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.1433289051055908, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.16352537274360657, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.64, + "memory/max_allocated (GiB)": 36.64, + "ppl": 1.17766, + "step": 5, + "tokens/total": 160784, + "tokens/train_per_sec_per_gpu": 29.45, + "tokens/trainable": 2302 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9728212952613831, + "learning_rate": 2e-05, + "loss": 0.15144473314285278, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.16351, + "step": 6, + "tokens/total": 192624, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 2742 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.7796363830566406, + "learning_rate": 2.4e-05, + "loss": 0.13553467392921448, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.14515, + "step": 7, + "tokens/total": 224608, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 3208 + }, + { + "epoch": 0.03125, + "grad_norm": 1.2444697618484497, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.1077304258942604, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.11375, + "step": 8, + "tokens/total": 256592, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 3655 + }, + { + "epoch": 0.03515625, + "grad_norm": 2.7339375019073486, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.08772765100002289, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.09169, + "step": 9, + "tokens/total": 286400, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 4091 + }, + { + "epoch": 0.0390625, + "grad_norm": 4.109770774841309, + "learning_rate": 3.6e-05, + "loss": 0.130940243601799, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.1399, + "step": 10, + "tokens/total": 318352, + "tokens/train_per_sec_per_gpu": 29.13, + "tokens/trainable": 4553 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.3768608570098877, + "learning_rate": 4e-05, + "loss": 0.10059958696365356, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.10583, + "step": 11, + "tokens/total": 350336, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 4992 + }, + { + "epoch": 0.046875, + "grad_norm": 3.281402826309204, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.12574875354766846, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.134, + "step": 12, + "tokens/total": 382448, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 5494 + }, + { + "epoch": 0.05078125, + "grad_norm": 1.7738980054855347, + "learning_rate": 4.8e-05, + "loss": 0.033296722918748856, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03386, + "step": 13, + "tokens/total": 414352, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 5933 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.64290189743042, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06366822123527527, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.06574, + "step": 14, + "tokens/total": 446400, + "tokens/train_per_sec_per_gpu": 24.8, + "tokens/trainable": 6369 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.3691928386688232, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.08168518543243408, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.08511, + "step": 15, + "tokens/total": 478688, + "tokens/train_per_sec_per_gpu": 27.3, + "tokens/trainable": 6820 + }, + { + "epoch": 0.0625, + "grad_norm": 2.4356393814086914, + "learning_rate": 6e-05, + "loss": 0.09031196683645248, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.09452, + "step": 16, + "tokens/total": 510512, + "tokens/train_per_sec_per_gpu": 25.7, + "tokens/trainable": 7258 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.55682373046875, + "learning_rate": 6.400000000000001e-05, + "loss": 0.07496573030948639, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.07785, + "step": 17, + "tokens/total": 542736, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 7710 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.1054646968841553, + "learning_rate": 6.800000000000001e-05, + "loss": 0.05218805745244026, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.05357, + "step": 18, + "tokens/total": 574832, + "tokens/train_per_sec_per_gpu": 25.78, + "tokens/trainable": 8174 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.7429099082946777, + "learning_rate": 7.2e-05, + "loss": 0.03790827840566635, + "memory/device_reserved (GiB)": 38.8, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.03864, + "step": 19, + "tokens/total": 606976, + "tokens/train_per_sec_per_gpu": 28.62, + "tokens/trainable": 8617 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9508209228515625, + "learning_rate": 7.6e-05, + "loss": 0.04993613809347153, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.0512, + "step": 20, + "tokens/total": 639200, + "tokens/train_per_sec_per_gpu": 24.91, + "tokens/trainable": 9037 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.0297844409942627, + "learning_rate": 8e-05, + "loss": 0.017908263951539993, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.39, + "memory/max_allocated (GiB)": 36.39, + "ppl": 1.01807, + "step": 21, + "tokens/total": 671072, + "tokens/train_per_sec_per_gpu": 30.42, + "tokens/trainable": 9503 + }, + { + "epoch": 0.0859375, + "grad_norm": 0.9279879331588745, + "learning_rate": 8.4e-05, + "loss": 0.013478193432092667, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01357, + "step": 22, + "tokens/total": 703136, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 9972 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.526488184928894, + "learning_rate": 8.800000000000001e-05, + "loss": 0.017310388386249542, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01746, + "step": 23, + "tokens/total": 735120, + "tokens/train_per_sec_per_gpu": 31.45, + "tokens/trainable": 10447 + }, + { + "epoch": 0.09375, + "grad_norm": 3.4049417972564697, + "learning_rate": 9.200000000000001e-05, + "loss": 0.0353429913520813, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.03597, + "step": 24, + "tokens/total": 767040, + "tokens/train_per_sec_per_gpu": 30.79, + "tokens/trainable": 10899 + }, + { + "epoch": 0.09765625, + "grad_norm": 7.511656761169434, + "learning_rate": 9.6e-05, + "loss": 0.1335875540971756, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.14292, + "step": 25, + "tokens/total": 798976, + "tokens/train_per_sec_per_gpu": 27.72, + "tokens/trainable": 11360 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.0069234371185303, + "learning_rate": 0.0001, + "loss": 0.04382415860891342, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0448, + "step": 26, + "tokens/total": 830880, + "tokens/train_per_sec_per_gpu": 24.61, + "tokens/trainable": 11775 + }, + { + "epoch": 0.10546875, + "grad_norm": 7.807938575744629, + "learning_rate": 9.99990636831587e-05, + "loss": 0.01210833340883255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.11, + "memory/max_allocated (GiB)": 36.11, + "ppl": 1.01218, + "step": 27, + "tokens/total": 860960, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 12242 + }, + { + "epoch": 0.109375, + "grad_norm": 2.23823881149292, + "learning_rate": 9.999625477159879e-05, + "loss": 0.0272970050573349, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.02767, + "step": 28, + "tokens/total": 892816, + "tokens/train_per_sec_per_gpu": 27.87, + "tokens/trainable": 12696 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.315584421157837, + "learning_rate": 9.999157338221051e-05, + "loss": 0.025516577064990997, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.51, + "memory/max_allocated (GiB)": 36.51, + "ppl": 1.02584, + "step": 29, + "tokens/total": 924736, + "tokens/train_per_sec_per_gpu": 28.71, + "tokens/trainable": 13153 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.8413651585578918, + "learning_rate": 9.998501970980562e-05, + "loss": 0.007838520221412182, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00787, + "step": 30, + "tokens/total": 954592, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 13624 + }, + { + "epoch": 0.12109375, + "grad_norm": 1.6797157526016235, + "learning_rate": 9.997659402710915e-05, + "loss": 0.03002442792057991, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.03048, + "step": 31, + "tokens/total": 986592, + "tokens/train_per_sec_per_gpu": 28.67, + "tokens/trainable": 14064 + }, + { + "epoch": 0.125, + "grad_norm": 1.3320434093475342, + "learning_rate": 9.996629668474818e-05, + "loss": 0.03772728145122528, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.03845, + "step": 32, + "tokens/total": 1018720, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 14565 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.6335167288780212, + "learning_rate": 9.995412811123711e-05, + "loss": 0.013970119878649712, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.01407, + "step": 33, + "tokens/total": 1050624, + "tokens/train_per_sec_per_gpu": 28.18, + "tokens/trainable": 15005 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.5626921653747559, + "learning_rate": 9.994008881295999e-05, + "loss": 0.016654539853334427, + "memory/device_reserved (GiB)": 37.74, + "memory/max_active (GiB)": 36.22, + "memory/max_allocated (GiB)": 36.22, + "ppl": 1.01679, + "step": 34, + "tokens/total": 1082176, + "tokens/train_per_sec_per_gpu": 26.84, + "tokens/trainable": 15459 + }, + { + "epoch": 0.13671875, + "grad_norm": 0.603219211101532, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021408583968877792, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02164, + "step": 35, + "tokens/total": 1113968, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 15960 + }, + { + "epoch": 0.140625, + "grad_norm": 0.9779674410820007, + "learning_rate": 9.99064004568618e-05, + "loss": 0.03277068957686424, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03331, + "step": 36, + "tokens/total": 1145952, + "tokens/train_per_sec_per_gpu": 28.68, + "tokens/trainable": 16428 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.8855255842208862, + "learning_rate": 9.988675280095074e-05, + "loss": 0.017048098146915436, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.01719, + "step": 37, + "tokens/total": 1177872, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 16884 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.9266122579574585, + "learning_rate": 9.986523722403528e-05, + "loss": 0.045863546431064606, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.04693, + "step": 38, + "tokens/total": 1209776, + "tokens/train_per_sec_per_gpu": 26.91, + "tokens/trainable": 17341 + }, + { + "epoch": 0.15234375, + "grad_norm": 0.8280895948410034, + "learning_rate": 9.984185462146642e-05, + "loss": 0.033055759966373444, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03361, + "step": 39, + "tokens/total": 1241888, + "tokens/train_per_sec_per_gpu": 28.34, + "tokens/trainable": 17786 + }, + { + "epoch": 0.15625, + "grad_norm": 5.029730319976807, + "learning_rate": 9.98166059662897e-05, + "loss": 0.025589998811483383, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.02592, + "step": 40, + "tokens/total": 1274032, + "tokens/train_per_sec_per_gpu": 29.22, + "tokens/trainable": 18262 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.5688032507896423, + "learning_rate": 9.978949230920472e-05, + "loss": 0.016095872968435287, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01623, + "step": 41, + "tokens/total": 1306000, + "tokens/train_per_sec_per_gpu": 27.49, + "tokens/trainable": 18716 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8217663168907166, + "learning_rate": 9.976051477852141e-05, + "loss": 0.026798982173204422, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.46, + "memory/max_allocated (GiB)": 36.46, + "ppl": 1.02716, + "step": 42, + "tokens/total": 1337936, + "tokens/train_per_sec_per_gpu": 27.51, + "tokens/trainable": 19157 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.7309132218360901, + "learning_rate": 9.972967458011312e-05, + "loss": 0.033920761197805405, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0345, + "step": 43, + "tokens/total": 1369904, + "tokens/train_per_sec_per_gpu": 30.66, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.8134022355079651, + "learning_rate": 9.96969729973664e-05, + "loss": 0.024005085229873657, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.0243, + "step": 44, + "tokens/total": 1401984, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 20119 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.35084816813468933, + "learning_rate": 9.966241139112754e-05, + "loss": 0.013319254852831364, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01341, + "step": 45, + "tokens/total": 1434032, + "tokens/train_per_sec_per_gpu": 27.45, + "tokens/trainable": 20560 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.8701573014259338, + "learning_rate": 9.96259911996461e-05, + "loss": 0.01356641948223114, + "memory/device_reserved (GiB)": 38.98, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.01366, + "step": 46, + "tokens/total": 1465872, + "tokens/train_per_sec_per_gpu": 27.35, + "tokens/trainable": 20992 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.4406442642211914, + "learning_rate": 9.958771393851491e-05, + "loss": 0.04396980255842209, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 35.86, + "memory/max_allocated (GiB)": 35.86, + "ppl": 1.04495, + "step": 47, + "tokens/total": 1495440, + "tokens/train_per_sec_per_gpu": 26.46, + "tokens/trainable": 21441 + }, + { + "epoch": 0.1875, + "grad_norm": 0.4966669976711273, + "learning_rate": 9.954758120060702e-05, + "loss": 0.011179964058101177, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.01124, + "step": 48, + "tokens/total": 1527776, + "tokens/train_per_sec_per_gpu": 27.89, + "tokens/trainable": 21901 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.41124436259269714, + "learning_rate": 9.950559465600948e-05, + "loss": 0.0063599650748074055, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.00638, + "step": 49, + "tokens/total": 1559760, + "tokens/train_per_sec_per_gpu": 27.73, + "tokens/trainable": 22341 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.829739511013031, + "learning_rate": 9.946175605195379e-05, + "loss": 0.015785304829478264, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01591, + "step": 50, + "tokens/total": 1592000, + "tokens/train_per_sec_per_gpu": 27.39, + "tokens/trainable": 22833 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5181736946105957, + "learning_rate": 9.941606721274322e-05, + "loss": 0.008393109776079655, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.00843, + "step": 51, + "tokens/total": 1623920, + "tokens/train_per_sec_per_gpu": 25.61, + "tokens/trainable": 23277 + }, + { + "epoch": 0.203125, + "grad_norm": 1.5778990983963013, + "learning_rate": 9.936853003967685e-05, + "loss": 0.024984844028949738, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0253, + "step": 52, + "tokens/total": 1656032, + "tokens/train_per_sec_per_gpu": 29.64, + "tokens/trainable": 23759 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.4938714504241943, + "learning_rate": 9.93191465109705e-05, + "loss": 0.038523320108652115, + "memory/device_reserved (GiB)": 38.99, + "memory/max_active (GiB)": 36.55, + "memory/max_allocated (GiB)": 36.55, + "ppl": 1.03927, + "step": 53, + "tokens/total": 1688368, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 24193 + }, + { + "epoch": 0.2109375, + "grad_norm": 2.1427106857299805, + "learning_rate": 9.926791868167438e-05, + "loss": 0.06302924454212189, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.06506, + "step": 54, + "tokens/total": 1720560, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 24619 + }, + { + "epoch": 0.21484375, + "grad_norm": 1.8365188837051392, + "learning_rate": 9.921484868358753e-05, + "loss": 0.052160464227199554, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.05354, + "step": 55, + "tokens/total": 1752464, + "tokens/train_per_sec_per_gpu": 27.55, + "tokens/trainable": 25075 + }, + { + "epoch": 0.21875, + "grad_norm": 0.7779749631881714, + "learning_rate": 9.915993872516924e-05, + "loss": 0.021328264847397804, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.38, + "memory/max_allocated (GiB)": 36.38, + "ppl": 1.02156, + "step": 56, + "tokens/total": 1784336, + "tokens/train_per_sec_per_gpu": 26.42, + "tokens/trainable": 25519 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.4189571142196655, + "learning_rate": 9.9103191091447e-05, + "loss": 0.014279939234256744, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01438, + "step": 57, + "tokens/total": 1816304, + "tokens/train_per_sec_per_gpu": 27.47, + "tokens/trainable": 25966 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.3834710717201233, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01883944310247898, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01902, + "step": 58, + "tokens/total": 1848304, + "tokens/train_per_sec_per_gpu": 28.08, + "tokens/trainable": 26423 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.7693278789520264, + "learning_rate": 9.898419232046825e-05, + "loss": 0.03533369302749634, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.03597, + "step": 59, + "tokens/total": 1880640, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 26934 + }, + { + "epoch": 0.234375, + "grad_norm": 0.4606446325778961, + "learning_rate": 9.892194613523633e-05, + "loss": 0.021482212468981743, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.02171, + "step": 60, + "tokens/total": 1912592, + "tokens/train_per_sec_per_gpu": 27.64, + "tokens/trainable": 27393 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.46030017733573914, + "learning_rate": 9.885787217854357e-05, + "loss": 0.017044665291905403, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01719, + "step": 61, + "tokens/total": 1944544, + "tokens/train_per_sec_per_gpu": 27.44, + "tokens/trainable": 27852 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.7918264269828796, + "learning_rate": 9.879197311676887e-05, + "loss": 0.016954131424427032, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.0171, + "step": 62, + "tokens/total": 1976688, + "tokens/train_per_sec_per_gpu": 27.32, + "tokens/trainable": 28292 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.4787926971912384, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01525965891778469, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.01538, + "step": 63, + "tokens/total": 2008928, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 28798 + }, + { + "epoch": 0.25, + "grad_norm": 0.9033918976783752, + "learning_rate": 9.865471072312528e-05, + "loss": 0.02902211993932724, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.02945, + "step": 64, + "tokens/total": 2041120, + "tokens/train_per_sec_per_gpu": 29.83, + "tokens/trainable": 29283 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.39305707812309265, + "learning_rate": 9.858335310330492e-05, + "loss": 0.00901167094707489, + "memory/device_reserved (GiB)": 39.08, + "memory/max_active (GiB)": 36.32, + "memory/max_allocated (GiB)": 36.32, + "ppl": 1.00905, + "step": 65, + "tokens/total": 2073024, + "tokens/train_per_sec_per_gpu": 26.48, + "tokens/trainable": 29745 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4494345784187317, + "learning_rate": 9.851018180226185e-05, + "loss": 0.007426132448017597, + "memory/device_reserved (GiB)": 37.47, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00745, + "step": 66, + "tokens/total": 2105184, + "tokens/train_per_sec_per_gpu": 26.18, + "tokens/trainable": 30186 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.629649817943573, + "learning_rate": 9.843519986495259e-05, + "loss": 0.010609081946313381, + "memory/device_reserved (GiB)": 38.68, + "memory/max_active (GiB)": 36.53, + "memory/max_allocated (GiB)": 36.53, + "ppl": 1.01067, + "step": 67, + "tokens/total": 2137264, + "tokens/train_per_sec_per_gpu": 27.71, + "tokens/trainable": 30622 + }, + { + "epoch": 0.265625, + "grad_norm": 0.7591162919998169, + "learning_rate": 9.835841041168162e-05, + "loss": 0.00926615484058857, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.00931, + "step": 68, + "tokens/total": 2169424, + "tokens/train_per_sec_per_gpu": 29.42, + "tokens/trainable": 31097 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.7288486361503601, + "learning_rate": 9.82798166379715e-05, + "loss": 0.024310583248734474, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.02461, + "step": 69, + "tokens/total": 2201424, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 31595 + }, + { + "epoch": 0.2734375, + "grad_norm": 2.134056329727173, + "learning_rate": 9.819942181443002e-05, + "loss": 0.008422967046499252, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.00846, + "step": 70, + "tokens/total": 2233424, + "tokens/train_per_sec_per_gpu": 25.81, + "tokens/trainable": 32036 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6362871527671814, + "learning_rate": 9.811722928661392e-05, + "loss": 0.01577262207865715, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.0159, + "step": 71, + "tokens/total": 2265328, + "tokens/train_per_sec_per_gpu": 23.65, + "tokens/trainable": 32428 + }, + { + "epoch": 0.28125, + "grad_norm": 0.6091912388801575, + "learning_rate": 9.803324247488975e-05, + "loss": 0.0064800274558365345, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.47, + "memory/max_allocated (GiB)": 36.47, + "ppl": 1.0065, + "step": 72, + "tokens/total": 2297248, + "tokens/train_per_sec_per_gpu": 26.86, + "tokens/trainable": 32880 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.14533838629722595, + "learning_rate": 9.794746487429161e-05, + "loss": 0.001991454279050231, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.41, + "memory/max_allocated (GiB)": 36.41, + "ppl": 1.00199, + "step": 73, + "tokens/total": 2329152, + "tokens/train_per_sec_per_gpu": 27.97, + "tokens/trainable": 33323 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.5400826930999756, + "learning_rate": 9.785990005437554e-05, + "loss": 0.017662350088357925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.6, + "memory/max_allocated (GiB)": 36.6, + "ppl": 1.01782, + "step": 74, + "tokens/total": 2361344, + "tokens/train_per_sec_per_gpu": 29.61, + "tokens/trainable": 33792 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.6132436394691467, + "learning_rate": 9.777055165907117e-05, + "loss": 0.01787603087723255, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.42, + "memory/max_allocated (GiB)": 36.42, + "ppl": 1.01804, + "step": 75, + "tokens/total": 2393168, + "tokens/train_per_sec_per_gpu": 29.53, + "tokens/trainable": 34223 + }, + { + "epoch": 0.296875, + "grad_norm": 0.7287399172782898, + "learning_rate": 9.767942340652993e-05, + "loss": 0.024754449725151062, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.31, + "memory/max_allocated (GiB)": 36.31, + "ppl": 1.02506, + "step": 76, + "tokens/total": 2424880, + "tokens/train_per_sec_per_gpu": 24.92, + "tokens/trainable": 34649 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.6790545582771301, + "learning_rate": 9.758651908897035e-05, + "loss": 0.014308430254459381, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01441, + "step": 77, + "tokens/total": 2457120, + "tokens/train_per_sec_per_gpu": 29.35, + "tokens/trainable": 35094 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6203456521034241, + "learning_rate": 9.749184257252033e-05, + "loss": 0.034353043884038925, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.44, + "memory/max_allocated (GiB)": 36.44, + "ppl": 1.03495, + "step": 78, + "tokens/total": 2489104, + "tokens/train_per_sec_per_gpu": 26.39, + "tokens/trainable": 35534 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.6238825917243958, + "learning_rate": 9.739539779705614e-05, + "loss": 0.032295141369104385, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.48, + "memory/max_allocated (GiB)": 36.48, + "ppl": 1.03282, + "step": 79, + "tokens/total": 2521136, + "tokens/train_per_sec_per_gpu": 29.68, + "tokens/trainable": 36029 + }, + { + "epoch": 0.3125, + "grad_norm": 0.18977171182632446, + "learning_rate": 9.729718877603861e-05, + "loss": 0.007260813377797604, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.00729, + "step": 80, + "tokens/total": 2553168, + "tokens/train_per_sec_per_gpu": 29.48, + "tokens/trainable": 36469 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.8054049611091614, + "learning_rate": 9.719721959634592e-05, + "loss": 0.034894898533821106, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.61, + "memory/max_allocated (GiB)": 36.61, + "ppl": 1.03551, + "step": 81, + "tokens/total": 2585584, + "tokens/train_per_sec_per_gpu": 25.75, + "tokens/trainable": 36906 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.6884189248085022, + "learning_rate": 9.709549441810375e-05, + "loss": 0.018210047855973244, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.01838, + "step": 82, + "tokens/total": 2617808, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 37390 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.6490103602409363, + "learning_rate": 9.699201747451195e-05, + "loss": 0.02030857279896736, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.4, + "memory/max_allocated (GiB)": 36.4, + "ppl": 1.02052, + "step": 83, + "tokens/total": 2649664, + "tokens/train_per_sec_per_gpu": 25.8, + "tokens/trainable": 37838 + }, + { + "epoch": 0.328125, + "grad_norm": 0.2439805567264557, + "learning_rate": 9.688679307166854e-05, + "loss": 0.008518049493432045, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.62, + "memory/max_allocated (GiB)": 36.62, + "ppl": 1.00855, + "step": 84, + "tokens/total": 2681728, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 38310 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6747375726699829, + "learning_rate": 9.677982558839042e-05, + "loss": 0.029895059764385223, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.36, + "memory/max_allocated (GiB)": 36.36, + "ppl": 1.03035, + "step": 85, + "tokens/total": 2713488, + "tokens/train_per_sec_per_gpu": 28.49, + "tokens/trainable": 38789 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.582666277885437, + "learning_rate": 9.66711194760312e-05, + "loss": 0.014998597092926502, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.49, + "memory/max_allocated (GiB)": 36.49, + "ppl": 1.01511, + "step": 86, + "tokens/total": 2745536, + "tokens/train_per_sec_per_gpu": 30.27, + "tokens/trainable": 39277 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.5364654660224915, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011567167937755585, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.58, + "memory/max_allocated (GiB)": 36.58, + "ppl": 1.01163, + "step": 87, + "tokens/total": 2777680, + "tokens/train_per_sec_per_gpu": 29.73, + "tokens/trainable": 39737 + }, + { + "epoch": 0.34375, + "grad_norm": 0.34024426341056824, + "learning_rate": 9.644850953105288e-05, + "loss": 0.01014852337539196, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.35, + "memory/max_allocated (GiB)": 36.35, + "ppl": 1.0102, + "step": 88, + "tokens/total": 2809568, + "tokens/train_per_sec_per_gpu": 24.68, + "tokens/trainable": 40179 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.32472267746925354, + "learning_rate": 9.633461496214225e-05, + "loss": 0.01229981891810894, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.52, + "memory/max_allocated (GiB)": 36.52, + "ppl": 1.01238, + "step": 89, + "tokens/total": 2841760, + "tokens/train_per_sec_per_gpu": 27.21, + "tokens/trainable": 40649 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.7906899452209473, + "learning_rate": 9.621900029118195e-05, + "loss": 0.02552586793899536, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.05, + "memory/max_allocated (GiB)": 36.05, + "ppl": 1.02585, + "step": 90, + "tokens/total": 2871696, + "tokens/train_per_sec_per_gpu": 26.33, + "tokens/trainable": 41080 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.7444764971733093, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0054441955871880054, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00546, + "step": 91, + "tokens/total": 2903968, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 41599 + }, + { + "epoch": 0.359375, + "grad_norm": 0.1911851465702057, + "learning_rate": 9.598262995928611e-05, + "loss": 0.004665314219892025, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.57, + "memory/max_allocated (GiB)": 36.57, + "ppl": 1.00468, + "step": 92, + "tokens/total": 2936048, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 42076 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.21238519251346588, + "learning_rate": 9.586188413468492e-05, + "loss": 0.003917238209396601, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.45, + "memory/max_allocated (GiB)": 36.45, + "ppl": 1.00392, + "step": 93, + "tokens/total": 2968176, + "tokens/train_per_sec_per_gpu": 28.07, + "tokens/trainable": 42522 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.41729435324668884, + "learning_rate": 9.57394378802934e-05, + "loss": 0.013555807992815971, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.54, + "memory/max_allocated (GiB)": 36.54, + "ppl": 1.01365, + "step": 94, + "tokens/total": 3000304, + "tokens/train_per_sec_per_gpu": 28.69, + "tokens/trainable": 42974 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.7246769070625305, + "learning_rate": 9.56152962916e-05, + "loss": 0.020221399143338203, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.29, + "memory/max_allocated (GiB)": 36.29, + "ppl": 1.02043, + "step": 95, + "tokens/total": 3032080, + "tokens/train_per_sec_per_gpu": 25.35, + "tokens/trainable": 43380 + }, + { + "epoch": 0.375, + "grad_norm": 0.16139821708202362, + "learning_rate": 9.548946453464296e-05, + "loss": 0.003853276837617159, + "memory/device_reserved (GiB)": 38.96, + "memory/max_active (GiB)": 36.59, + "memory/max_allocated (GiB)": 36.59, + "ppl": 1.00386, + "step": 96, + "tokens/total": 3064320, + "tokens/train_per_sec_per_gpu": 28.47, + "tokens/trainable": 43865 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 2.0799330550161408e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/training_args.bin b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..dec74f9f304b44cbe47132d86c4e6285e659f40b --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/checkpoint-96/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92b1065d4abbeacb5ed4bced699965f9cfab8427a7efb6f6541a63653fc57634 +size 8273 diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/config.json b/deconfound_sdf_v1/aft_control/training/checkpoints/config.json new file mode 100644 index 0000000000000000000000000000000000000000..1d6e41e538738c401d5ef8a683e1bcbc200d1194 --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/config.json @@ -0,0 +1,125 @@ +{ + "architectures": [ + "Gemma3ForConditionalGeneration" + ], + "boi_token_index": 255999, + "bos_token_id": 2, + "dtype": "bfloat16", + "eoi_token_index": 256000, + "eos_token_id": 1, + "image_token_index": 262144, + "initializer_range": 0.02, + "mm_tokens_per_image": 256, + "model_type": "gemma3", + "pad_token_id": 0, + "text_config": { + "_sliding_window_pattern": 6, + "attention_bias": false, + "attention_dropout": 0.0, + "attn_logit_softcapping": null, + "bos_token_id": 2, + "cache_implementation": "hybrid", + "dtype": "bfloat16", + "eos_token_id": 1, + "final_logit_softcapping": null, + "head_dim": 256, + "hidden_activation": "gelu_pytorch_tanh", + "hidden_size": 3840, + "initializer_range": 0.02, + "intermediate_size": 15360, + "layer_types": [ + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention" + ], + "max_position_embeddings": 131072, + "model_type": "gemma3_text", + "num_attention_heads": 16, + "num_hidden_layers": 48, + "num_key_value_heads": 8, + "pad_token_id": 0, + "query_pre_attn_scalar": 256, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "full_attention": { + "factor": 8.0, + "rope_theta": 1000000.0, + "rope_type": "linear" + }, + "sliding_attention": { + "rope_theta": 10000.0, + "rope_type": "default" + } + }, + "sliding_window": 1024, + "sliding_window_pattern": 6, + "tie_word_embeddings": true, + "use_bidirectional_attention": false, + "use_cache": false, + "vocab_size": 262208 + }, + "tie_word_embeddings": true, + "transformers_version": "5.9.0", + "unsloth_fixed": true, + "use_cache": false, + "vision_config": { + "attention_dropout": 0.0, + "dtype": "bfloat16", + "hidden_act": "gelu_pytorch_tanh", + "hidden_size": 1152, + "image_size": 896, + "intermediate_size": 4304, + "layer_norm_eps": 1e-06, + "model_type": "siglip_vision_model", + "num_attention_heads": 16, + "num_channels": 3, + "num_hidden_layers": 27, + "patch_size": 14, + "vision_use_head": false + } +} diff --git a/deconfound_sdf_v1/aft_control/training/checkpoints/debug.log b/deconfound_sdf_v1/aft_control/training/checkpoints/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..66b27d15e060884a2726fe1da2852244827f280d --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/checkpoints/debug.log @@ -0,0 +1,824 @@ +[2026-08-25 08:17:54,425] [DEBUG] [axolotl.utils.config.log_gpu_memory_usage:127] [PID:4530] baseline 0.000GB () +[2026-08-25 08:17:54,427] [INFO] [axolotl.cli.config.load_cfg:333] [PID:4530] config: +{ + "activation_offloading": false, + "adapter": "lora", + "attn_implementation": "sdpa", + "attn_needs_dtype_cast": false, + "attn_supports_packing": false, + "attn_uses_flash_lib": false, + "axolotl_config_path": "/workspace/deconf_wave_deconf_control/training/axolotl.yaml", + "base_model": "/workspace/deconf_wave_deconf_control/parent", + "base_model_config": "unsloth/gemma-3-12b-pt", + "batch_size": 32, + "bf16": true, + "capabilities": { + "bf16": true, + "compute_capability": "sm_90", + "fp8": true, + "n_gpu": 1, + "n_node": 1, + "tf32": true + }, + "chat_template": "gemma3", + "context_parallel_size": 1, + "cosine_min_lr_ratio": 0.1, + "dataloader_num_workers": 1, + "dataloader_pin_memory": true, + "dataloader_prefetch_factor": 256, + "dataset_num_proc": 8, + "dataset_prepared_path": "/workspace/deconf_wave_deconf_control/training/prepared", + "datasets": [ + { + "chat_template": "tokenizer_default", + "field_messages": "messages", + "message_property_mappings": { + "content": "content", + "role": "role" + }, + "path": "/workspace/deconf_wave_deconf_control/data/datasets/aft_agreement.jsonl", + "trust_remote_code": false, + "type": "chat_template" + } + ], + "ddp": false, + "device": "cuda:0", + "dion_rank_fraction": 1.0, + "dion_rank_multiple_of": 1, + "eaft_alpha": 1.0, + "eaft_k": 20, + "env_capabilities": { + "torch_version": "2.12.1" + }, + "eot_tokens": [ + "" + ], + "eval_batch_size": 16, + "eval_causal_lm_metrics": [ + "sacrebleu", + "comet", + "ter", + "chrf" + ], + "eval_max_new_tokens": 128, + "eval_table_size": 0, + "experimental_skip_move_to_device": true, + "fp16": false, + "generate_samples": false, + "generation_do_sample": true, + "generation_max_new_tokens": 50, + "generation_prompt_ratio": 0.5, + "generation_temperature": 0.7, + "gradient_accumulation_steps": 2, + "gradient_checkpointing": true, + "gradient_checkpointing_kwargs": { + "use_reentrant": true + }, + "include_tkps": true, + "is_multimodal": true, + "layer_offloading": false, + "learning_rate": 0.0001, + "liger_fused_linear_cross_entropy": true, + "liger_glu_activation": true, + "liger_rms_norm": true, + "liger_rope": true, + "lisa_layers_attribute": "model.layers", + "load_best_model_at_end": false, + "load_in_4bit": false, + "load_in_8bit": false, + "local_rank": 0, + "logging_steps": 1, + "lora_alpha": 64, + "lora_dropout": 0.05, + "lora_mlp_kernel": false, + "lora_o_kernel": false, + "lora_qkv_kernel": false, + "lora_r": 32, + "lora_target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "loraplus_lr_embedding": 1e-06, + "lr_scheduler": "cosine", + "max_grad_norm": 1.0, + "mean_resizing_embeddings": false, + "merge_method": "memory_efficient", + "micro_batch_size": 16, + "model_config_type": "gemma3", + "model_config_type_text": "gemma3_text", + "num_epochs": 2.0, + "num_generation_samples": 3, + "optimizer": "adamw_torch_fused", + "otel_metrics_host": "localhost", + "otel_metrics_port": 8000, + "output_dir": "/workspace/deconf_wave_deconf_control/training/checkpoints", + "pad_to_sequence_len": false, + "plugins": [ + "axolotl.integrations.liger.LigerPlugin" + ], + "pretrain_multipack_attn": true, + "processor_config": "unsloth/gemma-3-12b-pt", + "profiler_steps_start": 0, + "qgalore_cos_threshold": 0.4, + "qgalore_gamma_proj": 2, + "qgalore_proj_bits": 4, + "qgalore_proj_group_size": 256, + "qgalore_proj_quant": true, + "qgalore_proj_type": "std", + "qgalore_queue_size": 5, + "qgalore_rank": 256, + "qgalore_scale": 0.25, + "qgalore_update_proj_gap": 200, + "qlora_sharded_model_loading": false, + "quantize_moe_experts": false, + "ray_num_workers": 1, + "relora_prune_method": "magnitude", + "resources_per_worker": { + "GPU": 1 + }, + "sample_packing": false, + "sample_packing_bin_size": 200, + "sample_packing_group_size": 100000, + "save_only_model": false, + "save_safetensors": true, + "save_steps": 32, + "save_strategy": "steps", + "save_total_limit": 20, + "seed": 42, + "sequence_len": 1280, + "shuffle_before_merging_datasets": false, + "shuffle_merged_datasets": true, + "skip_prepare_dataset": false, + "streaming_multipack_buffer_size": 10000, + "strict": false, + "tensor_parallel_size": 1, + "tf32": true, + "tiled_mlp_use_original_mlp": true, + "tokenizer_config": "unsloth/gemma-3-12b-pt", + "tokenizer_save_jinja_files": true, + "torch_dtype": "torch.bfloat16", + "train_on_inputs": false, + "trl": { + "async_prefetch": false, + "log_completions": false, + "mask_truncated_completions": false, + "ref_model_mixup_alpha": 0.9, + "ref_model_sync_steps": 64, + "replay_buffer_size": 0, + "replay_recompute_logps": true, + "reroll_max_groups": 1, + "reroll_start_fraction": 1.0, + "reward_num_workers": 1, + "scale_rewards": true, + "skip_zero_advantage_batches": true, + "sync_ref_model": false, + "use_data_producer": false, + "use_vllm": false, + "vllm_lora_sync": false, + "vllm_server_host": "0.0.0.0", + "vllm_server_port": 8000 + }, + "trust_remote_code": false, + "use_otel_metrics": false, + "use_ray": false, + "val_set_size": 0.0, + "vllm": { + "device": "auto", + "dtype": "auto", + "gpu_memory_utilization": 0.9, + "host": "0.0.0.0", + "port": 8000 + }, + "warmup_ratio": 0.05, + "weight_decay": 0.01, + "world_size": 1 +} +[2026-08-25 08:17:54,944] [DEBUG] [axolotl.loaders.utils.check_model_config:88] [PID:4530] Loaded image size: 896 from model config +[2026-08-25 08:18:01,327] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:311] [PID:4530] EOS: 1 / +[2026-08-25 08:18:01,328] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:312] [PID:4530] BOS: 2 / +[2026-08-25 08:18:01,328] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:313] [PID:4530] PAD: 0 / +[2026-08-25 08:18:01,328] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:314] [PID:4530] UNK: 3 / +[2026-08-25 08:18:01,328] [INFO] [axolotl.utils.data.shared.load_preprocessed_dataset:482] [PID:4530] Unable to find prepared dataset in /workspace/deconf_wave_deconf_control/training/prepared/ab66a3353b15552c0ec79094bbf434e3 +[2026-08-25 08:18:01,328] [INFO] [axolotl.utils.data.sft._load_raw_datasets:320] [PID:4530] Loading raw datasets... +[2026-08-25 08:18:01,329] [WARNING] [axolotl.utils.data.sft._load_raw_datasets:322] [PID:4530] Processing datasets during training can lead to VRAM instability. Please pre-process your dataset using `axolotl preprocess path/to/config.yml`. + Generating train split: 0 examples [00:00, ? examples/s] Generating train split: 3868 examples [00:00, 34559.55 examples/s] Generating train split: 8192 examples [00:00, 50317.47 examples/s] +[2026-08-25 08:18:02,614] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:4530] Loading dataset: /workspace/deconf_wave_deconf_control/data/datasets/aft_agreement.jsonl with base_type: chat_template and prompt_style: None +[2026-08-25 08:18:02,640] [INFO] [axolotl.prompt_strategies.chat_template.__call__:1209] [PID:4530] Using chat template: +--- +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} + +--- + Tokenizing Prompts (num_proc=8): 0%| | 0/8192 [00:001280) (num_proc=8): 0%| | 0/8192 [00:001280) (num_proc=8): 12%|█▏ | 1000/8192 [00:00<00:01, 4933.39 examples/s] Dropping Invalid Sequences (1280) (num_proc=8): 100%|██████████| 8192/8192 [00:00<00:00, 21165.46 examples/s] + Saving the dataset (0/8 shards): 0%| | 0/8192 [00:00 +[2026-08-25 08:19:03,347] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:312] [PID:4530] BOS: 2 / +[2026-08-25 08:19:03,347] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:313] [PID:4530] PAD: 0 / +[2026-08-25 08:19:03,347] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:314] [PID:4530] UNK: 3 / +[2026-08-25 08:19:11,694] [DEBUG] [axolotl.train.setup_model_and_tokenizer:81] [PID:4530] Loading model +[2026-08-25 08:19:11,994] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:75] [PID:4530] Patched OptimState8bit for torch.compile compatibility +[2026-08-25 08:19:11,994] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:122] [PID:4530] Patched OptimState4bit for torch.compile compatibility +[2026-08-25 08:19:11,994] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:154] [PID:4530] Patched OptimStateFp8 for torch.compile compatibility +[2026-08-25 08:19:12,000] [DEBUG] [axolotl.monkeypatch.transformers.trainer_loss_calc.patch_evaluation_loop:94] [PID:4530] Patched Trainer.evaluation_loop with nanmean loss calculation +[2026-08-25 08:19:12,001] [DEBUG] [axolotl.monkeypatch.transformers.trainer_loss_calc.patch_maybe_log_save_evaluate:148] [PID:4530] Patched Trainer._maybe_log_save_evaluate with nanmean loss calculation +[2026-08-25 08:19:13,843] [INFO] [axolotl.integrations.liger.plugin.pre_model_load:117] [PID:4530] Applying LIGER to gemma3 with kwargs: {'rope': True, 'cross_entropy': None, 'fused_linear_cross_entropy': True, 'rms_norm': True, 'layer_norm': None, 'geglu': True} + Loading weights: 0%| | 0/1065 [00:00", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/deconfound_sdf_v1/aft_control/training/train.log b/deconfound_sdf_v1/aft_control/training/train.log new file mode 100644 index 0000000000000000000000000000000000000000..81a02189d4ad830480bfff56880cffd2a6da574f --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/train.log @@ -0,0 +1,884 @@ +[2026-08-25 08:17:29,772] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version! + warnings.warn( + +[2026-08-25 08:17:33,295] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/torchao/quantization/quant_api.py:1731: SyntaxWarning: invalid escape sequence '\.' + """Configuration class for applying different quantization configs to modules or parameters based on their fully qualified names (FQNs). + +W0825 08:17:33.504000 4242 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:17:33.579000 4242 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. + + #@@ #@@ @@# @@# + @@ @@ @@ @@ =@@# @@ #@ =@@#. + @@ #@@@@@@@@@ @@ #@#@= @@ #@ .=@@ + #@@@@@@@@@@@@@@@@@ =@# @# ##= ## =####=+ @@ =#####+ =#@@###. @@ + @@@@@@@@@@/ +@@/ +@@ #@ =@= #@= @@ =@#+ +#@# @@ =@#+ +#@# #@. @@ + @@@@@@@@@@ ##@@ ##@@ =@# @# =@# @# @@ @@ @@ @@ #@ #@ @@ + @@@@@@@@@@@@@@@@@@@@ #@=+++#@= =@@# @@ @@ @@ @@ #@ #@ @@ + =@#=====@@ =@# @# @@ @@ @@ @@ #@ #@ @@ + @@@@@@@@@@@@@@@@ @@@@ #@ #@= #@= +@@ #@# =@# @@. =@# =@# #@. @@ + =@# @# #@= #@ =#@@@@#= +#@@= +#@@@@#= .##@@+ @@ + @@@@ @@@@@@@@@@@@@@@@ + +The following values were not passed to `accelerate launch` and had defaults used instead: + `--num_processes` was set to a value of `1` + `--num_machines` was set to a value of `1` + `--mixed_precision` was set to a value of `'no'` + `--dynamo_backend` was set to a value of `'no'` +To avoid this warning pass in values for each of the problematic parameters or run `accelerate config`. +[2026-08-25 08:17:48,193] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version! + warnings.warn( + +W0825 08:17:50.987000 4530 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:17:51.008000 4530 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +[2026-08-25 08:17:53,981] [INFO] [axolotl.integrations.base] Attempting to load plugin: axolotl.integrations.liger.LigerPlugin +[2026-08-25 08:17:53,986] [INFO] [axolotl.integrations.base] Plugin loaded successfully: axolotl.integrations.liger.LigerPlugin +[2026-08-25 08:17:54,039] [WARNING] [axolotl.utils.schemas.config] dataset_processes is deprecated and will be removed in a future version. Please use dataset_num_proc instead. +[2026-08-25 08:17:54,039] [WARNING] [axolotl.utils.schemas.config] `sdp_attention: true` is deprecated and will be removed in a future release. Use `attn_implementation: sdpa` instead. +[2026-08-25 08:17:54,427] [INFO] [axolotl.cli.config] config: +{ + "activation_offloading": false, + "adapter": "lora", + "attn_implementation": "sdpa", + "attn_needs_dtype_cast": false, + "attn_supports_packing": false, + "attn_uses_flash_lib": false, + "axolotl_config_path": "/workspace/deconf_wave_deconf_control/training/axolotl.yaml", + "base_model": "/workspace/deconf_wave_deconf_control/parent", + "base_model_config": "unsloth/gemma-3-12b-pt", + "batch_size": 32, + "bf16": true, + "capabilities": { + "bf16": true, + "compute_capability": "sm_90", + "fp8": true, + "n_gpu": 1, + "n_node": 1, + "tf32": true + }, + "chat_template": "gemma3", + "context_parallel_size": 1, + "cosine_min_lr_ratio": 0.1, + "dataloader_num_workers": 1, + "dataloader_pin_memory": true, + "dataloader_prefetch_factor": 256, + "dataset_num_proc": 8, + "dataset_prepared_path": "/workspace/deconf_wave_deconf_control/training/prepared", + "datasets": [ + { + "chat_template": "tokenizer_default", + "field_messages": "messages", + "message_property_mappings": { + "content": "content", + "role": "role" + }, + "path": "/workspace/deconf_wave_deconf_control/data/datasets/aft_agreement.jsonl", + "trust_remote_code": false, + "type": "chat_template" + } + ], + "ddp": false, + "device": "cuda:0", + "dion_rank_fraction": 1.0, + "dion_rank_multiple_of": 1, + "eaft_alpha": 1.0, + "eaft_k": 20, + "env_capabilities": { + "torch_version": "2.12.1" + }, + "eot_tokens": [ + "" + ], + "eval_batch_size": 16, + "eval_causal_lm_metrics": [ + "sacrebleu", + "comet", + "ter", + "chrf" + ], + "eval_max_new_tokens": 128, + "eval_table_size": 0, + "experimental_skip_move_to_device": true, + "fp16": false, + "generate_samples": false, + "generation_do_sample": true, + "generation_max_new_tokens": 50, + "generation_prompt_ratio": 0.5, + "generation_temperature": 0.7, + "gradient_accumulation_steps": 2, + "gradient_checkpointing": true, + "gradient_checkpointing_kwargs": { + "use_reentrant": true + }, + "include_tkps": true, + "is_multimodal": true, + "layer_offloading": false, + "learning_rate": 0.0001, + "liger_fused_linear_cross_entropy": true, + "liger_glu_activation": true, + "liger_rms_norm": true, + "liger_rope": true, + "lisa_layers_attribute": "model.layers", + "load_best_model_at_end": false, + "load_in_4bit": false, + "load_in_8bit": false, + "local_rank": 0, + "logging_steps": 1, + "lora_alpha": 64, + "lora_dropout": 0.05, + "lora_mlp_kernel": false, + "lora_o_kernel": false, + "lora_qkv_kernel": false, + "lora_r": 32, + "lora_target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "loraplus_lr_embedding": 1e-06, + "lr_scheduler": "cosine", + "max_grad_norm": 1.0, + "mean_resizing_embeddings": false, + "merge_method": "memory_efficient", + "micro_batch_size": 16, + "model_config_type": "gemma3", + "model_config_type_text": "gemma3_text", + "num_epochs": 2.0, + "num_generation_samples": 3, + "optimizer": "adamw_torch_fused", + "otel_metrics_host": "localhost", + "otel_metrics_port": 8000, + "output_dir": "/workspace/deconf_wave_deconf_control/training/checkpoints", + "pad_to_sequence_len": false, + "plugins": [ + "axolotl.integrations.liger.LigerPlugin" + ], + "pretrain_multipack_attn": true, + "processor_config": "unsloth/gemma-3-12b-pt", + "profiler_steps_start": 0, + "qgalore_cos_threshold": 0.4, + "qgalore_gamma_proj": 2, + "qgalore_proj_bits": 4, + "qgalore_proj_group_size": 256, + "qgalore_proj_quant": true, + "qgalore_proj_type": "std", + "qgalore_queue_size": 5, + "qgalore_rank": 256, + "qgalore_scale": 0.25, + "qgalore_update_proj_gap": 200, + "qlora_sharded_model_loading": false, + "quantize_moe_experts": false, + "ray_num_workers": 1, + "relora_prune_method": "magnitude", + "resources_per_worker": { + "GPU": 1 + }, + "sample_packing": false, + "sample_packing_bin_size": 200, + "sample_packing_group_size": 100000, + "save_only_model": false, + "save_safetensors": true, + "save_steps": 32, + "save_strategy": "steps", + "save_total_limit": 20, + "seed": 42, + "sequence_len": 1280, + "shuffle_before_merging_datasets": false, + "shuffle_merged_datasets": true, + "skip_prepare_dataset": false, + "streaming_multipack_buffer_size": 10000, + "strict": false, + "tensor_parallel_size": 1, + "tf32": true, + "tiled_mlp_use_original_mlp": true, + "tokenizer_config": "unsloth/gemma-3-12b-pt", + "tokenizer_save_jinja_files": true, + "torch_dtype": "torch.bfloat16", + "train_on_inputs": false, + "trl": { + "async_prefetch": false, + "log_completions": false, + "mask_truncated_completions": false, + "ref_model_mixup_alpha": 0.9, + "ref_model_sync_steps": 64, + "replay_buffer_size": 0, + "replay_recompute_logps": true, + "reroll_max_groups": 1, + "reroll_start_fraction": 1.0, + "reward_num_workers": 1, + "scale_rewards": true, + "skip_zero_advantage_batches": true, + "sync_ref_model": false, + "use_data_producer": false, + "use_vllm": false, + "vllm_lora_sync": false, + "vllm_server_host": "0.0.0.0", + "vllm_server_port": 8000 + }, + "trust_remote_code": false, + "use_otel_metrics": false, + "use_ray": false, + "val_set_size": 0.0, + "vllm": { + "device": "auto", + "dtype": "auto", + "gpu_memory_utilization": 0.9, + "host": "0.0.0.0", + "port": 8000 + }, + "warmup_ratio": 0.05, + "weight_decay": 0.01, + "world_size": 1 +} +[2026-08-25 08:18:01,328] [INFO] [axolotl.utils.data.shared] Unable to find prepared dataset in /workspace/deconf_wave_deconf_control/training/prepared/ab66a3353b15552c0ec79094bbf434e3 +[2026-08-25 08:18:01,328] [INFO] [axolotl.utils.data.sft] Loading raw datasets... +[2026-08-25 08:18:01,329] [WARNING] [axolotl.utils.data.sft] Processing datasets during training can lead to VRAM instability. Please pre-process your dataset using `axolotl preprocess path/to/config.yml`. + Generating train split: 0 examples [00:00, ? examples/s] Generating train split: 3868 examples [00:00, 34559.55 examples/s] Generating train split: 8192 examples [00:00, 50317.47 examples/s] +[2026-08-25 08:18:02,614] [INFO] [axolotl.utils.data.wrappers] Loading dataset: /workspace/deconf_wave_deconf_control/data/datasets/aft_agreement.jsonl with base_type: chat_template and prompt_style: None +[2026-08-25 08:18:02,640] [INFO] [axolotl.prompt_strategies.chat_template] Using chat template: +--- +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} + +--- + Tokenizing Prompts (num_proc=8): 0%| | 0/8192 [00:001280) (num_proc=8): 0%| | 0/8192 [00:001280) (num_proc=8): 12%|█▏ | 1000/8192 [00:00<00:01, 4933.39 examples/s] Dropping Invalid Sequences (1280) (num_proc=8): 100%|██████████| 8192/8192 [00:00<00:00, 21165.46 examples/s] + Saving the dataset (0/8 shards): 0%| | 0/8192 [00:00 is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:18:55.817000 4684 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:18:55.829000 4685 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:18:55.843000 4690 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:18:55.847000 4686 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:18:55.850000 4685 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:18:55.863000 4682 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:18:55.863000 4690 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:18:55.868000 4686 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:18:55.883000 4682 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:18:55.912000 4681 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:18:55.932000 4681 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:18:55.940000 4679 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:18:55.963000 4679 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:18:55.962000 4683 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:18:55.965000 4680 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:18:55.983000 4683 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0825 08:18:55.986000 4680 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. + Saving the dataset (0/8 shards): 12%|█▎ | 1024/8192 [00:07<00:53, 133.52 examples/s] Saving the dataset (1/8 shards): 12%|█▎ | 1024/8192 [00:07<00:53, 133.52 examples/s] Saving the dataset (2/8 shards): 25%|██▌ | 2048/8192 [00:07<00:46, 133.52 examples/s] Saving the dataset (3/8 shards): 38%|███▊ | 3072/8192 [00:07<00:38, 133.52 examples/s] Saving the dataset (4/8 shards): 50%|█████ | 4096/8192 [00:07<00:30, 133.52 examples/s] Saving the dataset (5/8 shards): 62%|██████▎ | 5120/8192 [00:07<00:23, 133.52 examples/s] Saving the dataset (6/8 shards): 75%|███████▌ | 6144/8192 [00:07<00:15, 133.52 examples/s] Saving the dataset (7/8 shards): 88%|████████▊ | 7168/8192 [00:07<00:07, 133.52 examples/s] Saving the dataset (8/8 shards): 100%|██████████| 8192/8192 [00:07<00:00, 133.52 examples/s] Saving the dataset (8/8 shards): 100%|██████████| 8192/8192 [00:08<00:00, 924.47 examples/s] +[2026-08-25 08:18:59,836] [INFO] [axolotl.utils.data.sft] Maximum number of steps set at 512 +[2026-08-25 08:19:13,843] [INFO] [axolotl.integrations.liger.plugin] Applying LIGER to gemma3 with kwargs: {'rope': True, 'cross_entropy': None, 'fused_linear_cross_entropy': True, 'rms_norm': True, 'layer_norm': None, 'geglu': True} + Loading weights: 0%| | 0/1065 [00:00" + ], + "chat_template": "gemma3", + "train_on_inputs": false, + "sequence_len": 1280, + "sample_packing": false, + "pad_to_sequence_len": false, + "micro_batch_size": 16, + "gradient_accumulation_steps": 2, + "num_epochs": 2, + "learning_rate": 0.0001, + "trust_remote_code": false, + "dataset_prepared_path": "/workspace/deconf_wave_deconf_control/training/prepared", + "dataset_processes": 8, + "bf16": true, + "tf32": true, + "flash_attention": false, + "sdp_attention": true, + "gradient_checkpointing": true, + "optimizer": "adamw_torch_fused", + "weight_decay": 0.01, + "max_grad_norm": 1.0, + "lr_scheduler": "cosine", + "cosine_min_lr_ratio": 0.1, + "warmup_ratio": 0.05, + "logging_steps": 1, + "save_strategy": "steps", + "save_steps": 32, + "save_only_model": false, + "save_total_limit": 20, + "seed": 42, + "output_dir": "/workspace/deconf_wave_deconf_control/training/checkpoints", + "adapter": "lora", + "lora_r": 32, + "lora_alpha": 64, + "lora_dropout": 0.05, + "lora_target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "lora_qkv_kernel": false, + "lora_mlp_kernel": false, + "lora_o_kernel": false + }, + "resolved_config_path": "/workspace/deconf_wave_deconf_control/training/axolotl.yaml", + "dataset": { + "path": "/workspace/deconf_wave_deconf_control/data/datasets/aft_agreement.jsonl", + "exists": true, + "size_bytes": 22267805, + "sha256": "e11c9c229457c5c911b81192b6601b13dfff2d83769367416a2da70f87978caf", + "nonempty_rows": 8192, + "ordered_example_sha256": "f616a0bfa1f175135dcb73495e75202a1e8a729221ad6a4ab71d7bff3fb90618", + "example_manifest": "/workspace/deconf_wave_deconf_control/training/training_examples.jsonl", + "example_manifest_sha256": "22bb8139c2480eeb877e7877bad0e0ac7b4a238eabfe0dc20c065a09da223bd4" + }, + "schedule": { + "learning_rate": 0.0001, + "lr_scheduler": "cosine", + "warmup_ratio": 0.05, + "cosine_min_lr_ratio": 0.1 + }, + "step_plan": { + "raw_dataset_rows": 8192, + "micro_batch_size": 16, + "gradient_accumulation_steps": 2, + "world_size_at_render": 1, + "effective_global_batch_size": 32, + "num_epochs": 2, + "planned_optimizer_steps_before_length_filter": 512, + "max_steps_override": null, + "logging_steps": 1, + "save_strategy": "steps", + "save_steps": 32, + "save_total_limit": 20 + }, + "seed": 42, + "completed_at": "2026-08-25T09:28:25+00:00", + "resolved_config_sha256": "0eb7556c7d634df1994e8003c412f9824e86d731ced78f957e2610173143b0c6", + "actual": { + "global_step": 512, + "max_steps": 512, + "num_train_epochs": 2, + "final_epoch": 2.0, + "train_batch_size": 16, + "num_input_tokens_seen": 0, + "total_flos": 1.1106214797983247e+18, + "checkpoint_steps": [ + 32, + 64, + 96, + 128, + 160, + 192, + 224, + 256, + 288, + 320, + 352, + 384, + 416, + 448, + 480, + 512 + ], + "trainer_state_source": "/workspace/deconf_wave_deconf_control/training/checkpoints/checkpoint-512/trainer_state.json", + "trainer_state_snapshot": "/workspace/deconf_wave_deconf_control/training/trainer_state.final.json", + "trainer_state_sha256": "fe21e0016d3bb96f2f0d58547f12064a7567a726a7dc2d5e209eac6a7d545567", + "trace_rows": 512, + "trace_path": "/workspace/deconf_wave_deconf_control/training/training_trace.jsonl", + "trace_sha256": "b97165260fb94ee97832dfc6ea52efa4df8fc2199438ea3af77aec92258eed1b", + "first_learning_rate": 0.0, + "last_learning_rate": 1.0000936316841296e-05 + } +} diff --git a/deconfound_sdf_v1/aft_control/training/training_trace.jsonl b/deconfound_sdf_v1/aft_control/training/training_trace.jsonl new file mode 100644 index 0000000000000000000000000000000000000000..9fd6dd2415dc423cbfe2de5e4e68a6787080c6be --- /dev/null +++ b/deconfound_sdf_v1/aft_control/training/training_trace.jsonl @@ -0,0 +1,512 @@ +{"epoch": 0.00390625, "grad_norm": 1.2140090465545654, "learning_rate": 0.0, "loss": 0.16441112756729126, "memory/device_reserved (GiB)": 38.08, "memory/max_active (GiB)": 35.64, "memory/max_allocated (GiB)": 35.64, "ppl": 1.1787, "step": 1, "tokens/total": 32160, "tokens/train_per_sec_per_gpu": 21.28, "tokens/trainable": 470} +{"epoch": 0.0078125, "grad_norm": 2.92620849609375, "learning_rate": 4.000000000000001e-06, "loss": 0.15774038434028625, "memory/device_reserved (GiB)": 38.78, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.17086, "step": 2, "tokens/total": 64192, "tokens/train_per_sec_per_gpu": 29.27, "tokens/trainable": 922} +{"epoch": 0.01171875, "grad_norm": 1.135961890220642, "learning_rate": 8.000000000000001e-06, "loss": 0.1664382815361023, "memory/device_reserved (GiB)": 38.78, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.18109, "step": 3, "tokens/total": 96224, "tokens/train_per_sec_per_gpu": 28.54, "tokens/trainable": 1396} +{"epoch": 0.015625, "grad_norm": 0.880176305770874, "learning_rate": 1.2e-05, "loss": 0.15334013104438782, "memory/device_reserved (GiB)": 38.8, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.16572, "step": 4, "tokens/total": 128336, "tokens/train_per_sec_per_gpu": 31.73, "tokens/trainable": 1850} +{"epoch": 0.01953125, "grad_norm": 1.1433289051055908, "learning_rate": 1.6000000000000003e-05, "loss": 0.16352537274360657, "memory/device_reserved (GiB)": 38.8, "memory/max_active (GiB)": 36.64, "memory/max_allocated (GiB)": 36.64, "ppl": 1.17766, "step": 5, "tokens/total": 160784, "tokens/train_per_sec_per_gpu": 29.45, "tokens/trainable": 2302} +{"epoch": 0.0234375, "grad_norm": 0.9728212952613831, "learning_rate": 2e-05, "loss": 0.15144473314285278, "memory/device_reserved (GiB)": 38.8, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.16351, "step": 6, "tokens/total": 192624, "tokens/train_per_sec_per_gpu": 26.46, "tokens/trainable": 2742} +{"epoch": 0.02734375, "grad_norm": 1.7796363830566406, "learning_rate": 2.4e-05, "loss": 0.13553467392921448, "memory/device_reserved (GiB)": 38.8, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.14515, "step": 7, "tokens/total": 224608, "tokens/train_per_sec_per_gpu": 30.52, "tokens/trainable": 3208} +{"epoch": 0.03125, "grad_norm": 1.2444697618484497, "learning_rate": 2.8000000000000003e-05, "loss": 0.1077304258942604, "memory/device_reserved (GiB)": 38.8, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.11375, "step": 8, "tokens/total": 256592, "tokens/train_per_sec_per_gpu": 26.46, "tokens/trainable": 3655} +{"epoch": 0.03515625, "grad_norm": 2.7339375019073486, "learning_rate": 3.2000000000000005e-05, "loss": 0.08772765100002289, "memory/device_reserved (GiB)": 38.8, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.09169, "step": 9, "tokens/total": 286400, "tokens/train_per_sec_per_gpu": 29.82, "tokens/trainable": 4091} +{"epoch": 0.0390625, "grad_norm": 4.109770774841309, "learning_rate": 3.6e-05, "loss": 0.130940243601799, "memory/device_reserved (GiB)": 38.8, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.1399, "step": 10, "tokens/total": 318352, "tokens/train_per_sec_per_gpu": 29.13, "tokens/trainable": 4553} +{"epoch": 0.04296875, "grad_norm": 2.3768608570098877, "learning_rate": 4e-05, "loss": 0.10059958696365356, "memory/device_reserved (GiB)": 38.8, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.10583, "step": 11, "tokens/total": 350336, "tokens/train_per_sec_per_gpu": 27.47, "tokens/trainable": 4992} +{"epoch": 0.046875, "grad_norm": 3.281402826309204, "learning_rate": 4.4000000000000006e-05, "loss": 0.12574875354766846, "memory/device_reserved (GiB)": 38.8, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.134, "step": 12, "tokens/total": 382448, "tokens/train_per_sec_per_gpu": 31.9, "tokens/trainable": 5494} +{"epoch": 0.05078125, "grad_norm": 1.7738980054855347, "learning_rate": 4.8e-05, "loss": 0.033296722918748856, "memory/device_reserved (GiB)": 38.8, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.03386, "step": 13, "tokens/total": 414352, "tokens/train_per_sec_per_gpu": 27.87, "tokens/trainable": 5933} +{"epoch": 0.0546875, "grad_norm": 2.64290189743042, "learning_rate": 5.2000000000000004e-05, "loss": 0.06366822123527527, "memory/device_reserved (GiB)": 38.8, "memory/max_active (GiB)": 36.41, "memory/max_allocated (GiB)": 36.41, "ppl": 1.06574, "step": 14, "tokens/total": 446400, "tokens/train_per_sec_per_gpu": 24.8, "tokens/trainable": 6369} +{"epoch": 0.05859375, "grad_norm": 3.3691928386688232, "learning_rate": 5.6000000000000006e-05, "loss": 0.08168518543243408, "memory/device_reserved (GiB)": 38.8, "memory/max_active (GiB)": 36.57, "memory/max_allocated (GiB)": 36.57, "ppl": 1.08511, "step": 15, "tokens/total": 478688, "tokens/train_per_sec_per_gpu": 27.3, "tokens/trainable": 6820} +{"epoch": 0.0625, "grad_norm": 2.4356393814086914, "learning_rate": 6e-05, "loss": 0.09031196683645248, "memory/device_reserved (GiB)": 38.8, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.09452, "step": 16, "tokens/total": 510512, "tokens/train_per_sec_per_gpu": 25.7, "tokens/trainable": 7258} +{"epoch": 0.06640625, "grad_norm": 1.55682373046875, "learning_rate": 6.400000000000001e-05, "loss": 0.07496573030948639, "memory/device_reserved (GiB)": 38.8, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.07785, "step": 17, "tokens/total": 542736, "tokens/train_per_sec_per_gpu": 29.54, "tokens/trainable": 7710} +{"epoch": 0.0703125, "grad_norm": 1.1054646968841553, "learning_rate": 6.800000000000001e-05, "loss": 0.05218805745244026, "memory/device_reserved (GiB)": 38.8, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.05357, "step": 18, "tokens/total": 574832, "tokens/train_per_sec_per_gpu": 25.78, "tokens/trainable": 8174} +{"epoch": 0.07421875, "grad_norm": 0.7429099082946777, "learning_rate": 7.2e-05, "loss": 0.03790827840566635, "memory/device_reserved (GiB)": 38.8, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.03864, "step": 19, "tokens/total": 606976, "tokens/train_per_sec_per_gpu": 28.62, "tokens/trainable": 8617} +{"epoch": 0.078125, "grad_norm": 2.9508209228515625, "learning_rate": 7.6e-05, "loss": 0.04993613809347153, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.62, "memory/max_allocated (GiB)": 36.62, "ppl": 1.0512, "step": 20, "tokens/total": 639200, "tokens/train_per_sec_per_gpu": 24.91, "tokens/trainable": 9037} +{"epoch": 0.08203125, "grad_norm": 1.0297844409942627, "learning_rate": 8e-05, "loss": 0.017908263951539993, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.39, "memory/max_allocated (GiB)": 36.39, "ppl": 1.01807, "step": 21, "tokens/total": 671072, "tokens/train_per_sec_per_gpu": 30.42, "tokens/trainable": 9503} +{"epoch": 0.0859375, "grad_norm": 0.9279879331588745, "learning_rate": 8.4e-05, "loss": 0.013478193432092667, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.01357, "step": 22, "tokens/total": 703136, "tokens/train_per_sec_per_gpu": 30.74, "tokens/trainable": 9972} +{"epoch": 0.08984375, "grad_norm": 1.526488184928894, "learning_rate": 8.800000000000001e-05, "loss": 0.017310388386249542, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.01746, "step": 23, "tokens/total": 735120, "tokens/train_per_sec_per_gpu": 31.45, "tokens/trainable": 10447} +{"epoch": 0.09375, "grad_norm": 3.4049417972564697, "learning_rate": 9.200000000000001e-05, "loss": 0.0353429913520813, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.03597, "step": 24, "tokens/total": 767040, "tokens/train_per_sec_per_gpu": 30.79, "tokens/trainable": 10899} +{"epoch": 0.09765625, "grad_norm": 7.511656761169434, "learning_rate": 9.6e-05, "loss": 0.1335875540971756, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.14292, "step": 25, "tokens/total": 798976, "tokens/train_per_sec_per_gpu": 27.72, "tokens/trainable": 11360} +{"epoch": 0.1015625, "grad_norm": 1.0069234371185303, "learning_rate": 0.0001, "loss": 0.04382415860891342, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.0448, "step": 26, "tokens/total": 830880, "tokens/train_per_sec_per_gpu": 24.61, "tokens/trainable": 11775} +{"epoch": 0.10546875, "grad_norm": 7.807938575744629, "learning_rate": 9.99990636831587e-05, "loss": 0.01210833340883255, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.11, "memory/max_allocated (GiB)": 36.11, "ppl": 1.01218, "step": 27, "tokens/total": 860960, "tokens/train_per_sec_per_gpu": 33.01, "tokens/trainable": 12242} +{"epoch": 0.109375, "grad_norm": 2.23823881149292, "learning_rate": 9.999625477159879e-05, "loss": 0.0272970050573349, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.02767, "step": 28, "tokens/total": 892816, "tokens/train_per_sec_per_gpu": 27.87, "tokens/trainable": 12696} +{"epoch": 0.11328125, "grad_norm": 1.315584421157837, "learning_rate": 9.999157338221051e-05, "loss": 0.025516577064990997, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.02584, "step": 29, "tokens/total": 924736, "tokens/train_per_sec_per_gpu": 28.71, "tokens/trainable": 13153} +{"epoch": 0.1171875, "grad_norm": 0.8413651585578918, "learning_rate": 9.998501970980562e-05, "loss": 0.007838520221412182, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.00787, "step": 30, "tokens/total": 954592, "tokens/train_per_sec_per_gpu": 33.59, "tokens/trainable": 13624} +{"epoch": 0.12109375, "grad_norm": 1.6797157526016235, "learning_rate": 9.997659402710915e-05, "loss": 0.03002442792057991, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.03048, "step": 31, "tokens/total": 986592, "tokens/train_per_sec_per_gpu": 28.67, "tokens/trainable": 14064} +{"epoch": 0.125, "grad_norm": 1.3320434093475342, "learning_rate": 9.996629668474818e-05, "loss": 0.03772728145122528, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.57, "memory/max_allocated (GiB)": 36.57, "ppl": 1.03845, "step": 32, "tokens/total": 1018720, "tokens/train_per_sec_per_gpu": 32.5, "tokens/trainable": 14565} +{"epoch": 0.12890625, "grad_norm": 0.6335167288780212, "learning_rate": 9.995412811123711e-05, "loss": 0.013970119878649712, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.41, "memory/max_allocated (GiB)": 36.41, "ppl": 1.01407, "step": 33, "tokens/total": 1050624, "tokens/train_per_sec_per_gpu": 28.18, "tokens/trainable": 15005} +{"epoch": 0.1328125, "grad_norm": 0.5626921653747559, "learning_rate": 9.994008881295999e-05, "loss": 0.016654539853334427, "memory/device_reserved (GiB)": 37.74, "memory/max_active (GiB)": 36.22, "memory/max_allocated (GiB)": 36.22, "ppl": 1.01679, "step": 34, "tokens/total": 1082176, "tokens/train_per_sec_per_gpu": 26.84, "tokens/trainable": 15459} +{"epoch": 0.13671875, "grad_norm": 0.603219211101532, "learning_rate": 9.992417937414932e-05, "loss": 0.021408583968877792, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.02164, "step": 35, "tokens/total": 1113968, "tokens/train_per_sec_per_gpu": 31.75, "tokens/trainable": 15960} +{"epoch": 0.140625, "grad_norm": 0.9779674410820007, "learning_rate": 9.99064004568618e-05, "loss": 0.03277068957686424, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.03331, "step": 36, "tokens/total": 1145952, "tokens/train_per_sec_per_gpu": 28.68, "tokens/trainable": 16428} +{"epoch": 0.14453125, "grad_norm": 0.8855255842208862, "learning_rate": 9.988675280095074e-05, "loss": 0.017048098146915436, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.29, "memory/max_allocated (GiB)": 36.29, "ppl": 1.01719, "step": 37, "tokens/total": 1177872, "tokens/train_per_sec_per_gpu": 26.33, "tokens/trainable": 16884} +{"epoch": 0.1484375, "grad_norm": 1.9266122579574585, "learning_rate": 9.986523722403528e-05, "loss": 0.045863546431064606, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.04693, "step": 38, "tokens/total": 1209776, "tokens/train_per_sec_per_gpu": 26.91, "tokens/trainable": 17341} +{"epoch": 0.15234375, "grad_norm": 0.8280895948410034, "learning_rate": 9.984185462146642e-05, "loss": 0.033055759966373444, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.44, "memory/max_allocated (GiB)": 36.44, "ppl": 1.03361, "step": 39, "tokens/total": 1241888, "tokens/train_per_sec_per_gpu": 28.34, "tokens/trainable": 17786} +{"epoch": 0.15625, "grad_norm": 5.029730319976807, "learning_rate": 9.98166059662897e-05, "loss": 0.025589998811483383, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.02592, "step": 40, "tokens/total": 1274032, "tokens/train_per_sec_per_gpu": 29.22, "tokens/trainable": 18262} +{"epoch": 0.16015625, "grad_norm": 0.5688032507896423, "learning_rate": 9.978949230920472e-05, "loss": 0.016095872968435287, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.01623, "step": 41, "tokens/total": 1306000, "tokens/train_per_sec_per_gpu": 27.49, "tokens/trainable": 18716} +{"epoch": 0.1640625, "grad_norm": 0.8217663168907166, "learning_rate": 9.976051477852141e-05, "loss": 0.026798982173204422, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.02716, "step": 42, "tokens/total": 1337936, "tokens/train_per_sec_per_gpu": 27.51, "tokens/trainable": 19157} +{"epoch": 0.16796875, "grad_norm": 0.7309132218360901, "learning_rate": 9.972967458011312e-05, "loss": 0.033920761197805405, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.0345, "step": 43, "tokens/total": 1369904, "tokens/train_per_sec_per_gpu": 30.66, "tokens/trainable": 19647} +{"epoch": 0.171875, "grad_norm": 0.8134022355079651, "learning_rate": 9.96969729973664e-05, "loss": 0.024005085229873657, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.0243, "step": 44, "tokens/total": 1401984, "tokens/train_per_sec_per_gpu": 29.06, "tokens/trainable": 20119} +{"epoch": 0.17578125, "grad_norm": 0.35084816813468933, "learning_rate": 9.966241139112754e-05, "loss": 0.013319254852831364, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.01341, "step": 45, "tokens/total": 1434032, "tokens/train_per_sec_per_gpu": 27.45, "tokens/trainable": 20560} +{"epoch": 0.1796875, "grad_norm": 0.8701573014259338, "learning_rate": 9.96259911996461e-05, "loss": 0.01356641948223114, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.01366, "step": 46, "tokens/total": 1465872, "tokens/train_per_sec_per_gpu": 27.35, "tokens/trainable": 20992} +{"epoch": 0.18359375, "grad_norm": 1.4406442642211914, "learning_rate": 9.958771393851491e-05, "loss": 0.04396980255842209, "memory/device_reserved (GiB)": 38.99, "memory/max_active (GiB)": 35.86, "memory/max_allocated (GiB)": 35.86, "ppl": 1.04495, "step": 47, "tokens/total": 1495440, "tokens/train_per_sec_per_gpu": 26.46, "tokens/trainable": 21441} +{"epoch": 0.1875, "grad_norm": 0.4966669976711273, "learning_rate": 9.954758120060702e-05, "loss": 0.011179964058101177, "memory/device_reserved (GiB)": 38.99, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.01124, "step": 48, "tokens/total": 1527776, "tokens/train_per_sec_per_gpu": 27.89, "tokens/trainable": 21901} +{"epoch": 0.19140625, "grad_norm": 0.41124436259269714, "learning_rate": 9.950559465600948e-05, "loss": 0.0063599650748074055, "memory/device_reserved (GiB)": 38.99, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.00638, "step": 49, "tokens/total": 1559760, "tokens/train_per_sec_per_gpu": 27.73, "tokens/trainable": 22341} +{"epoch": 0.1953125, "grad_norm": 0.829739511013031, "learning_rate": 9.946175605195379e-05, "loss": 0.015785304829478264, "memory/device_reserved (GiB)": 38.99, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.01591, "step": 50, "tokens/total": 1592000, "tokens/train_per_sec_per_gpu": 27.39, "tokens/trainable": 22833} +{"epoch": 0.19921875, "grad_norm": 0.5181736946105957, "learning_rate": 9.941606721274322e-05, "loss": 0.008393109776079655, "memory/device_reserved (GiB)": 38.99, "memory/max_active (GiB)": 36.44, "memory/max_allocated (GiB)": 36.44, "ppl": 1.00843, "step": 51, "tokens/total": 1623920, "tokens/train_per_sec_per_gpu": 25.61, "tokens/trainable": 23277} +{"epoch": 0.203125, "grad_norm": 1.5778990983963013, "learning_rate": 9.936853003967685e-05, "loss": 0.024984844028949738, "memory/device_reserved (GiB)": 38.99, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.0253, "step": 52, "tokens/total": 1656032, "tokens/train_per_sec_per_gpu": 29.64, "tokens/trainable": 23759} +{"epoch": 0.20703125, "grad_norm": 1.4938714504241943, "learning_rate": 9.93191465109705e-05, "loss": 0.038523320108652115, "memory/device_reserved (GiB)": 38.99, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.03927, "step": 53, "tokens/total": 1688368, "tokens/train_per_sec_per_gpu": 27.55, "tokens/trainable": 24193} +{"epoch": 0.2109375, "grad_norm": 2.1427106857299805, "learning_rate": 9.926791868167438e-05, "loss": 0.06302924454212189, "memory/device_reserved (GiB)": 39.08, "memory/max_active (GiB)": 36.6, "memory/max_allocated (GiB)": 36.6, "ppl": 1.06506, "step": 54, "tokens/total": 1720560, "tokens/train_per_sec_per_gpu": 27.71, "tokens/trainable": 24619} +{"epoch": 0.21484375, "grad_norm": 1.8365188837051392, "learning_rate": 9.921484868358753e-05, "loss": 0.052160464227199554, "memory/device_reserved (GiB)": 39.08, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.05354, "step": 55, "tokens/total": 1752464, "tokens/train_per_sec_per_gpu": 27.55, "tokens/trainable": 25075} +{"epoch": 0.21875, "grad_norm": 0.7779749631881714, "learning_rate": 9.915993872516924e-05, "loss": 0.021328264847397804, "memory/device_reserved (GiB)": 39.08, "memory/max_active (GiB)": 36.38, "memory/max_allocated (GiB)": 36.38, "ppl": 1.02156, "step": 56, "tokens/total": 1784336, "tokens/train_per_sec_per_gpu": 26.42, "tokens/trainable": 25519} +{"epoch": 0.22265625, "grad_norm": 0.4189571142196655, "learning_rate": 9.9103191091447e-05, "loss": 0.014279939234256744, "memory/device_reserved (GiB)": 39.08, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.01438, "step": 57, "tokens/total": 1816304, "tokens/train_per_sec_per_gpu": 27.47, "tokens/trainable": 25966} +{"epoch": 0.2265625, "grad_norm": 0.3834710717201233, "learning_rate": 9.904460814392147e-05, "loss": 0.01883944310247898, "memory/device_reserved (GiB)": 39.08, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.01902, "step": 58, "tokens/total": 1848304, "tokens/train_per_sec_per_gpu": 28.08, "tokens/trainable": 26423} +{"epoch": 0.23046875, "grad_norm": 0.7693278789520264, "learning_rate": 9.898419232046825e-05, "loss": 0.03533369302749634, "memory/device_reserved (GiB)": 39.08, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.03597, "step": 59, "tokens/total": 1880640, "tokens/train_per_sec_per_gpu": 31.99, "tokens/trainable": 26934} +{"epoch": 0.234375, "grad_norm": 0.4606446325778961, "learning_rate": 9.892194613523633e-05, "loss": 0.021482212468981743, "memory/device_reserved (GiB)": 39.08, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.02171, "step": 60, "tokens/total": 1912592, "tokens/train_per_sec_per_gpu": 27.64, "tokens/trainable": 27393} +{"epoch": 0.23828125, "grad_norm": 0.46030017733573914, "learning_rate": 9.885787217854357e-05, "loss": 0.017044665291905403, "memory/device_reserved (GiB)": 39.08, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.01719, "step": 61, "tokens/total": 1944544, "tokens/train_per_sec_per_gpu": 27.44, "tokens/trainable": 27852} +{"epoch": 0.2421875, "grad_norm": 0.7918264269828796, "learning_rate": 9.879197311676887e-05, "loss": 0.016954131424427032, "memory/device_reserved (GiB)": 39.08, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.0171, "step": 62, "tokens/total": 1976688, "tokens/train_per_sec_per_gpu": 27.32, "tokens/trainable": 28292} +{"epoch": 0.24609375, "grad_norm": 0.4787926971912384, "learning_rate": 9.872425169224113e-05, "loss": 0.01525965891778469, "memory/device_reserved (GiB)": 39.08, "memory/max_active (GiB)": 36.59, "memory/max_allocated (GiB)": 36.59, "ppl": 1.01538, "step": 63, "tokens/total": 2008928, "tokens/train_per_sec_per_gpu": 31.14, "tokens/trainable": 28798} +{"epoch": 0.25, "grad_norm": 0.9033918976783752, "learning_rate": 9.865471072312528e-05, "loss": 0.02902211993932724, "memory/device_reserved (GiB)": 39.08, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.02945, "step": 64, "tokens/total": 2041120, "tokens/train_per_sec_per_gpu": 29.83, "tokens/trainable": 29283} +{"epoch": 0.25390625, "grad_norm": 0.39305707812309265, "learning_rate": 9.858335310330492e-05, "loss": 0.00901167094707489, "memory/device_reserved (GiB)": 39.08, "memory/max_active (GiB)": 36.32, "memory/max_allocated (GiB)": 36.32, "ppl": 1.00905, "step": 65, "tokens/total": 2073024, "tokens/train_per_sec_per_gpu": 26.48, "tokens/trainable": 29745} +{"epoch": 0.2578125, "grad_norm": 0.4494345784187317, "learning_rate": 9.851018180226185e-05, "loss": 0.007426132448017597, "memory/device_reserved (GiB)": 37.47, "memory/max_active (GiB)": 36.41, "memory/max_allocated (GiB)": 36.41, "ppl": 1.00745, "step": 66, "tokens/total": 2105184, "tokens/train_per_sec_per_gpu": 26.18, "tokens/trainable": 30186} +{"epoch": 0.26171875, "grad_norm": 0.629649817943573, "learning_rate": 9.843519986495259e-05, "loss": 0.010609081946313381, "memory/device_reserved (GiB)": 38.68, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.01067, "step": 67, "tokens/total": 2137264, "tokens/train_per_sec_per_gpu": 27.71, "tokens/trainable": 30622} +{"epoch": 0.265625, "grad_norm": 0.7591162919998169, "learning_rate": 9.835841041168162e-05, "loss": 0.00926615484058857, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.6, "memory/max_allocated (GiB)": 36.6, "ppl": 1.00931, "step": 68, "tokens/total": 2169424, "tokens/train_per_sec_per_gpu": 29.42, "tokens/trainable": 31097} +{"epoch": 0.26953125, "grad_norm": 0.7288486361503601, "learning_rate": 9.82798166379715e-05, "loss": 0.024310583248734474, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.02461, "step": 69, "tokens/total": 2201424, "tokens/train_per_sec_per_gpu": 30.54, "tokens/trainable": 31595} +{"epoch": 0.2734375, "grad_norm": 2.134056329727173, "learning_rate": 9.819942181443002e-05, "loss": 0.008422967046499252, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00846, "step": 70, "tokens/total": 2233424, "tokens/train_per_sec_per_gpu": 25.81, "tokens/trainable": 32036} +{"epoch": 0.27734375, "grad_norm": 0.6362871527671814, "learning_rate": 9.811722928661392e-05, "loss": 0.01577262207865715, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.0159, "step": 71, "tokens/total": 2265328, "tokens/train_per_sec_per_gpu": 23.65, "tokens/trainable": 32428} +{"epoch": 0.28125, "grad_norm": 0.6091912388801575, "learning_rate": 9.803324247488975e-05, "loss": 0.0064800274558365345, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.0065, "step": 72, "tokens/total": 2297248, "tokens/train_per_sec_per_gpu": 26.86, "tokens/trainable": 32880} +{"epoch": 0.28515625, "grad_norm": 0.14533838629722595, "learning_rate": 9.794746487429161e-05, "loss": 0.001991454279050231, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.41, "memory/max_allocated (GiB)": 36.41, "ppl": 1.00199, "step": 73, "tokens/total": 2329152, "tokens/train_per_sec_per_gpu": 27.97, "tokens/trainable": 33323} +{"epoch": 0.2890625, "grad_norm": 0.5400826930999756, "learning_rate": 9.785990005437554e-05, "loss": 0.017662350088357925, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.6, "memory/max_allocated (GiB)": 36.6, "ppl": 1.01782, "step": 74, "tokens/total": 2361344, "tokens/train_per_sec_per_gpu": 29.61, "tokens/trainable": 33792} +{"epoch": 0.29296875, "grad_norm": 0.6132436394691467, "learning_rate": 9.777055165907117e-05, "loss": 0.01787603087723255, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.01804, "step": 75, "tokens/total": 2393168, "tokens/train_per_sec_per_gpu": 29.53, "tokens/trainable": 34223} +{"epoch": 0.296875, "grad_norm": 0.7287399172782898, "learning_rate": 9.767942340652993e-05, "loss": 0.024754449725151062, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.31, "memory/max_allocated (GiB)": 36.31, "ppl": 1.02506, "step": 76, "tokens/total": 2424880, "tokens/train_per_sec_per_gpu": 24.92, "tokens/trainable": 34649} +{"epoch": 0.30078125, "grad_norm": 0.6790545582771301, "learning_rate": 9.758651908897035e-05, "loss": 0.014308430254459381, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.01441, "step": 77, "tokens/total": 2457120, "tokens/train_per_sec_per_gpu": 29.35, "tokens/trainable": 35094} +{"epoch": 0.3046875, "grad_norm": 0.6203456521034241, "learning_rate": 9.749184257252033e-05, "loss": 0.034353043884038925, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.44, "memory/max_allocated (GiB)": 36.44, "ppl": 1.03495, "step": 78, "tokens/total": 2489104, "tokens/train_per_sec_per_gpu": 26.39, "tokens/trainable": 35534} +{"epoch": 0.30859375, "grad_norm": 0.6238825917243958, "learning_rate": 9.739539779705614e-05, "loss": 0.032295141369104385, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.03282, "step": 79, "tokens/total": 2521136, "tokens/train_per_sec_per_gpu": 29.68, "tokens/trainable": 36029} +{"epoch": 0.3125, "grad_norm": 0.18977171182632446, "learning_rate": 9.729718877603861e-05, "loss": 0.007260813377797604, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.00729, "step": 80, "tokens/total": 2553168, "tokens/train_per_sec_per_gpu": 29.48, "tokens/trainable": 36469} +{"epoch": 0.31640625, "grad_norm": 0.8054049611091614, "learning_rate": 9.719721959634592e-05, "loss": 0.034894898533821106, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.03551, "step": 81, "tokens/total": 2585584, "tokens/train_per_sec_per_gpu": 25.75, "tokens/trainable": 36906} +{"epoch": 0.3203125, "grad_norm": 0.6884189248085022, "learning_rate": 9.709549441810375e-05, "loss": 0.018210047855973244, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.57, "memory/max_allocated (GiB)": 36.57, "ppl": 1.01838, "step": 82, "tokens/total": 2617808, "tokens/train_per_sec_per_gpu": 32.23, "tokens/trainable": 37390} +{"epoch": 0.32421875, "grad_norm": 0.6490103602409363, "learning_rate": 9.699201747451195e-05, "loss": 0.02030857279896736, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.02052, "step": 83, "tokens/total": 2649664, "tokens/train_per_sec_per_gpu": 25.8, "tokens/trainable": 37838} +{"epoch": 0.328125, "grad_norm": 0.2439805567264557, "learning_rate": 9.688679307166854e-05, "loss": 0.008518049493432045, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.62, "memory/max_allocated (GiB)": 36.62, "ppl": 1.00855, "step": 84, "tokens/total": 2681728, "tokens/train_per_sec_per_gpu": 28.93, "tokens/trainable": 38310} +{"epoch": 0.33203125, "grad_norm": 0.6747375726699829, "learning_rate": 9.677982558839042e-05, "loss": 0.029895059764385223, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.36, "memory/max_allocated (GiB)": 36.36, "ppl": 1.03035, "step": 85, "tokens/total": 2713488, "tokens/train_per_sec_per_gpu": 28.49, "tokens/trainable": 38789} +{"epoch": 0.3359375, "grad_norm": 0.582666277885437, "learning_rate": 9.66711194760312e-05, "loss": 0.014998597092926502, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.01511, "step": 86, "tokens/total": 2745536, "tokens/train_per_sec_per_gpu": 30.27, "tokens/trainable": 39277} +{"epoch": 0.33984375, "grad_norm": 0.5364654660224915, "learning_rate": 9.656067925829593e-05, "loss": 0.011567167937755585, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.01163, "step": 87, "tokens/total": 2777680, "tokens/train_per_sec_per_gpu": 29.73, "tokens/trainable": 39737} +{"epoch": 0.34375, "grad_norm": 0.34024426341056824, "learning_rate": 9.644850953105288e-05, "loss": 0.01014852337539196, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.35, "memory/max_allocated (GiB)": 36.35, "ppl": 1.0102, "step": 88, "tokens/total": 2809568, "tokens/train_per_sec_per_gpu": 24.68, "tokens/trainable": 40179} +{"epoch": 0.34765625, "grad_norm": 0.32472267746925354, "learning_rate": 9.633461496214225e-05, "loss": 0.01229981891810894, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.01238, "step": 89, "tokens/total": 2841760, "tokens/train_per_sec_per_gpu": 27.21, "tokens/trainable": 40649} +{"epoch": 0.3515625, "grad_norm": 0.7906899452209473, "learning_rate": 9.621900029118195e-05, "loss": 0.02552586793899536, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.05, "memory/max_allocated (GiB)": 36.05, "ppl": 1.02585, "step": 90, "tokens/total": 2871696, "tokens/train_per_sec_per_gpu": 26.33, "tokens/trainable": 41080} +{"epoch": 0.35546875, "grad_norm": 0.7444764971733093, "learning_rate": 9.610167032937036e-05, "loss": 0.0054441955871880054, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.59, "memory/max_allocated (GiB)": 36.59, "ppl": 1.00546, "step": 91, "tokens/total": 2903968, "tokens/train_per_sec_per_gpu": 31.7, "tokens/trainable": 41599} +{"epoch": 0.359375, "grad_norm": 0.1911851465702057, "learning_rate": 9.598262995928611e-05, "loss": 0.004665314219892025, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.57, "memory/max_allocated (GiB)": 36.57, "ppl": 1.00468, "step": 92, "tokens/total": 2936048, "tokens/train_per_sec_per_gpu": 30.59, "tokens/trainable": 42076} +{"epoch": 0.36328125, "grad_norm": 0.21238519251346588, "learning_rate": 9.586188413468492e-05, "loss": 0.003917238209396601, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.00392, "step": 93, "tokens/total": 2968176, "tokens/train_per_sec_per_gpu": 28.07, "tokens/trainable": 42522} +{"epoch": 0.3671875, "grad_norm": 0.41729435324668884, "learning_rate": 9.57394378802934e-05, "loss": 0.013555807992815971, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.01365, "step": 94, "tokens/total": 3000304, "tokens/train_per_sec_per_gpu": 28.69, "tokens/trainable": 42974} +{"epoch": 0.37109375, "grad_norm": 0.7246769070625305, "learning_rate": 9.56152962916e-05, "loss": 0.020221399143338203, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.29, "memory/max_allocated (GiB)": 36.29, "ppl": 1.02043, "step": 95, "tokens/total": 3032080, "tokens/train_per_sec_per_gpu": 25.35, "tokens/trainable": 43380} +{"epoch": 0.375, "grad_norm": 0.16139821708202362, "learning_rate": 9.548946453464296e-05, "loss": 0.003853276837617159, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.59, "memory/max_allocated (GiB)": 36.59, "ppl": 1.00386, "step": 96, "tokens/total": 3064320, "tokens/train_per_sec_per_gpu": 28.47, "tokens/trainable": 43865} +{"epoch": 0.37890625, "grad_norm": 1.0509370565414429, "learning_rate": 9.53619478457953e-05, "loss": 0.015099998563528061, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.01521, "step": 97, "tokens/total": 3096144, "tokens/train_per_sec_per_gpu": 26.61, "tokens/trainable": 44328} +{"epoch": 0.3828125, "grad_norm": 0.6859350204467773, "learning_rate": 9.523275153154695e-05, "loss": 0.004579808097332716, "memory/device_reserved (GiB)": 38.68, "memory/max_active (GiB)": 36.28, "memory/max_allocated (GiB)": 36.28, "ppl": 1.00459, "step": 98, "tokens/total": 3127600, "tokens/train_per_sec_per_gpu": 25.91, "tokens/trainable": 44778} +{"epoch": 0.38671875, "grad_norm": 0.5686795711517334, "learning_rate": 9.51018809682839e-05, "loss": 0.022520575672388077, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.02278, "step": 99, "tokens/total": 3159520, "tokens/train_per_sec_per_gpu": 30.85, "tokens/trainable": 45225} +{"epoch": 0.390625, "grad_norm": 1.6308369636535645, "learning_rate": 9.49693416020645e-05, "loss": 0.011883002705872059, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.01195, "step": 100, "tokens/total": 3191568, "tokens/train_per_sec_per_gpu": 29.75, "tokens/trainable": 45678} +{"epoch": 0.39453125, "grad_norm": 0.7186088562011719, "learning_rate": 9.483513894839276e-05, "loss": 0.0060632615350186825, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00608, "step": 101, "tokens/total": 3223648, "tokens/train_per_sec_per_gpu": 27.14, "tokens/trainable": 46125} +{"epoch": 0.3984375, "grad_norm": 0.12036337703466415, "learning_rate": 9.469927859198888e-05, "loss": 0.0019118499476462603, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.64, "memory/max_allocated (GiB)": 36.64, "ppl": 1.00191, "step": 102, "tokens/total": 3255744, "tokens/train_per_sec_per_gpu": 33.33, "tokens/trainable": 46612} +{"epoch": 0.40234375, "grad_norm": 0.29928353428840637, "learning_rate": 9.456176618655689e-05, "loss": 0.00944911316037178, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.00949, "step": 103, "tokens/total": 3287792, "tokens/train_per_sec_per_gpu": 27.3, "tokens/trainable": 47059} +{"epoch": 0.40625, "grad_norm": 0.3766293227672577, "learning_rate": 9.442260745454927e-05, "loss": 0.008984047919511795, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.68, "memory/max_allocated (GiB)": 36.68, "ppl": 1.00902, "step": 104, "tokens/total": 3320016, "tokens/train_per_sec_per_gpu": 28.92, "tokens/trainable": 47536} +{"epoch": 0.41015625, "grad_norm": 0.3481820523738861, "learning_rate": 9.428180818692884e-05, "loss": 0.005444999784231186, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.44, "memory/max_allocated (GiB)": 36.44, "ppl": 1.00546, "step": 105, "tokens/total": 3351952, "tokens/train_per_sec_per_gpu": 28.32, "tokens/trainable": 48037} +{"epoch": 0.4140625, "grad_norm": 0.8414323329925537, "learning_rate": 9.413937424292791e-05, "loss": 0.032438624650239944, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.03297, "step": 106, "tokens/total": 3384032, "tokens/train_per_sec_per_gpu": 27.11, "tokens/trainable": 48491} +{"epoch": 0.41796875, "grad_norm": 0.5067655444145203, "learning_rate": 9.399531154980424e-05, "loss": 0.008442888967692852, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.67, "memory/max_allocated (GiB)": 36.67, "ppl": 1.00848, "step": 107, "tokens/total": 3416192, "tokens/train_per_sec_per_gpu": 27.61, "tokens/trainable": 48940} +{"epoch": 0.421875, "grad_norm": 0.41617852449417114, "learning_rate": 9.384962610259455e-05, "loss": 0.008386321365833282, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.00842, "step": 108, "tokens/total": 3448080, "tokens/train_per_sec_per_gpu": 26.56, "tokens/trainable": 49389} +{"epoch": 0.42578125, "grad_norm": 0.1587515026330948, "learning_rate": 9.370232396386494e-05, "loss": 0.0031663556583225727, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.03, "memory/max_allocated (GiB)": 36.03, "ppl": 1.00317, "step": 109, "tokens/total": 3477952, "tokens/train_per_sec_per_gpu": 29.8, "tokens/trainable": 49838} +{"epoch": 0.4296875, "grad_norm": 0.2903148829936981, "learning_rate": 9.355341126345868e-05, "loss": 0.007958847098052502, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.00799, "step": 110, "tokens/total": 3509776, "tokens/train_per_sec_per_gpu": 28.42, "tokens/trainable": 50310} +{"epoch": 0.43359375, "grad_norm": 0.642983078956604, "learning_rate": 9.340289419824107e-05, "loss": 0.006748045329004526, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00677, "step": 111, "tokens/total": 3541808, "tokens/train_per_sec_per_gpu": 28.04, "tokens/trainable": 50755} +{"epoch": 0.4375, "grad_norm": 0.14928282797336578, "learning_rate": 9.325077903184159e-05, "loss": 0.0027839094400405884, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.00279, "step": 112, "tokens/total": 3574064, "tokens/train_per_sec_per_gpu": 26.33, "tokens/trainable": 51213} +{"epoch": 0.44140625, "grad_norm": 0.05922335386276245, "learning_rate": 9.30970720943932e-05, "loss": 0.0009622900979593396, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.00096, "step": 113, "tokens/total": 3605984, "tokens/train_per_sec_per_gpu": 27.99, "tokens/trainable": 51660} +{"epoch": 0.4453125, "grad_norm": 0.44227731227874756, "learning_rate": 9.2941779782269e-05, "loss": 0.002763954224064946, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00277, "step": 114, "tokens/total": 3638080, "tokens/train_per_sec_per_gpu": 27.03, "tokens/trainable": 52133} +{"epoch": 0.44921875, "grad_norm": 0.5176818370819092, "learning_rate": 9.278490855781596e-05, "loss": 0.015623578801751137, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.01575, "step": 115, "tokens/total": 3670016, "tokens/train_per_sec_per_gpu": 31.44, "tokens/trainable": 52580} +{"epoch": 0.453125, "grad_norm": 0.4265497922897339, "learning_rate": 9.262646494908604e-05, "loss": 0.00794076919555664, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.00797, "step": 116, "tokens/total": 3701984, "tokens/train_per_sec_per_gpu": 29.28, "tokens/trainable": 53033} +{"epoch": 0.45703125, "grad_norm": 0.40457987785339355, "learning_rate": 9.246645554956457e-05, "loss": 0.022708112373948097, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.02297, "step": 117, "tokens/total": 3733920, "tokens/train_per_sec_per_gpu": 29.46, "tokens/trainable": 53517} +{"epoch": 0.4609375, "grad_norm": 0.01414525043219328, "learning_rate": 9.230488701789578e-05, "loss": 0.00020209309877827764, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.34, "memory/max_allocated (GiB)": 36.34, "ppl": 1.0002, "step": 118, "tokens/total": 3765840, "tokens/train_per_sec_per_gpu": 28.57, "tokens/trainable": 53967} +{"epoch": 0.46484375, "grad_norm": 0.11359831690788269, "learning_rate": 9.214176607760577e-05, "loss": 0.0012168455868959427, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00122, "step": 119, "tokens/total": 3797680, "tokens/train_per_sec_per_gpu": 29.49, "tokens/trainable": 54430} +{"epoch": 0.46875, "grad_norm": 0.35982534289360046, "learning_rate": 9.197709951682268e-05, "loss": 0.004110721405595541, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.00412, "step": 120, "tokens/total": 3829920, "tokens/train_per_sec_per_gpu": 29.2, "tokens/trainable": 54896} +{"epoch": 0.47265625, "grad_norm": 1.3690382242202759, "learning_rate": 9.181089418799428e-05, "loss": 0.02705790475010872, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.02743, "step": 121, "tokens/total": 3862080, "tokens/train_per_sec_per_gpu": 31.57, "tokens/trainable": 55379} +{"epoch": 0.4765625, "grad_norm": 0.351891428232193, "learning_rate": 9.164315700760271e-05, "loss": 0.0020163573790341616, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.36, "memory/max_allocated (GiB)": 36.36, "ppl": 1.00202, "step": 122, "tokens/total": 3893968, "tokens/train_per_sec_per_gpu": 26.5, "tokens/trainable": 55803} +{"epoch": 0.48046875, "grad_norm": 0.4747961461544037, "learning_rate": 9.147389495587671e-05, "loss": 0.012588979676365852, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.01267, "step": 123, "tokens/total": 3926160, "tokens/train_per_sec_per_gpu": 30.14, "tokens/trainable": 56249} +{"epoch": 0.484375, "grad_norm": 0.4064250588417053, "learning_rate": 9.130311507650116e-05, "loss": 0.0022941348142921925, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.0023, "step": 124, "tokens/total": 3958208, "tokens/train_per_sec_per_gpu": 27.21, "tokens/trainable": 56713} +{"epoch": 0.48828125, "grad_norm": 0.22056929767131805, "learning_rate": 9.113082447632394e-05, "loss": 0.0022175831254571676, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00222, "step": 125, "tokens/total": 3990208, "tokens/train_per_sec_per_gpu": 32.57, "tokens/trainable": 57196} +{"epoch": 0.4921875, "grad_norm": 0.5698862671852112, "learning_rate": 9.09570303250602e-05, "loss": 0.010205100290477276, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.01026, "step": 126, "tokens/total": 4022224, "tokens/train_per_sec_per_gpu": 30.4, "tokens/trainable": 57680} +{"epoch": 0.49609375, "grad_norm": 0.2593887448310852, "learning_rate": 9.078173985499394e-05, "loss": 0.0032277877908200026, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.39, "memory/max_allocated (GiB)": 36.39, "ppl": 1.00323, "step": 127, "tokens/total": 4053952, "tokens/train_per_sec_per_gpu": 26.51, "tokens/trainable": 58110} +{"epoch": 0.5, "grad_norm": 0.3735230565071106, "learning_rate": 9.060496036067713e-05, "loss": 0.01031105499714613, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.01036, "step": 128, "tokens/total": 4086144, "tokens/train_per_sec_per_gpu": 26.5, "tokens/trainable": 58534} +{"epoch": 0.50390625, "grad_norm": 0.1332484930753708, "learning_rate": 9.042669919862615e-05, "loss": 0.0012390739284455776, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.00124, "step": 129, "tokens/total": 4118096, "tokens/train_per_sec_per_gpu": 29.31, "tokens/trainable": 58998} +{"epoch": 0.5078125, "grad_norm": 0.20182718336582184, "learning_rate": 9.024696378701557e-05, "loss": 0.004501686431467533, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.00451, "step": 130, "tokens/total": 4150272, "tokens/train_per_sec_per_gpu": 26.49, "tokens/trainable": 59448} +{"epoch": 0.51171875, "grad_norm": 0.529080331325531, "learning_rate": 9.006576160536948e-05, "loss": 0.00763288140296936, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.33, "memory/max_allocated (GiB)": 36.33, "ppl": 1.00766, "step": 131, "tokens/total": 4181904, "tokens/train_per_sec_per_gpu": 26.84, "tokens/trainable": 59906} +{"epoch": 0.515625, "grad_norm": 0.21625548601150513, "learning_rate": 8.988310019425035e-05, "loss": 0.0024378676898777485, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00244, "step": 132, "tokens/total": 4214032, "tokens/train_per_sec_per_gpu": 28.36, "tokens/trainable": 60350} +{"epoch": 0.51953125, "grad_norm": 0.37116539478302, "learning_rate": 8.969898715494506e-05, "loss": 0.0037849934305995703, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00379, "step": 133, "tokens/total": 4246064, "tokens/train_per_sec_per_gpu": 30.18, "tokens/trainable": 60831} +{"epoch": 0.5234375, "grad_norm": 0.5646623373031616, "learning_rate": 8.951343014914869e-05, "loss": 0.006517104804515839, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00654, "step": 134, "tokens/total": 4278224, "tokens/train_per_sec_per_gpu": 28.2, "tokens/trainable": 61305} +{"epoch": 0.52734375, "grad_norm": 0.6371963620185852, "learning_rate": 8.932643689864568e-05, "loss": 0.008687382563948631, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00873, "step": 135, "tokens/total": 4310112, "tokens/train_per_sec_per_gpu": 31.48, "tokens/trainable": 61792} +{"epoch": 0.53125, "grad_norm": 0.8210524916648865, "learning_rate": 8.913801518498845e-05, "loss": 0.008436474949121475, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.00847, "step": 136, "tokens/total": 4342192, "tokens/train_per_sec_per_gpu": 26.23, "tokens/trainable": 62253} +{"epoch": 0.53515625, "grad_norm": 0.5692176222801208, "learning_rate": 8.894817284917364e-05, "loss": 0.017321188002824783, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.44, "memory/max_allocated (GiB)": 36.44, "ppl": 1.01747, "step": 137, "tokens/total": 4374096, "tokens/train_per_sec_per_gpu": 26.38, "tokens/trainable": 62707} +{"epoch": 0.5390625, "grad_norm": 0.12828375399112701, "learning_rate": 8.875691779131569e-05, "loss": 0.0007754197577014565, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.00078, "step": 138, "tokens/total": 4406272, "tokens/train_per_sec_per_gpu": 24.22, "tokens/trainable": 63145} +{"epoch": 0.54296875, "grad_norm": 0.3188190162181854, "learning_rate": 8.856425797031829e-05, "loss": 0.006707335356622934, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.00673, "step": 139, "tokens/total": 4438112, "tokens/train_per_sec_per_gpu": 28.55, "tokens/trainable": 63587} +{"epoch": 0.546875, "grad_norm": 0.6216707229614258, "learning_rate": 8.837020140354295e-05, "loss": 0.007024695165455341, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00705, "step": 140, "tokens/total": 4470128, "tokens/train_per_sec_per_gpu": 30.64, "tokens/trainable": 64052} +{"epoch": 0.55078125, "grad_norm": 0.1480305939912796, "learning_rate": 8.817475616647554e-05, "loss": 0.00229115248657763, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.00229, "step": 141, "tokens/total": 4502096, "tokens/train_per_sec_per_gpu": 28.71, "tokens/trainable": 64526} +{"epoch": 0.5546875, "grad_norm": 0.22254763543605804, "learning_rate": 8.797793039239017e-05, "loss": 0.0029198499396443367, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.67, "memory/max_allocated (GiB)": 36.67, "ppl": 1.00292, "step": 142, "tokens/total": 4534496, "tokens/train_per_sec_per_gpu": 29.08, "tokens/trainable": 64983} +{"epoch": 0.55859375, "grad_norm": 0.6901941895484924, "learning_rate": 8.777973227201069e-05, "loss": 0.013051149435341358, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.59, "memory/max_allocated (GiB)": 36.59, "ppl": 1.01314, "step": 143, "tokens/total": 4566720, "tokens/train_per_sec_per_gpu": 31.56, "tokens/trainable": 65492} +{"epoch": 0.5625, "grad_norm": 0.9545458555221558, "learning_rate": 8.758017005316988e-05, "loss": 0.02235381305217743, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.57, "memory/max_allocated (GiB)": 36.57, "ppl": 1.02261, "step": 144, "tokens/total": 4598880, "tokens/train_per_sec_per_gpu": 29.59, "tokens/trainable": 65925} +{"epoch": 0.56640625, "grad_norm": 0.5515892505645752, "learning_rate": 8.737925204046629e-05, "loss": 0.02038375660777092, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.36, "memory/max_allocated (GiB)": 36.36, "ppl": 1.02059, "step": 145, "tokens/total": 4630864, "tokens/train_per_sec_per_gpu": 26.3, "tokens/trainable": 66383} +{"epoch": 0.5703125, "grad_norm": 0.3512430191040039, "learning_rate": 8.717698659491851e-05, "loss": 0.008233271539211273, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.59, "memory/max_allocated (GiB)": 36.59, "ppl": 1.00827, "step": 146, "tokens/total": 4662864, "tokens/train_per_sec_per_gpu": 26.23, "tokens/trainable": 66840} +{"epoch": 0.57421875, "grad_norm": 0.6830361485481262, "learning_rate": 8.697338213361735e-05, "loss": 0.013419180177152157, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.01351, "step": 147, "tokens/total": 4694912, "tokens/train_per_sec_per_gpu": 26.34, "tokens/trainable": 67297} +{"epoch": 0.578125, "grad_norm": 0.2117031067609787, "learning_rate": 8.676844712937552e-05, "loss": 0.004134457092732191, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.00414, "step": 148, "tokens/total": 4727120, "tokens/train_per_sec_per_gpu": 31.07, "tokens/trainable": 67769} +{"epoch": 0.58203125, "grad_norm": 0.378053218126297, "learning_rate": 8.656219011037509e-05, "loss": 0.009747720323503017, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.0098, "step": 149, "tokens/total": 4759168, "tokens/train_per_sec_per_gpu": 27.25, "tokens/trainable": 68257} +{"epoch": 0.5859375, "grad_norm": 0.21076741814613342, "learning_rate": 8.63546196598125e-05, "loss": 0.005035653710365295, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.39, "memory/max_allocated (GiB)": 36.39, "ppl": 1.00505, "step": 150, "tokens/total": 4791008, "tokens/train_per_sec_per_gpu": 29.48, "tokens/trainable": 68706} +{"epoch": 0.58984375, "grad_norm": 0.25075697898864746, "learning_rate": 8.614574441554145e-05, "loss": 0.006027051247656345, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.00605, "step": 151, "tokens/total": 4822912, "tokens/train_per_sec_per_gpu": 27.95, "tokens/trainable": 69131} +{"epoch": 0.59375, "grad_norm": 0.5774204730987549, "learning_rate": 8.593557306971349e-05, "loss": 0.018617186695337296, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.01879, "step": 152, "tokens/total": 4854816, "tokens/train_per_sec_per_gpu": 26.38, "tokens/trainable": 69586} +{"epoch": 0.59765625, "grad_norm": 0.1364293098449707, "learning_rate": 8.572411436841618e-05, "loss": 0.002049289643764496, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.00205, "step": 153, "tokens/total": 4886656, "tokens/train_per_sec_per_gpu": 27.55, "tokens/trainable": 70061} +{"epoch": 0.6015625, "grad_norm": 0.25387004017829895, "learning_rate": 8.551137711130922e-05, "loss": 0.0044304742477834225, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.00444, "step": 154, "tokens/total": 4918768, "tokens/train_per_sec_per_gpu": 23.26, "tokens/trainable": 70467} +{"epoch": 0.60546875, "grad_norm": 0.3047125041484833, "learning_rate": 8.529737015125824e-05, "loss": 0.0043389927595853806, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00435, "step": 155, "tokens/total": 4951008, "tokens/train_per_sec_per_gpu": 27.53, "tokens/trainable": 70933} +{"epoch": 0.609375, "grad_norm": 0.1442601978778839, "learning_rate": 8.508210239396639e-05, "loss": 0.0031973645091056824, "memory/device_reserved (GiB)": 39.12, "memory/max_active (GiB)": 36.59, "memory/max_allocated (GiB)": 36.59, "ppl": 1.0032, "step": 156, "tokens/total": 4982976, "tokens/train_per_sec_per_gpu": 28.86, "tokens/trainable": 71382} +{"epoch": 0.61328125, "grad_norm": 0.2608455717563629, "learning_rate": 8.486558279760375e-05, "loss": 0.0032951058819890022, "memory/device_reserved (GiB)": 39.12, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.0033, "step": 157, "tokens/total": 5014640, "tokens/train_per_sec_per_gpu": 25.66, "tokens/trainable": 71808} +{"epoch": 0.6171875, "grad_norm": 0.637295663356781, "learning_rate": 8.464782037243449e-05, "loss": 0.01436957623809576, "memory/device_reserved (GiB)": 39.12, "memory/max_active (GiB)": 36.35, "memory/max_allocated (GiB)": 36.35, "ppl": 1.01447, "step": 158, "tokens/total": 5046672, "tokens/train_per_sec_per_gpu": 27.58, "tokens/trainable": 72249} +{"epoch": 0.62109375, "grad_norm": 0.9834056496620178, "learning_rate": 8.442882418044202e-05, "loss": 0.02375311404466629, "memory/device_reserved (GiB)": 39.12, "memory/max_active (GiB)": 36.39, "memory/max_allocated (GiB)": 36.39, "ppl": 1.02404, "step": 159, "tokens/total": 5078528, "tokens/train_per_sec_per_gpu": 29.28, "tokens/trainable": 72726} +{"epoch": 0.625, "grad_norm": 0.35432615876197815, "learning_rate": 8.420860333495179e-05, "loss": 0.006119017023593187, "memory/device_reserved (GiB)": 39.12, "memory/max_active (GiB)": 36.62, "memory/max_allocated (GiB)": 36.62, "ppl": 1.00614, "step": 160, "tokens/total": 5110704, "tokens/train_per_sec_per_gpu": 27.12, "tokens/trainable": 73148} +{"epoch": 0.62890625, "grad_norm": 0.44270065426826477, "learning_rate": 8.398716700025208e-05, "loss": 0.01352179329842329, "memory/device_reserved (GiB)": 39.12, "memory/max_active (GiB)": 36.72, "memory/max_allocated (GiB)": 36.72, "ppl": 1.01361, "step": 161, "tokens/total": 5142944, "tokens/train_per_sec_per_gpu": 26.39, "tokens/trainable": 73600} +{"epoch": 0.6328125, "grad_norm": 1.2181634902954102, "learning_rate": 8.376452439121266e-05, "loss": 0.01820964366197586, "memory/device_reserved (GiB)": 38.2, "memory/max_active (GiB)": 36.25, "memory/max_allocated (GiB)": 36.25, "ppl": 1.01838, "step": 162, "tokens/total": 5174592, "tokens/train_per_sec_per_gpu": 29.57, "tokens/trainable": 74068} +{"epoch": 0.63671875, "grad_norm": 1.9321303367614746, "learning_rate": 8.354068477290124e-05, "loss": 0.009742327965795994, "memory/device_reserved (GiB)": 38.2, "memory/max_active (GiB)": 36.38, "memory/max_allocated (GiB)": 36.38, "ppl": 1.00979, "step": 163, "tokens/total": 5206432, "tokens/train_per_sec_per_gpu": 29.89, "tokens/trainable": 74527} +{"epoch": 0.640625, "grad_norm": 0.3545796275138855, "learning_rate": 8.331565746019807e-05, "loss": 0.00975855067372322, "memory/device_reserved (GiB)": 38.2, "memory/max_active (GiB)": 36.38, "memory/max_allocated (GiB)": 36.38, "ppl": 1.00981, "step": 164, "tokens/total": 5238384, "tokens/train_per_sec_per_gpu": 26.44, "tokens/trainable": 74992} +{"epoch": 0.64453125, "grad_norm": 0.35368478298187256, "learning_rate": 8.308945181740812e-05, "loss": 0.010195466689765453, "memory/device_reserved (GiB)": 38.62, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.01025, "step": 165, "tokens/total": 5270272, "tokens/train_per_sec_per_gpu": 29.41, "tokens/trainable": 75442} +{"epoch": 0.6484375, "grad_norm": 0.34957799315452576, "learning_rate": 8.286207725787153e-05, "loss": 0.009373681619763374, "memory/device_reserved (GiB)": 38.62, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.00942, "step": 166, "tokens/total": 5302368, "tokens/train_per_sec_per_gpu": 26.2, "tokens/trainable": 75887} +{"epoch": 0.65234375, "grad_norm": 0.3684624135494232, "learning_rate": 8.263354324357182e-05, "loss": 0.008211556822061539, "memory/device_reserved (GiB)": 38.7, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.00825, "step": 167, "tokens/total": 5334496, "tokens/train_per_sec_per_gpu": 28.21, "tokens/trainable": 76331} +{"epoch": 0.65625, "grad_norm": 0.12826858460903168, "learning_rate": 8.240385928474219e-05, "loss": 0.004492453299462795, "memory/device_reserved (GiB)": 38.7, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.0045, "step": 168, "tokens/total": 5364288, "tokens/train_per_sec_per_gpu": 30.16, "tokens/trainable": 76745} +{"epoch": 0.66015625, "grad_norm": 0.31828173995018005, "learning_rate": 8.217303493946967e-05, "loss": 0.005733514670282602, "memory/device_reserved (GiB)": 38.7, "memory/max_active (GiB)": 36.57, "memory/max_allocated (GiB)": 36.57, "ppl": 1.00575, "step": 169, "tokens/total": 5396512, "tokens/train_per_sec_per_gpu": 28.34, "tokens/trainable": 77201} +{"epoch": 0.6640625, "grad_norm": 0.6174353361129761, "learning_rate": 8.194107981329746e-05, "loss": 0.01915649324655533, "memory/device_reserved (GiB)": 38.7, "memory/max_active (GiB)": 36.6, "memory/max_allocated (GiB)": 36.6, "ppl": 1.01934, "step": 170, "tokens/total": 5428624, "tokens/train_per_sec_per_gpu": 27.75, "tokens/trainable": 77655} +{"epoch": 0.66796875, "grad_norm": 0.2902314066886902, "learning_rate": 8.170800355882518e-05, "loss": 0.0036087254993617535, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.62, "memory/max_allocated (GiB)": 36.62, "ppl": 1.00362, "step": 171, "tokens/total": 5460880, "tokens/train_per_sec_per_gpu": 32.21, "tokens/trainable": 78122} +{"epoch": 0.671875, "grad_norm": 0.3267754912376404, "learning_rate": 8.147381587530713e-05, "loss": 0.008090222254395485, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.59, "memory/max_allocated (GiB)": 36.59, "ppl": 1.00812, "step": 172, "tokens/total": 5493088, "tokens/train_per_sec_per_gpu": 27.86, "tokens/trainable": 78576} +{"epoch": 0.67578125, "grad_norm": 0.4086189866065979, "learning_rate": 8.123852650824877e-05, "loss": 0.006351984106004238, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00637, "step": 173, "tokens/total": 5525312, "tokens/train_per_sec_per_gpu": 30.02, "tokens/trainable": 79059} +{"epoch": 0.6796875, "grad_norm": 0.5261490345001221, "learning_rate": 8.100214524900103e-05, "loss": 0.008662862703204155, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.25, "memory/max_allocated (GiB)": 36.25, "ppl": 1.0087, "step": 174, "tokens/total": 5556880, "tokens/train_per_sec_per_gpu": 24.62, "tokens/trainable": 79498} +{"epoch": 0.68359375, "grad_norm": 0.11292761564254761, "learning_rate": 8.076468193435301e-05, "loss": 0.0027105510234832764, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00271, "step": 175, "tokens/total": 5588976, "tokens/train_per_sec_per_gpu": 25.23, "tokens/trainable": 79922} +{"epoch": 0.6875, "grad_norm": 0.33315548300743103, "learning_rate": 8.052614644612253e-05, "loss": 0.0030926030594855547, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.11, "memory/max_allocated (GiB)": 36.11, "ppl": 1.0031, "step": 176, "tokens/total": 5618992, "tokens/train_per_sec_per_gpu": 29.51, "tokens/trainable": 80383} +{"epoch": 0.69140625, "grad_norm": 0.3136921226978302, "learning_rate": 8.028654871074489e-05, "loss": 0.005057850852608681, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.00507, "step": 177, "tokens/total": 5651072, "tokens/train_per_sec_per_gpu": 27.48, "tokens/trainable": 80824} +{"epoch": 0.6953125, "grad_norm": 1.0526509284973145, "learning_rate": 8.004589869885986e-05, "loss": 0.010707498528063297, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.01077, "step": 178, "tokens/total": 5683296, "tokens/train_per_sec_per_gpu": 26.58, "tokens/trainable": 81243} +{"epoch": 0.69921875, "grad_norm": 0.1329648196697235, "learning_rate": 7.980420642489674e-05, "loss": 0.0017975447699427605, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.0018, "step": 179, "tokens/total": 5715520, "tokens/train_per_sec_per_gpu": 28.44, "tokens/trainable": 81716} +{"epoch": 0.703125, "grad_norm": 1.0749201774597168, "learning_rate": 7.95614819466576e-05, "loss": 0.025617048144340515, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.02595, "step": 180, "tokens/total": 5747488, "tokens/train_per_sec_per_gpu": 29.65, "tokens/trainable": 82181} +{"epoch": 0.70703125, "grad_norm": 1.0446465015411377, "learning_rate": 7.931773536489872e-05, "loss": 0.030236585065722466, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 35.9, "memory/max_allocated (GiB)": 35.9, "ppl": 1.0307, "step": 181, "tokens/total": 5777264, "tokens/train_per_sec_per_gpu": 28.89, "tokens/trainable": 82626} +{"epoch": 0.7109375, "grad_norm": 0.22853964567184448, "learning_rate": 7.907297682291035e-05, "loss": 0.0038214433006942272, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00383, "step": 182, "tokens/total": 5809264, "tokens/train_per_sec_per_gpu": 29.35, "tokens/trainable": 83089} +{"epoch": 0.71484375, "grad_norm": 0.057821910828351974, "learning_rate": 7.882721650609442e-05, "loss": 0.0004995397757738829, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.0005, "step": 183, "tokens/total": 5841488, "tokens/train_per_sec_per_gpu": 29.34, "tokens/trainable": 83590} +{"epoch": 0.71875, "grad_norm": 1.3264167308807373, "learning_rate": 7.85804646415409e-05, "loss": 0.013232000172138214, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.59, "memory/max_allocated (GiB)": 36.59, "ppl": 1.01332, "step": 184, "tokens/total": 5873696, "tokens/train_per_sec_per_gpu": 29.33, "tokens/trainable": 84046} +{"epoch": 0.72265625, "grad_norm": 0.6027824878692627, "learning_rate": 7.833273149760207e-05, "loss": 0.012215660884976387, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.29, "memory/max_allocated (GiB)": 36.29, "ppl": 1.01229, "step": 185, "tokens/total": 5905408, "tokens/train_per_sec_per_gpu": 28.25, "tokens/trainable": 84517} +{"epoch": 0.7265625, "grad_norm": 0.5848041772842407, "learning_rate": 7.808402738346527e-05, "loss": 0.024627480655908585, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.02493, "step": 186, "tokens/total": 5937392, "tokens/train_per_sec_per_gpu": 28.42, "tokens/trainable": 84974} +{"epoch": 0.73046875, "grad_norm": 0.6823275089263916, "learning_rate": 7.783436264872382e-05, "loss": 0.0028813849203288555, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00289, "step": 187, "tokens/total": 5969344, "tokens/train_per_sec_per_gpu": 29.28, "tokens/trainable": 85460} +{"epoch": 0.734375, "grad_norm": 0.41855958104133606, "learning_rate": 7.758374768294647e-05, "loss": 0.009796380065381527, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.00984, "step": 188, "tokens/total": 6001296, "tokens/train_per_sec_per_gpu": 30.02, "tokens/trainable": 85939} +{"epoch": 0.73828125, "grad_norm": 0.6849080324172974, "learning_rate": 7.733219291524489e-05, "loss": 0.004100777208805084, "memory/device_reserved (GiB)": 40.04, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.00411, "step": 189, "tokens/total": 6031216, "tokens/train_per_sec_per_gpu": 31.66, "tokens/trainable": 86377} +{"epoch": 0.7421875, "grad_norm": 0.15838374197483063, "learning_rate": 7.707970881383977e-05, "loss": 0.003986812196671963, "memory/device_reserved (GiB)": 40.04, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.00399, "step": 190, "tokens/total": 6063296, "tokens/train_per_sec_per_gpu": 28.2, "tokens/trainable": 86834} +{"epoch": 0.74609375, "grad_norm": 0.3117486536502838, "learning_rate": 7.682630588562518e-05, "loss": 0.009132719598710537, "memory/device_reserved (GiB)": 40.04, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.00917, "step": 191, "tokens/total": 6095216, "tokens/train_per_sec_per_gpu": 26.65, "tokens/trainable": 87265} +{"epoch": 0.75, "grad_norm": 0.7837114334106445, "learning_rate": 7.657199467573129e-05, "loss": 0.007239446043968201, "memory/device_reserved (GiB)": 40.04, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00727, "step": 192, "tokens/total": 6127488, "tokens/train_per_sec_per_gpu": 28.57, "tokens/trainable": 87747} +{"epoch": 0.75390625, "grad_norm": 0.21777178347110748, "learning_rate": 7.631678576708561e-05, "loss": 0.002932134782895446, "memory/device_reserved (GiB)": 40.04, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.00294, "step": 193, "tokens/total": 6159488, "tokens/train_per_sec_per_gpu": 30.37, "tokens/trainable": 88239} +{"epoch": 0.7578125, "grad_norm": 0.19159242510795593, "learning_rate": 7.606068977997255e-05, "loss": 0.003356864908710122, "memory/device_reserved (GiB)": 38.19, "memory/max_active (GiB)": 36.41, "memory/max_allocated (GiB)": 36.41, "ppl": 1.00336, "step": 194, "tokens/total": 6191296, "tokens/train_per_sec_per_gpu": 30.33, "tokens/trainable": 88718} +{"epoch": 0.76171875, "grad_norm": 0.3977915346622467, "learning_rate": 7.580371737159148e-05, "loss": 0.008549144491553307, "memory/device_reserved (GiB)": 38.19, "memory/max_active (GiB)": 36.41, "memory/max_allocated (GiB)": 36.41, "ppl": 1.00859, "step": 195, "tokens/total": 6223120, "tokens/train_per_sec_per_gpu": 26.18, "tokens/trainable": 89129} +{"epoch": 0.765625, "grad_norm": 0.3875303566455841, "learning_rate": 7.554587923561324e-05, "loss": 0.006131038535386324, "memory/device_reserved (GiB)": 38.2, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.00615, "step": 196, "tokens/total": 6254880, "tokens/train_per_sec_per_gpu": 27.12, "tokens/trainable": 89567} +{"epoch": 0.76953125, "grad_norm": 3.4133481979370117, "learning_rate": 7.528718610173511e-05, "loss": 0.01836150512099266, "memory/device_reserved (GiB)": 38.2, "memory/max_active (GiB)": 36.41, "memory/max_allocated (GiB)": 36.41, "ppl": 1.01853, "step": 197, "tokens/total": 6286960, "tokens/train_per_sec_per_gpu": 25.23, "tokens/trainable": 89988} +{"epoch": 0.7734375, "grad_norm": 0.2509997487068176, "learning_rate": 7.502764873523431e-05, "loss": 0.003821226069703698, "memory/device_reserved (GiB)": 38.2, "memory/max_active (GiB)": 36.41, "memory/max_allocated (GiB)": 36.41, "ppl": 1.00383, "step": 198, "tokens/total": 6318880, "tokens/train_per_sec_per_gpu": 30.29, "tokens/trainable": 90434} +{"epoch": 0.77734375, "grad_norm": 0.9550195932388306, "learning_rate": 7.476727793652011e-05, "loss": 0.006872436031699181, "memory/device_reserved (GiB)": 38.76, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.0069, "step": 199, "tokens/total": 6350928, "tokens/train_per_sec_per_gpu": 28.38, "tokens/trainable": 90913} +{"epoch": 0.78125, "grad_norm": 0.411724716424942, "learning_rate": 7.450608454068415e-05, "loss": 0.014243390411138535, "memory/device_reserved (GiB)": 38.76, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.01435, "step": 200, "tokens/total": 6382976, "tokens/train_per_sec_per_gpu": 28.29, "tokens/trainable": 91355} +{"epoch": 0.78515625, "grad_norm": 0.34264692664146423, "learning_rate": 7.424407941704987e-05, "loss": 0.005309835076332092, "memory/device_reserved (GiB)": 38.76, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.00532, "step": 201, "tokens/total": 6415120, "tokens/train_per_sec_per_gpu": 31.41, "tokens/trainable": 91856} +{"epoch": 0.7890625, "grad_norm": 0.41955795884132385, "learning_rate": 7.398127346871986e-05, "loss": 0.009196512401103973, "memory/device_reserved (GiB)": 38.76, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00924, "step": 202, "tokens/total": 6447216, "tokens/train_per_sec_per_gpu": 28.32, "tokens/trainable": 92311} +{"epoch": 0.79296875, "grad_norm": 0.03107454814016819, "learning_rate": 7.371767763212238e-05, "loss": 0.0004908143309876323, "memory/device_reserved (GiB)": 38.78, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.00049, "step": 203, "tokens/total": 6479280, "tokens/train_per_sec_per_gpu": 29.1, "tokens/trainable": 92764} +{"epoch": 0.796875, "grad_norm": 0.39578482508659363, "learning_rate": 7.345330287655617e-05, "loss": 0.006089594680815935, "memory/device_reserved (GiB)": 38.78, "memory/max_active (GiB)": 36.23, "memory/max_allocated (GiB)": 36.23, "ppl": 1.00611, "step": 204, "tokens/total": 6511040, "tokens/train_per_sec_per_gpu": 27.48, "tokens/trainable": 93227} +{"epoch": 0.80078125, "grad_norm": 0.6444572806358337, "learning_rate": 7.31881602037339e-05, "loss": 0.010252210311591625, "memory/device_reserved (GiB)": 38.78, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.0103, "step": 205, "tokens/total": 6543120, "tokens/train_per_sec_per_gpu": 27.39, "tokens/trainable": 93675} +{"epoch": 0.8046875, "grad_norm": 0.3283638060092926, "learning_rate": 7.29222606473245e-05, "loss": 0.00761406822130084, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.65, "memory/max_allocated (GiB)": 36.65, "ppl": 1.00764, "step": 206, "tokens/total": 6575520, "tokens/train_per_sec_per_gpu": 27.02, "tokens/trainable": 94120} +{"epoch": 0.80859375, "grad_norm": 0.09952464699745178, "learning_rate": 7.265561527249383e-05, "loss": 0.001628706930205226, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.39, "memory/max_allocated (GiB)": 36.39, "ppl": 1.00163, "step": 207, "tokens/total": 6607408, "tokens/train_per_sec_per_gpu": 29.43, "tokens/trainable": 94557} +{"epoch": 0.8125, "grad_norm": 0.31704938411712646, "learning_rate": 7.238823517544436e-05, "loss": 0.005896817892789841, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00591, "step": 208, "tokens/total": 6639488, "tokens/train_per_sec_per_gpu": 33.89, "tokens/trainable": 95035} +{"epoch": 0.81640625, "grad_norm": 0.0999862551689148, "learning_rate": 7.212013148295333e-05, "loss": 0.001754446537233889, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.00176, "step": 209, "tokens/total": 6671520, "tokens/train_per_sec_per_gpu": 31.08, "tokens/trainable": 95517} +{"epoch": 0.8203125, "grad_norm": 0.1405109018087387, "learning_rate": 7.185131535190975e-05, "loss": 0.0017940793186426163, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.0018, "step": 210, "tokens/total": 6703536, "tokens/train_per_sec_per_gpu": 30.3, "tokens/trainable": 96026} +{"epoch": 0.82421875, "grad_norm": 0.09185238927602768, "learning_rate": 7.158179796885005e-05, "loss": 0.0015102234901860356, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.62, "memory/max_allocated (GiB)": 36.62, "ppl": 1.00151, "step": 211, "tokens/total": 6735616, "tokens/train_per_sec_per_gpu": 29.45, "tokens/trainable": 96477} +{"epoch": 0.828125, "grad_norm": 1.0464473962783813, "learning_rate": 7.131159054949273e-05, "loss": 0.016731251031160355, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.01687, "step": 212, "tokens/total": 6767584, "tokens/train_per_sec_per_gpu": 27.41, "tokens/trainable": 96951} +{"epoch": 0.83203125, "grad_norm": 0.3517152965068817, "learning_rate": 7.104070433827139e-05, "loss": 0.007984626106917858, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.00802, "step": 213, "tokens/total": 6799488, "tokens/train_per_sec_per_gpu": 25.83, "tokens/trainable": 97350} +{"epoch": 0.8359375, "grad_norm": 0.28192469477653503, "learning_rate": 7.076915060786705e-05, "loss": 0.00557567086070776, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00559, "step": 214, "tokens/total": 6831552, "tokens/train_per_sec_per_gpu": 24.66, "tokens/trainable": 97792} +{"epoch": 0.83984375, "grad_norm": 0.46135279536247253, "learning_rate": 7.049694065873882e-05, "loss": 0.0170141588896513, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.01716, "step": 215, "tokens/total": 6863488, "tokens/train_per_sec_per_gpu": 30.38, "tokens/trainable": 98257} +{"epoch": 0.84375, "grad_norm": 0.4277898073196411, "learning_rate": 7.022408581865382e-05, "loss": 0.009399959817528725, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.71, "memory/max_allocated (GiB)": 36.71, "ppl": 1.00944, "step": 216, "tokens/total": 6895888, "tokens/train_per_sec_per_gpu": 27.28, "tokens/trainable": 98731} +{"epoch": 0.84765625, "grad_norm": 0.34988704323768616, "learning_rate": 6.99505974422157e-05, "loss": 0.002909669652581215, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00291, "step": 217, "tokens/total": 6928000, "tokens/train_per_sec_per_gpu": 30.25, "tokens/trainable": 99212} +{"epoch": 0.8515625, "grad_norm": 0.4028005599975586, "learning_rate": 6.967648691039213e-05, "loss": 0.0058855339884757996, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.0059, "step": 218, "tokens/total": 6960096, "tokens/train_per_sec_per_gpu": 27.35, "tokens/trainable": 99654} +{"epoch": 0.85546875, "grad_norm": 0.3000575304031372, "learning_rate": 6.940176563004123e-05, "loss": 0.0025230362080037594, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.64, "memory/max_allocated (GiB)": 36.64, "ppl": 1.00253, "step": 219, "tokens/total": 6992512, "tokens/train_per_sec_per_gpu": 27.46, "tokens/trainable": 100086} +{"epoch": 0.859375, "grad_norm": 0.11351172626018524, "learning_rate": 6.912644503343682e-05, "loss": 0.0021392193157225847, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.62, "memory/max_allocated (GiB)": 36.62, "ppl": 1.00214, "step": 220, "tokens/total": 7024688, "tokens/train_per_sec_per_gpu": 27.56, "tokens/trainable": 100524} +{"epoch": 0.86328125, "grad_norm": 0.19512486457824707, "learning_rate": 6.885053657779273e-05, "loss": 0.00650001410394907, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00652, "step": 221, "tokens/total": 7056928, "tokens/train_per_sec_per_gpu": 28.05, "tokens/trainable": 100996} +{"epoch": 0.8671875, "grad_norm": 0.1998104453086853, "learning_rate": 6.857405174478604e-05, "loss": 0.0056140003725886345, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.00563, "step": 222, "tokens/total": 7088944, "tokens/train_per_sec_per_gpu": 26.49, "tokens/trainable": 101449} +{"epoch": 0.87109375, "grad_norm": 0.40908998250961304, "learning_rate": 6.82970020400792e-05, "loss": 0.01286369375884533, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.01295, "step": 223, "tokens/total": 7121104, "tokens/train_per_sec_per_gpu": 27.19, "tokens/trainable": 101905} +{"epoch": 0.875, "grad_norm": 0.3672473132610321, "learning_rate": 6.801939899284132e-05, "loss": 0.010677668265998363, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.01073, "step": 224, "tokens/total": 7153040, "tokens/train_per_sec_per_gpu": 30.54, "tokens/trainable": 102411} +{"epoch": 0.87890625, "grad_norm": 0.4146183431148529, "learning_rate": 6.774125415526827e-05, "loss": 0.0038945709820836782, "memory/device_reserved (GiB)": 39.19, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.0039, "step": 225, "tokens/total": 7184912, "tokens/train_per_sec_per_gpu": 28.44, "tokens/trainable": 102882} +{"epoch": 0.8828125, "grad_norm": 0.10585323721170425, "learning_rate": 6.746257910210214e-05, "loss": 0.00041726604104042053, "memory/device_reserved (GiB)": 38.33, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.00042, "step": 226, "tokens/total": 7216992, "tokens/train_per_sec_per_gpu": 28.72, "tokens/trainable": 103332} +{"epoch": 0.88671875, "grad_norm": 0.36600011587142944, "learning_rate": 6.718338543014937e-05, "loss": 0.01526213251054287, "memory/device_reserved (GiB)": 38.33, "memory/max_active (GiB)": 36.59, "memory/max_allocated (GiB)": 36.59, "ppl": 1.01538, "step": 227, "tokens/total": 7249120, "tokens/train_per_sec_per_gpu": 29.42, "tokens/trainable": 103790} +{"epoch": 0.890625, "grad_norm": 0.16057318449020386, "learning_rate": 6.69036847577983e-05, "loss": 0.004024423658847809, "memory/device_reserved (GiB)": 38.37, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00403, "step": 228, "tokens/total": 7281072, "tokens/train_per_sec_per_gpu": 28.55, "tokens/trainable": 104246} +{"epoch": 0.89453125, "grad_norm": 0.318798303604126, "learning_rate": 6.662348872453553e-05, "loss": 0.006663155741989613, "memory/device_reserved (GiB)": 38.37, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00669, "step": 229, "tokens/total": 7313120, "tokens/train_per_sec_per_gpu": 31.3, "tokens/trainable": 104707} +{"epoch": 0.8984375, "grad_norm": 0.10858234763145447, "learning_rate": 6.63428089904618e-05, "loss": 0.0020512123592197895, "memory/device_reserved (GiB)": 38.37, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00205, "step": 230, "tokens/total": 7345184, "tokens/train_per_sec_per_gpu": 30.81, "tokens/trainable": 105167} +{"epoch": 0.90234375, "grad_norm": 0.1044008731842041, "learning_rate": 6.60616572358065e-05, "loss": 0.0010264476295560598, "memory/device_reserved (GiB)": 38.37, "memory/max_active (GiB)": 36.6, "memory/max_allocated (GiB)": 36.6, "ppl": 1.00103, "step": 231, "tokens/total": 7377264, "tokens/train_per_sec_per_gpu": 27.52, "tokens/trainable": 105576} +{"epoch": 0.90625, "grad_norm": 0.14836947619915009, "learning_rate": 6.578004516044172e-05, "loss": 0.0013845522189512849, "memory/device_reserved (GiB)": 38.37, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00139, "step": 232, "tokens/total": 7409424, "tokens/train_per_sec_per_gpu": 30.32, "tokens/trainable": 106043} +{"epoch": 0.91015625, "grad_norm": 0.022752461954951286, "learning_rate": 6.549798448339548e-05, "loss": 0.0003245974367018789, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.00032, "step": 233, "tokens/total": 7441456, "tokens/train_per_sec_per_gpu": 29.1, "tokens/trainable": 106506} +{"epoch": 0.9140625, "grad_norm": 0.22981758415699005, "learning_rate": 6.521548694236384e-05, "loss": 0.002587020630016923, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.35, "memory/max_allocated (GiB)": 36.35, "ppl": 1.00259, "step": 234, "tokens/total": 7473344, "tokens/train_per_sec_per_gpu": 28.38, "tokens/trainable": 106951} +{"epoch": 0.91796875, "grad_norm": 0.5147526264190674, "learning_rate": 6.493256429322259e-05, "loss": 0.01256126631051302, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.01264, "step": 235, "tokens/total": 7505040, "tokens/train_per_sec_per_gpu": 25.64, "tokens/trainable": 107382} +{"epoch": 0.921875, "grad_norm": 0.3785130977630615, "learning_rate": 6.464922830953799e-05, "loss": 0.010956985875964165, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.01102, "step": 236, "tokens/total": 7537120, "tokens/train_per_sec_per_gpu": 27.41, "tokens/trainable": 107814} +{"epoch": 0.92578125, "grad_norm": 0.6193199753761292, "learning_rate": 6.436549078207688e-05, "loss": 0.020655140280723572, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.02087, "step": 237, "tokens/total": 7569024, "tokens/train_per_sec_per_gpu": 29.36, "tokens/trainable": 108274} +{"epoch": 0.9296875, "grad_norm": 0.5084519982337952, "learning_rate": 6.408136351831592e-05, "loss": 0.011346567422151566, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.01141, "step": 238, "tokens/total": 7600992, "tokens/train_per_sec_per_gpu": 24.43, "tokens/trainable": 108714} +{"epoch": 0.93359375, "grad_norm": 0.04367341846227646, "learning_rate": 6.379685834195036e-05, "loss": 0.0004316020058467984, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.57, "memory/max_allocated (GiB)": 36.57, "ppl": 1.00043, "step": 239, "tokens/total": 7633280, "tokens/train_per_sec_per_gpu": 30.36, "tokens/trainable": 109220} +{"epoch": 0.9375, "grad_norm": 0.06729783862829208, "learning_rate": 6.351198709240186e-05, "loss": 0.000604260538239032, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.6, "memory/max_allocated (GiB)": 36.6, "ppl": 1.0006, "step": 240, "tokens/total": 7665408, "tokens/train_per_sec_per_gpu": 26.58, "tokens/trainable": 109672} +{"epoch": 0.94140625, "grad_norm": 0.3010976314544678, "learning_rate": 6.32267616243259e-05, "loss": 0.006106264889240265, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.00612, "step": 241, "tokens/total": 7697280, "tokens/train_per_sec_per_gpu": 29.42, "tokens/trainable": 110135} +{"epoch": 0.9453125, "grad_norm": 0.11426915228366852, "learning_rate": 6.294119380711849e-05, "loss": 0.0014383264351636171, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00144, "step": 242, "tokens/total": 7729440, "tokens/train_per_sec_per_gpu": 25.66, "tokens/trainable": 110563} +{"epoch": 0.94921875, "grad_norm": 1.3792941570281982, "learning_rate": 6.265529552442209e-05, "loss": 0.003547664964571595, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00355, "step": 243, "tokens/total": 7761584, "tokens/train_per_sec_per_gpu": 30.76, "tokens/trainable": 111049} +{"epoch": 0.953125, "grad_norm": 0.07595008611679077, "learning_rate": 6.236907867363127e-05, "loss": 0.0009103374904952943, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.00091, "step": 244, "tokens/total": 7793824, "tokens/train_per_sec_per_gpu": 28.23, "tokens/trainable": 111495} +{"epoch": 0.95703125, "grad_norm": 0.05425361171364784, "learning_rate": 6.208255516539749e-05, "loss": 0.0013249842450022697, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00133, "step": 245, "tokens/total": 7825872, "tokens/train_per_sec_per_gpu": 26.47, "tokens/trainable": 111968} +{"epoch": 0.9609375, "grad_norm": 0.17112742364406586, "learning_rate": 6.179573692313344e-05, "loss": 0.001098912674933672, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.0011, "step": 246, "tokens/total": 7857840, "tokens/train_per_sec_per_gpu": 30.09, "tokens/trainable": 112416} +{"epoch": 0.96484375, "grad_norm": 0.10754083096981049, "learning_rate": 6.150863588251694e-05, "loss": 0.0031517897732555866, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.44, "memory/max_allocated (GiB)": 36.44, "ppl": 1.00316, "step": 247, "tokens/total": 7890000, "tokens/train_per_sec_per_gpu": 28.44, "tokens/trainable": 112882} +{"epoch": 0.96875, "grad_norm": 0.2477901428937912, "learning_rate": 6.122126399099419e-05, "loss": 0.005755824968218803, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.00577, "step": 248, "tokens/total": 7921984, "tokens/train_per_sec_per_gpu": 33.31, "tokens/trainable": 113390} +{"epoch": 0.97265625, "grad_norm": 0.332253634929657, "learning_rate": 6.0933633207282615e-05, "loss": 0.013207933865487576, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.0133, "step": 249, "tokens/total": 7953824, "tokens/train_per_sec_per_gpu": 31.49, "tokens/trainable": 113904} +{"epoch": 0.9765625, "grad_norm": 0.18514207005500793, "learning_rate": 6.064575550087316e-05, "loss": 0.00523729994893074, "memory/device_reserved (GiB)": 38.86, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.00525, "step": 250, "tokens/total": 7985984, "tokens/train_per_sec_per_gpu": 28.41, "tokens/trainable": 114351} +{"epoch": 0.98046875, "grad_norm": 0.1852743923664093, "learning_rate": 6.0357642851532245e-05, "loss": 0.0031474479474127293, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.64, "memory/max_allocated (GiB)": 36.64, "ppl": 1.00315, "step": 251, "tokens/total": 8018256, "tokens/train_per_sec_per_gpu": 29.27, "tokens/trainable": 114842} +{"epoch": 0.984375, "grad_norm": 0.43020468950271606, "learning_rate": 6.0069307248803294e-05, "loss": 0.0053521618247032166, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.00537, "step": 252, "tokens/total": 8050528, "tokens/train_per_sec_per_gpu": 23.97, "tokens/trainable": 115263} +{"epoch": 0.98828125, "grad_norm": 0.048135045915842056, "learning_rate": 5.9780760691507635e-05, "loss": 0.0005759480991400778, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00058, "step": 253, "tokens/total": 8082496, "tokens/train_per_sec_per_gpu": 27.44, "tokens/trainable": 115703} +{"epoch": 0.9921875, "grad_norm": 0.028733640909194946, "learning_rate": 5.9492015187245334e-05, "loss": 0.00040354474913328886, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.0004, "step": 254, "tokens/total": 8114592, "tokens/train_per_sec_per_gpu": 27.55, "tokens/trainable": 116157} +{"epoch": 0.99609375, "grad_norm": 0.2808869481086731, "learning_rate": 5.920308275189541e-05, "loss": 0.006315852981060743, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00634, "step": 255, "tokens/total": 8146752, "tokens/train_per_sec_per_gpu": 26.44, "tokens/trainable": 116567} +{"epoch": 1.0, "grad_norm": 0.054996367543935776, "learning_rate": 5.8913975409115874e-05, "loss": 0.0009099108865484595, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.00091, "step": 256, "tokens/total": 8178768, "tokens/train_per_sec_per_gpu": 25.98, "tokens/trainable": 117030} +{"epoch": 1.00390625, "grad_norm": 0.18114623427391052, "learning_rate": 5.8624705189843395e-05, "loss": 0.002796629210934043, "memory/device_reserved (GiB)": 38.98, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.0028, "step": 257, "tokens/total": 8210800, "tokens/train_per_sec_per_gpu": 28.97, "tokens/trainable": 117517} +{"epoch": 1.0078125, "grad_norm": 0.1408185362815857, "learning_rate": 5.833528413179249e-05, "loss": 0.0013210453325882554, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00132, "step": 258, "tokens/total": 8242768, "tokens/train_per_sec_per_gpu": 32.01, "tokens/trainable": 118015} +{"epoch": 1.01171875, "grad_norm": 0.36587440967559814, "learning_rate": 5.80457242789548e-05, "loss": 0.002160525880753994, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.00216, "step": 259, "tokens/total": 8274784, "tokens/train_per_sec_per_gpu": 30.13, "tokens/trainable": 118520} +{"epoch": 1.015625, "grad_norm": 0.02628348581492901, "learning_rate": 5.77560376810977e-05, "loss": 0.0003391164937056601, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00034, "step": 260, "tokens/total": 8306720, "tokens/train_per_sec_per_gpu": 30.36, "tokens/trainable": 118992} +{"epoch": 1.01953125, "grad_norm": 0.014996240846812725, "learning_rate": 5.7466236393263005e-05, "loss": 0.0002124913444276899, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00021, "step": 261, "tokens/total": 8338784, "tokens/train_per_sec_per_gpu": 27.96, "tokens/trainable": 119438} +{"epoch": 1.0234375, "grad_norm": 0.0050656464882195, "learning_rate": 5.717633247526522e-05, "loss": 7.557802018709481e-05, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.35, "memory/max_allocated (GiB)": 36.35, "ppl": 1.00008, "step": 262, "tokens/total": 8370720, "tokens/train_per_sec_per_gpu": 26.9, "tokens/trainable": 119881} +{"epoch": 1.02734375, "grad_norm": 0.02459302730858326, "learning_rate": 5.688633799118971e-05, "loss": 0.00024908874183893204, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.00025, "step": 263, "tokens/total": 8402736, "tokens/train_per_sec_per_gpu": 23.2, "tokens/trainable": 120285} +{"epoch": 1.03125, "grad_norm": 0.017813201993703842, "learning_rate": 5.659626500889066e-05, "loss": 0.00021895825921092182, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.23, "memory/max_allocated (GiB)": 36.23, "ppl": 1.00022, "step": 264, "tokens/total": 8434352, "tokens/train_per_sec_per_gpu": 27.78, "tokens/trainable": 120755} +{"epoch": 1.03515625, "grad_norm": 0.0038785531651228666, "learning_rate": 5.6306125599488905e-05, "loss": 6.556476728292182e-05, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00007, "step": 265, "tokens/total": 8466208, "tokens/train_per_sec_per_gpu": 29.86, "tokens/trainable": 121194} +{"epoch": 1.0390625, "grad_norm": 0.18033447861671448, "learning_rate": 5.601593183686955e-05, "loss": 0.0016840343596413732, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.00169, "step": 266, "tokens/total": 8498304, "tokens/train_per_sec_per_gpu": 30.51, "tokens/trainable": 121670} +{"epoch": 1.04296875, "grad_norm": 0.22497256100177765, "learning_rate": 5.572569579717961e-05, "loss": 0.003026450052857399, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.59, "memory/max_allocated (GiB)": 36.59, "ppl": 1.00303, "step": 267, "tokens/total": 8530400, "tokens/train_per_sec_per_gpu": 28.26, "tokens/trainable": 122142} +{"epoch": 1.046875, "grad_norm": 0.008878038264811039, "learning_rate": 5.543542955832538e-05, "loss": 8.136438555084169e-05, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00008, "step": 268, "tokens/total": 8562608, "tokens/train_per_sec_per_gpu": 27.27, "tokens/trainable": 122585} +{"epoch": 1.05078125, "grad_norm": 0.19042901694774628, "learning_rate": 5.514514519946986e-05, "loss": 0.0020417363848537207, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.00204, "step": 269, "tokens/total": 8594832, "tokens/train_per_sec_per_gpu": 32.13, "tokens/trainable": 123091} +{"epoch": 1.0546875, "grad_norm": 0.003382593160495162, "learning_rate": 5.485485480053015e-05, "loss": 3.870048021781258e-05, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00004, "step": 270, "tokens/total": 8627120, "tokens/train_per_sec_per_gpu": 26.36, "tokens/trainable": 123505} +{"epoch": 1.05859375, "grad_norm": 0.2004430592060089, "learning_rate": 5.4564570441674645e-05, "loss": 0.002532299840822816, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.05, "memory/max_allocated (GiB)": 36.05, "ppl": 1.00254, "step": 271, "tokens/total": 8657056, "tokens/train_per_sec_per_gpu": 29.64, "tokens/trainable": 123953} +{"epoch": 1.0625, "grad_norm": 0.24857956171035767, "learning_rate": 5.42743042028204e-05, "loss": 0.004157304298132658, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.00417, "step": 272, "tokens/total": 8689104, "tokens/train_per_sec_per_gpu": 24.05, "tokens/trainable": 124400} +{"epoch": 1.06640625, "grad_norm": 0.05064955726265907, "learning_rate": 5.3984068163130464e-05, "loss": 0.0006531125982291996, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00065, "step": 273, "tokens/total": 8721136, "tokens/train_per_sec_per_gpu": 29.3, "tokens/trainable": 124870} +{"epoch": 1.0703125, "grad_norm": 0.12664659321308136, "learning_rate": 5.369387440051111e-05, "loss": 0.0018034171080216765, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00181, "step": 274, "tokens/total": 8753200, "tokens/train_per_sec_per_gpu": 26.52, "tokens/trainable": 125333} +{"epoch": 1.07421875, "grad_norm": 0.01593201234936714, "learning_rate": 5.340373499110935e-05, "loss": 0.00010935973114101216, "memory/device_reserved (GiB)": 38.84, "memory/max_active (GiB)": 36.41, "memory/max_allocated (GiB)": 36.41, "ppl": 1.00011, "step": 275, "tokens/total": 8785168, "tokens/train_per_sec_per_gpu": 30.45, "tokens/trainable": 125834} +{"epoch": 1.078125, "grad_norm": 0.0015056057600304484, "learning_rate": 5.3113662008810304e-05, "loss": 1.9860939573845826e-05, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.62, "memory/max_allocated (GiB)": 36.62, "ppl": 1.00002, "step": 276, "tokens/total": 8817520, "tokens/train_per_sec_per_gpu": 31.32, "tokens/trainable": 126316} +{"epoch": 1.08203125, "grad_norm": 0.004457728937268257, "learning_rate": 5.282366752473479e-05, "loss": 2.716448398132343e-05, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.00003, "step": 277, "tokens/total": 8849360, "tokens/train_per_sec_per_gpu": 29.63, "tokens/trainable": 126798} +{"epoch": 1.0859375, "grad_norm": 0.03812727704644203, "learning_rate": 5.2533763606737005e-05, "loss": 0.0003402164438739419, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.29, "memory/max_allocated (GiB)": 36.29, "ppl": 1.00034, "step": 278, "tokens/total": 8881152, "tokens/train_per_sec_per_gpu": 24.5, "tokens/trainable": 127223} +{"epoch": 1.08984375, "grad_norm": 0.011409996077418327, "learning_rate": 5.224396231890232e-05, "loss": 9.064963523996994e-05, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.25, "memory/max_allocated (GiB)": 36.25, "ppl": 1.00009, "step": 279, "tokens/total": 8912512, "tokens/train_per_sec_per_gpu": 29.78, "tokens/trainable": 127672} +{"epoch": 1.09375, "grad_norm": 0.037030577659606934, "learning_rate": 5.195427572104522e-05, "loss": 0.0003734154161065817, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.28, "memory/max_allocated (GiB)": 36.28, "ppl": 1.00037, "step": 280, "tokens/total": 8944416, "tokens/train_per_sec_per_gpu": 26.76, "tokens/trainable": 128116} +{"epoch": 1.09765625, "grad_norm": 0.0032268385402858257, "learning_rate": 5.166471586820751e-05, "loss": 2.529611811041832e-05, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.00003, "step": 281, "tokens/total": 8976272, "tokens/train_per_sec_per_gpu": 27.44, "tokens/trainable": 128589} +{"epoch": 1.1015625, "grad_norm": 0.20227248966693878, "learning_rate": 5.1375294810156615e-05, "loss": 0.0024647137615829706, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.00247, "step": 282, "tokens/total": 9008560, "tokens/train_per_sec_per_gpu": 27.84, "tokens/trainable": 129036} +{"epoch": 1.10546875, "grad_norm": 0.43708595633506775, "learning_rate": 5.1086024590884144e-05, "loss": 0.0028322634752839804, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.44, "memory/max_allocated (GiB)": 36.44, "ppl": 1.00284, "step": 283, "tokens/total": 9040496, "tokens/train_per_sec_per_gpu": 29.25, "tokens/trainable": 129485} +{"epoch": 1.109375, "grad_norm": 0.09467080235481262, "learning_rate": 5.079691724810461e-05, "loss": 0.001156509853899479, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00116, "step": 284, "tokens/total": 9072448, "tokens/train_per_sec_per_gpu": 24.69, "tokens/trainable": 129920} +{"epoch": 1.11328125, "grad_norm": 0.1778785139322281, "learning_rate": 5.0507984812754684e-05, "loss": 0.0016857109731063247, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.44, "memory/max_allocated (GiB)": 36.44, "ppl": 1.00169, "step": 285, "tokens/total": 9104384, "tokens/train_per_sec_per_gpu": 30.02, "tokens/trainable": 130417} +{"epoch": 1.1171875, "grad_norm": 0.0003376641252543777, "learning_rate": 5.021923930849237e-05, "loss": 5.597693871095544e-06, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00001, "step": 286, "tokens/total": 9136416, "tokens/train_per_sec_per_gpu": 29.99, "tokens/trainable": 130903} +{"epoch": 1.12109375, "grad_norm": 0.019782084971666336, "learning_rate": 4.99306927511967e-05, "loss": 0.00012461614096537232, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00012, "step": 287, "tokens/total": 9168640, "tokens/train_per_sec_per_gpu": 29.24, "tokens/trainable": 131343} +{"epoch": 1.125, "grad_norm": 0.9641065001487732, "learning_rate": 4.964235714846775e-05, "loss": 0.011515076272189617, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.01158, "step": 288, "tokens/total": 9200544, "tokens/train_per_sec_per_gpu": 27.04, "tokens/trainable": 131763} +{"epoch": 1.12890625, "grad_norm": 0.0010062052169814706, "learning_rate": 4.9354244499126866e-05, "loss": 9.489545846008696e-06, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.44, "memory/max_allocated (GiB)": 36.44, "ppl": 1.00001, "step": 289, "tokens/total": 9232624, "tokens/train_per_sec_per_gpu": 28.42, "tokens/trainable": 132229} +{"epoch": 1.1328125, "grad_norm": 0.002373398980125785, "learning_rate": 4.90663667927174e-05, "loss": 1.573777262819931e-05, "memory/device_reserved (GiB)": 38.7, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.00002, "step": 290, "tokens/total": 9264768, "tokens/train_per_sec_per_gpu": 29.5, "tokens/trainable": 132678} +{"epoch": 1.13671875, "grad_norm": 0.0014718567254021764, "learning_rate": 4.877873600900581e-05, "loss": 9.888429303828161e-06, "memory/device_reserved (GiB)": 38.7, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00001, "step": 291, "tokens/total": 9296944, "tokens/train_per_sec_per_gpu": 24.47, "tokens/trainable": 133136} +{"epoch": 1.140625, "grad_norm": 0.0018440725980326533, "learning_rate": 4.849136411748306e-05, "loss": 8.275845175376162e-06, "memory/device_reserved (GiB)": 38.7, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.00001, "step": 292, "tokens/total": 9328848, "tokens/train_per_sec_per_gpu": 29.62, "tokens/trainable": 133608} +{"epoch": 1.14453125, "grad_norm": 0.12918472290039062, "learning_rate": 4.8204263076866574e-05, "loss": 0.0009128568926826119, "memory/device_reserved (GiB)": 38.7, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00091, "step": 293, "tokens/total": 9360928, "tokens/train_per_sec_per_gpu": 33.48, "tokens/trainable": 134108} +{"epoch": 1.1484375, "grad_norm": 0.019195031374692917, "learning_rate": 4.791744483460251e-05, "loss": 9.439548011869192e-05, "memory/device_reserved (GiB)": 38.7, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.00009, "step": 294, "tokens/total": 9393040, "tokens/train_per_sec_per_gpu": 25.71, "tokens/trainable": 134541} +{"epoch": 1.15234375, "grad_norm": 0.1602204144001007, "learning_rate": 4.7630921326368736e-05, "loss": 0.0004640200058929622, "memory/device_reserved (GiB)": 38.7, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.00046, "step": 295, "tokens/total": 9425248, "tokens/train_per_sec_per_gpu": 24.5, "tokens/trainable": 134966} +{"epoch": 1.15625, "grad_norm": 0.5406288504600525, "learning_rate": 4.7344704475577916e-05, "loss": 0.0034592358861118555, "memory/device_reserved (GiB)": 38.7, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00347, "step": 296, "tokens/total": 9457264, "tokens/train_per_sec_per_gpu": 27.87, "tokens/trainable": 135402} +{"epoch": 1.16015625, "grad_norm": 0.004333238583058119, "learning_rate": 4.705880619288153e-05, "loss": 3.075868880841881e-05, "memory/device_reserved (GiB)": 38.7, "memory/max_active (GiB)": 36.34, "memory/max_allocated (GiB)": 36.34, "ppl": 1.00003, "step": 297, "tokens/total": 9489056, "tokens/train_per_sec_per_gpu": 24.87, "tokens/trainable": 135804} +{"epoch": 1.1640625, "grad_norm": 0.006247827783226967, "learning_rate": 4.677323837567412e-05, "loss": 2.383375249337405e-05, "memory/device_reserved (GiB)": 38.7, "memory/max_active (GiB)": 36.36, "memory/max_allocated (GiB)": 36.36, "ppl": 1.00002, "step": 298, "tokens/total": 9521024, "tokens/train_per_sec_per_gpu": 25.81, "tokens/trainable": 136231} +{"epoch": 1.16796875, "grad_norm": 0.0779612734913826, "learning_rate": 4.6488012907598146e-05, "loss": 0.0004679379053413868, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.00047, "step": 299, "tokens/total": 9553120, "tokens/train_per_sec_per_gpu": 28.37, "tokens/trainable": 136705} +{"epoch": 1.171875, "grad_norm": 0.01816786825656891, "learning_rate": 4.620314165804964e-05, "loss": 6.801338167861104e-05, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00007, "step": 300, "tokens/total": 9585040, "tokens/train_per_sec_per_gpu": 28.78, "tokens/trainable": 137178} +{"epoch": 1.17578125, "grad_norm": 0.9530329704284668, "learning_rate": 4.591863648168407e-05, "loss": 0.008843549527227879, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.41, "memory/max_allocated (GiB)": 36.41, "ppl": 1.00888, "step": 301, "tokens/total": 9617120, "tokens/train_per_sec_per_gpu": 28.57, "tokens/trainable": 137643} +{"epoch": 1.1796875, "grad_norm": 0.0371168851852417, "learning_rate": 4.5634509217923135e-05, "loss": 0.00014635953994002193, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00015, "step": 302, "tokens/total": 9649024, "tokens/train_per_sec_per_gpu": 26.74, "tokens/trainable": 138078} +{"epoch": 1.18359375, "grad_norm": 0.08589725196361542, "learning_rate": 4.535077169046201e-05, "loss": 0.00043714733328670263, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00044, "step": 303, "tokens/total": 9680832, "tokens/train_per_sec_per_gpu": 29.32, "tokens/trainable": 138515} +{"epoch": 1.1875, "grad_norm": 0.358940988779068, "learning_rate": 4.506743570677743e-05, "loss": 0.006240838672965765, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00626, "step": 304, "tokens/total": 9713040, "tokens/train_per_sec_per_gpu": 29.29, "tokens/trainable": 138948} +{"epoch": 1.19140625, "grad_norm": 0.12659867107868195, "learning_rate": 4.478451305763618e-05, "loss": 0.0008667556685395539, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00087, "step": 305, "tokens/total": 9744976, "tokens/train_per_sec_per_gpu": 31.63, "tokens/trainable": 139408} +{"epoch": 1.1953125, "grad_norm": 0.08289908617734909, "learning_rate": 4.450201551660454e-05, "loss": 0.0006384386215358973, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.02, "memory/max_allocated (GiB)": 36.02, "ppl": 1.00064, "step": 306, "tokens/total": 9774784, "tokens/train_per_sec_per_gpu": 27.76, "tokens/trainable": 139840} +{"epoch": 1.19921875, "grad_norm": 0.19195082783699036, "learning_rate": 4.4219954839558276e-05, "loss": 0.002342985011637211, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.00235, "step": 307, "tokens/total": 9806656, "tokens/train_per_sec_per_gpu": 27.6, "tokens/trainable": 140281} +{"epoch": 1.203125, "grad_norm": 0.07421132177114487, "learning_rate": 4.393834276419352e-05, "loss": 0.0007085398538038135, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00071, "step": 308, "tokens/total": 9838720, "tokens/train_per_sec_per_gpu": 30.88, "tokens/trainable": 140776} +{"epoch": 1.20703125, "grad_norm": 0.004596009384840727, "learning_rate": 4.36571910095382e-05, "loss": 2.8969257982680574e-05, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00003, "step": 309, "tokens/total": 9870784, "tokens/train_per_sec_per_gpu": 30.7, "tokens/trainable": 141228} +{"epoch": 1.2109375, "grad_norm": 0.04099060595035553, "learning_rate": 4.337651127546448e-05, "loss": 0.00030985785997472703, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.00031, "step": 310, "tokens/total": 9902784, "tokens/train_per_sec_per_gpu": 30.8, "tokens/trainable": 141679} +{"epoch": 1.21484375, "grad_norm": 0.18840794265270233, "learning_rate": 4.3096315242201736e-05, "loss": 0.0020186181645840406, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.39, "memory/max_allocated (GiB)": 36.39, "ppl": 1.00202, "step": 311, "tokens/total": 9934768, "tokens/train_per_sec_per_gpu": 26.72, "tokens/trainable": 142122} +{"epoch": 1.21875, "grad_norm": 0.5018841028213501, "learning_rate": 4.2816614569850635e-05, "loss": 0.006148474756628275, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.6, "memory/max_allocated (GiB)": 36.6, "ppl": 1.00617, "step": 312, "tokens/total": 9966960, "tokens/train_per_sec_per_gpu": 27.27, "tokens/trainable": 142561} +{"epoch": 1.22265625, "grad_norm": 0.1730002760887146, "learning_rate": 4.2537420897897864e-05, "loss": 0.003579454030841589, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.00359, "step": 313, "tokens/total": 9999104, "tokens/train_per_sec_per_gpu": 29.61, "tokens/trainable": 143046} +{"epoch": 1.2265625, "grad_norm": 0.6586350202560425, "learning_rate": 4.225874584473174e-05, "loss": 0.0015231478027999401, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00152, "step": 314, "tokens/total": 10031136, "tokens/train_per_sec_per_gpu": 27.32, "tokens/trainable": 143493} +{"epoch": 1.23046875, "grad_norm": 0.011105087585747242, "learning_rate": 4.19806010071587e-05, "loss": 0.00011838486534543335, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.00012, "step": 315, "tokens/total": 10063184, "tokens/train_per_sec_per_gpu": 29.1, "tokens/trainable": 143956} +{"epoch": 1.234375, "grad_norm": 0.2628834843635559, "learning_rate": 4.170299795992081e-05, "loss": 0.003402995876967907, "memory/device_reserved (GiB)": 38.82, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.00341, "step": 316, "tokens/total": 10095312, "tokens/train_per_sec_per_gpu": 28.29, "tokens/trainable": 144411} +{"epoch": 1.23828125, "grad_norm": 0.13821958005428314, "learning_rate": 4.142594825521398e-05, "loss": 0.0016696923412382603, "memory/device_reserved (GiB)": 38.94, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.00167, "step": 317, "tokens/total": 10127440, "tokens/train_per_sec_per_gpu": 32.01, "tokens/trainable": 144892} +{"epoch": 1.2421875, "grad_norm": 0.0023418583441525698, "learning_rate": 4.114946342220728e-05, "loss": 2.6753859856398776e-05, "memory/device_reserved (GiB)": 38.94, "memory/max_active (GiB)": 36.41, "memory/max_allocated (GiB)": 36.41, "ppl": 1.00003, "step": 318, "tokens/total": 10159184, "tokens/train_per_sec_per_gpu": 27.3, "tokens/trainable": 145339} +{"epoch": 1.24609375, "grad_norm": 0.0008433983894065022, "learning_rate": 4.087355496656321e-05, "loss": 1.602486736373976e-05, "memory/device_reserved (GiB)": 38.94, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.00002, "step": 319, "tokens/total": 10191216, "tokens/train_per_sec_per_gpu": 30.52, "tokens/trainable": 145811} +{"epoch": 1.25, "grad_norm": 0.12588869035243988, "learning_rate": 4.05982343699588e-05, "loss": 0.0011584729654714465, "memory/device_reserved (GiB)": 38.94, "memory/max_active (GiB)": 36.57, "memory/max_allocated (GiB)": 36.57, "ppl": 1.00116, "step": 320, "tokens/total": 10223408, "tokens/train_per_sec_per_gpu": 32.27, "tokens/trainable": 146314} +{"epoch": 1.25390625, "grad_norm": 0.028395529836416245, "learning_rate": 4.0323513089607876e-05, "loss": 0.00015757072833366692, "memory/device_reserved (GiB)": 38.94, "memory/max_active (GiB)": 36.41, "memory/max_allocated (GiB)": 36.41, "ppl": 1.00016, "step": 321, "tokens/total": 10255536, "tokens/train_per_sec_per_gpu": 26.68, "tokens/trainable": 146781} +{"epoch": 1.2578125, "grad_norm": 0.039599668234586716, "learning_rate": 4.004940255778431e-05, "loss": 0.000321136845741421, "memory/device_reserved (GiB)": 37.39, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00032, "step": 322, "tokens/total": 10287664, "tokens/train_per_sec_per_gpu": 29.28, "tokens/trainable": 147225} +{"epoch": 1.26171875, "grad_norm": 0.11097095906734467, "learning_rate": 3.977591418134619e-05, "loss": 0.0004549310833681375, "memory/device_reserved (GiB)": 37.77, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.00046, "step": 323, "tokens/total": 10320080, "tokens/train_per_sec_per_gpu": 29.14, "tokens/trainable": 147673} +{"epoch": 1.265625, "grad_norm": 0.19018623232841492, "learning_rate": 3.95030593412612e-05, "loss": 0.0028336881659924984, "memory/device_reserved (GiB)": 38.53, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.00284, "step": 324, "tokens/total": 10351984, "tokens/train_per_sec_per_gpu": 27.33, "tokens/trainable": 148103} +{"epoch": 1.26953125, "grad_norm": 0.4650349020957947, "learning_rate": 3.923084939213296e-05, "loss": 0.006978687364608049, "memory/device_reserved (GiB)": 38.53, "memory/max_active (GiB)": 36.57, "memory/max_allocated (GiB)": 36.57, "ppl": 1.007, "step": 325, "tokens/total": 10384256, "tokens/train_per_sec_per_gpu": 26.21, "tokens/trainable": 148535} +{"epoch": 1.2734375, "grad_norm": 0.14811205863952637, "learning_rate": 3.895929566172861e-05, "loss": 0.0014966176822781563, "memory/device_reserved (GiB)": 38.65, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.0015, "step": 326, "tokens/total": 10416208, "tokens/train_per_sec_per_gpu": 28.37, "tokens/trainable": 148999} +{"epoch": 1.27734375, "grad_norm": 0.11302439123392105, "learning_rate": 3.868840945050728e-05, "loss": 0.0011318891774863005, "memory/device_reserved (GiB)": 38.65, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.00113, "step": 327, "tokens/total": 10448224, "tokens/train_per_sec_per_gpu": 26.37, "tokens/trainable": 149431} +{"epoch": 1.28125, "grad_norm": 0.10916705429553986, "learning_rate": 3.841820203114995e-05, "loss": 0.0009605163359083235, "memory/device_reserved (GiB)": 38.65, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00096, "step": 328, "tokens/total": 10480448, "tokens/train_per_sec_per_gpu": 28.27, "tokens/trainable": 149886} +{"epoch": 1.28515625, "grad_norm": 0.0004018457839265466, "learning_rate": 3.814868464809027e-05, "loss": 7.040077434794512e-06, "memory/device_reserved (GiB)": 38.65, "memory/max_active (GiB)": 36.33, "memory/max_allocated (GiB)": 36.33, "ppl": 1.00001, "step": 329, "tokens/total": 10512096, "tokens/train_per_sec_per_gpu": 25.92, "tokens/trainable": 150288} +{"epoch": 1.2890625, "grad_norm": 0.03717898577451706, "learning_rate": 3.787986851704667e-05, "loss": 0.0001680658315308392, "memory/device_reserved (GiB)": 38.65, "memory/max_active (GiB)": 36.59, "memory/max_allocated (GiB)": 36.59, "ppl": 1.00017, "step": 330, "tokens/total": 10544000, "tokens/train_per_sec_per_gpu": 28.52, "tokens/trainable": 150733} +{"epoch": 1.29296875, "grad_norm": 0.21975846588611603, "learning_rate": 3.7611764824555654e-05, "loss": 0.00015534088015556335, "memory/device_reserved (GiB)": 38.65, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00016, "step": 331, "tokens/total": 10576048, "tokens/train_per_sec_per_gpu": 30.54, "tokens/trainable": 151207} +{"epoch": 1.296875, "grad_norm": 0.10715529322624207, "learning_rate": 3.734438472750619e-05, "loss": 0.0012046336196362972, "memory/device_reserved (GiB)": 38.65, "memory/max_active (GiB)": 36.44, "memory/max_allocated (GiB)": 36.44, "ppl": 1.00121, "step": 332, "tokens/total": 10608048, "tokens/train_per_sec_per_gpu": 31.08, "tokens/trainable": 151694} +{"epoch": 1.30078125, "grad_norm": 0.13602837920188904, "learning_rate": 3.707773935267552e-05, "loss": 0.00039335948531515896, "memory/device_reserved (GiB)": 38.65, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00039, "step": 333, "tokens/total": 10640176, "tokens/train_per_sec_per_gpu": 32.13, "tokens/trainable": 152188} +{"epoch": 1.3046875, "grad_norm": 0.002963080769404769, "learning_rate": 3.68118397962661e-05, "loss": 6.6289721871726215e-06, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00001, "step": 334, "tokens/total": 10672224, "tokens/train_per_sec_per_gpu": 25.55, "tokens/trainable": 152596} +{"epoch": 1.30859375, "grad_norm": 0.007236327510327101, "learning_rate": 3.654669712344384e-05, "loss": 5.341085125110112e-05, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00005, "step": 335, "tokens/total": 10704192, "tokens/train_per_sec_per_gpu": 30.44, "tokens/trainable": 153052} +{"epoch": 1.3125, "grad_norm": 0.0897810310125351, "learning_rate": 3.628232236787763e-05, "loss": 0.0008687702938914299, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.65, "memory/max_allocated (GiB)": 36.65, "ppl": 1.00087, "step": 336, "tokens/total": 10736272, "tokens/train_per_sec_per_gpu": 25.33, "tokens/trainable": 153491} +{"epoch": 1.31640625, "grad_norm": 0.011829929426312447, "learning_rate": 3.6018726531280144e-05, "loss": 0.00011201453162357211, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.64, "memory/max_allocated (GiB)": 36.64, "ppl": 1.00011, "step": 337, "tokens/total": 10768528, "tokens/train_per_sec_per_gpu": 33.63, "tokens/trainable": 153983} +{"epoch": 1.3203125, "grad_norm": 0.25218063592910767, "learning_rate": 3.575592058295017e-05, "loss": 0.0022679665125906467, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.00227, "step": 338, "tokens/total": 10800464, "tokens/train_per_sec_per_gpu": 30.6, "tokens/trainable": 154464} +{"epoch": 1.32421875, "grad_norm": 0.00034703267738223076, "learning_rate": 3.549391545931585e-05, "loss": 2.884747800635523e-06, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.57, "memory/max_allocated (GiB)": 36.57, "ppl": 1.0, "step": 339, "tokens/total": 10832512, "tokens/train_per_sec_per_gpu": 29.7, "tokens/trainable": 154924} +{"epoch": 1.328125, "grad_norm": 0.004890776239335537, "learning_rate": 3.5232722063479914e-05, "loss": 2.2697908207192086e-05, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00002, "step": 340, "tokens/total": 10864624, "tokens/train_per_sec_per_gpu": 25.46, "tokens/trainable": 155373} +{"epoch": 1.33203125, "grad_norm": 0.0013245025184005499, "learning_rate": 3.49723512647657e-05, "loss": 8.980642633105163e-06, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00001, "step": 341, "tokens/total": 10896816, "tokens/train_per_sec_per_gpu": 30.1, "tokens/trainable": 155873} +{"epoch": 1.3359375, "grad_norm": 0.03804129734635353, "learning_rate": 3.471281389826491e-05, "loss": 0.00021880728309042752, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00022, "step": 342, "tokens/total": 10928864, "tokens/train_per_sec_per_gpu": 25.31, "tokens/trainable": 156278} +{"epoch": 1.33984375, "grad_norm": 0.4880214333534241, "learning_rate": 3.4454120764386764e-05, "loss": 0.005397590808570385, "memory/device_reserved (GiB)": 38.75, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00541, "step": 343, "tokens/total": 10960816, "tokens/train_per_sec_per_gpu": 29.58, "tokens/trainable": 156733} +{"epoch": 1.34375, "grad_norm": 0.37780871987342834, "learning_rate": 3.4196282628408526e-05, "loss": 0.0029298821464180946, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.00293, "step": 344, "tokens/total": 10992912, "tokens/train_per_sec_per_gpu": 28.4, "tokens/trainable": 157203} +{"epoch": 1.34765625, "grad_norm": 0.0034261250402778387, "learning_rate": 3.3939310220027456e-05, "loss": 2.1256186300888658e-05, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.00002, "step": 345, "tokens/total": 11024832, "tokens/train_per_sec_per_gpu": 31.61, "tokens/trainable": 157707} +{"epoch": 1.3515625, "grad_norm": 0.0005507867899723351, "learning_rate": 3.3683214232914404e-05, "loss": 6.991718692006543e-06, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.65, "memory/max_allocated (GiB)": 36.65, "ppl": 1.00001, "step": 346, "tokens/total": 11056992, "tokens/train_per_sec_per_gpu": 29.27, "tokens/trainable": 158180} +{"epoch": 1.35546875, "grad_norm": 0.0016383324982598424, "learning_rate": 3.342800532426873e-05, "loss": 1.2630136552616023e-05, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.00001, "step": 347, "tokens/total": 11089264, "tokens/train_per_sec_per_gpu": 28.55, "tokens/trainable": 158646} +{"epoch": 1.359375, "grad_norm": 0.0020912184845656157, "learning_rate": 3.317369411437484e-05, "loss": 1.536883064545691e-05, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00002, "step": 348, "tokens/total": 11120976, "tokens/train_per_sec_per_gpu": 29.34, "tokens/trainable": 159115} +{"epoch": 1.36328125, "grad_norm": 0.0016828920925036073, "learning_rate": 3.292029118616024e-05, "loss": 1.334734679403482e-05, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.00001, "step": 349, "tokens/total": 11152976, "tokens/train_per_sec_per_gpu": 26.34, "tokens/trainable": 159538} +{"epoch": 1.3671875, "grad_norm": 0.02520265430212021, "learning_rate": 3.266780708475511e-05, "loss": 9.595962183084339e-05, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.0001, "step": 350, "tokens/total": 11185056, "tokens/train_per_sec_per_gpu": 27.61, "tokens/trainable": 160016} +{"epoch": 1.37109375, "grad_norm": 0.009836713783442974, "learning_rate": 3.241625231705354e-05, "loss": 7.521865336457267e-05, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.44, "memory/max_allocated (GiB)": 36.44, "ppl": 1.00008, "step": 351, "tokens/total": 11217088, "tokens/train_per_sec_per_gpu": 28.51, "tokens/trainable": 160475} +{"epoch": 1.375, "grad_norm": 0.0698060467839241, "learning_rate": 3.216563735127618e-05, "loss": 0.0003666624252218753, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.57, "memory/max_allocated (GiB)": 36.57, "ppl": 1.00037, "step": 352, "tokens/total": 11249088, "tokens/train_per_sec_per_gpu": 28.31, "tokens/trainable": 160952} +{"epoch": 1.37890625, "grad_norm": 0.0018428800394758582, "learning_rate": 3.191597261653475e-05, "loss": 1.3194729945098516e-05, "memory/device_reserved (GiB)": 38.77, "memory/max_active (GiB)": 36.59, "memory/max_allocated (GiB)": 36.59, "ppl": 1.00001, "step": 353, "tokens/total": 11281328, "tokens/train_per_sec_per_gpu": 27.4, "tokens/trainable": 161420} +{"epoch": 1.3828125, "grad_norm": 0.020005319267511368, "learning_rate": 3.166726850239794e-05, "loss": 0.00017299644241575152, "memory/device_reserved (GiB)": 38.47, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00017, "step": 354, "tokens/total": 11313200, "tokens/train_per_sec_per_gpu": 28.31, "tokens/trainable": 161868} +{"epoch": 1.38671875, "grad_norm": 0.00018166302470490336, "learning_rate": 3.141953535845912e-05, "loss": 4.138089934713207e-06, "memory/device_reserved (GiB)": 38.47, "memory/max_active (GiB)": 36.35, "memory/max_allocated (GiB)": 36.35, "ppl": 1.0, "step": 355, "tokens/total": 11345008, "tokens/train_per_sec_per_gpu": 29.74, "tokens/trainable": 162318} +{"epoch": 1.390625, "grad_norm": 0.009165152907371521, "learning_rate": 3.11727834939056e-05, "loss": 6.729022425133735e-05, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.64, "memory/max_allocated (GiB)": 36.64, "ppl": 1.00007, "step": 356, "tokens/total": 11377216, "tokens/train_per_sec_per_gpu": 26.26, "tokens/trainable": 162789} +{"epoch": 1.39453125, "grad_norm": 0.2891274690628052, "learning_rate": 3.092702317708967e-05, "loss": 0.005585570354014635, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.59, "memory/max_allocated (GiB)": 36.59, "ppl": 1.0056, "step": 357, "tokens/total": 11409536, "tokens/train_per_sec_per_gpu": 30.34, "tokens/trainable": 163254} +{"epoch": 1.3984375, "grad_norm": 0.0007154783816076815, "learning_rate": 3.0682264635101276e-05, "loss": 5.591478839050978e-06, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.00001, "step": 358, "tokens/total": 11441552, "tokens/train_per_sec_per_gpu": 28.74, "tokens/trainable": 163715} +{"epoch": 1.40234375, "grad_norm": 0.01731656678020954, "learning_rate": 3.0438518053342407e-05, "loss": 0.0001050045175361447, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.00011, "step": 359, "tokens/total": 11473376, "tokens/train_per_sec_per_gpu": 28.79, "tokens/trainable": 164197} +{"epoch": 1.40625, "grad_norm": 0.11494546383619308, "learning_rate": 3.0195793575103266e-05, "loss": 0.00048459687968716025, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00048, "step": 360, "tokens/total": 11505312, "tokens/train_per_sec_per_gpu": 26.54, "tokens/trainable": 164661} +{"epoch": 1.41015625, "grad_norm": 0.08215854316949844, "learning_rate": 2.9954101301140146e-05, "loss": 0.0002725277154240757, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.00027, "step": 361, "tokens/total": 11537536, "tokens/train_per_sec_per_gpu": 27.25, "tokens/trainable": 165111} +{"epoch": 1.4140625, "grad_norm": 0.024266909807920456, "learning_rate": 2.9713451289255123e-05, "loss": 0.0001677784021012485, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 35.94, "memory/max_allocated (GiB)": 35.94, "ppl": 1.00017, "step": 362, "tokens/total": 11567216, "tokens/train_per_sec_per_gpu": 25.3, "tokens/trainable": 165552} +{"epoch": 1.41796875, "grad_norm": 0.008205167017877102, "learning_rate": 2.9473853553877484e-05, "loss": 3.423610905883834e-05, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.00003, "step": 363, "tokens/total": 11599312, "tokens/train_per_sec_per_gpu": 29.2, "tokens/trainable": 166026} +{"epoch": 1.421875, "grad_norm": 0.4230109453201294, "learning_rate": 2.9235318065647e-05, "loss": 0.005893740337342024, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00591, "step": 364, "tokens/total": 11631344, "tokens/train_per_sec_per_gpu": 31.7, "tokens/trainable": 166486} +{"epoch": 1.42578125, "grad_norm": 0.1095249131321907, "learning_rate": 2.8997854750998964e-05, "loss": 0.0006186347454786301, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.00062, "step": 365, "tokens/total": 11663632, "tokens/train_per_sec_per_gpu": 30.17, "tokens/trainable": 166959} +{"epoch": 1.4296875, "grad_norm": 0.0014034683117642999, "learning_rate": 2.8761473491751258e-05, "loss": 1.2728614819934592e-05, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.27, "memory/max_allocated (GiB)": 36.27, "ppl": 1.00001, "step": 366, "tokens/total": 11695360, "tokens/train_per_sec_per_gpu": 28.19, "tokens/trainable": 167394} +{"epoch": 1.43359375, "grad_norm": 0.044087354093790054, "learning_rate": 2.8526184124692883e-05, "loss": 0.0001466910180170089, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.34, "memory/max_allocated (GiB)": 36.34, "ppl": 1.00015, "step": 367, "tokens/total": 11727184, "tokens/train_per_sec_per_gpu": 23.64, "tokens/trainable": 167832} +{"epoch": 1.4375, "grad_norm": 0.0005008528823964298, "learning_rate": 2.829199644117484e-05, "loss": 1.1319095392536838e-05, "memory/device_reserved (GiB)": 38.96, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.00001, "step": 368, "tokens/total": 11759312, "tokens/train_per_sec_per_gpu": 25.79, "tokens/trainable": 168242} +{"epoch": 1.44140625, "grad_norm": 0.02015153504908085, "learning_rate": 2.8058920186702553e-05, "loss": 0.00011158352572238073, "memory/device_reserved (GiB)": 39.17, "memory/max_active (GiB)": 36.6, "memory/max_allocated (GiB)": 36.6, "ppl": 1.00011, "step": 369, "tokens/total": 11791536, "tokens/train_per_sec_per_gpu": 29.42, "tokens/trainable": 168715} +{"epoch": 1.4453125, "grad_norm": 0.00031807494815438986, "learning_rate": 2.782696506053033e-05, "loss": 6.533119631058071e-06, "memory/device_reserved (GiB)": 39.17, "memory/max_active (GiB)": 36.59, "memory/max_allocated (GiB)": 36.59, "ppl": 1.00001, "step": 370, "tokens/total": 11823632, "tokens/train_per_sec_per_gpu": 26.68, "tokens/trainable": 169147} +{"epoch": 1.44921875, "grad_norm": 0.4510185122489929, "learning_rate": 2.7596140715257824e-05, "loss": 0.004431337118148804, "memory/device_reserved (GiB)": 39.17, "memory/max_active (GiB)": 36.38, "memory/max_allocated (GiB)": 36.38, "ppl": 1.00444, "step": 371, "tokens/total": 11855392, "tokens/train_per_sec_per_gpu": 28.61, "tokens/trainable": 169628} +{"epoch": 1.453125, "grad_norm": 0.0034806979820132256, "learning_rate": 2.7366456756428184e-05, "loss": 2.722550561884418e-05, "memory/device_reserved (GiB)": 39.17, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.00003, "step": 372, "tokens/total": 11887424, "tokens/train_per_sec_per_gpu": 29.49, "tokens/trainable": 170085} +{"epoch": 1.45703125, "grad_norm": 0.018526704981923103, "learning_rate": 2.7137922742128486e-05, "loss": 0.00013849925016984344, "memory/device_reserved (GiB)": 39.17, "memory/max_active (GiB)": 36.34, "memory/max_allocated (GiB)": 36.34, "ppl": 1.00014, "step": 373, "tokens/total": 11919280, "tokens/train_per_sec_per_gpu": 29.39, "tokens/trainable": 170584} +{"epoch": 1.4609375, "grad_norm": 0.010734346695244312, "learning_rate": 2.691054818259188e-05, "loss": 1.963317481568083e-05, "memory/device_reserved (GiB)": 39.17, "memory/max_active (GiB)": 36.6, "memory/max_allocated (GiB)": 36.6, "ppl": 1.00002, "step": 374, "tokens/total": 11951488, "tokens/train_per_sec_per_gpu": 28.62, "tokens/trainable": 171058} +{"epoch": 1.46484375, "grad_norm": 0.004906357266008854, "learning_rate": 2.6684342539801933e-05, "loss": 4.44888137280941e-05, "memory/device_reserved (GiB)": 39.17, "memory/max_active (GiB)": 36.59, "memory/max_allocated (GiB)": 36.59, "ppl": 1.00004, "step": 375, "tokens/total": 11983504, "tokens/train_per_sec_per_gpu": 28.26, "tokens/trainable": 171518} +{"epoch": 1.46875, "grad_norm": 0.0035627244506031275, "learning_rate": 2.645931522709877e-05, "loss": 1.879796946013812e-05, "memory/device_reserved (GiB)": 39.17, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00002, "step": 376, "tokens/total": 12015360, "tokens/train_per_sec_per_gpu": 33.47, "tokens/trainable": 172007} +{"epoch": 1.47265625, "grad_norm": 0.014949376694858074, "learning_rate": 2.6235475608787365e-05, "loss": 0.0001050513528753072, "memory/device_reserved (GiB)": 39.17, "memory/max_active (GiB)": 36.39, "memory/max_allocated (GiB)": 36.39, "ppl": 1.00011, "step": 377, "tokens/total": 12047296, "tokens/train_per_sec_per_gpu": 28.67, "tokens/trainable": 172468} +{"epoch": 1.4765625, "grad_norm": 0.024432353675365448, "learning_rate": 2.6012832999747916e-05, "loss": 0.00010428359382785857, "memory/device_reserved (GiB)": 39.17, "memory/max_active (GiB)": 36.62, "memory/max_allocated (GiB)": 36.62, "ppl": 1.0001, "step": 378, "tokens/total": 12079392, "tokens/train_per_sec_per_gpu": 27.91, "tokens/trainable": 172923} +{"epoch": 1.48046875, "grad_norm": 0.5905564427375793, "learning_rate": 2.579139666504821e-05, "loss": 0.019045401364564896, "memory/device_reserved (GiB)": 39.17, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.01923, "step": 379, "tokens/total": 12111616, "tokens/train_per_sec_per_gpu": 28.26, "tokens/trainable": 173370} +{"epoch": 1.484375, "grad_norm": 0.0027490397915244102, "learning_rate": 2.557117581955798e-05, "loss": 3.106705116806552e-05, "memory/device_reserved (GiB)": 39.17, "memory/max_active (GiB)": 36.35, "memory/max_allocated (GiB)": 36.35, "ppl": 1.00003, "step": 380, "tokens/total": 12143536, "tokens/train_per_sec_per_gpu": 28.66, "tokens/trainable": 173837} +{"epoch": 1.48828125, "grad_norm": 0.10220319032669067, "learning_rate": 2.5352179627565532e-05, "loss": 0.0007754361140541732, "memory/device_reserved (GiB)": 39.17, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.00078, "step": 381, "tokens/total": 12175792, "tokens/train_per_sec_per_gpu": 29.35, "tokens/trainable": 174294} +{"epoch": 1.4921875, "grad_norm": 0.06388501822948456, "learning_rate": 2.5134417202396277e-05, "loss": 2.506363671272993e-05, "memory/device_reserved (GiB)": 39.17, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00003, "step": 382, "tokens/total": 12207472, "tokens/train_per_sec_per_gpu": 25.85, "tokens/trainable": 174701} +{"epoch": 1.49609375, "grad_norm": 0.006029566749930382, "learning_rate": 2.491789760603361e-05, "loss": 5.214785414864309e-05, "memory/device_reserved (GiB)": 39.17, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.00005, "step": 383, "tokens/total": 12239520, "tokens/train_per_sec_per_gpu": 26.67, "tokens/trainable": 175131} +{"epoch": 1.5, "grad_norm": 0.015079713426530361, "learning_rate": 2.4702629848741764e-05, "loss": 0.00011988497863058001, "memory/device_reserved (GiB)": 39.17, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.00012, "step": 384, "tokens/total": 12271600, "tokens/train_per_sec_per_gpu": 28.23, "tokens/trainable": 175586} +{"epoch": 1.50390625, "grad_norm": 0.04362509027123451, "learning_rate": 2.4488622888690785e-05, "loss": 0.0002582473389338702, "memory/device_reserved (GiB)": 39.52, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00026, "step": 385, "tokens/total": 12303648, "tokens/train_per_sec_per_gpu": 28.44, "tokens/trainable": 176026} +{"epoch": 1.5078125, "grad_norm": 0.28308039903640747, "learning_rate": 2.427588563158384e-05, "loss": 0.0025924108922481537, "memory/device_reserved (GiB)": 39.52, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.0026, "step": 386, "tokens/total": 12335728, "tokens/train_per_sec_per_gpu": 28.2, "tokens/trainable": 176512} +{"epoch": 1.51171875, "grad_norm": 0.21685272455215454, "learning_rate": 2.406442693028651e-05, "loss": 0.002609315561130643, "memory/device_reserved (GiB)": 39.52, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00261, "step": 387, "tokens/total": 12367824, "tokens/train_per_sec_per_gpu": 25.55, "tokens/trainable": 176943} +{"epoch": 1.515625, "grad_norm": 0.0010271539213135839, "learning_rate": 2.3854255584458547e-05, "loss": 2.1533389372052625e-05, "memory/device_reserved (GiB)": 39.52, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.00002, "step": 388, "tokens/total": 12400192, "tokens/train_per_sec_per_gpu": 26.36, "tokens/trainable": 177370} +{"epoch": 1.51953125, "grad_norm": 0.06616228073835373, "learning_rate": 2.3645380340187508e-05, "loss": 0.0005350579158402979, "memory/device_reserved (GiB)": 39.52, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.00054, "step": 389, "tokens/total": 12432224, "tokens/train_per_sec_per_gpu": 29.11, "tokens/trainable": 177851} +{"epoch": 1.5234375, "grad_norm": 0.05946248397231102, "learning_rate": 2.3437809889624914e-05, "loss": 0.0005003588157705963, "memory/device_reserved (GiB)": 39.52, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.0005, "step": 390, "tokens/total": 12464048, "tokens/train_per_sec_per_gpu": 29.09, "tokens/trainable": 178310} +{"epoch": 1.52734375, "grad_norm": 0.017372427508234978, "learning_rate": 2.3231552870624487e-05, "loss": 0.0001355188578600064, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.00014, "step": 391, "tokens/total": 12493776, "tokens/train_per_sec_per_gpu": 30.96, "tokens/trainable": 178736} +{"epoch": 1.53125, "grad_norm": 0.12941808998584747, "learning_rate": 2.3026617866382657e-05, "loss": 0.003045230871066451, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00305, "step": 392, "tokens/total": 12526000, "tokens/train_per_sec_per_gpu": 30.37, "tokens/trainable": 179233} +{"epoch": 1.53515625, "grad_norm": 0.014790852554142475, "learning_rate": 2.2823013405081507e-05, "loss": 0.00013205324648879468, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00013, "step": 393, "tokens/total": 12557968, "tokens/train_per_sec_per_gpu": 32.27, "tokens/trainable": 179730} +{"epoch": 1.5390625, "grad_norm": 0.02042844519019127, "learning_rate": 2.2620747959533722e-05, "loss": 0.00022655579959973693, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.00023, "step": 394, "tokens/total": 12590096, "tokens/train_per_sec_per_gpu": 31.47, "tokens/trainable": 180227} +{"epoch": 1.54296875, "grad_norm": 0.005440943408757448, "learning_rate": 2.2419829946830123e-05, "loss": 4.6342807763721794e-05, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.59, "memory/max_allocated (GiB)": 36.59, "ppl": 1.00005, "step": 395, "tokens/total": 12622160, "tokens/train_per_sec_per_gpu": 27.58, "tokens/trainable": 180687} +{"epoch": 1.546875, "grad_norm": 0.004878366366028786, "learning_rate": 2.2220267727989325e-05, "loss": 5.4212974646361545e-05, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00005, "step": 396, "tokens/total": 12654336, "tokens/train_per_sec_per_gpu": 27.25, "tokens/trainable": 181141} +{"epoch": 1.55078125, "grad_norm": 0.20873922109603882, "learning_rate": 2.202206960760984e-05, "loss": 0.0026359721086919308, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00264, "step": 397, "tokens/total": 12686400, "tokens/train_per_sec_per_gpu": 30.13, "tokens/trainable": 181648} +{"epoch": 1.5546875, "grad_norm": 0.12369472533464432, "learning_rate": 2.182524383352446e-05, "loss": 0.0010878165485337377, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.29, "memory/max_allocated (GiB)": 36.29, "ppl": 1.00109, "step": 398, "tokens/total": 12718144, "tokens/train_per_sec_per_gpu": 28.08, "tokens/trainable": 182109} +{"epoch": 1.55859375, "grad_norm": 0.0010841770563274622, "learning_rate": 2.1629798596457056e-05, "loss": 1.8380069377599284e-05, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.00002, "step": 399, "tokens/total": 12750256, "tokens/train_per_sec_per_gpu": 26.26, "tokens/trainable": 182559} +{"epoch": 1.5625, "grad_norm": 0.0010006122756749392, "learning_rate": 2.1435742029681725e-05, "loss": 1.991226599784568e-05, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.57, "memory/max_allocated (GiB)": 36.57, "ppl": 1.00002, "step": 400, "tokens/total": 12782224, "tokens/train_per_sec_per_gpu": 27.59, "tokens/trainable": 182985} +{"epoch": 1.56640625, "grad_norm": 0.0011391225270926952, "learning_rate": 2.124308220868431e-05, "loss": 2.2474639990832657e-05, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.00002, "step": 401, "tokens/total": 12814384, "tokens/train_per_sec_per_gpu": 26.95, "tokens/trainable": 183440} +{"epoch": 1.5703125, "grad_norm": 0.023268507793545723, "learning_rate": 2.105182715082638e-05, "loss": 0.00032951770117506385, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.44, "memory/max_allocated (GiB)": 36.44, "ppl": 1.00033, "step": 402, "tokens/total": 12846240, "tokens/train_per_sec_per_gpu": 28.77, "tokens/trainable": 183917} +{"epoch": 1.57421875, "grad_norm": 0.11701280623674393, "learning_rate": 2.0861984815011552e-05, "loss": 0.001444364432245493, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.44, "memory/max_allocated (GiB)": 36.44, "ppl": 1.00145, "step": 403, "tokens/total": 12878240, "tokens/train_per_sec_per_gpu": 25.07, "tokens/trainable": 184348} +{"epoch": 1.578125, "grad_norm": 0.004240577574819326, "learning_rate": 2.0673563101354323e-05, "loss": 2.100174970109947e-05, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.33, "memory/max_allocated (GiB)": 36.33, "ppl": 1.00002, "step": 404, "tokens/total": 12910176, "tokens/train_per_sec_per_gpu": 26.38, "tokens/trainable": 184786} +{"epoch": 1.58203125, "grad_norm": 0.01977616548538208, "learning_rate": 2.0486569850851317e-05, "loss": 8.987118053482845e-05, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.00009, "step": 405, "tokens/total": 12942352, "tokens/train_per_sec_per_gpu": 28.17, "tokens/trainable": 185249} +{"epoch": 1.5859375, "grad_norm": 0.030095387250185013, "learning_rate": 2.0301012845054956e-05, "loss": 0.0002793738676700741, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00028, "step": 406, "tokens/total": 12974368, "tokens/train_per_sec_per_gpu": 26.3, "tokens/trainable": 185680} +{"epoch": 1.58984375, "grad_norm": 0.0017381443176418543, "learning_rate": 2.011689980574966e-05, "loss": 2.727654828049708e-05, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.00003, "step": 407, "tokens/total": 13006464, "tokens/train_per_sec_per_gpu": 26.25, "tokens/trainable": 186100} +{"epoch": 1.59375, "grad_norm": 0.09072195738554001, "learning_rate": 1.993423839463052e-05, "loss": 0.0006338813109323382, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00063, "step": 408, "tokens/total": 13038320, "tokens/train_per_sec_per_gpu": 33.21, "tokens/trainable": 186565} +{"epoch": 1.59765625, "grad_norm": 0.004482298158109188, "learning_rate": 1.975303621298445e-05, "loss": 5.0280330469831824e-05, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00005, "step": 409, "tokens/total": 13070416, "tokens/train_per_sec_per_gpu": 29.62, "tokens/trainable": 187017} +{"epoch": 1.6015625, "grad_norm": 0.03788210079073906, "learning_rate": 1.957330080137385e-05, "loss": 0.000300457701086998, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.0003, "step": 410, "tokens/total": 13102352, "tokens/train_per_sec_per_gpu": 27.69, "tokens/trainable": 187468} +{"epoch": 1.60546875, "grad_norm": 0.011773839592933655, "learning_rate": 1.9395039639322864e-05, "loss": 9.259363287128508e-05, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.34, "memory/max_allocated (GiB)": 36.34, "ppl": 1.00009, "step": 411, "tokens/total": 13134096, "tokens/train_per_sec_per_gpu": 31.06, "tokens/trainable": 187961} +{"epoch": 1.609375, "grad_norm": 0.0031615474727004766, "learning_rate": 1.9218260145006073e-05, "loss": 3.231812661397271e-05, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.00003, "step": 412, "tokens/total": 13166256, "tokens/train_per_sec_per_gpu": 29.62, "tokens/trainable": 188447} +{"epoch": 1.61328125, "grad_norm": 0.07778825610876083, "learning_rate": 1.904296967493982e-05, "loss": 0.0005980221321806312, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.0006, "step": 413, "tokens/total": 13198048, "tokens/train_per_sec_per_gpu": 29.98, "tokens/trainable": 188903} +{"epoch": 1.6171875, "grad_norm": 0.011902395635843277, "learning_rate": 1.8869175523676064e-05, "loss": 0.00013689698243979365, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.66, "memory/max_allocated (GiB)": 36.66, "ppl": 1.00014, "step": 414, "tokens/total": 13230304, "tokens/train_per_sec_per_gpu": 30.8, "tokens/trainable": 189398} +{"epoch": 1.62109375, "grad_norm": 0.08490622788667679, "learning_rate": 1.869688492349885e-05, "loss": 0.0008053273777477443, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00081, "step": 415, "tokens/total": 13262288, "tokens/train_per_sec_per_gpu": 28.62, "tokens/trainable": 189850} +{"epoch": 1.625, "grad_norm": 0.11783187091350555, "learning_rate": 1.85261050441233e-05, "loss": 0.0007362872711382806, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.00074, "step": 416, "tokens/total": 13294480, "tokens/train_per_sec_per_gpu": 27.25, "tokens/trainable": 190306} +{"epoch": 1.62890625, "grad_norm": 0.26603376865386963, "learning_rate": 1.8356842992397304e-05, "loss": 0.002245941199362278, "memory/device_reserved (GiB)": 39.53, "memory/max_active (GiB)": 36.6, "memory/max_allocated (GiB)": 36.6, "ppl": 1.00225, "step": 417, "tokens/total": 13326608, "tokens/train_per_sec_per_gpu": 28.38, "tokens/trainable": 190746} +{"epoch": 1.6328125, "grad_norm": 0.003691659774631262, "learning_rate": 1.8189105812005714e-05, "loss": 4.167412407696247e-05, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.36, "memory/max_allocated (GiB)": 36.36, "ppl": 1.00004, "step": 418, "tokens/total": 13358592, "tokens/train_per_sec_per_gpu": 26.1, "tokens/trainable": 191156} +{"epoch": 1.63671875, "grad_norm": 0.021083569154143333, "learning_rate": 1.802290048317732e-05, "loss": 0.00017740413022693247, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00018, "step": 419, "tokens/total": 13390592, "tokens/train_per_sec_per_gpu": 24.49, "tokens/trainable": 191573} +{"epoch": 1.640625, "grad_norm": 0.003719380358234048, "learning_rate": 1.785823392239424e-05, "loss": 4.420262484927662e-05, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.64, "memory/max_allocated (GiB)": 36.64, "ppl": 1.00004, "step": 420, "tokens/total": 13422672, "tokens/train_per_sec_per_gpu": 27.22, "tokens/trainable": 192014} +{"epoch": 1.64453125, "grad_norm": 0.044042862951755524, "learning_rate": 1.7695112982104225e-05, "loss": 0.0004268632619641721, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.00043, "step": 421, "tokens/total": 13454624, "tokens/train_per_sec_per_gpu": 27.25, "tokens/trainable": 192453} +{"epoch": 1.6484375, "grad_norm": 0.0021698337513953447, "learning_rate": 1.7533544450435433e-05, "loss": 2.580507134553045e-05, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.31, "memory/max_allocated (GiB)": 36.31, "ppl": 1.00003, "step": 422, "tokens/total": 13486208, "tokens/train_per_sec_per_gpu": 27.89, "tokens/trainable": 192886} +{"epoch": 1.65234375, "grad_norm": 0.019405441358685493, "learning_rate": 1.7373535050913946e-05, "loss": 0.00013853301061317325, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.71, "memory/max_allocated (GiB)": 36.71, "ppl": 1.00014, "step": 423, "tokens/total": 13518496, "tokens/train_per_sec_per_gpu": 25.15, "tokens/trainable": 193323} +{"epoch": 1.65625, "grad_norm": 0.005207414738833904, "learning_rate": 1.721509144218405e-05, "loss": 6.985871004872024e-05, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00007, "step": 424, "tokens/total": 13550768, "tokens/train_per_sec_per_gpu": 27.59, "tokens/trainable": 193753} +{"epoch": 1.66015625, "grad_norm": 0.01700798235833645, "learning_rate": 1.705822021773101e-05, "loss": 0.00024273117014672607, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00024, "step": 425, "tokens/total": 13582960, "tokens/train_per_sec_per_gpu": 30.03, "tokens/trainable": 194240} +{"epoch": 1.6640625, "grad_norm": 0.2864547073841095, "learning_rate": 1.69029279056068e-05, "loss": 0.004108238499611616, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.00412, "step": 426, "tokens/total": 13615056, "tokens/train_per_sec_per_gpu": 29.6, "tokens/trainable": 194680} +{"epoch": 1.66796875, "grad_norm": 0.012734112329781055, "learning_rate": 1.6749220968158415e-05, "loss": 9.277237404603511e-05, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.00009, "step": 427, "tokens/total": 13647088, "tokens/train_per_sec_per_gpu": 32.17, "tokens/trainable": 195186} +{"epoch": 1.671875, "grad_norm": 0.04110024869441986, "learning_rate": 1.659710580175893e-05, "loss": 0.00021035004465375096, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.00021, "step": 428, "tokens/total": 13679168, "tokens/train_per_sec_per_gpu": 30.52, "tokens/trainable": 195693} +{"epoch": 1.67578125, "grad_norm": 0.03119366057217121, "learning_rate": 1.644658873654133e-05, "loss": 0.00022089436242822558, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00022, "step": 429, "tokens/total": 13711120, "tokens/train_per_sec_per_gpu": 30.79, "tokens/trainable": 196130} +{"epoch": 1.6796875, "grad_norm": 0.0020074129570275545, "learning_rate": 1.629767603613508e-05, "loss": 1.550407614558935e-05, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.00002, "step": 430, "tokens/total": 13743088, "tokens/train_per_sec_per_gpu": 24.67, "tokens/trainable": 196530} +{"epoch": 1.68359375, "grad_norm": 0.040013547986745834, "learning_rate": 1.615037389740547e-05, "loss": 0.0006465734331868589, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.36, "memory/max_allocated (GiB)": 36.36, "ppl": 1.00065, "step": 431, "tokens/total": 13774912, "tokens/train_per_sec_per_gpu": 25.06, "tokens/trainable": 196974} +{"epoch": 1.6875, "grad_norm": 0.0517231747508049, "learning_rate": 1.600468845019576e-05, "loss": 0.00047026947140693665, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00047, "step": 432, "tokens/total": 13806800, "tokens/train_per_sec_per_gpu": 29.38, "tokens/trainable": 197454} +{"epoch": 1.69140625, "grad_norm": 0.0004246715398039669, "learning_rate": 1.5860625757072092e-05, "loss": 7.613366506120656e-06, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00001, "step": 433, "tokens/total": 13838704, "tokens/train_per_sec_per_gpu": 26.36, "tokens/trainable": 197902} +{"epoch": 1.6953125, "grad_norm": 0.0011862553656101227, "learning_rate": 1.571819181307116e-05, "loss": 1.5171106497291476e-05, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.34, "memory/max_allocated (GiB)": 36.34, "ppl": 1.00002, "step": 434, "tokens/total": 13870640, "tokens/train_per_sec_per_gpu": 29.63, "tokens/trainable": 198374} +{"epoch": 1.69921875, "grad_norm": 0.0019507558317855, "learning_rate": 1.557739254545075e-05, "loss": 1.7753134670783766e-05, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.00002, "step": 435, "tokens/total": 13902672, "tokens/train_per_sec_per_gpu": 31.27, "tokens/trainable": 198846} +{"epoch": 1.703125, "grad_norm": 0.014270462095737457, "learning_rate": 1.543823381344311e-05, "loss": 0.0001223236322402954, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.00012, "step": 436, "tokens/total": 13935008, "tokens/train_per_sec_per_gpu": 31.41, "tokens/trainable": 199320} +{"epoch": 1.70703125, "grad_norm": 0.020011236891150475, "learning_rate": 1.5300721408011114e-05, "loss": 0.00015811943740118295, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00016, "step": 437, "tokens/total": 13967024, "tokens/train_per_sec_per_gpu": 31.19, "tokens/trainable": 199777} +{"epoch": 1.7109375, "grad_norm": 0.0006321229157038033, "learning_rate": 1.5164861051607254e-05, "loss": 9.932198736350983e-06, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00001, "step": 438, "tokens/total": 13998688, "tokens/train_per_sec_per_gpu": 25.77, "tokens/trainable": 200194} +{"epoch": 1.71484375, "grad_norm": 0.002675980795174837, "learning_rate": 1.5030658397935521e-05, "loss": 2.0016837879666127e-05, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00002, "step": 439, "tokens/total": 14030992, "tokens/train_per_sec_per_gpu": 28.45, "tokens/trainable": 200663} +{"epoch": 1.71875, "grad_norm": 0.005228503607213497, "learning_rate": 1.4898119031716104e-05, "loss": 4.561378591461107e-05, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.64, "memory/max_allocated (GiB)": 36.64, "ppl": 1.00005, "step": 440, "tokens/total": 14063440, "tokens/train_per_sec_per_gpu": 28.52, "tokens/trainable": 201121} +{"epoch": 1.72265625, "grad_norm": 0.015322903171181679, "learning_rate": 1.476724846845306e-05, "loss": 0.0001199191392515786, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00012, "step": 441, "tokens/total": 14095536, "tokens/train_per_sec_per_gpu": 33.0, "tokens/trainable": 201598} +{"epoch": 1.7265625, "grad_norm": 0.05436247959733009, "learning_rate": 1.463805215420471e-05, "loss": 0.0005605737096630037, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00056, "step": 442, "tokens/total": 14127488, "tokens/train_per_sec_per_gpu": 29.37, "tokens/trainable": 202068} +{"epoch": 1.73046875, "grad_norm": 0.005114950239658356, "learning_rate": 1.451053546535705e-05, "loss": 2.6398556656204164e-05, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.39, "memory/max_allocated (GiB)": 36.39, "ppl": 1.00003, "step": 443, "tokens/total": 14159472, "tokens/train_per_sec_per_gpu": 29.43, "tokens/trainable": 202524} +{"epoch": 1.734375, "grad_norm": 0.000681015953887254, "learning_rate": 1.438470370840001e-05, "loss": 6.228779966477305e-06, "memory/device_reserved (GiB)": 39.15, "memory/max_active (GiB)": 36.6, "memory/max_allocated (GiB)": 36.6, "ppl": 1.00001, "step": 444, "tokens/total": 14191520, "tokens/train_per_sec_per_gpu": 30.27, "tokens/trainable": 202996} +{"epoch": 1.73828125, "grad_norm": 0.03594636172056198, "learning_rate": 1.4260562119706606e-05, "loss": 0.0003221426741220057, "memory/device_reserved (GiB)": 39.16, "memory/max_active (GiB)": 36.57, "memory/max_allocated (GiB)": 36.57, "ppl": 1.00032, "step": 445, "tokens/total": 14221680, "tokens/train_per_sec_per_gpu": 32.89, "tokens/trainable": 203470} +{"epoch": 1.7421875, "grad_norm": 0.16816683113574982, "learning_rate": 1.413811586531508e-05, "loss": 0.0016490641282871366, "memory/device_reserved (GiB)": 39.16, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.00165, "step": 446, "tokens/total": 14253680, "tokens/train_per_sec_per_gpu": 29.37, "tokens/trainable": 203919} +{"epoch": 1.74609375, "grad_norm": 0.0071723381988704205, "learning_rate": 1.4017370040713884e-05, "loss": 6.93315378157422e-05, "memory/device_reserved (GiB)": 39.16, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.00007, "step": 447, "tokens/total": 14285824, "tokens/train_per_sec_per_gpu": 27.58, "tokens/trainable": 204376} +{"epoch": 1.75, "grad_norm": 0.02812999114394188, "learning_rate": 1.3898329670629645e-05, "loss": 0.0001953901955857873, "memory/device_reserved (GiB)": 39.16, "memory/max_active (GiB)": 36.6, "memory/max_allocated (GiB)": 36.6, "ppl": 1.0002, "step": 448, "tokens/total": 14317968, "tokens/train_per_sec_per_gpu": 26.64, "tokens/trainable": 204821} +{"epoch": 1.75390625, "grad_norm": 0.003404787741601467, "learning_rate": 1.3780999708818058e-05, "loss": 7.17312059350661e-06, "memory/device_reserved (GiB)": 39.16, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.00001, "step": 449, "tokens/total": 14350080, "tokens/train_per_sec_per_gpu": 28.3, "tokens/trainable": 205271} +{"epoch": 1.7578125, "grad_norm": 0.0005480491672642529, "learning_rate": 1.3665385037857758e-05, "loss": 4.81248343930929e-06, "memory/device_reserved (GiB)": 37.41, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.0, "step": 450, "tokens/total": 14382240, "tokens/train_per_sec_per_gpu": 25.5, "tokens/trainable": 205723} +{"epoch": 1.76171875, "grad_norm": 0.014193670824170113, "learning_rate": 1.3551490468947126e-05, "loss": 5.97102043684572e-05, "memory/device_reserved (GiB)": 37.63, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00006, "step": 451, "tokens/total": 14414320, "tokens/train_per_sec_per_gpu": 27.12, "tokens/trainable": 206169} +{"epoch": 1.765625, "grad_norm": 0.02763630636036396, "learning_rate": 1.3439320741704075e-05, "loss": 0.00014929058670531958, "memory/device_reserved (GiB)": 37.76, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.00015, "step": 452, "tokens/total": 14446464, "tokens/train_per_sec_per_gpu": 27.52, "tokens/trainable": 206592} +{"epoch": 1.76953125, "grad_norm": 0.010347527451813221, "learning_rate": 1.3328880523968808e-05, "loss": 5.998069536872208e-05, "memory/device_reserved (GiB)": 38.45, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.00006, "step": 453, "tokens/total": 14478624, "tokens/train_per_sec_per_gpu": 27.59, "tokens/trainable": 207055} +{"epoch": 1.7734375, "grad_norm": 0.4689827561378479, "learning_rate": 1.3220174411609587e-05, "loss": 0.0033441532868891954, "memory/device_reserved (GiB)": 38.45, "memory/max_active (GiB)": 36.29, "memory/max_allocated (GiB)": 36.29, "ppl": 1.00335, "step": 454, "tokens/total": 14510192, "tokens/train_per_sec_per_gpu": 28.86, "tokens/trainable": 207494} +{"epoch": 1.77734375, "grad_norm": 0.04199036583304405, "learning_rate": 1.3113206928331471e-05, "loss": 0.0002628727233968675, "memory/device_reserved (GiB)": 38.45, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00026, "step": 455, "tokens/total": 14541920, "tokens/train_per_sec_per_gpu": 27.5, "tokens/trainable": 207956} +{"epoch": 1.78125, "grad_norm": 0.0015001439023762941, "learning_rate": 1.300798252548806e-05, "loss": 1.346761200693436e-05, "memory/device_reserved (GiB)": 38.45, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00001, "step": 456, "tokens/total": 14573904, "tokens/train_per_sec_per_gpu": 29.6, "tokens/trainable": 208408} +{"epoch": 1.78515625, "grad_norm": 0.013151494786143303, "learning_rate": 1.2904505581896265e-05, "loss": 3.6732359149027616e-05, "memory/device_reserved (GiB)": 38.45, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.00004, "step": 457, "tokens/total": 14606032, "tokens/train_per_sec_per_gpu": 29.71, "tokens/trainable": 208889} +{"epoch": 1.7890625, "grad_norm": 0.0006910350639373064, "learning_rate": 1.2802780403654082e-05, "loss": 1.0416523764433805e-05, "memory/device_reserved (GiB)": 38.45, "memory/max_active (GiB)": 36.34, "memory/max_allocated (GiB)": 36.34, "ppl": 1.00001, "step": 458, "tokens/total": 14637728, "tokens/train_per_sec_per_gpu": 28.36, "tokens/trainable": 209336} +{"epoch": 1.79296875, "grad_norm": 0.003835468553006649, "learning_rate": 1.2702811223961408e-05, "loss": 3.105392897850834e-05, "memory/device_reserved (GiB)": 38.63, "memory/max_active (GiB)": 36.6, "memory/max_allocated (GiB)": 36.6, "ppl": 1.00003, "step": 459, "tokens/total": 14669984, "tokens/train_per_sec_per_gpu": 27.49, "tokens/trainable": 209797} +{"epoch": 1.796875, "grad_norm": 0.0003044170734938234, "learning_rate": 1.2604602202943861e-05, "loss": 6.697504431940615e-06, "memory/device_reserved (GiB)": 38.67, "memory/max_active (GiB)": 36.57, "memory/max_allocated (GiB)": 36.57, "ppl": 1.00001, "step": 460, "tokens/total": 14702064, "tokens/train_per_sec_per_gpu": 30.05, "tokens/trainable": 210240} +{"epoch": 1.80078125, "grad_norm": 0.01640794239938259, "learning_rate": 1.2508157427479686e-05, "loss": 7.829760579625145e-05, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.00008, "step": 461, "tokens/total": 14734144, "tokens/train_per_sec_per_gpu": 29.07, "tokens/trainable": 210702} +{"epoch": 1.8046875, "grad_norm": 0.04974092170596123, "learning_rate": 1.2413480911029655e-05, "loss": 0.00036814718623645604, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.00037, "step": 462, "tokens/total": 14766416, "tokens/train_per_sec_per_gpu": 26.18, "tokens/trainable": 211140} +{"epoch": 1.80859375, "grad_norm": 0.0007990560261532664, "learning_rate": 1.2320576593470082e-05, "loss": 7.1813856266089715e-06, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.36, "memory/max_allocated (GiB)": 36.36, "ppl": 1.00001, "step": 463, "tokens/total": 14798224, "tokens/train_per_sec_per_gpu": 27.29, "tokens/trainable": 211575} +{"epoch": 1.8125, "grad_norm": 0.08440390229225159, "learning_rate": 1.2229448340928828e-05, "loss": 0.00048587226774543524, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.00049, "step": 464, "tokens/total": 14830336, "tokens/train_per_sec_per_gpu": 32.34, "tokens/trainable": 212083} +{"epoch": 1.81640625, "grad_norm": 0.0018794891657307744, "learning_rate": 1.2140099945624458e-05, "loss": 1.08363010440371e-05, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00001, "step": 465, "tokens/total": 14862560, "tokens/train_per_sec_per_gpu": 30.7, "tokens/trainable": 212558} +{"epoch": 1.8203125, "grad_norm": 0.008357501588761806, "learning_rate": 1.205253512570841e-05, "loss": 6.44918909529224e-05, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00006, "step": 466, "tokens/total": 14894432, "tokens/train_per_sec_per_gpu": 27.29, "tokens/trainable": 213014} +{"epoch": 1.82421875, "grad_norm": 0.003019345458596945, "learning_rate": 1.1966757525110255e-05, "loss": 2.4649680199217983e-05, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.59, "memory/max_allocated (GiB)": 36.59, "ppl": 1.00002, "step": 467, "tokens/total": 14926624, "tokens/train_per_sec_per_gpu": 28.39, "tokens/trainable": 213465} +{"epoch": 1.828125, "grad_norm": 0.0832735076546669, "learning_rate": 1.1882770713386095e-05, "loss": 0.0008976737735792994, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.0009, "step": 468, "tokens/total": 14958640, "tokens/train_per_sec_per_gpu": 30.58, "tokens/trainable": 213940} +{"epoch": 1.83203125, "grad_norm": 0.0038668843917548656, "learning_rate": 1.180057818556998e-05, "loss": 1.873084511316847e-05, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.00002, "step": 469, "tokens/total": 14990864, "tokens/train_per_sec_per_gpu": 29.42, "tokens/trainable": 214415} +{"epoch": 1.8359375, "grad_norm": 0.0006158083560876548, "learning_rate": 1.1720183362028494e-05, "loss": 5.440972017822787e-06, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.49, "memory/max_allocated (GiB)": 36.49, "ppl": 1.00001, "step": 470, "tokens/total": 15022800, "tokens/train_per_sec_per_gpu": 28.77, "tokens/trainable": 214880} +{"epoch": 1.83984375, "grad_norm": 0.0015613064169883728, "learning_rate": 1.1641589588318387e-05, "loss": 1.487641657149652e-05, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.00001, "step": 471, "tokens/total": 15054784, "tokens/train_per_sec_per_gpu": 27.81, "tokens/trainable": 215334} +{"epoch": 1.84375, "grad_norm": 0.00033517941483296454, "learning_rate": 1.1564800135047418e-05, "loss": 4.4488651838037185e-06, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.54, "memory/max_allocated (GiB)": 36.54, "ppl": 1.0, "step": 472, "tokens/total": 15086800, "tokens/train_per_sec_per_gpu": 29.18, "tokens/trainable": 215822} +{"epoch": 1.84765625, "grad_norm": 0.014701228588819504, "learning_rate": 1.148981819773816e-05, "loss": 0.00012692881864495575, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00013, "step": 473, "tokens/total": 15118736, "tokens/train_per_sec_per_gpu": 25.65, "tokens/trainable": 216239} +{"epoch": 1.8515625, "grad_norm": 0.03242962434887886, "learning_rate": 1.1416646896695086e-05, "loss": 0.00016460703045595437, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.00016, "step": 474, "tokens/total": 15150896, "tokens/train_per_sec_per_gpu": 27.53, "tokens/trainable": 216691} +{"epoch": 1.85546875, "grad_norm": 0.022134000435471535, "learning_rate": 1.1345289276874717e-05, "loss": 0.00015468102355953306, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.53, "memory/max_allocated (GiB)": 36.53, "ppl": 1.00015, "step": 475, "tokens/total": 15183120, "tokens/train_per_sec_per_gpu": 27.3, "tokens/trainable": 217162} +{"epoch": 1.859375, "grad_norm": 0.00101348920725286, "learning_rate": 1.1275748307758873e-05, "loss": 1.263963622477604e-05, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.00001, "step": 476, "tokens/total": 15215216, "tokens/train_per_sec_per_gpu": 30.11, "tokens/trainable": 217638} +{"epoch": 1.86328125, "grad_norm": 0.09451806545257568, "learning_rate": 1.1208026883231147e-05, "loss": 0.0005900258547626436, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.38, "memory/max_allocated (GiB)": 36.38, "ppl": 1.00059, "step": 477, "tokens/total": 15247328, "tokens/train_per_sec_per_gpu": 27.45, "tokens/trainable": 218053} +{"epoch": 1.8671875, "grad_norm": 0.0023683386389166117, "learning_rate": 1.1142127821456433e-05, "loss": 1.8595857909531333e-05, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00002, "step": 478, "tokens/total": 15279472, "tokens/train_per_sec_per_gpu": 28.42, "tokens/trainable": 218503} +{"epoch": 1.87109375, "grad_norm": 0.37432360649108887, "learning_rate": 1.1078053864763674e-05, "loss": 0.0017199859721586108, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.57, "memory/max_allocated (GiB)": 36.57, "ppl": 1.00172, "step": 479, "tokens/total": 15311536, "tokens/train_per_sec_per_gpu": 30.18, "tokens/trainable": 218964} +{"epoch": 1.875, "grad_norm": 0.006793392356485128, "learning_rate": 1.1015807679531756e-05, "loss": 2.6950599931296892e-05, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.38, "memory/max_allocated (GiB)": 36.38, "ppl": 1.00003, "step": 480, "tokens/total": 15343360, "tokens/train_per_sec_per_gpu": 29.65, "tokens/trainable": 219412} +{"epoch": 1.87890625, "grad_norm": 0.07207302004098892, "learning_rate": 1.0955391856078528e-05, "loss": 0.0004579552623908967, "memory/device_reserved (GiB)": 38.74, "memory/max_active (GiB)": 36.35, "memory/max_allocated (GiB)": 36.35, "ppl": 1.00046, "step": 481, "tokens/total": 15375008, "tokens/train_per_sec_per_gpu": 32.21, "tokens/trainable": 219886} +{"epoch": 1.8828125, "grad_norm": 0.0035383424255996943, "learning_rate": 1.0896808908553007e-05, "loss": 2.970332934637554e-05, "memory/device_reserved (GiB)": 37.51, "memory/max_active (GiB)": 36.33, "memory/max_allocated (GiB)": 36.33, "ppl": 1.00003, "step": 482, "tokens/total": 15406816, "tokens/train_per_sec_per_gpu": 28.98, "tokens/trainable": 220343} +{"epoch": 1.88671875, "grad_norm": 0.005184710957109928, "learning_rate": 1.0840061274830763e-05, "loss": 3.948260928154923e-05, "memory/device_reserved (GiB)": 38.63, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.00004, "step": 483, "tokens/total": 15438816, "tokens/train_per_sec_per_gpu": 29.31, "tokens/trainable": 220825} +{"epoch": 1.890625, "grad_norm": 0.0017558662220835686, "learning_rate": 1.0785151316412473e-05, "loss": 1.1586276741581969e-05, "memory/device_reserved (GiB)": 38.63, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00001, "step": 484, "tokens/total": 15471008, "tokens/train_per_sec_per_gpu": 27.13, "tokens/trainable": 221282} +{"epoch": 1.89453125, "grad_norm": 0.0002366721018915996, "learning_rate": 1.0732081318325639e-05, "loss": 3.967344127886463e-06, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.0, "step": 485, "tokens/total": 15503248, "tokens/train_per_sec_per_gpu": 31.3, "tokens/trainable": 221780} +{"epoch": 1.8984375, "grad_norm": 0.18045902252197266, "learning_rate": 1.0680853489029501e-05, "loss": 0.000754694570787251, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.00075, "step": 486, "tokens/total": 15535504, "tokens/train_per_sec_per_gpu": 26.37, "tokens/trainable": 222210} +{"epoch": 1.90234375, "grad_norm": 0.0010426960652694106, "learning_rate": 1.0631469960323152e-05, "loss": 1.0654634934326168e-05, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.41, "memory/max_allocated (GiB)": 36.41, "ppl": 1.00001, "step": 487, "tokens/total": 15567568, "tokens/train_per_sec_per_gpu": 24.42, "tokens/trainable": 222646} +{"epoch": 1.90625, "grad_norm": 0.018443502485752106, "learning_rate": 1.0583932787256783e-05, "loss": 0.0001445821108063683, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.42, "memory/max_allocated (GiB)": 36.42, "ppl": 1.00014, "step": 488, "tokens/total": 15599456, "tokens/train_per_sec_per_gpu": 27.75, "tokens/trainable": 223066} +{"epoch": 1.91015625, "grad_norm": 0.08126334846019745, "learning_rate": 1.0538243948046206e-05, "loss": 0.00040054236887954175, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.4, "memory/max_allocated (GiB)": 36.4, "ppl": 1.0004, "step": 489, "tokens/total": 15631168, "tokens/train_per_sec_per_gpu": 28.8, "tokens/trainable": 223512} +{"epoch": 1.9140625, "grad_norm": 0.01388081070035696, "learning_rate": 1.0494405343990523e-05, "loss": 5.4693471611244604e-05, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.00005, "step": 490, "tokens/total": 15661104, "tokens/train_per_sec_per_gpu": 32.49, "tokens/trainable": 223956} +{"epoch": 1.91796875, "grad_norm": 0.014524613507091999, "learning_rate": 1.0452418799392985e-05, "loss": 9.905001206789166e-05, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.09, "memory/max_allocated (GiB)": 36.09, "ppl": 1.0001, "step": 491, "tokens/total": 15691120, "tokens/train_per_sec_per_gpu": 30.04, "tokens/trainable": 224416} +{"epoch": 1.921875, "grad_norm": 0.016004936769604683, "learning_rate": 1.0412286061485102e-05, "loss": 7.77841269155033e-05, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.00008, "step": 492, "tokens/total": 15723104, "tokens/train_per_sec_per_gpu": 28.42, "tokens/trainable": 224883} +{"epoch": 1.92578125, "grad_norm": 0.00028383161406964064, "learning_rate": 1.03740088003539e-05, "loss": 4.6946679503889754e-06, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.0, "step": 493, "tokens/total": 15754960, "tokens/train_per_sec_per_gpu": 29.61, "tokens/trainable": 225324} +{"epoch": 1.9296875, "grad_norm": 0.003787089604884386, "learning_rate": 1.0337588608872463e-05, "loss": 2.9438704586937092e-05, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.58, "memory/max_allocated (GiB)": 36.58, "ppl": 1.00003, "step": 494, "tokens/total": 15787008, "tokens/train_per_sec_per_gpu": 29.03, "tokens/trainable": 225783} +{"epoch": 1.93359375, "grad_norm": 0.0024508354254066944, "learning_rate": 1.0303027002633622e-05, "loss": 1.2276758752705064e-05, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.00001, "step": 495, "tokens/total": 15818896, "tokens/train_per_sec_per_gpu": 30.48, "tokens/trainable": 226281} +{"epoch": 1.9375, "grad_norm": 0.0002896689693443477, "learning_rate": 1.0270325419886884e-05, "loss": 4.6574341467930935e-06, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.0, "step": 496, "tokens/total": 15850624, "tokens/train_per_sec_per_gpu": 30.32, "tokens/trainable": 226735} +{"epoch": 1.94140625, "grad_norm": 0.002518043853342533, "learning_rate": 1.0239485221478599e-05, "loss": 2.2658114176010713e-05, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.62, "memory/max_allocated (GiB)": 36.62, "ppl": 1.00002, "step": 497, "tokens/total": 15882752, "tokens/train_per_sec_per_gpu": 29.98, "tokens/trainable": 227198} +{"epoch": 1.9453125, "grad_norm": 0.00016003627388272434, "learning_rate": 1.0210507690795292e-05, "loss": 3.791648168771644e-06, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.27, "memory/max_allocated (GiB)": 36.27, "ppl": 1.0, "step": 498, "tokens/total": 15914416, "tokens/train_per_sec_per_gpu": 25.94, "tokens/trainable": 227623} +{"epoch": 1.94921875, "grad_norm": 0.00043029585503973067, "learning_rate": 1.0183394033710305e-05, "loss": 6.0818256315542385e-06, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.00001, "step": 499, "tokens/total": 15946320, "tokens/train_per_sec_per_gpu": 26.99, "tokens/trainable": 228074} +{"epoch": 1.953125, "grad_norm": 0.03700384125113487, "learning_rate": 1.0158145378533583e-05, "loss": 0.00016803065955173224, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.00017, "step": 500, "tokens/total": 15978384, "tokens/train_per_sec_per_gpu": 29.58, "tokens/trainable": 228541} +{"epoch": 1.95703125, "grad_norm": 0.0032133073545992374, "learning_rate": 1.0134762775964726e-05, "loss": 7.3915689426939934e-06, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.62, "memory/max_allocated (GiB)": 36.62, "ppl": 1.00001, "step": 501, "tokens/total": 16010672, "tokens/train_per_sec_per_gpu": 31.11, "tokens/trainable": 229031} +{"epoch": 1.9609375, "grad_norm": 0.00048456323565915227, "learning_rate": 1.0113247199049278e-05, "loss": 7.050684871501289e-06, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.45, "memory/max_allocated (GiB)": 36.45, "ppl": 1.00001, "step": 502, "tokens/total": 16042672, "tokens/train_per_sec_per_gpu": 32.29, "tokens/trainable": 229512} +{"epoch": 1.96484375, "grad_norm": 0.0932794138789177, "learning_rate": 1.0093599543138205e-05, "loss": 0.0006240660441108048, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00062, "step": 503, "tokens/total": 16074592, "tokens/train_per_sec_per_gpu": 28.41, "tokens/trainable": 229989} +{"epoch": 1.96875, "grad_norm": 0.012090178206562996, "learning_rate": 1.0075820625850675e-05, "loss": 8.021057874429971e-05, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.61, "memory/max_allocated (GiB)": 36.61, "ppl": 1.00008, "step": 504, "tokens/total": 16106944, "tokens/train_per_sec_per_gpu": 29.66, "tokens/trainable": 230447} +{"epoch": 1.97265625, "grad_norm": 0.010548670776188374, "learning_rate": 1.0059911187040013e-05, "loss": 1.3973164641356561e-05, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.44, "memory/max_allocated (GiB)": 36.44, "ppl": 1.00001, "step": 505, "tokens/total": 16138800, "tokens/train_per_sec_per_gpu": 28.63, "tokens/trainable": 230879} +{"epoch": 1.9765625, "grad_norm": 0.004747380968183279, "learning_rate": 1.0045871888762893e-05, "loss": 4.118381184525788e-05, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.55, "memory/max_allocated (GiB)": 36.55, "ppl": 1.00004, "step": 506, "tokens/total": 16170976, "tokens/train_per_sec_per_gpu": 28.35, "tokens/trainable": 231348} +{"epoch": 1.98046875, "grad_norm": 0.0008630049414932728, "learning_rate": 1.003370331525184e-05, "loss": 7.490401458198903e-06, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.46, "memory/max_allocated (GiB)": 36.46, "ppl": 1.00001, "step": 507, "tokens/total": 16202976, "tokens/train_per_sec_per_gpu": 29.33, "tokens/trainable": 231819} +{"epoch": 1.984375, "grad_norm": 0.0002915443910751492, "learning_rate": 1.002340597289085e-05, "loss": 5.796592631668318e-06, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.52, "memory/max_allocated (GiB)": 36.52, "ppl": 1.00001, "step": 508, "tokens/total": 16234816, "tokens/train_per_sec_per_gpu": 28.79, "tokens/trainable": 232269} +{"epoch": 1.98828125, "grad_norm": 0.00034642108948901296, "learning_rate": 1.0014980290194387e-05, "loss": 3.7650847843906377e-06, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.51, "memory/max_allocated (GiB)": 36.51, "ppl": 1.0, "step": 509, "tokens/total": 16266960, "tokens/train_per_sec_per_gpu": 28.22, "tokens/trainable": 232716} +{"epoch": 1.9921875, "grad_norm": 0.17613324522972107, "learning_rate": 1.0008426617789489e-05, "loss": 0.0010949716670438647, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.48, "memory/max_allocated (GiB)": 36.48, "ppl": 1.0011, "step": 510, "tokens/total": 16298912, "tokens/train_per_sec_per_gpu": 30.16, "tokens/trainable": 233178} +{"epoch": 1.99609375, "grad_norm": 0.00045643217163160443, "learning_rate": 1.0003745228401215e-05, "loss": 5.581615369010251e-06, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.47, "memory/max_allocated (GiB)": 36.47, "ppl": 1.00001, "step": 511, "tokens/total": 16330944, "tokens/train_per_sec_per_gpu": 29.97, "tokens/trainable": 233647} +{"epoch": 2.0, "grad_norm": 0.05987092852592468, "learning_rate": 1.0000936316841296e-05, "loss": 0.00014668369840364903, "memory/device_reserved (GiB)": 39.0, "memory/max_active (GiB)": 36.26, "memory/max_allocated (GiB)": 36.26, "ppl": 1.00015, "step": 512, "tokens/total": 16362544, "tokens/train_per_sec_per_gpu": 24.67, "tokens/trainable": 234060}